gitnexus 1.6.6-rc.103 → 1.6.6-rc.104
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/core/ingestion/call-processor.js +42 -6
- package/dist/core/ingestion/pipeline-phases/parse-impl.d.ts +23 -0
- package/dist/core/ingestion/pipeline-phases/parse-impl.js +88 -16
- package/dist/core/ingestion/utils/deferred-resolution-profile.d.ts +10 -0
- package/dist/core/ingestion/utils/deferred-resolution-profile.js +30 -0
- package/dist/core/ingestion/workers/parse-worker.js +28 -1
- package/dist/core/ingestion/workers/worker-pool.d.ts +38 -3
- package/dist/core/ingestion/workers/worker-pool.js +282 -43
- package/dist/core/run-analyze.js +4 -1
- package/package.json +1 -1
|
@@ -28,7 +28,7 @@ import { generateId } from '../../lib/utils.js';
|
|
|
28
28
|
import { getLanguageFromFilename, SupportedLanguages } from '../../_shared/index.js';
|
|
29
29
|
import { isRegistryPrimary } from './registry-primary-flag.js';
|
|
30
30
|
import { isVerboseIngestionEnabled } from './utils/verbose.js';
|
|
31
|
-
import { deferredCallFileSlowMs, deferredCallLogEveryN, getDeferredProfileDroppedCount, isDeferredResolutionProfileEnabled, logDeferredProfile, profileElapsedMs, resetDeferredProfileDroppedCount, startTimer, } from './utils/deferred-resolution-profile.js';
|
|
31
|
+
import { ALWAYS_ON_SLOW_FILE_WARN_THROTTLE_MS, alwaysOnSlowFileWarnMs, deferredCallFileSlowMs, deferredCallLogEveryN, getDeferredProfileDroppedCount, isDeferredResolutionProfileEnabled, logDeferredProfile, profileElapsedMs, resetDeferredProfileDroppedCount, startTimer, } from './utils/deferred-resolution-profile.js';
|
|
32
32
|
import { yieldToEventLoop } from './utils/event-loop.js';
|
|
33
33
|
import { parseSourceSafe } from '../tree-sitter/safe-parse.js';
|
|
34
34
|
import { CLASS_CONTAINER_TYPES, FUNCTION_NODE_TYPES, findEnclosingClassInfo, genericFuncName, inferFunctionLabel, } from './utils/ast-helpers.js';
|
|
@@ -2303,6 +2303,14 @@ export const processCallsFromExtracted = async (graph, extractedCalls, ctx, onPr
|
|
|
2303
2303
|
const slowFileMs = profileCalls ? deferredCallFileSlowMs() : 0;
|
|
2304
2304
|
const logEveryN = profileCalls ? deferredCallLogEveryN() : 0;
|
|
2305
2305
|
let skippedRegistryPrimaryFiles = 0;
|
|
2306
|
+
// Always-on slow-file watchdog (#1741). Independent of the verbose/profile
|
|
2307
|
+
// gate above: even a plain `analyze` run surfaces ONE actionable warning
|
|
2308
|
+
// when a single file's call resolution is pathologically slow — turning the
|
|
2309
|
+
// silent "stuck at Resolving calls (N/M)" symptom into a named culprit.
|
|
2310
|
+
// Throttled so a genuinely slow repo can't produce a warn storm.
|
|
2311
|
+
const alwaysSlowFileMs = alwaysOnSlowFileWarnMs();
|
|
2312
|
+
let lastSlowFileWarnAt = 0;
|
|
2313
|
+
let suppressedSlowFileWarnings = 0;
|
|
2306
2314
|
// Fresh dropped-log counter per analyze run — the module-private counter
|
|
2307
2315
|
// in deferred-resolution-profile.ts is process-lived, so without a reset
|
|
2308
2316
|
// here it would accumulate across consecutive analyze invocations in the
|
|
@@ -2314,12 +2322,16 @@ export const processCallsFromExtracted = async (graph, extractedCalls, ctx, onPr
|
|
|
2314
2322
|
// denominator stays stable as the loop iterates. Otherwise `${totalFiles -
|
|
2315
2323
|
// skippedRegistryPrimaryFiles}` drifts upward — files iterated before later
|
|
2316
2324
|
// registry-primary skips have been seen carry an inflated denominator, and
|
|
2317
|
-
// the ratio only self-corrects after every file has been classified.
|
|
2318
|
-
//
|
|
2319
|
-
//
|
|
2320
|
-
//
|
|
2325
|
+
// the ratio only self-corrects after every file has been classified.
|
|
2326
|
+
//
|
|
2327
|
+
// Runs whenever its result will actually be read: on the profile path (the
|
|
2328
|
+
// live deferred-profile log) OR when the always-on slow-file watchdog is
|
|
2329
|
+
// active (#1741) — the watchdog's warning prints `${resolvedFiles}/${resolvedTotal}`
|
|
2330
|
+
// unconditionally, so leaving resolvedTotal at 0 on a plain run produced a
|
|
2331
|
+
// bogus "Resolved N/0 files" denominator on exactly the unprofiled runs the
|
|
2332
|
+
// watchdog exists for. When both gates are off, skip the extra Map pass.
|
|
2321
2333
|
let resolvedTotal = 0;
|
|
2322
|
-
if (profileCalls) {
|
|
2334
|
+
if (profileCalls || alwaysSlowFileMs > 0) {
|
|
2323
2335
|
for (const filePath of byFile.keys()) {
|
|
2324
2336
|
const lang = getLanguageFromFilename(filePath);
|
|
2325
2337
|
if (!lang || !isRegistryPrimary(lang))
|
|
@@ -2341,6 +2353,9 @@ export const processCallsFromExtracted = async (graph, extractedCalls, ctx, onPr
|
|
|
2341
2353
|
}
|
|
2342
2354
|
resolvedFiles++;
|
|
2343
2355
|
const tFile = startTimer(profileCalls);
|
|
2356
|
+
// Always-on timer (cheap: one hrtime read) feeding the slow-file watchdog
|
|
2357
|
+
// below. Distinct from `tFile`, which is null unless profiling is on.
|
|
2358
|
+
const tFileAlways = alwaysSlowFileMs > 0 ? process.hrtime.bigint() : null;
|
|
2344
2359
|
if (profileCalls && (resolvedFiles === 1 || resolvedFiles % logEveryN === 0)) {
|
|
2345
2360
|
logDeferredProfile(`calls ${resolvedFiles}/${resolvedTotal} file=${filePath} sites=${calls.length}`);
|
|
2346
2361
|
}
|
|
@@ -2455,6 +2470,27 @@ export const processCallsFromExtracted = async (graph, extractedCalls, ctx, onPr
|
|
|
2455
2470
|
logDeferredProfile(`slow file ${elapsed.toFixed(0)}ms path=${filePath} calls=${calls.length} lang=${fileLanguage ?? 'unknown'}`);
|
|
2456
2471
|
}
|
|
2457
2472
|
}
|
|
2473
|
+
// Always-on slow-file watchdog (#1741) — fires regardless of verbose.
|
|
2474
|
+
if (tFileAlways !== null) {
|
|
2475
|
+
const elapsedAlways = profileElapsedMs(tFileAlways);
|
|
2476
|
+
if (elapsedAlways >= alwaysSlowFileMs) {
|
|
2477
|
+
const now = Date.now();
|
|
2478
|
+
if (now - lastSlowFileWarnAt >= ALWAYS_ON_SLOW_FILE_WARN_THROTTLE_MS) {
|
|
2479
|
+
lastSlowFileWarnAt = now;
|
|
2480
|
+
const suppressedNote = suppressedSlowFileWarnings > 0
|
|
2481
|
+
? ` (+${suppressedSlowFileWarnings} more slow files since the last warning)`
|
|
2482
|
+
: '';
|
|
2483
|
+
logger.warn(`⏳ Call resolution for ${filePath} took ${(elapsedAlways / 1000).toFixed(1)}s ` +
|
|
2484
|
+
`(${calls.length} call sites, ${fileLanguage ?? 'unknown'}). The run is not frozen — ` +
|
|
2485
|
+
`this file is unusually expensive to resolve. Resolved ${resolvedFiles}/${resolvedTotal} ` +
|
|
2486
|
+
`files so far.${suppressedNote} Pass -v for per-file deferred-resolution timing.`);
|
|
2487
|
+
suppressedSlowFileWarnings = 0;
|
|
2488
|
+
}
|
|
2489
|
+
else {
|
|
2490
|
+
suppressedSlowFileWarnings++;
|
|
2491
|
+
}
|
|
2492
|
+
}
|
|
2493
|
+
}
|
|
2458
2494
|
}
|
|
2459
2495
|
if (profileCalls) {
|
|
2460
2496
|
logDeferredProfile(`processCallsFromExtracted done: ${totalFiles} files, ${extractedCalls.length} call sites, skipped registry-primary files=${skippedRegistryPrimaryFiles}`);
|
|
@@ -24,6 +24,29 @@ type ScannedFile = {
|
|
|
24
24
|
size: number;
|
|
25
25
|
};
|
|
26
26
|
type ProgressFn = (progress: PipelineProgress) => void;
|
|
27
|
+
/**
|
|
28
|
+
* Handle a worker-pool startup failure by FAILING FAST with the captured cause
|
|
29
|
+
* (#1741). The pool self-heals *transient* worker crashes on its own — a
|
|
30
|
+
* bounded, jittered startup restart loop (see worker-pool.ts) — so this is
|
|
31
|
+
* reached only when that self-heal is EXHAUSTED, or a deterministic crash-loop
|
|
32
|
+
* was detected, or the pool could not even be constructed. In every such case
|
|
33
|
+
* the workers genuinely cannot start.
|
|
34
|
+
*
|
|
35
|
+
* Rather than silently degrade to the ~10× slower sequential parser — which
|
|
36
|
+
* masked a worker-startup regression as a 2-hour "stuck" run in #1741 (rc99:
|
|
37
|
+
* the failure was a dropped `logger.warn` and an unbounded sequential grind) —
|
|
38
|
+
* GitNexus surfaces the real crash and aborts. An operator who genuinely wants
|
|
39
|
+
* sequential parsing asks for it explicitly with `--workers 0`.
|
|
40
|
+
*
|
|
41
|
+
* The decision is automatic: NO `--allow-sequential-fallback` or pool-sizing
|
|
42
|
+
* flag participates. The pool's own crash classification (`crashClass` on
|
|
43
|
+
* WorkerPoolInitializationError) only sharpens the message.
|
|
44
|
+
*
|
|
45
|
+
* @throws always — an actionable Error carrying the captured worker crash.
|
|
46
|
+
* @internal Exported for unit tests; production callers are the parse loop's
|
|
47
|
+
* two worker-startup catch sites below.
|
|
48
|
+
*/
|
|
49
|
+
export declare function handleWorkerStartupFailure(err: Error): never;
|
|
27
50
|
/**
|
|
28
51
|
* Chunked parse + resolve loop.
|
|
29
52
|
*
|
|
@@ -25,7 +25,7 @@ import { getLanguageFromFilename } from '../../../_shared/index.js';
|
|
|
25
25
|
import { isRegistryPrimary } from '../registry-primary-flag.js';
|
|
26
26
|
import { readFileContents } from '../filesystem-walker.js';
|
|
27
27
|
import { isLanguageAvailable } from '../../tree-sitter/parser-loader.js';
|
|
28
|
-
import { createWorkerPool, WorkerPoolInitializationError } from '../workers/worker-pool.js';
|
|
28
|
+
import { createWorkerPool, workerPoolDisabledByEnv, WorkerPoolInitializationError, } from '../workers/worker-pool.js';
|
|
29
29
|
import { extractFetchCallsFromFiles } from '../call-processor.js';
|
|
30
30
|
import fs from 'node:fs';
|
|
31
31
|
import path from 'node:path';
|
|
@@ -66,6 +66,64 @@ function resolveChunkByteBudget(options) {
|
|
|
66
66
|
return env;
|
|
67
67
|
return DEFAULT_CHUNK_BYTE_BUDGET;
|
|
68
68
|
}
|
|
69
|
+
/**
|
|
70
|
+
* Handle a worker-pool startup failure by FAILING FAST with the captured cause
|
|
71
|
+
* (#1741). The pool self-heals *transient* worker crashes on its own — a
|
|
72
|
+
* bounded, jittered startup restart loop (see worker-pool.ts) — so this is
|
|
73
|
+
* reached only when that self-heal is EXHAUSTED, or a deterministic crash-loop
|
|
74
|
+
* was detected, or the pool could not even be constructed. In every such case
|
|
75
|
+
* the workers genuinely cannot start.
|
|
76
|
+
*
|
|
77
|
+
* Rather than silently degrade to the ~10× slower sequential parser — which
|
|
78
|
+
* masked a worker-startup regression as a 2-hour "stuck" run in #1741 (rc99:
|
|
79
|
+
* the failure was a dropped `logger.warn` and an unbounded sequential grind) —
|
|
80
|
+
* GitNexus surfaces the real crash and aborts. An operator who genuinely wants
|
|
81
|
+
* sequential parsing asks for it explicitly with `--workers 0`.
|
|
82
|
+
*
|
|
83
|
+
* The decision is automatic: NO `--allow-sequential-fallback` or pool-sizing
|
|
84
|
+
* flag participates. The pool's own crash classification (`crashClass` on
|
|
85
|
+
* WorkerPoolInitializationError) only sharpens the message.
|
|
86
|
+
*
|
|
87
|
+
* @throws always — an actionable Error carrying the captured worker crash.
|
|
88
|
+
* @internal Exported for unit tests; production callers are the parse loop's
|
|
89
|
+
* two worker-startup catch sites below.
|
|
90
|
+
*/
|
|
91
|
+
export function handleWorkerStartupFailure(err) {
|
|
92
|
+
const isInit = err instanceof WorkerPoolInitializationError;
|
|
93
|
+
const readinessFailures = isInit ? err.readinessFailures : [];
|
|
94
|
+
const crashClass = isInit ? err.crashClass : undefined;
|
|
95
|
+
// Surface the real cause verbatim: readiness failures for an init crash, or
|
|
96
|
+
// the construction error message (e.g. "Worker script not found: …") when the
|
|
97
|
+
// pool never got to spawn workers.
|
|
98
|
+
const failureDetail = readinessFailures.length > 0
|
|
99
|
+
? ` Underlying worker failure(s): ${readinessFailures.join(' | ')}`
|
|
100
|
+
: isInit
|
|
101
|
+
? ''
|
|
102
|
+
: ` Underlying error: ${err.message}`;
|
|
103
|
+
// Always surface the real crash — never let a startup failure pass silently.
|
|
104
|
+
logger.error({ err: err.message, readinessFailures, crashClass }, 'Worker pool failed to start — workers could not start (bounded self-heal exhausted).');
|
|
105
|
+
const cause = crashClass === 'deterministic-startup'
|
|
106
|
+
? `every worker crashed identically during startup (a deterministic ` +
|
|
107
|
+
`crash-loop — retrying cannot help), so the pool has no usable workers.`
|
|
108
|
+
: isInit
|
|
109
|
+
? `workers exhausted the bounded startup retry budget without reporting ` +
|
|
110
|
+
`ready, so the pool has no usable workers.`
|
|
111
|
+
: `the worker pool could not be constructed.`;
|
|
112
|
+
// Class-aware fix hint: a missing/broken native binding is the likely cause
|
|
113
|
+
// when workers crashed during init, but it is the WRONG guess for a pool that
|
|
114
|
+
// never constructed (commonly a missing build / unresolvable worker path).
|
|
115
|
+
const fixHint = isInit
|
|
116
|
+
? `Fix the worker startup failure shown above (often a missing/broken native ` +
|
|
117
|
+
`binding or a top-of-script import error in parse-worker).`
|
|
118
|
+
: `Fix the worker pool construction error shown above (commonly a missing ` +
|
|
119
|
+
`build, so dist/ has no parse-worker, or an unresolvable worker path).`;
|
|
120
|
+
throw new Error(`Worker pool failed to start: ${cause}${failureDetail}\n\n` +
|
|
121
|
+
`GitNexus will NOT silently fall back to the (much slower) sequential ` +
|
|
122
|
+
`parser and hide this crash — that masked a worker-startup regression as ` +
|
|
123
|
+
`a 2-hour "stuck" run in #1741. Options:\n` +
|
|
124
|
+
` • ${fixHint}\n` +
|
|
125
|
+
` • Re-run with --workers 0 to parse sequentially without the worker pool.`);
|
|
126
|
+
}
|
|
69
127
|
/**
|
|
70
128
|
* Chunked parse + resolve loop.
|
|
71
129
|
*
|
|
@@ -168,13 +226,26 @@ export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, t
|
|
|
168
226
|
// intentionally NOT created before parse-cache lookup: a warm-cache
|
|
169
227
|
// all-hit run should replay cached worker output without loading
|
|
170
228
|
// parse-worker.js or any tree-sitter/N-API native bindings.
|
|
229
|
+
// `--workers 0` (workerPoolSize === 0) and `GITNEXUS_WORKER_POOL_SIZE=0` both
|
|
230
|
+
// mean "no pool, parse sequentially". The env channel is consulted ONLY when
|
|
231
|
+
// no explicit `--workers <N>` was given, so an explicit positive size always
|
|
232
|
+
// wins over an ambient env=0 (#1741). Without this, env=0 built a size-0 pool
|
|
233
|
+
// that failed fast with a fabricated "retry budget exhausted" crash.
|
|
234
|
+
const envDisablesWorkers = options?.workerPoolSize === undefined && workerPoolDisabledByEnv();
|
|
235
|
+
const meetsWorkerThreshold = totalParseable >= MIN_FILES_FOR_WORKERS || totalBytes >= MIN_BYTES_FOR_WORKERS;
|
|
236
|
+
// Log only when env=0 actually skips a pool we'd otherwise have used, so the
|
|
237
|
+
// undocumented (possibly accidental) env=0 case is observable instead of a
|
|
238
|
+
// silent degrade — small repos go sequential anyway and need no notice.
|
|
239
|
+
if (envDisablesWorkers && meetsWorkerThreshold) {
|
|
240
|
+
logger.warn('GITNEXUS_WORKER_POOL_SIZE=0 → parsing sequentially; unset it or pass --workers <N> to use the worker pool.');
|
|
241
|
+
}
|
|
171
242
|
const shouldUseWorkers = !options?.skipWorkers &&
|
|
172
243
|
options?.workerPoolSize !== 0 &&
|
|
173
|
-
|
|
244
|
+
!envDisablesWorkers &&
|
|
245
|
+
meetsWorkerThreshold;
|
|
174
246
|
let workerPool;
|
|
175
|
-
let workerPoolDisabled = false;
|
|
176
247
|
const getOrCreateWorkerPool = () => {
|
|
177
|
-
if (!shouldUseWorkers
|
|
248
|
+
if (!shouldUseWorkers)
|
|
178
249
|
return undefined;
|
|
179
250
|
if (workerPool)
|
|
180
251
|
return workerPool;
|
|
@@ -199,9 +270,11 @@ export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, t
|
|
|
199
270
|
return workerPool;
|
|
200
271
|
}
|
|
201
272
|
catch (err) {
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
273
|
+
// Pool *construction* failed (e.g. the worker script is missing — a
|
|
274
|
+
// broken install). Fail fast with the cause rather than silently
|
|
275
|
+
// degrading to the slow sequential parser (#1741); `--workers 0` is the
|
|
276
|
+
// explicit opt-out for anyone who genuinely wants sequential parsing.
|
|
277
|
+
handleWorkerStartupFailure(err);
|
|
205
278
|
}
|
|
206
279
|
};
|
|
207
280
|
let filesParsedSoFar = 0;
|
|
@@ -400,16 +473,15 @@ export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, t
|
|
|
400
473
|
catch (err) {
|
|
401
474
|
if (!(err instanceof WorkerPoolInitializationError))
|
|
402
475
|
throw err;
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
476
|
+
// Every worker crashed during startup and the pool's bounded
|
|
477
|
+
// self-heal (jittered restart, deterministic crash-loop detection —
|
|
478
|
+
// see worker-pool.ts) was exhausted. Fail fast with the captured
|
|
479
|
+
// cause rather than silently degrading to the ~10× slower sequential
|
|
480
|
+
// parser, which masked this exact regression as a 2-hour "stuck" run
|
|
481
|
+
// in #1741. The failed (zero-worker) pool is torn down by the outer
|
|
482
|
+
// finally. `--workers 0` is the explicit opt-in to sequential.
|
|
407
483
|
rawResults.length = 0;
|
|
408
|
-
|
|
409
|
-
const failedPool = workerPool;
|
|
410
|
-
workerPool = undefined;
|
|
411
|
-
await failedPool?.terminate().catch(() => undefined);
|
|
412
|
-
chunkWorkerData = await processParsing(graph, chunkFiles, symbolTable, astCache, scopeTreeCache, progressForChunk, undefined, undefined);
|
|
484
|
+
handleWorkerStartupFailure(err); // always throws
|
|
413
485
|
}
|
|
414
486
|
// Persist the raw results for this chunk hash. Sequential path
|
|
415
487
|
// doesn't populate rawResults (it writes directly to graph), so
|
|
@@ -10,12 +10,22 @@
|
|
|
10
10
|
* because the UI progress bar updates every 100 files and intermediate
|
|
11
11
|
* stages emit little to the log.
|
|
12
12
|
*/
|
|
13
|
+
/** Min wall-clock gap between always-on slow-file warnings (throttle). */
|
|
14
|
+
export declare const ALWAYS_ON_SLOW_FILE_WARN_THROTTLE_MS = 30000;
|
|
13
15
|
/** True when deferred-stage timing / progress logs should emit. */
|
|
14
16
|
export declare const isDeferredResolutionProfileEnabled: () => boolean;
|
|
15
17
|
/** Log a call-resolution progress line every N files (finer when verbose). */
|
|
16
18
|
export declare const deferredCallLogEveryN: () => number;
|
|
17
19
|
/** Per-file call-resolution log threshold (ms). Lower default when verbose. */
|
|
18
20
|
export declare const deferredCallFileSlowMs: () => number;
|
|
21
|
+
/**
|
|
22
|
+
* Always-on per-file slow threshold (ms) for the `logger.warn` watchdog in
|
|
23
|
+
* `processCallsFromExtracted`. Unlike {@link deferredCallFileSlowMs} this is
|
|
24
|
+
* NOT gated on verbose/profile — it fires on every run. `0` (or a negative /
|
|
25
|
+
* non-finite override) disables the watchdog entirely. Override via
|
|
26
|
+
* `GITNEXUS_SLOW_FILE_WARN_MS`.
|
|
27
|
+
*/
|
|
28
|
+
export declare const alwaysOnSlowFileWarnMs: () => number;
|
|
19
29
|
export declare const profileNow: () => bigint;
|
|
20
30
|
export declare const profileElapsedMs: (start: bigint) => number;
|
|
21
31
|
/**
|
|
@@ -19,6 +19,19 @@ const LOG_EVERY_N_VERBOSE = 10;
|
|
|
19
19
|
const LOG_EVERY_N_PROFILE = 100;
|
|
20
20
|
const DEFAULT_SLOW_MS_VERBOSE = 3_000;
|
|
21
21
|
const DEFAULT_SLOW_MS = 5_000;
|
|
22
|
+
/**
|
|
23
|
+
* Always-on (NOT gated on verbose/profile) threshold above which a single
|
|
24
|
+
* file's deferred call resolution earns a `logger.warn`. The verbose
|
|
25
|
+
* slow-file profile (above) only fires with `-v`/`GITNEXUS_PROFILE_DEFERRED`;
|
|
26
|
+
* a plain `analyze` run that hangs in "Resolving calls" (the #1741 symptom)
|
|
27
|
+
* gives the user a frozen progress bar and nothing in the log. This higher
|
|
28
|
+
* default (15s — never hit by a healthy file) turns that silence into one
|
|
29
|
+
* actionable line naming the expensive file. Override via
|
|
30
|
+
* `GITNEXUS_SLOW_FILE_WARN_MS`; the throttle in the caller bounds volume.
|
|
31
|
+
*/
|
|
32
|
+
const DEFAULT_ALWAYS_ON_SLOW_FILE_WARN_MS = 15_000;
|
|
33
|
+
/** Min wall-clock gap between always-on slow-file warnings (throttle). */
|
|
34
|
+
export const ALWAYS_ON_SLOW_FILE_WARN_THROTTLE_MS = 30_000;
|
|
22
35
|
/** True when deferred-stage timing / progress logs should emit. */
|
|
23
36
|
export const isDeferredResolutionProfileEnabled = () => isVerboseIngestionEnabled() || parseTruthyEnv(process.env.GITNEXUS_PROFILE_DEFERRED);
|
|
24
37
|
/** Log a call-resolution progress line every N files (finer when verbose). */
|
|
@@ -35,6 +48,23 @@ export const deferredCallFileSlowMs = () => {
|
|
|
35
48
|
}
|
|
36
49
|
return isVerboseIngestionEnabled() ? DEFAULT_SLOW_MS_VERBOSE : DEFAULT_SLOW_MS;
|
|
37
50
|
};
|
|
51
|
+
/**
|
|
52
|
+
* Always-on per-file slow threshold (ms) for the `logger.warn` watchdog in
|
|
53
|
+
* `processCallsFromExtracted`. Unlike {@link deferredCallFileSlowMs} this is
|
|
54
|
+
* NOT gated on verbose/profile — it fires on every run. `0` (or a negative /
|
|
55
|
+
* non-finite override) disables the watchdog entirely. Override via
|
|
56
|
+
* `GITNEXUS_SLOW_FILE_WARN_MS`.
|
|
57
|
+
*/
|
|
58
|
+
export const alwaysOnSlowFileWarnMs = () => {
|
|
59
|
+
const raw = process.env.GITNEXUS_SLOW_FILE_WARN_MS;
|
|
60
|
+
if (raw !== undefined) {
|
|
61
|
+
const n = Number(raw);
|
|
62
|
+
// 0 / negative / NaN → disabled. Use Number() not parseInt (see
|
|
63
|
+
// deferredCallFileSlowMs for the '1e9' prefix-parse hazard).
|
|
64
|
+
return Number.isFinite(n) && n > 0 ? n : 0;
|
|
65
|
+
}
|
|
66
|
+
return DEFAULT_ALWAYS_ON_SLOW_FILE_WARN_MS;
|
|
67
|
+
};
|
|
38
68
|
export const profileNow = () => process.hrtime.bigint();
|
|
39
69
|
export const profileElapsedMs = (start) => Number(process.hrtime.bigint() - start) / 1e6;
|
|
40
70
|
// Module-private counter for `[deferred-profile]` log lines the underlying
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { parentPort } from 'node:worker_threads';
|
|
1
|
+
import { parentPort, threadId } from 'node:worker_threads';
|
|
2
2
|
import Parser from 'tree-sitter';
|
|
3
3
|
import JavaScript from 'tree-sitter-javascript';
|
|
4
4
|
import TypeScript from 'tree-sitter-typescript';
|
|
@@ -49,6 +49,27 @@ import { extractTemplateArguments, templateArgumentsIdTag } from '../utils/templ
|
|
|
49
49
|
import { extractParsedFile } from '../scope-extractor-bridge.js';
|
|
50
50
|
import { extractLaravelRoutes } from '../route-extractors/laravel.js';
|
|
51
51
|
import { logger } from '../../logger.js';
|
|
52
|
+
// ── Bootstrap-stage diagnostics (#1741) ────────────────────────────────────
|
|
53
|
+
// When GITNEXUS_WORKER_BOOTSTRAP=1 (or --verbose sets GITNEXUS_VERBOSE), each
|
|
54
|
+
// worker reports its startup stage timings to stderr — which the pool tees
|
|
55
|
+
// and captures (worker-pool.ts captureWorkerStderr). This makes a slow or
|
|
56
|
+
// crashing startup diagnosable: you can see whether a worker reached
|
|
57
|
+
// "grammars loaded", "ready sent", or never emitted a line at all (=> it
|
|
58
|
+
// crashed in a native binding load before this code ran). The pool then
|
|
59
|
+
// attaches whatever stderr it captured to its readiness-failure message,
|
|
60
|
+
// so the operator sees the real cause instead of "did not report ready".
|
|
61
|
+
const BOOTSTRAP_LOG = process.env.GITNEXUS_WORKER_BOOTSTRAP === '1' || process.env.GITNEXUS_VERBOSE === '1';
|
|
62
|
+
const bootstrapStart = performance.now();
|
|
63
|
+
const bootstrapLog = (stage) => {
|
|
64
|
+
if (!BOOTSTRAP_LOG)
|
|
65
|
+
return;
|
|
66
|
+
const ms = Math.round(performance.now() - bootstrapStart);
|
|
67
|
+
process.stderr.write(`[parse-worker bootstrap] thread=${threadId} ${stage} (+${ms}ms)\n`);
|
|
68
|
+
};
|
|
69
|
+
// First line we can emit: every static import above (tree-sitter native
|
|
70
|
+
// bindings, language grammars, helper modules) has already resolved by the
|
|
71
|
+
// time this module-body statement runs.
|
|
72
|
+
bootstrapLog('imports + grammars loaded');
|
|
52
73
|
// ============================================================================
|
|
53
74
|
// Worker-local parser + language map
|
|
54
75
|
// ============================================================================
|
|
@@ -1688,6 +1709,7 @@ const mergeResult = (target, src) => {
|
|
|
1688
1709
|
// `WORKER_READY_TIMEOUT_MS` (5s), so emitting it AFTER all top-of-script
|
|
1689
1710
|
// init (imports, native binding loads, type-env setup) completes is the
|
|
1690
1711
|
// load-bearing signal that this worker is ready for dispatch.
|
|
1712
|
+
bootstrapLog('ready sent');
|
|
1691
1713
|
parentPort.postMessage({ type: 'ready' });
|
|
1692
1714
|
// Module-scope `TextDecoder` for sub-batch content. The pool sends each
|
|
1693
1715
|
// file's content as a `Uint8Array` (zero-copy ArrayBuffer transfer); we
|
|
@@ -1713,7 +1735,12 @@ function decodeSubBatchFiles(files) {
|
|
|
1713
1735
|
content: typeof f.content === 'string' ? f.content : sharedContentDecoder.decode(f.content),
|
|
1714
1736
|
}));
|
|
1715
1737
|
}
|
|
1738
|
+
let firstTaskLogged = false;
|
|
1716
1739
|
parentPort.on('message', (msg) => {
|
|
1740
|
+
if (!firstTaskLogged) {
|
|
1741
|
+
firstTaskLogged = true;
|
|
1742
|
+
bootstrapLog('first task received');
|
|
1743
|
+
}
|
|
1717
1744
|
try {
|
|
1718
1745
|
// Sub-batch mode: { type: 'sub-batch', files: [...] }
|
|
1719
1746
|
if (msg.type === 'sub-batch') {
|
|
@@ -149,10 +149,37 @@ export declare class WorkerPoolDispatchError extends Error {
|
|
|
149
149
|
readonly quarantinedPaths: readonly string[];
|
|
150
150
|
constructor(message: string, quarantinedPaths?: readonly string[]);
|
|
151
151
|
}
|
|
152
|
+
/**
|
|
153
|
+
* How a total worker-startup failure was classified by the pool's bounded
|
|
154
|
+
* self-heal (#1741). Lets the caller render an accurate cause without
|
|
155
|
+
* inspecting any operator flag:
|
|
156
|
+
* - 'deterministic-startup': ≥2 fresh workers crashed with the SAME signature
|
|
157
|
+
* before any reached ready (e.g. a missing native binding) — retrying is
|
|
158
|
+
* futile, so the pool short-circuited fast.
|
|
159
|
+
* - 'transient-exhausted': workers crashed variably and exhausted the bounded
|
|
160
|
+
* startup retry budget without ever reaching ready.
|
|
161
|
+
*/
|
|
162
|
+
export type StartupCrashClass = 'deterministic-startup' | 'transient-exhausted';
|
|
152
163
|
export declare class WorkerPoolInitializationError extends WorkerPoolDispatchError {
|
|
153
164
|
readonly readinessFailures: readonly string[];
|
|
154
|
-
|
|
165
|
+
/** Pool's automatic classification of the startup crash (#1741). */
|
|
166
|
+
readonly crashClass: StartupCrashClass;
|
|
167
|
+
constructor(message: string, quarantinedPaths?: readonly string[], readinessFailures?: readonly string[], crashClass?: StartupCrashClass);
|
|
155
168
|
}
|
|
169
|
+
/**
|
|
170
|
+
* Normalize a worker crash message into a stable signature so two instances of
|
|
171
|
+
* the SAME deterministic crash compare equal while unrelated crashes don't.
|
|
172
|
+
* Strips hex addresses, digit runs (pids / line numbers / timestamps) and
|
|
173
|
+
* absolute paths. Best-effort by design: the deterministic classification's
|
|
174
|
+
* correctness rests on the STRUCTURAL signal (zero workers ever ready + startup
|
|
175
|
+
* budget exhausted), so an imperfect signature only changes how fast the
|
|
176
|
+
* short-circuit fires, never whether the pool ultimately fails fast. Even a
|
|
177
|
+
* stderr-less crash normalizes its "exited with code N" message to a stable
|
|
178
|
+
* key, so the empty-stderr timing case still groups.
|
|
179
|
+
*
|
|
180
|
+
* @internal Exported for unit tests; production callers are in this module.
|
|
181
|
+
*/
|
|
182
|
+
export declare function crashSignature(message: string): string;
|
|
156
183
|
interface ResolvedWorkerPoolOptions {
|
|
157
184
|
subBatchSize: number;
|
|
158
185
|
subBatchMaxBytes: number;
|
|
@@ -164,12 +191,20 @@ interface ResolvedWorkerPoolOptions {
|
|
|
164
191
|
consecutiveFailureThreshold: number;
|
|
165
192
|
}
|
|
166
193
|
export declare function resolveWorkerPoolOptions(options?: WorkerPoolOptions, poolSize?: number): ResolvedWorkerPoolOptions;
|
|
194
|
+
/**
|
|
195
|
+
* True when the operator explicitly disabled the worker pool via
|
|
196
|
+
* `GITNEXUS_WORKER_POOL_SIZE=0` — the env-channel equivalent of `--workers 0`.
|
|
197
|
+
* The parse phase's `shouldUseWorkers` gate consults this (only when no
|
|
198
|
+
* explicit `--workers <N>` was passed) to route to sequential parsing instead
|
|
199
|
+
* of constructing a useless size-0 pool that would fail fast on a phantom
|
|
200
|
+
* crash (#1741). An explicit positive `--workers N` always wins.
|
|
201
|
+
*/
|
|
202
|
+
export declare function workerPoolDisabledByEnv(): boolean;
|
|
167
203
|
/**
|
|
168
204
|
* Resolve the auto-default worker pool size when no explicit `poolSize`
|
|
169
205
|
* arg is passed to `createWorkerPool`. Precedence:
|
|
170
206
|
*
|
|
171
|
-
* 1. `GITNEXUS_WORKER_POOL_SIZE` env var (operator override
|
|
172
|
-
* `--workers <N>` on the CLI).
|
|
207
|
+
* 1. `GITNEXUS_WORKER_POOL_SIZE` env var (operator override).
|
|
173
208
|
* 2. `os.cpus().length - 1`, clamped to `[1, DEFAULT_POOL_SIZE_CAP]`.
|
|
174
209
|
*
|
|
175
210
|
* The cap exists because past ~16 workers the main-thread merge /
|
|
@@ -79,10 +79,13 @@ export class WorkerPoolDispatchError extends Error {
|
|
|
79
79
|
}
|
|
80
80
|
export class WorkerPoolInitializationError extends WorkerPoolDispatchError {
|
|
81
81
|
readinessFailures;
|
|
82
|
-
|
|
82
|
+
/** Pool's automatic classification of the startup crash (#1741). */
|
|
83
|
+
crashClass;
|
|
84
|
+
constructor(message, quarantinedPaths = [], readinessFailures = [], crashClass = 'transient-exhausted') {
|
|
83
85
|
super(message, quarantinedPaths);
|
|
84
86
|
this.name = 'WorkerPoolInitializationError';
|
|
85
87
|
this.readinessFailures = readinessFailures;
|
|
88
|
+
this.crashClass = crashClass;
|
|
86
89
|
}
|
|
87
90
|
}
|
|
88
91
|
/**
|
|
@@ -117,6 +120,89 @@ const WORKER_READY_TIMEOUT_MS = 5_000;
|
|
|
117
120
|
* `GITNEXUS_WORKER_POOL_SIZE` or `--workers <N>`.
|
|
118
121
|
*/
|
|
119
122
|
const DEFAULT_POOL_SIZE_CAP = 16;
|
|
123
|
+
// ── Self-healing startup restart policy (#1741) ──────────────────────────────
|
|
124
|
+
// A worker that crashes during top-of-script init (broken native binding, bad
|
|
125
|
+
// import) is retried a BOUNDED number of times with jittered backoff before
|
|
126
|
+
// its slot is dropped, so a transient blip self-heals with no operator
|
|
127
|
+
// intervention. The bound is the whole point of #1741: recovery must never
|
|
128
|
+
// become a silent, unbounded "stuck" run. When the budget is exhausted (or a
|
|
129
|
+
// deterministic crash-loop is detected), the slot is dropped; if every slot is
|
|
130
|
+
// dropped the first dispatch fails fast with the captured cause.
|
|
131
|
+
/** Retries beyond the first attempt, per slot, to bring a startup worker ready. */
|
|
132
|
+
const STARTUP_RESTART_BUDGET = 2;
|
|
133
|
+
const RESTART_BACKOFF_BASE_MS = 250;
|
|
134
|
+
const RESTART_BACKOFF_CAP_MS = 2_000;
|
|
135
|
+
/**
|
|
136
|
+
* When this many freshly-spawned workers crash with the SAME crash signature
|
|
137
|
+
* before ANY worker reaches the `{type:'ready'}` handshake, the failure is
|
|
138
|
+
* deterministic (the #1741 missing-binding case: every worker prints a
|
|
139
|
+
* byte-identical native-binding stack). The pool stops retrying immediately
|
|
140
|
+
* instead of burning every slot's budget, and fails fast with the cause.
|
|
141
|
+
*/
|
|
142
|
+
const DETERMINISTIC_STARTUP_FINGERPRINT_THRESHOLD = 2;
|
|
143
|
+
/**
|
|
144
|
+
* Capped exponential backoff with FULL jitter (AWS "Exponential Backoff And
|
|
145
|
+
* Jitter"): random(0, min(CAP, BASE·2^attempt)). Full jitter de-synchronizes
|
|
146
|
+
* the N workers that crash near-simultaneously on a shared startup fault so
|
|
147
|
+
* their respawns don't re-storm in lockstep (Google SRE thundering herd).
|
|
148
|
+
*/
|
|
149
|
+
function startupBackoffMs(attempt) {
|
|
150
|
+
const ceil = Math.min(RESTART_BACKOFF_CAP_MS, RESTART_BACKOFF_BASE_MS * 2 ** attempt);
|
|
151
|
+
return Math.floor(Math.random() * (ceil + 1));
|
|
152
|
+
}
|
|
153
|
+
/**
|
|
154
|
+
* Sleep used between startup self-heal retries. The timer is intentionally NOT
|
|
155
|
+
* `unref`'d: a pending retry is necessary work, so it must keep the event loop
|
|
156
|
+
* alive long enough to actually respawn — otherwise a pool whose only live work
|
|
157
|
+
* is a startup backoff could let the process exit mid-recovery (#1741). To
|
|
158
|
+
* avoid wedging shutdown, the timer registers a cancel function in `pending`;
|
|
159
|
+
* `terminate()` invokes those cancels to `clearTimeout` and resolve early, and
|
|
160
|
+
* a normally-fired timer removes its own cancel. `aborted()` is checked once up
|
|
161
|
+
* front; the CALLER re-checks after wake (it owns the terminated/deterministic
|
|
162
|
+
* decision) — this function does not itself re-evaluate abort on wake.
|
|
163
|
+
*/
|
|
164
|
+
function abortableSleep(ms, aborted, pending) {
|
|
165
|
+
return new Promise((resolve) => {
|
|
166
|
+
if (ms <= 0 || aborted()) {
|
|
167
|
+
resolve();
|
|
168
|
+
return;
|
|
169
|
+
}
|
|
170
|
+
// `cancel` is registered so terminate() can clear a pending backoff; it is
|
|
171
|
+
// also the timer's own callback, so a normally-fired sleep self-deregisters.
|
|
172
|
+
const cancel = () => {
|
|
173
|
+
clearTimeout(timer);
|
|
174
|
+
pending.delete(cancel);
|
|
175
|
+
resolve();
|
|
176
|
+
};
|
|
177
|
+
const timer = setTimeout(cancel, ms);
|
|
178
|
+
pending.add(cancel);
|
|
179
|
+
});
|
|
180
|
+
}
|
|
181
|
+
/**
|
|
182
|
+
* Normalize a worker crash message into a stable signature so two instances of
|
|
183
|
+
* the SAME deterministic crash compare equal while unrelated crashes don't.
|
|
184
|
+
* Strips hex addresses, digit runs (pids / line numbers / timestamps) and
|
|
185
|
+
* absolute paths. Best-effort by design: the deterministic classification's
|
|
186
|
+
* correctness rests on the STRUCTURAL signal (zero workers ever ready + startup
|
|
187
|
+
* budget exhausted), so an imperfect signature only changes how fast the
|
|
188
|
+
* short-circuit fires, never whether the pool ultimately fails fast. Even a
|
|
189
|
+
* stderr-less crash normalizes its "exited with code N" message to a stable
|
|
190
|
+
* key, so the empty-stderr timing case still groups.
|
|
191
|
+
*
|
|
192
|
+
* @internal Exported for unit tests; production callers are in this module.
|
|
193
|
+
*/
|
|
194
|
+
export function crashSignature(message) {
|
|
195
|
+
return (message
|
|
196
|
+
.replace(/0x[0-9a-fA-F]+/g, '0xADDR') // 0x-prefixed addresses
|
|
197
|
+
// Windows backslash paths (optional drive letter), e.g. C:\Users\ci\Temp\w-7f3a.js
|
|
198
|
+
.replace(/(?:[A-Za-z]:)?(?:\\[^\s\\'"]+)+/g, '\\PATH')
|
|
199
|
+
.replace(/(?:\/[^\s:'"]+)+/g, '/PATH') // POSIX paths
|
|
200
|
+
.replace(/\b[0-9a-fA-F]{6,}\b/g, 'HEX') // bare hex runs (ASLR addrs / backtrace tokens)
|
|
201
|
+
.replace(/[0-9]+/g, 'N') // pids / line numbers / exit codes / timestamps
|
|
202
|
+
.replace(/\s+/g, ' ')
|
|
203
|
+
.trim()
|
|
204
|
+
.slice(0, 300));
|
|
205
|
+
}
|
|
120
206
|
function positiveInteger(value) {
|
|
121
207
|
const parsed = typeof value === 'string' ? Number(value) : value;
|
|
122
208
|
return typeof parsed === 'number' && Number.isFinite(parsed) && parsed > 0
|
|
@@ -152,12 +238,38 @@ export function resolveWorkerPoolOptions(options = {}, poolSize) {
|
|
|
152
238
|
Math.max(DEFAULT_CONSECUTIVE_FAILURE_THRESHOLD_FLOOR, poolSize ?? 0),
|
|
153
239
|
};
|
|
154
240
|
}
|
|
241
|
+
/**
|
|
242
|
+
* The pool size requested via the `GITNEXUS_WORKER_POOL_SIZE` env var, or
|
|
243
|
+
* `undefined` when unset, empty/whitespace, or invalid. Module-internal sizing
|
|
244
|
+
* reader consumed by {@link resolveAutoPoolSize} (the env override) and
|
|
245
|
+
* {@link workerPoolDisabledByEnv} (the sequential-routing gate). Reads only —
|
|
246
|
+
* never mutates `process.env`. Empty/whitespace is treated as *unset* (falls
|
|
247
|
+
* through to the auto formula), not as 0 — an empty assignment (`export
|
|
248
|
+
* GITNEXUS_WORKER_POOL_SIZE=`) is an accident, not a request for zero workers;
|
|
249
|
+
* only a literal `0` disables the pool.
|
|
250
|
+
*/
|
|
251
|
+
function envWorkerPoolSize() {
|
|
252
|
+
const raw = process.env.GITNEXUS_WORKER_POOL_SIZE;
|
|
253
|
+
if (raw === undefined || raw.trim() === '')
|
|
254
|
+
return undefined;
|
|
255
|
+
return nonNegativeInteger(raw);
|
|
256
|
+
}
|
|
257
|
+
/**
|
|
258
|
+
* True when the operator explicitly disabled the worker pool via
|
|
259
|
+
* `GITNEXUS_WORKER_POOL_SIZE=0` — the env-channel equivalent of `--workers 0`.
|
|
260
|
+
* The parse phase's `shouldUseWorkers` gate consults this (only when no
|
|
261
|
+
* explicit `--workers <N>` was passed) to route to sequential parsing instead
|
|
262
|
+
* of constructing a useless size-0 pool that would fail fast on a phantom
|
|
263
|
+
* crash (#1741). An explicit positive `--workers N` always wins.
|
|
264
|
+
*/
|
|
265
|
+
export function workerPoolDisabledByEnv() {
|
|
266
|
+
return envWorkerPoolSize() === 0;
|
|
267
|
+
}
|
|
155
268
|
/**
|
|
156
269
|
* Resolve the auto-default worker pool size when no explicit `poolSize`
|
|
157
270
|
* arg is passed to `createWorkerPool`. Precedence:
|
|
158
271
|
*
|
|
159
|
-
* 1. `GITNEXUS_WORKER_POOL_SIZE` env var (operator override
|
|
160
|
-
* `--workers <N>` on the CLI).
|
|
272
|
+
* 1. `GITNEXUS_WORKER_POOL_SIZE` env var (operator override).
|
|
161
273
|
* 2. `os.cpus().length - 1`, clamped to `[1, DEFAULT_POOL_SIZE_CAP]`.
|
|
162
274
|
*
|
|
163
275
|
* The cap exists because past ~16 workers the main-thread merge /
|
|
@@ -170,7 +282,7 @@ export function resolveWorkerPoolOptions(options = {}, poolSize) {
|
|
|
170
282
|
* on the env / default.
|
|
171
283
|
*/
|
|
172
284
|
export function resolveAutoPoolSize() {
|
|
173
|
-
const envOverride =
|
|
285
|
+
const envOverride = envWorkerPoolSize();
|
|
174
286
|
if (envOverride !== undefined)
|
|
175
287
|
return envOverride;
|
|
176
288
|
// Prefer os.availableParallelism (Node 18.14+) so cgroup CPU limits
|
|
@@ -184,6 +296,51 @@ export function resolveAutoPoolSize() {
|
|
|
184
296
|
const cores = typeof os.availableParallelism === 'function' ? os.availableParallelism() : os.cpus().length;
|
|
185
297
|
return Math.min(DEFAULT_POOL_SIZE_CAP, Math.max(1, cores - 1));
|
|
186
298
|
}
|
|
299
|
+
/**
|
|
300
|
+
* Max characters of a worker's stderr retained for crash diagnostics. A
|
|
301
|
+
* native-binding load failure or a top-of-script throw prints a stack to
|
|
302
|
+
* stderr; we keep the tail so `waitForWorkerReady` can attach the real
|
|
303
|
+
* reason to its rejection instead of the generic "did not report ready".
|
|
304
|
+
*/
|
|
305
|
+
const WORKER_STDERR_TAIL_LIMIT = 4000;
|
|
306
|
+
/**
|
|
307
|
+
* Per-worker captured stderr tail. Populated only for workers spawned with
|
|
308
|
+
* `{ stderr: true }` (the production factory below). Test-injected workers
|
|
309
|
+
* via `workerFactory` typically inherit the parent's stderr and have no
|
|
310
|
+
* `worker.stderr` stream — those are simply skipped (empty tail). A WeakMap
|
|
311
|
+
* so the buffer is released when the worker is GC'd.
|
|
312
|
+
*/
|
|
313
|
+
const workerStderrTails = new WeakMap();
|
|
314
|
+
/**
|
|
315
|
+
* Tee a worker's stderr into a bounded in-memory tail (for surfacing the
|
|
316
|
+
* real crash on a startup failure) while still mirroring it to the parent
|
|
317
|
+
* process's stderr — preserving the live-diagnostics behavior workers had
|
|
318
|
+
* when they inherited stderr, before `{ stderr: true }` redirected it to a
|
|
319
|
+
* stream. No-op when the worker has no `stderr` stream (test factories).
|
|
320
|
+
*/
|
|
321
|
+
function captureWorkerStderr(worker) {
|
|
322
|
+
const stream = worker.stderr;
|
|
323
|
+
if (!stream)
|
|
324
|
+
return;
|
|
325
|
+
const buf = { text: '' };
|
|
326
|
+
workerStderrTails.set(worker, buf);
|
|
327
|
+
stream.on('data', (chunk) => {
|
|
328
|
+
const s = typeof chunk === 'string' ? chunk : chunk.toString('utf8');
|
|
329
|
+
process.stderr.write(s);
|
|
330
|
+
buf.text = (buf.text + s).slice(-WORKER_STDERR_TAIL_LIMIT);
|
|
331
|
+
});
|
|
332
|
+
// A stderr stream error must never crash the pool.
|
|
333
|
+
stream.on('error', () => undefined);
|
|
334
|
+
}
|
|
335
|
+
/** Captured stderr tail for a worker, trimmed; '' when nothing was captured. */
|
|
336
|
+
function workerStderrTail(worker) {
|
|
337
|
+
return workerStderrTails.get(worker)?.text.trim() ?? '';
|
|
338
|
+
}
|
|
339
|
+
/** Append the worker's captured stderr to a readiness-failure message. */
|
|
340
|
+
function withStderr(worker, message) {
|
|
341
|
+
const tail = workerStderrTail(worker);
|
|
342
|
+
return tail ? `${message}. Worker stderr:\n${tail}` : message;
|
|
343
|
+
}
|
|
187
344
|
/**
|
|
188
345
|
* Wait for a freshly-spawned replacement worker to emit the
|
|
189
346
|
* `{type:'ready'}` handshake from `parse-worker.ts` before treating its
|
|
@@ -220,22 +377,24 @@ function waitForWorkerReady(worker) {
|
|
|
220
377
|
};
|
|
221
378
|
const onError = (err) => {
|
|
222
379
|
cleanup();
|
|
223
|
-
|
|
380
|
+
// The 'error' event carries the real top-of-script exception; enrich it
|
|
381
|
+
// with the worker's stderr tail (native-binding stacks land there).
|
|
382
|
+
reject(new Error(withStderr(worker, err.message)));
|
|
224
383
|
};
|
|
225
384
|
const onExit = (code) => {
|
|
226
385
|
cleanup();
|
|
227
|
-
reject(new Error(`Replacement worker exited with code ${code} before reporting ready`));
|
|
386
|
+
reject(new Error(withStderr(worker, `Replacement worker exited with code ${code} before reporting ready`)));
|
|
228
387
|
};
|
|
229
388
|
const onMessageError = (err) => {
|
|
230
389
|
cleanup();
|
|
231
|
-
reject(new Error(`Replacement worker emitted messageerror before reporting ready: ${err.message}`));
|
|
390
|
+
reject(new Error(withStderr(worker, `Replacement worker emitted messageerror before reporting ready: ${err.message}`)));
|
|
232
391
|
};
|
|
233
392
|
// `timer` is declared after `cleanup` so the cleanup closure can reference
|
|
234
393
|
// it. The const is reached before any handler attaches below, so no TDZ
|
|
235
394
|
// access can fire from the listeners.
|
|
236
395
|
const timer = setTimeout(() => {
|
|
237
396
|
cleanup();
|
|
238
|
-
reject(new Error(`Replacement worker did not report ready within ${WORKER_READY_TIMEOUT_MS}ms — likely crashed during top-of-script init`));
|
|
397
|
+
reject(new Error(withStderr(worker, `Replacement worker did not report ready within ${WORKER_READY_TIMEOUT_MS}ms — likely crashed during top-of-script init`)));
|
|
239
398
|
}, WORKER_READY_TIMEOUT_MS);
|
|
240
399
|
worker.on('message', onMessage);
|
|
241
400
|
worker.once('error', onError);
|
|
@@ -340,7 +499,18 @@ export const createWorkerPool = (workerUrl, poolSize, options) => {
|
|
|
340
499
|
}
|
|
341
500
|
const size = poolSize ?? resolveAutoPoolSize();
|
|
342
501
|
const poolOptions = resolveWorkerPoolOptions(options, size);
|
|
343
|
-
|
|
502
|
+
// Production factory spawns with `{ stderr: true }` so a worker's crash
|
|
503
|
+
// output is redirected to a `worker.stderr` stream we can tee + capture
|
|
504
|
+
// (see captureWorkerStderr) and attach to readiness-failure messages —
|
|
505
|
+
// instead of the generic "did not report ready" that hid the real cause
|
|
506
|
+
// in #1741. Test factories (workerFactory) are used verbatim.
|
|
507
|
+
const spawnWorker = options?.workerFactory ?? ((url) => new Worker(url, { stderr: true }));
|
|
508
|
+
/** Spawn + wire stderr capture in one step (used by all spawn sites). */
|
|
509
|
+
const spawnAndCapture = (url) => {
|
|
510
|
+
const worker = spawnWorker(url);
|
|
511
|
+
captureWorkerStderr(worker);
|
|
512
|
+
return worker;
|
|
513
|
+
};
|
|
344
514
|
const workers = new Array(size);
|
|
345
515
|
const retiredWorkers = new Set();
|
|
346
516
|
const respawnCount = new Array(size).fill(0);
|
|
@@ -369,6 +539,9 @@ export const createWorkerPool = (workerUrl, poolSize, options) => {
|
|
|
369
539
|
const slotGenerations = new Array(size).fill(0);
|
|
370
540
|
let poolBroken = false;
|
|
371
541
|
let poolFailure;
|
|
542
|
+
// Set by `terminate()` (below). Also read by the self-healing startup loop so
|
|
543
|
+
// a terminate during startup aborts pending backoff/retries (#1741).
|
|
544
|
+
let terminated = false;
|
|
372
545
|
const terminateTrackedWorkers = async (liveWorkers) => {
|
|
373
546
|
const retired = Array.from(retiredWorkers);
|
|
374
547
|
await Promise.all([
|
|
@@ -378,41 +551,96 @@ export const createWorkerPool = (workerUrl, poolSize, options) => {
|
|
|
378
551
|
retiredWorkers.clear();
|
|
379
552
|
};
|
|
380
553
|
for (let i = 0; i < size; i++) {
|
|
381
|
-
workers[i] =
|
|
554
|
+
workers[i] = spawnAndCapture(workerUrl);
|
|
382
555
|
activeSlots.add(i);
|
|
383
556
|
}
|
|
384
|
-
//
|
|
385
|
-
//
|
|
386
|
-
//
|
|
387
|
-
//
|
|
388
|
-
//
|
|
389
|
-
// missing dependency) would only be noticed at the first dispatch's
|
|
390
|
-
// 30s idle timeout, vs the 5s WORKER_READY_TIMEOUT_MS bound that
|
|
391
|
-
// replacements enjoy.
|
|
557
|
+
// ── Self-healing startup readiness (#1741) ────────────────────────────────
|
|
558
|
+
// Bring every initial slot to readiness with a BOUNDED, jittered retry loop
|
|
559
|
+
// instead of dropping it on the first crash. This symmetrizes the gate with
|
|
560
|
+
// the runtime `replaceWorker` path (which already respawns a crashed slot),
|
|
561
|
+
// and adds genuine self-healing at startup:
|
|
392
562
|
//
|
|
393
|
-
//
|
|
394
|
-
//
|
|
395
|
-
//
|
|
396
|
-
//
|
|
397
|
-
//
|
|
398
|
-
//
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
563
|
+
// - TRANSIENT crash (a one-off OS hiccup / fork throttle): the slot is
|
|
564
|
+
// respawned after jittered backoff and retried, up to STARTUP_RESTART_BUDGET
|
|
565
|
+
// — so a blip heals itself with no operator intervention.
|
|
566
|
+
// - DETERMINISTIC crash-loop (every worker dies with the SAME signature
|
|
567
|
+
// before any reaches ready — the #1741 missing-binding case): detected via
|
|
568
|
+
// `crashSignature` and short-circuited immediately, so the pool gives up in
|
|
569
|
+
// ~1s rather than burning every slot's budget.
|
|
570
|
+
//
|
|
571
|
+
// When the loop exhausts, the slot is dropped from `activeSlots`. If EVERY
|
|
572
|
+
// slot is dropped, the first dispatch throws WorkerPoolInitializationError
|
|
573
|
+
// carrying the captured crash cause + classification — never a silent hang.
|
|
574
|
+
// Correctness of the deterministic short-circuit rests on the STRUCTURAL
|
|
575
|
+
// signal (zero workers ever ready + budget exhausted), not on signature
|
|
576
|
+
// matching alone: a missed match only costs a few seconds of extra retrying.
|
|
577
|
+
// Deterministic crash-loop detection (#1741). A crash counts toward
|
|
578
|
+
// "deterministic" ONLY after its signature reproduces across a respawn on the
|
|
579
|
+
// same slot — so every slot is guaranteed at least one self-heal attempt and
|
|
580
|
+
// a simultaneous attempt-0 crash storm (e.g. transient `spawn EAGAIN` under
|
|
581
|
+
// fork pressure) cannot be misclassified as deterministic. We short-circuit
|
|
582
|
+
// once enough DISTINCT slots have each reproduced: ≥2 normally, or 1 for a
|
|
583
|
+
// size-1 pool. Until then the structural floor (every slot exhausts its
|
|
584
|
+
// budget) still bounds the worst case, so a missed match only costs retries.
|
|
585
|
+
const lastStartupSignature = new Map();
|
|
586
|
+
const reproducedStartupSlots = new Set();
|
|
587
|
+
const deterministicSlotThreshold = Math.min(DETERMINISTIC_STARTUP_FINGERPRINT_THRESHOLD, size);
|
|
588
|
+
let deterministicStartupDetected = false;
|
|
589
|
+
let anyWorkerReachedReady = false;
|
|
590
|
+
// Cancel functions for in-flight startup backoffs (see abortableSleep). The
|
|
591
|
+
// backoff timer is ref'd so a retry actually runs; terminate() invokes these
|
|
592
|
+
// to clear pending backoffs and resolve their sleeps so the slot loops wake,
|
|
593
|
+
// see `terminated`, and give up — instead of the process staying pinned for
|
|
594
|
+
// the backoff cap after terminate (#1741).
|
|
595
|
+
const pendingStartupTimers = new Set();
|
|
596
|
+
const bringSlotReady = async (i) => {
|
|
597
|
+
for (let attempt = 0;; attempt++) {
|
|
598
|
+
const worker = workers[i];
|
|
599
|
+
if (!worker)
|
|
600
|
+
return; // terminated mid-startup
|
|
601
|
+
try {
|
|
602
|
+
await waitForWorkerReady(worker);
|
|
603
|
+
anyWorkerReachedReady = true;
|
|
604
|
+
return; // ready — slot stays in activeSlots
|
|
605
|
+
}
|
|
606
|
+
catch (err) {
|
|
607
|
+
const msg = err instanceof Error ? err.message : String(err);
|
|
608
|
+
const sig = crashSignature(msg);
|
|
609
|
+
// Same signature as this slot's previous attempt => it survived a
|
|
610
|
+
// respawn, so retrying this slot is futile. (First crash has no prior
|
|
611
|
+
// signature, so attempt 0 never counts — every slot self-heals once.)
|
|
612
|
+
if (lastStartupSignature.get(i) === sig)
|
|
613
|
+
reproducedStartupSlots.add(i);
|
|
614
|
+
lastStartupSignature.set(i, sig);
|
|
615
|
+
if (!anyWorkerReachedReady && reproducedStartupSlots.size >= deterministicSlotThreshold) {
|
|
616
|
+
deterministicStartupDetected = true;
|
|
617
|
+
}
|
|
618
|
+
await worker.terminate().catch(() => undefined);
|
|
619
|
+
workers[i] = undefined;
|
|
620
|
+
const giveUp = terminated || deterministicStartupDetected || attempt >= STARTUP_RESTART_BUDGET;
|
|
621
|
+
if (giveUp) {
|
|
622
|
+
initialReadinessFailures.push(msg);
|
|
623
|
+
activeSlots.delete(i);
|
|
624
|
+
logger.warn({ workerIndex: i, attempt, err: msg, deterministic: deterministicStartupDetected }, deterministicStartupDetected
|
|
625
|
+
? `Worker ${i} hit a deterministic startup crash-loop; dropping slot without further retries.`
|
|
626
|
+
: `Worker ${i} did not report ready after ${attempt + 1} attempt(s); dropping slot.`);
|
|
627
|
+
return;
|
|
628
|
+
}
|
|
629
|
+
// Transient: jittered backoff, then respawn the slot and retry.
|
|
630
|
+
await abortableSleep(startupBackoffMs(attempt), () => terminated || deterministicStartupDetected, pendingStartupTimers);
|
|
631
|
+
if (terminated || deterministicStartupDetected) {
|
|
632
|
+
initialReadinessFailures.push(msg);
|
|
633
|
+
activeSlots.delete(i);
|
|
634
|
+
return;
|
|
635
|
+
}
|
|
636
|
+
logger.warn({ workerIndex: i, attempt: attempt + 1 }, `Worker ${i} crashed during startup; respawning slot (self-heal attempt ${attempt + 1}/${STARTUP_RESTART_BUDGET}).`);
|
|
637
|
+
workers[i] = spawnAndCapture(workerUrl);
|
|
638
|
+
}
|
|
414
639
|
}
|
|
415
|
-
}
|
|
640
|
+
};
|
|
641
|
+
// First dispatch awaits this; it settles every slot's bounded retry loop in
|
|
642
|
+
// parallel and drops the unrecoverable ones before any dispatch can fire.
|
|
643
|
+
const initialReadyGate = Promise.allSettled(workers.map((_, i) => bringSlotReady(i))).then(() => undefined);
|
|
416
644
|
const dispatch = async (items, onProgress) => {
|
|
417
645
|
// Await the initial-spawn readiness gate (F13). On first dispatch
|
|
418
646
|
// this blocks for up to WORKER_READY_TIMEOUT_MS while every initial
|
|
@@ -434,7 +662,13 @@ export const createWorkerPool = (workerUrl, poolSize, options) => {
|
|
|
434
662
|
const detail = initialReadinessFailures.length > 0
|
|
435
663
|
? ` after initial ready handshake: ${initialReadinessFailures.join('; ')}`
|
|
436
664
|
: '';
|
|
437
|
-
|
|
665
|
+
// The bounded self-heal exhausted (or short-circuited a deterministic
|
|
666
|
+
// crash-loop). Classify automatically so the caller renders the real
|
|
667
|
+
// cause without consulting any operator flag (#1741).
|
|
668
|
+
const crashClass = deterministicStartupDetected
|
|
669
|
+
? 'deterministic-startup'
|
|
670
|
+
: 'transient-exhausted';
|
|
671
|
+
throw new WorkerPoolInitializationError(`Worker pool has no active workers${detail}`, [], initialReadinessFailures, crashClass);
|
|
438
672
|
}
|
|
439
673
|
// Layer 3: filter out quarantined paths so a known-bad file never reaches
|
|
440
674
|
// a worker again this pool lifetime. The caller queries
|
|
@@ -549,7 +783,7 @@ export const createWorkerPool = (workerUrl, poolSize, options) => {
|
|
|
549
783
|
await removeWorkerFromSlot(workerIndex, mode, reason);
|
|
550
784
|
if (stopped)
|
|
551
785
|
return false;
|
|
552
|
-
const replacement =
|
|
786
|
+
const replacement = spawnAndCapture(workerUrl);
|
|
553
787
|
try {
|
|
554
788
|
await waitForWorkerReady(replacement);
|
|
555
789
|
}
|
|
@@ -1165,9 +1399,13 @@ export const createWorkerPool = (workerUrl, poolSize, options) => {
|
|
|
1165
1399
|
runWorker(slotIndex);
|
|
1166
1400
|
});
|
|
1167
1401
|
};
|
|
1168
|
-
let terminated = false;
|
|
1169
1402
|
const terminate = async () => {
|
|
1170
1403
|
terminated = true;
|
|
1404
|
+
// Cancel any in-flight startup backoff so its ref'd timer doesn't keep the
|
|
1405
|
+
// event loop alive after terminate; each cancel resolves the awaiting sleep
|
|
1406
|
+
// and the slot loop then sees `terminated` and gives up (#1741).
|
|
1407
|
+
for (const cancel of [...pendingStartupTimers])
|
|
1408
|
+
cancel();
|
|
1171
1409
|
// `.catch(() => undefined)` per-worker matches every other terminate
|
|
1172
1410
|
// site in this file. Without it, a hung/OOM-killed worker's terminate
|
|
1173
1411
|
// rejection escapes `Promise.all` and replaces the original pipeline
|
|
@@ -1190,6 +1428,7 @@ export const createWorkerPool = (workerUrl, poolSize, options) => {
|
|
|
1190
1428
|
quarantined: quarantine.size,
|
|
1191
1429
|
poolBroken,
|
|
1192
1430
|
terminated,
|
|
1431
|
+
pendingStartupTimers: pendingStartupTimers.size,
|
|
1193
1432
|
slotGenerations: slotGenerations.slice(),
|
|
1194
1433
|
}),
|
|
1195
1434
|
};
|
package/dist/core/run-analyze.js
CHANGED
|
@@ -296,7 +296,10 @@ export async function runFullAnalysis(repoPath, options, callbacks) {
|
|
|
296
296
|
? `${p.message || phaseLabel} (${p.detail})`
|
|
297
297
|
: p.message || phaseLabel;
|
|
298
298
|
progress(p.phase, scaled, message);
|
|
299
|
-
}, {
|
|
299
|
+
}, {
|
|
300
|
+
parseCache,
|
|
301
|
+
workerPoolSize: options.workerPoolSize,
|
|
302
|
+
});
|
|
300
303
|
// ── Phase 2: LadybugDB (60–85%) ──────────────────────────────────
|
|
301
304
|
progress('lbug', 60, 'Loading into LadybugDB...');
|
|
302
305
|
// Compute current per-file content hashes from the pipeline's File nodes.
|
package/package.json
CHANGED