gitnexus 1.6.12-rc.2 → 1.6.12-rc.21
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/_shared/graph/types.d.ts +1 -1
- package/dist/_shared/graph/types.d.ts.map +1 -1
- package/dist/_shared/language-detection.d.ts.map +1 -1
- package/dist/_shared/language-detection.js +2 -0
- package/dist/_shared/language-detection.js.map +1 -1
- package/dist/_shared/languages.d.ts +1 -0
- package/dist/_shared/languages.d.ts.map +1 -1
- package/dist/_shared/languages.js +1 -0
- package/dist/_shared/languages.js.map +1 -1
- package/dist/_shared/lbug/schema-constants.d.ts +1 -1
- package/dist/_shared/lbug/schema-constants.d.ts.map +1 -1
- package/dist/_shared/lbug/schema-constants.js +2 -0
- package/dist/_shared/lbug/schema-constants.js.map +1 -1
- package/dist/_shared/scope-resolution/language-classification.d.ts +2 -1
- package/dist/_shared/scope-resolution/language-classification.d.ts.map +1 -1
- package/dist/_shared/scope-resolution/language-classification.js +3 -1
- package/dist/_shared/scope-resolution/language-classification.js.map +1 -1
- package/dist/_shared/scope-resolution/registries/class-registry.d.ts +3 -3
- package/dist/_shared/scope-resolution/registries/class-registry.js +3 -3
- package/dist/_shared/scope-resolution/registries/context.d.ts.map +1 -1
- package/dist/_shared/scope-resolution/registries/context.js +2 -0
- package/dist/_shared/scope-resolution/registries/context.js.map +1 -1
- package/dist/_shared/scope-resolution/resolve-type-ref.d.ts.map +1 -1
- package/dist/_shared/scope-resolution/resolve-type-ref.js +2 -0
- package/dist/_shared/scope-resolution/resolve-type-ref.js.map +1 -1
- package/dist/cli/analyze-config.d.ts +2 -2
- package/dist/cli/analyze-config.js +10 -33
- package/dist/cli/doctor.js +6 -4
- package/dist/cli/eval-server.js +13 -1
- package/dist/cli/i18n/en.d.ts +4 -2
- package/dist/cli/i18n/en.js +4 -2
- package/dist/cli/i18n/resources.d.ts +6 -2
- package/dist/cli/i18n/zh-CN.d.ts +2 -0
- package/dist/cli/i18n/zh-CN.js +4 -2
- package/dist/core/analysis-feature-registry.d.ts +1 -1
- package/dist/core/analysis-feature-registry.js +2 -0
- package/dist/core/auto-sync/starter.js +2 -2
- package/dist/core/embeddings/ast-utils.js +8 -3
- package/dist/core/embeddings/chunker.js +67 -10
- package/dist/core/embeddings/embedding-pipeline.d.ts +1 -1
- package/dist/core/embeddings/embedding-pipeline.js +1 -1
- package/dist/core/embeddings/structural-extractor.js +2 -2
- package/dist/core/embeddings/types.d.ts +4 -2
- package/dist/core/embeddings/types.js +18 -0
- package/dist/core/git-ref.d.ts +22 -0
- package/dist/core/git-ref.js +114 -0
- package/dist/core/git-staleness.js +14 -0
- package/dist/core/group/extractors/graphql-extractor.d.ts +1 -1
- package/dist/core/group/extractors/graphql-extractor.js +149 -13
- package/dist/core/group/extractors/manifest-extractor.d.ts +1 -1
- package/dist/core/group/extractors/manifest-extractor.js +2 -2
- package/dist/core/ingestion/call-extractors/zig-static-gating.d.ts +5 -8
- package/dist/core/ingestion/content-language-classification.d.ts +10 -0
- package/dist/core/ingestion/content-language-classification.js +30 -0
- package/dist/core/ingestion/filesystem-walker.d.ts +1 -0
- package/dist/core/ingestion/filesystem-walker.js +1 -1
- package/dist/core/ingestion/import-resolvers/node-workspace-packages.js +3 -2
- package/dist/core/ingestion/language-config.d.ts +82 -13
- package/dist/core/ingestion/language-config.js +217 -13
- package/dist/core/ingestion/language-provider.d.ts +88 -1
- package/dist/core/ingestion/languages/index.d.ts +7 -0
- package/dist/core/ingestion/languages/index.js +30 -1
- package/dist/core/ingestion/languages/objective-c/analysis-features.d.ts +10 -0
- package/dist/core/ingestion/languages/objective-c/analysis-features.js +19 -0
- package/dist/core/ingestion/languages/objective-c/compilation-unit-siblings.d.ts +13 -0
- package/dist/core/ingestion/languages/objective-c/compilation-unit-siblings.js +124 -0
- package/dist/core/ingestion/languages/objective-c/facts.d.ts +148 -0
- package/dist/core/ingestion/languages/objective-c/facts.js +1150 -0
- package/dist/core/ingestion/languages/objective-c/import-target.d.ts +11 -0
- package/dist/core/ingestion/languages/objective-c/import-target.js +122 -0
- package/dist/core/ingestion/languages/objective-c/macro-marker-preprocess.d.ts +24 -0
- package/dist/core/ingestion/languages/objective-c/macro-marker-preprocess.js +261 -0
- package/dist/core/ingestion/languages/objective-c/resolution-config.d.ts +41 -0
- package/dist/core/ingestion/languages/objective-c/resolution-config.js +282 -0
- package/dist/core/ingestion/languages/objective-c/scope-resolver.d.ts +2 -0
- package/dist/core/ingestion/languages/objective-c/scope-resolver.js +503 -0
- package/dist/core/ingestion/languages/objective-c.d.ts +2 -0
- package/dist/core/ingestion/languages/objective-c.js +300 -0
- package/dist/core/ingestion/languages/typescript/query.d.ts +31 -0
- package/dist/core/ingestion/languages/typescript/query.js +7 -1
- package/dist/core/ingestion/languages/typescript/tsconfig.js +3 -2
- package/dist/core/ingestion/languages/typescript.js +1 -1
- package/dist/core/ingestion/languages/zig/query.d.ts +26 -0
- package/dist/core/ingestion/languages/zig/query.js +61 -1
- package/dist/core/ingestion/languages/zig/range-binding.d.ts +8 -0
- package/dist/core/ingestion/languages/zig/range-binding.js +24 -0
- package/dist/core/ingestion/languages/zig/scope-resolver.d.ts +3 -2
- package/dist/core/ingestion/languages/zig/scope-resolver.js +18 -5
- package/dist/core/ingestion/languages/zig/this-alias-bindings.d.ts +58 -0
- package/dist/core/ingestion/languages/zig/this-alias-bindings.js +176 -0
- package/dist/core/ingestion/languages/zig/workspace-static-gating.d.ts +8 -0
- package/dist/core/ingestion/languages/zig/workspace-static-gating.js +82 -0
- package/dist/core/ingestion/model/registration-table.d.ts +3 -3
- package/dist/core/ingestion/model/registration-table.js +9 -5
- package/dist/core/ingestion/model/symbol-table.d.ts +1 -1
- package/dist/core/ingestion/model/symbol-table.js +2 -0
- package/dist/core/ingestion/model/type-registry.d.ts +1 -1
- package/dist/core/ingestion/parsing-processor.d.ts +22 -6
- package/dist/core/ingestion/parsing-processor.js +39 -20
- package/dist/core/ingestion/pipeline-phases/parse-impl.d.ts +9 -1
- package/dist/core/ingestion/pipeline-phases/parse-impl.js +307 -96
- package/dist/core/ingestion/pipeline-phases/parse-round-budget.d.ts +42 -0
- package/dist/core/ingestion/pipeline-phases/parse-round-budget.js +50 -0
- package/dist/core/ingestion/pipeline-phases/parse.d.ts +7 -1
- package/dist/core/ingestion/scope-extractor.js +4 -0
- package/dist/core/ingestion/scope-resolution/contract/scope-resolver.d.ts +13 -0
- package/dist/core/ingestion/scope-resolution/graph-bridge/ids.js +2 -0
- package/dist/core/ingestion/scope-resolution/graph-bridge/node-lookup.js +2 -0
- package/dist/core/ingestion/scope-resolution/passes/property-dispatch.d.ts +78 -5
- package/dist/core/ingestion/scope-resolution/passes/property-dispatch.js +262 -7
- package/dist/core/ingestion/scope-resolution/pipeline/phase.js +5 -2
- package/dist/core/ingestion/scope-resolution/pipeline/registry.js +2 -0
- package/dist/core/ingestion/scope-resolution/pipeline/run.js +10 -1
- package/dist/core/ingestion/scope-resolution/scope/walkers.d.ts +39 -3
- package/dist/core/ingestion/scope-resolution/scope/walkers.js +82 -3
- package/dist/core/ingestion/scope-resolution/value-ref-edges.d.ts +14 -0
- package/dist/core/ingestion/scope-resolution/value-ref-edges.js +14 -0
- package/dist/core/ingestion/tree-sitter-queries.js +2 -0
- package/dist/core/ingestion/utils/symbol-labels.js +2 -0
- package/dist/core/ingestion/workers/parse-worker.d.ts +1 -1
- package/dist/core/ingestion/workers/parse-worker.js +63 -3
- package/dist/core/ingestion/workers/worker-pool.d.ts +56 -2
- package/dist/core/ingestion/workers/worker-pool.js +117 -26
- package/dist/core/lbug/csv-generator.js +2 -0
- package/dist/core/lbug/schema.d.ts +3 -1
- package/dist/core/lbug/schema.js +27 -0
- package/dist/core/search/fts-indexes.js +13 -4
- package/dist/core/search/fts-schema.js +2 -0
- package/dist/core/tree-sitter/parser-loader.js +10 -0
- package/dist/core/tree-sitter/vendored-grammars.d.ts +1 -1
- package/dist/core/tree-sitter/vendored-grammars.js +2 -1
- package/dist/mcp/local/local-backend.d.ts +33 -0
- package/dist/mcp/local/local-backend.js +166 -5
- package/dist/mcp/tools.js +6 -4
- package/dist/server/analyze-job.d.ts +19 -1
- package/dist/server/analyze-job.js +14 -3
- package/dist/server/analyze-launch.d.ts +14 -0
- package/dist/server/analyze-launch.js +59 -14
- package/dist/server/analyze-worker-ipc.d.ts +6 -5
- package/dist/server/analyze-worker-ipc.js +4 -0
- package/dist/server/api.js +68 -21
- package/dist/server/git-clone.d.ts +38 -5
- package/dist/server/git-clone.js +216 -38
- package/dist/server/repo-projection.d.ts +88 -0
- package/dist/server/repo-projection.js +53 -0
- package/dist/storage/file-lock.js +5 -3
- package/dist/storage/parse-cache.js +32 -1
- package/dist/storage/parsedfile-store.js +53 -1
- package/dist/storage/v8-sidecar.d.ts +11 -0
- package/dist/storage/v8-sidecar.js +15 -7
- package/dist/utils/process-identity.d.ts +13 -0
- package/dist/utils/process-identity.js +18 -0
- package/package.json +1 -1
- package/scripts/build-tree-sitter-grammars.cjs +4 -3
- package/vendor/tree-sitter-objc/LICENSE +21 -0
- package/vendor/tree-sitter-objc/README.md +18 -0
- package/vendor/tree-sitter-objc/binding.gyp +35 -0
- package/vendor/tree-sitter-objc/bindings/node/binding.cc +19 -0
- package/vendor/tree-sitter-objc/bindings/node/binding_test.js +9 -0
- package/vendor/tree-sitter-objc/bindings/node/index.d.ts +27 -0
- package/vendor/tree-sitter-objc/bindings/node/index.js +11 -0
- package/vendor/tree-sitter-objc/grammar.js +1286 -0
- package/vendor/tree-sitter-objc/package.json +56 -0
- package/vendor/tree-sitter-objc/prebuilds/SHA256SUMS +6 -0
- package/vendor/tree-sitter-objc/prebuilds/darwin-arm64/tree-sitter-objc.node +0 -0
- package/vendor/tree-sitter-objc/prebuilds/darwin-x64/tree-sitter-objc.node +0 -0
- package/vendor/tree-sitter-objc/prebuilds/linux-arm64/tree-sitter-objc.node +0 -0
- package/vendor/tree-sitter-objc/prebuilds/linux-x64/tree-sitter-objc.node +0 -0
- package/vendor/tree-sitter-objc/prebuilds/win32-arm64/tree-sitter-objc.node +0 -0
- package/vendor/tree-sitter-objc/prebuilds/win32-x64/tree-sitter-objc.node +0 -0
- package/vendor/tree-sitter-objc/queries/folds.scm +20 -0
- package/vendor/tree-sitter-objc/queries/highlights.scm +216 -0
- package/vendor/tree-sitter-objc/queries/indents.scm +1 -0
- package/vendor/tree-sitter-objc/queries/injections.scm +10 -0
- package/vendor/tree-sitter-objc/queries/locals.scm +1 -0
- package/vendor/tree-sitter-objc/src/grammar.json +16292 -0
- package/vendor/tree-sitter-objc/src/node-types.json +7524 -0
- package/vendor/tree-sitter-objc/src/parser.c +684023 -0
- package/vendor/tree-sitter-objc/src/tree_sitter/alloc.h +54 -0
- package/vendor/tree-sitter-objc/src/tree_sitter/array.h +291 -0
- package/vendor/tree-sitter-objc/src/tree_sitter/parser.h +266 -0
- package/vendor/tree-sitter-objc/tree-sitter-objc.wasm +0 -0
- package/vendor/tree-sitter-objc/tree-sitter.json +38 -0
- package/web/assets/{agent-BMRqXorA.js → agent-C2n33ANd.js} +1 -1
- package/web/assets/{index-Dj3vQOHK.js → index-CHJoW_P1.js} +18 -18
- package/web/assets/src-jc8Ffy3T.js +1 -0
- package/web/index.html +2 -2
- package/web/assets/src-Df5C1Nz4.js +0 -1
|
@@ -15,7 +15,7 @@
|
|
|
15
15
|
* @module
|
|
16
16
|
*/
|
|
17
17
|
import { BindingAccumulator, enrichExportedTypeMap, } from '../binding-accumulator.js';
|
|
18
|
-
import { mergeChunkResults,
|
|
18
|
+
import { mergeChunkResults, dispatchChunkParseRound } from '../parsing-processor.js';
|
|
19
19
|
import { fileContentHash, computeChunkHash, loadParseCacheChunk, persistParseCacheChunk, PARSE_CACHE_VERSION, packParseCacheChunks, } from '../../../storage/parse-cache.js';
|
|
20
20
|
import { clearParsedFileStore, persistParsedFileChunk, loadParsedFilesForPaths, getDurableParsedFileDir, loadDurableParsedFileIndex, prepareDurableParsedFileChunk, durableChunkHasShards, } from '../../../storage/parsedfile-store.js';
|
|
21
21
|
import { DEFAULT_PDG_MAX_FUNCTION_LINES } from '../cfg/collect.js';
|
|
@@ -25,10 +25,11 @@ import { getLanguageFromFilename, } from '../../../_shared/index.js';
|
|
|
25
25
|
import { readFileContents } from '../filesystem-walker.js';
|
|
26
26
|
import { isLanguageAvailable, isGrammarRuntimeSkipped, createParserForLanguage, } from '../../tree-sitter/parser-loader.js';
|
|
27
27
|
import { parseSourceSafe } from '../../tree-sitter/safe-parse.js';
|
|
28
|
-
import { getProvider, getProviderForFile, providers } from '../languages/index.js';
|
|
28
|
+
import { getProvider, getProviderForFile, needsContentLanguageClassification, providers, } from '../languages/index.js';
|
|
29
|
+
import { classifyContentLanguages } from '../content-language-classification.js';
|
|
29
30
|
import { SCOPE_RESOLVERS } from '../scope-resolution/pipeline/registry.js';
|
|
30
31
|
import { DATA_ROUTE_TABLE_SOURCE } from '../route-extractors/data-route-table.js';
|
|
31
|
-
import { createWorkerPool, workerPoolDisabledByEnv, resolveAutoPoolSize, WorkerPoolInitializationError, WorkerPoolDisabledError, } from '../workers/worker-pool.js';
|
|
32
|
+
import { createWorkerPool, workerPoolDisabledByEnv, resolveAutoPoolSize, envWorkerPoolSize, resolveHostParallelism, WorkerPoolInitializationError, WorkerPoolDisabledError, } from '../workers/worker-pool.js';
|
|
32
33
|
import { normalizeExtractedRoutePath } from '../route-extractors/route-path.js';
|
|
33
34
|
import { resolveOperands } from '../route-extractors/python-const-resolver.js';
|
|
34
35
|
import { prepareRouteConstantsByProvider } from '../language-provider.js';
|
|
@@ -43,6 +44,8 @@ import { isVerboseIngestionEnabled } from '../utils/verbose.js';
|
|
|
43
44
|
import { endTimer, isDeferredResolutionProfileEnabled, logDeferredProfile, startTimer, } from '../utils/deferred-resolution-profile.js';
|
|
44
45
|
import { isDebugHeapEnabled, logHeapProbe } from '../utils/heap-probe.js';
|
|
45
46
|
import { logger } from '../../logger.js';
|
|
47
|
+
import { mapConcurrent } from '../../../lib/utils.js';
|
|
48
|
+
import { createRoundBudget } from './parse-round-budget.js';
|
|
46
49
|
// ── Constants ──────────────────────────────────────────────────────────────
|
|
47
50
|
/**
|
|
48
51
|
* Heap-scale guardrail constants (#2649). Measured on a Linux-kernel analyze:
|
|
@@ -130,8 +133,36 @@ const CHUNK_BYTES_PER_WORKER = DEFAULT_CHUNK_BYTE_BUDGET;
|
|
|
130
133
|
* while the rest finish early). Drives the derived `subBatchMaxBytes`.
|
|
131
134
|
*/
|
|
132
135
|
const TARGET_JOBS_PER_WORKER = 3;
|
|
136
|
+
/**
|
|
137
|
+
* Concurrent durable ParsedFile directory resets per round. Matches the file
|
|
138
|
+
* reader's `READ_CONCURRENCY`, because both compete for the same descriptors.
|
|
139
|
+
*/
|
|
140
|
+
const DURABLE_RESET_CONCURRENCY = 32;
|
|
133
141
|
/** Floor for a derived sub-batch so jobs don't shrink to per-file IPC churn. */
|
|
134
142
|
const MIN_SUB_BATCH_BYTES = 256 * 1024;
|
|
143
|
+
/**
|
|
144
|
+
* Source bytes an open round may HOLD — cache hits and misses alike — before
|
|
145
|
+
* it is dispatched and drained.
|
|
146
|
+
*
|
|
147
|
+
* A `dispatch` is a barrier, so one round-trip per cache pack leaves most slots
|
|
148
|
+
* idle: packs are keyed by `(language, hash(path) % 128)` and routinely land far
|
|
149
|
+
* under {@link DEFAULT_CHUNK_BYTE_BUDGET} (this repo: 1285 packs where the byte
|
|
150
|
+
* budget alone needs 16, 549 of them holding a single file). Rounds batch packs
|
|
151
|
+
* into one `dispatchGroups` call without touching pack identity.
|
|
152
|
+
*
|
|
153
|
+
* This is the in-flight cap, the same role Piscina's `maxQueue` plays: bigger
|
|
154
|
+
* rounds remove more barriers but hold more file content and more un-merged
|
|
155
|
+
* worker output on the main thread at once. Defaulting to one chunk budget
|
|
156
|
+
* keeps in-flight source bytes at the magnitude the loop already prefetched
|
|
157
|
+
* (`parseChunkConcurrency`, 2 chunks ahead). Override via
|
|
158
|
+
* `GITNEXUS_PARSE_ROUND_BYTES`.
|
|
159
|
+
*/
|
|
160
|
+
function resolveParseRoundByteBudget(options) {
|
|
161
|
+
const env = Number(process.env.GITNEXUS_PARSE_ROUND_BYTES);
|
|
162
|
+
if (Number.isFinite(env) && env > 0)
|
|
163
|
+
return env;
|
|
164
|
+
return resolveChunkByteBudget(options);
|
|
165
|
+
}
|
|
135
166
|
function resolveChunkByteBudget(options) {
|
|
136
167
|
const opt = options?.chunkByteBudget;
|
|
137
168
|
if (typeof opt === 'number' && Number.isFinite(opt) && opt > 0)
|
|
@@ -321,14 +352,25 @@ export function handleWorkerStartupFailure(err) {
|
|
|
321
352
|
export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, totalFiles, repoPath, pipelineStart, onProgress, options) {
|
|
322
353
|
const model = createSemanticModel();
|
|
323
354
|
const symbolTable = model.symbols;
|
|
355
|
+
const contentClassifiedPaths = scannedFiles
|
|
356
|
+
.map((file) => file.path)
|
|
357
|
+
.filter(needsContentLanguageClassification);
|
|
358
|
+
const contentLanguageByPath = contentClassifiedPaths.length > 0
|
|
359
|
+
? await classifyContentLanguages(repoPath, contentClassifiedPaths)
|
|
360
|
+
: new Map();
|
|
361
|
+
const languageForScannedFile = (file) => {
|
|
362
|
+
return contentLanguageByPath.has(file.path)
|
|
363
|
+
? (contentLanguageByPath.get(file.path) ?? null)
|
|
364
|
+
: getLanguageFromFilename(file.path);
|
|
365
|
+
};
|
|
324
366
|
const parseableScanned = scannedFiles.filter((f) => {
|
|
325
|
-
const lang =
|
|
367
|
+
const lang = languageForScannedFile(f);
|
|
326
368
|
return lang && isLanguageAvailable(lang);
|
|
327
369
|
});
|
|
328
370
|
// Warn about files skipped due to unavailable parsers
|
|
329
371
|
const skippedByLang = new Map();
|
|
330
372
|
for (const f of scannedFiles) {
|
|
331
|
-
const lang =
|
|
373
|
+
const lang = languageForScannedFile(f);
|
|
332
374
|
const provider = lang === null ? undefined : getProvider(lang);
|
|
333
375
|
if (lang && provider?.parseStrategy !== 'standalone' && !isLanguageAvailable(lang)) {
|
|
334
376
|
skippedByLang.set(lang, (skippedByLang.get(lang) || 0) + 1);
|
|
@@ -391,11 +433,38 @@ export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, t
|
|
|
391
433
|
// cores-based auto size is capped by source bytes / CHUNK_BYTES_PER_WORKER
|
|
392
434
|
// so a tiny repo does not spawn a full idle pool. Cache pack membership
|
|
393
435
|
// is independent of this number (#3088).
|
|
394
|
-
|
|
436
|
+
// `--workers <N>` and `GITNEXUS_WORKER_POOL_SIZE` are both deliberate
|
|
437
|
+
// operator input, so both bypass the work-proportional cap below. Only the
|
|
438
|
+
// env path used to be clamped by it, which made the documented escape hatch
|
|
439
|
+
// silently do nothing: on a 30MB repo the cap resolves to 16, so an operator
|
|
440
|
+
// asking for 24 still got 16 with no warning, while `--workers 24` got 24.
|
|
441
|
+
const explicitPoolSize = options?.workerPoolSize ?? envWorkerPoolSize();
|
|
442
|
+
// Cores-based auto size, bounded by source bytes so a tiny repo does not
|
|
443
|
+
// spawn a full idle pool.
|
|
395
444
|
const workProportionalCap = Math.max(1, Math.ceil(totalBytes / CHUNK_BYTES_PER_WORKER));
|
|
445
|
+
// An operator's number is honored, but never exceeds the number of files
|
|
446
|
+
// there are to parse — `GITNEXUS_WORKER_POOL_SIZE=100000` on a five-file repo
|
|
447
|
+
// should not become the literal thread count. This bounds `--workers` and the
|
|
448
|
+
// env var identically, keeping the parity above intact. Note it does NOT
|
|
449
|
+
// shrink an incremental re-analyze: `totalParseable` counts every parseable
|
|
450
|
+
// file in the scan, not the changed ones, so a warm run of a large repo still
|
|
451
|
+
// spawns the full requested pool.
|
|
396
452
|
const effectivePoolSize = explicitPoolSize && explicitPoolSize > 0
|
|
397
|
-
? explicitPoolSize
|
|
453
|
+
? Math.min(explicitPoolSize, Math.max(1, totalParseable))
|
|
398
454
|
: Math.min(resolveAutoPoolSize(), workProportionalCap);
|
|
455
|
+
// Deliberate over-subscription is the operator's call, so this warns rather
|
|
456
|
+
// than caps — silently capping is what the override exists to stop. But an
|
|
457
|
+
// exported `GITNEXUS_WORKER_POOL_SIZE` applies to EVERY analyze in a
|
|
458
|
+
// long-lived caller (watch auto-sync, the MCP server), including small
|
|
459
|
+
// incremental ones, and that is easy to set once and forget.
|
|
460
|
+
if (explicitPoolSize && explicitPoolSize > 0) {
|
|
461
|
+
const hostParallelism = resolveHostParallelism();
|
|
462
|
+
if (effectivePoolSize > hostParallelism) {
|
|
463
|
+
logger.warn({ requested: explicitPoolSize, spawning: effectivePoolSize, hostParallelism }, `Worker pool size ${effectivePoolSize} exceeds this host's ${hostParallelism} usable core(s); ` +
|
|
464
|
+
`parsing is CPU-bound, so the extra workers add memory pressure without throughput. ` +
|
|
465
|
+
`This applies to every analyze while the override is set.`);
|
|
466
|
+
}
|
|
467
|
+
}
|
|
399
468
|
// Cache packs: stable (language, hash(path) mod 128) buckets, then the
|
|
400
469
|
// per-call byte budget inside each bucket (#3088). Pool size is used only
|
|
401
470
|
// for worker count and sub-batch fan-out, not membership.
|
|
@@ -596,13 +665,43 @@ export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, t
|
|
|
596
665
|
// the env can't change mid-run.
|
|
597
666
|
const verboseThroughputLog = isDev || isVerboseIngestionEnabled();
|
|
598
667
|
const heapProbeEveryN = isDebugHeapEnabled() ? 25 : 0;
|
|
599
|
-
|
|
668
|
+
/**
|
|
669
|
+
* Chunk hashes whose durable ParsedFile directory could not be reset. The
|
|
670
|
+
* old generation's shards are still on disk, so a warm hit would union
|
|
671
|
+
* stale shards with the new ones. Treated exactly like a quarantined chunk:
|
|
672
|
+
* skip the parse-cache write so the next run re-dispatches into a clean
|
|
673
|
+
* directory rather than trusting a generation we could not clear.
|
|
674
|
+
*/
|
|
675
|
+
const durablePrepareFailures = new Set();
|
|
676
|
+
const roundByteBudget = resolveParseRoundByteBudget(options);
|
|
677
|
+
let roundEntries = [];
|
|
678
|
+
/**
|
|
679
|
+
* Bytes an open round is HOLDING, counting hits as well as misses.
|
|
680
|
+
*
|
|
681
|
+
* Counting only the cache-MISSING bytes would bound just what the workers
|
|
682
|
+
* are asked to do, so a warm run — where nothing misses — would never reach
|
|
683
|
+
* the close condition and would buffer every chunk's cached output until
|
|
684
|
+
* the tail drain. That is the #2649 heap failure on a large repo. Counting
|
|
685
|
+
* both keeps a hits-only run draining at the same cadence as a cold one;
|
|
686
|
+
* `startRound` already supports a round with no misses.
|
|
687
|
+
*
|
|
688
|
+
* Measured in UTF-8 bytes, matching `estimateItemBytes` in the worker pool,
|
|
689
|
+
* so the cap means the same thing here as it does for a job's payload.
|
|
690
|
+
*/
|
|
691
|
+
const roundBudget = createRoundBudget(roundByteBudget);
|
|
692
|
+
/**
|
|
693
|
+
* Files QUEUED into rounds so far. `filesParsedSoFar` only advances when a
|
|
694
|
+
* round drains, so it is the right number for the throughput log but would
|
|
695
|
+
* pin a warm run's progress bar at the phase floor for the whole loop.
|
|
696
|
+
*/
|
|
697
|
+
let queuedFilesSoFar = 0;
|
|
698
|
+
let pendingRound = null;
|
|
600
699
|
// Apply one chunk's merged worker data: per-chunk aggregation into the
|
|
601
700
|
// run-level accumulators + the throughput log. Shared by the cache-hit
|
|
602
701
|
// (inline) and worker (deferred) paths. The `| null` guard is defensive —
|
|
603
702
|
// every live caller passes real worker data now that sequential parsing
|
|
604
703
|
// (which was the only path that passed null) is gone.
|
|
605
|
-
const applyChunkResults = async (chunkWorkerData, chunkIdx,
|
|
704
|
+
const applyChunkResults = async (chunkWorkerData, chunkIdx, fileCount, chunkStartMs) => {
|
|
606
705
|
if (chunkWorkerData) {
|
|
607
706
|
for (const filePath of chunkWorkerData.scopeExtractionFailures) {
|
|
608
707
|
scopeExtractionFailures.add(filePath);
|
|
@@ -690,30 +789,35 @@ export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, t
|
|
|
690
789
|
allORMQueries.push(item);
|
|
691
790
|
}
|
|
692
791
|
}
|
|
693
|
-
filesParsedSoFar +=
|
|
792
|
+
filesParsedSoFar += fileCount;
|
|
694
793
|
if (verboseThroughputLog && chunkStartMs !== null) {
|
|
695
794
|
const elapsedMs = Date.now() - chunkStartMs;
|
|
696
|
-
const filesPerSec = elapsedMs > 0 ? (
|
|
795
|
+
const filesPerSec = elapsedMs > 0 ? (fileCount * 1000) / elapsedMs : 0;
|
|
697
796
|
const stats = workerPool?.getStats?.();
|
|
698
797
|
const poolFrag = stats
|
|
699
798
|
? ` pool: ${stats.activeSlots}/${stats.size} active, ` +
|
|
700
799
|
`${stats.quarantined} quarantined${stats.poolBroken ? ', BROKEN' : ''}`
|
|
701
800
|
: ' (cache replay)';
|
|
702
|
-
logger.info(`📊 chunk ${chunkIdx + 1}/${numChunks}: ${
|
|
801
|
+
logger.info(`📊 chunk ${chunkIdx + 1}/${numChunks}: ${fileCount} files in ${elapsedMs}ms ` +
|
|
703
802
|
`(${filesPerSec.toFixed(1)} files/s)${poolFrag}`);
|
|
704
803
|
}
|
|
705
804
|
};
|
|
706
805
|
// Merge + finalize a parked worker chunk: graph merge (the overlapped
|
|
707
806
|
// main-thread step) → parse-cache write-guard → run-level aggregation.
|
|
708
|
-
const finalizeWorkerChunk = async (p) => {
|
|
709
|
-
const chunkWorkerData = mergeChunkResults(graph, symbolTable,
|
|
807
|
+
const finalizeWorkerChunk = async (p, rawResults) => {
|
|
808
|
+
const chunkWorkerData = mergeChunkResults(graph, symbolTable, rawResults, exportedTypeMap);
|
|
710
809
|
// Persist raw results for this chunk hash (skipping when any chunk file
|
|
711
810
|
// was worker-quarantined, so the narrower rawResults isn't cached under
|
|
712
811
|
// the full-chunk key — see the original inline note / U20.U2).
|
|
713
|
-
if (parseCache && p.chunkHash &&
|
|
812
|
+
if (parseCache && p.chunkHash && rawResults.length > 0) {
|
|
714
813
|
const quarantineSet = new Set(workerPool?.getQuarantinedPaths?.() ?? []);
|
|
715
814
|
const chunkHadQuarantine = p.chunkFiles.some((f) => quarantineSet.has(f.path));
|
|
716
|
-
|
|
815
|
+
const durableGenerationStale = durablePrepareFailures.has(p.chunkHash);
|
|
816
|
+
if (durableGenerationStale) {
|
|
817
|
+
logger.warn({ chunkHash: p.chunkHash.slice(0, 8) }, 'parse-cache SKIP: durable generation for this chunk could not be reset, ' +
|
|
818
|
+
'so its shards may be stale; next run will re-dispatch it');
|
|
819
|
+
}
|
|
820
|
+
else if (chunkHadQuarantine) {
|
|
717
821
|
if (isDev) {
|
|
718
822
|
const quarantinedInChunk = p.chunkFiles.filter((f) => quarantineSet.has(f.path)).length;
|
|
719
823
|
logger.info(`📦 parse-cache SKIP: chunk ${p.chunkIdx + 1}/${numChunks} ` +
|
|
@@ -722,13 +826,164 @@ export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, t
|
|
|
722
826
|
}
|
|
723
827
|
}
|
|
724
828
|
else {
|
|
725
|
-
await persistParseCacheChunk(parseCache, p.chunkHash,
|
|
829
|
+
await persistParseCacheChunk(parseCache, p.chunkHash, rawResults);
|
|
726
830
|
if (isDev) {
|
|
727
831
|
logger.info(`📦 parse-cache MISS+store: chunk ${p.chunkIdx + 1}/${numChunks} (${p.chunkFiles.length} files, ${p.chunkHash.slice(0, 8)})`);
|
|
728
832
|
}
|
|
729
833
|
}
|
|
730
834
|
}
|
|
731
|
-
await applyChunkResults(chunkWorkerData, p.chunkIdx, p.chunkFiles, p.chunkStartMs);
|
|
835
|
+
await applyChunkResults(chunkWorkerData, p.chunkIdx, p.chunkFiles.length, p.chunkStartMs);
|
|
836
|
+
};
|
|
837
|
+
/**
|
|
838
|
+
* Dispatch a round's cache misses as ONE pool round. Returns the parked
|
|
839
|
+
* round; the caller drains it after starting the next one so the workers
|
|
840
|
+
* parse round N+1 while the main thread merges round N (the same overlap
|
|
841
|
+
* the per-chunk loop had, at round granularity).
|
|
842
|
+
*/
|
|
843
|
+
const startRound = async (entries) => {
|
|
844
|
+
if (entries.length === 0)
|
|
845
|
+
return null;
|
|
846
|
+
const misses = entries.filter((entry) => entry.kind === 'miss');
|
|
847
|
+
if (misses.length === 0) {
|
|
848
|
+
return { entries, results: Promise.resolve([]) };
|
|
849
|
+
}
|
|
850
|
+
// Each chunk resets its own directory, so these are independent and run
|
|
851
|
+
// concurrently: serially they would sit on the critical path this round
|
|
852
|
+
// exists to shorten, with the pool idle and the previous round's merge
|
|
853
|
+
// waiting, once per miss.
|
|
854
|
+
//
|
|
855
|
+
// BOUNDED, though. A round can hold hundreds of small packs, and each
|
|
856
|
+
// reset is a recursive rm + mkdir. Firing all of them at once competes
|
|
857
|
+
// for descriptors with the chunk prefetch this loop already has in
|
|
858
|
+
// flight, and `readFileContents` degrades a losing read SILENTLY by
|
|
859
|
+
// contract — a dropped file would vanish from the chunk, from the graph,
|
|
860
|
+
// and from the chunk hash, shipping a narrowed index with exit 0. Same
|
|
861
|
+
// helper and width the file reads use.
|
|
862
|
+
await mapConcurrent(misses, async (miss) => {
|
|
863
|
+
if (durableParsedFileDir === undefined || miss.chunkHash === null)
|
|
864
|
+
return;
|
|
865
|
+
try {
|
|
866
|
+
await prepareDurableParsedFileChunk(durableParsedFileDir, miss.chunkHash);
|
|
867
|
+
}
|
|
868
|
+
catch (err) {
|
|
869
|
+
// The durable store is an optimization — degrade like the restore
|
|
870
|
+
// path does instead of failing the analyze. Workers recreate the
|
|
871
|
+
// directory on write, so at worst the old generation lingers.
|
|
872
|
+
// Caught per chunk so one failure cannot abort the others.
|
|
873
|
+
durablePrepareFailures.add(miss.chunkHash);
|
|
874
|
+
logger.warn({ err, chunkHash: miss.chunkHash.slice(0, 8) }, 'parsedfile-cache: could not reset durable chunk generation; ' +
|
|
875
|
+
'continuing without caching this chunk');
|
|
876
|
+
}
|
|
877
|
+
}, { concurrency: DURABLE_RESET_CONCURRENCY });
|
|
878
|
+
const roundFiles = misses.reduce((sum, miss) => sum + miss.chunkFiles.length, 0);
|
|
879
|
+
const firstIdx = misses[0].chunkIdx;
|
|
880
|
+
const lastIdx = misses[misses.length - 1].chunkIdx;
|
|
881
|
+
const progressForRound = (current, _total, filePath) => {
|
|
882
|
+
// Rounds queued before this one are already counted in
|
|
883
|
+
// `queuedFilesSoFar`; `current` is this round's own worker progress.
|
|
884
|
+
const globalCurrent = queuedFilesSoFar - roundFiles + current;
|
|
885
|
+
// Parse phase covers 20-70 (M2). Deferred extraction handles 70-95.
|
|
886
|
+
const parsingProgress = 20 + (globalCurrent / totalParseable) * 50;
|
|
887
|
+
onProgress({
|
|
888
|
+
phase: 'parsing',
|
|
889
|
+
percent: Math.round(parsingProgress),
|
|
890
|
+
message: firstIdx === lastIdx
|
|
891
|
+
? `Parsing chunk ${firstIdx + 1}/${numChunks}...`
|
|
892
|
+
: `Parsing chunks ${firstIdx + 1}-${lastIdx + 1}/${numChunks}...`,
|
|
893
|
+
detail: filePath,
|
|
894
|
+
stats: {
|
|
895
|
+
filesProcessed: globalCurrent,
|
|
896
|
+
totalFiles: totalParseable,
|
|
897
|
+
nodesCreated: graph.nodeCount,
|
|
898
|
+
},
|
|
899
|
+
});
|
|
900
|
+
};
|
|
901
|
+
const activeWorkerPool = getOrCreateWorkerPool();
|
|
902
|
+
if (verboseThroughputLog) {
|
|
903
|
+
logger.info(`🚚 round: ${misses.length} chunk(s) ${firstIdx + 1}-${lastIdx + 1}/${numChunks}, ` +
|
|
904
|
+
`${roundFiles} files in one dispatch`);
|
|
905
|
+
}
|
|
906
|
+
const results = dispatchChunkParseRound(misses.map((miss) => ({
|
|
907
|
+
items: miss.chunkFiles,
|
|
908
|
+
chunkHash: miss.chunkHash ?? undefined,
|
|
909
|
+
})), activeWorkerPool, progressForRound);
|
|
910
|
+
// Mark handled so a rejection during the overlap drain below isn't
|
|
911
|
+
// flagged as unhandled; the `await` in drainRound re-throws it for real
|
|
912
|
+
// handling.
|
|
913
|
+
results.catch(() => { });
|
|
914
|
+
return { entries, results };
|
|
915
|
+
};
|
|
916
|
+
/**
|
|
917
|
+
* Merge + finalize every chunk of a parked round, in `chunkIdx` order.
|
|
918
|
+
* Takes RESOLVED worker output: the round's dispatch must already have
|
|
919
|
+
* settled before this runs, because the pool allows only one dispatch in
|
|
920
|
+
* flight at a time (see `closeRound`).
|
|
921
|
+
*/
|
|
922
|
+
const drainRound = async (round) => {
|
|
923
|
+
const missResults = round.missResults;
|
|
924
|
+
const missCount = round.entries.filter((entry) => entry.kind === 'miss').length;
|
|
925
|
+
// `dispatchGroups` returns one array per input group. If that contract
|
|
926
|
+
// ever breaks, every later entry in this round would silently merge the
|
|
927
|
+
// wrong chunk's results and skip its cache write, with a clean exit.
|
|
928
|
+
if (missResults.length !== missCount) {
|
|
929
|
+
throw new Error(`Parse round result mismatch: ${missResults.length} result group(s) for ${missCount} dispatched chunk(s).`);
|
|
930
|
+
}
|
|
931
|
+
let missIdx = 0;
|
|
932
|
+
for (const entry of round.entries) {
|
|
933
|
+
if (entry.kind === 'hit') {
|
|
934
|
+
const chunkWorkerData = mergeChunkResults(graph, symbolTable, entry.cachedRaw, exportedTypeMap);
|
|
935
|
+
await applyChunkResults(chunkWorkerData, entry.chunkIdx, entry.fileCount, entry.chunkStartMs);
|
|
936
|
+
continue;
|
|
937
|
+
}
|
|
938
|
+
await finalizeWorkerChunk(entry, missResults[missIdx++]);
|
|
939
|
+
}
|
|
940
|
+
};
|
|
941
|
+
/**
|
|
942
|
+
* Close the accumulated round.
|
|
943
|
+
*
|
|
944
|
+
* `WorkerPool.dispatch`/`dispatchGroups` is NOT reentrant — concurrent
|
|
945
|
+
* calls race on the shared per-slot busy/in-flight state and wedge the
|
|
946
|
+
* pool until every worker idle-times out. So exactly one dispatch is in
|
|
947
|
+
* flight here: start this round, merge the PREVIOUS round (whose results
|
|
948
|
+
* are already resolved) while these workers run, then await this round and
|
|
949
|
+
* park it resolved for the next close to merge.
|
|
950
|
+
*/
|
|
951
|
+
const closeRound = async () => {
|
|
952
|
+
const started = await startRound(roundEntries);
|
|
953
|
+
roundEntries = [];
|
|
954
|
+
roundBudget.reset();
|
|
955
|
+
const previous = pendingRound;
|
|
956
|
+
pendingRound = null;
|
|
957
|
+
if (previous) {
|
|
958
|
+
try {
|
|
959
|
+
await drainRound(previous);
|
|
960
|
+
}
|
|
961
|
+
catch (err) {
|
|
962
|
+
// The round started above is still on the workers. Unwinding now
|
|
963
|
+
// reaches this function's `finally`, which calls `terminate()` — and
|
|
964
|
+
// terminate kills busy workers outright, which is the #2432
|
|
965
|
+
// mid-N-API SIGABRT hazard. Let the in-flight round settle first so
|
|
966
|
+
// the pool is idle, then propagate the original failure.
|
|
967
|
+
await started?.results.catch(() => undefined);
|
|
968
|
+
throw err;
|
|
969
|
+
}
|
|
970
|
+
}
|
|
971
|
+
if (!started)
|
|
972
|
+
return;
|
|
973
|
+
let missResults;
|
|
974
|
+
try {
|
|
975
|
+
missResults = await started.results;
|
|
976
|
+
}
|
|
977
|
+
catch (err) {
|
|
978
|
+
if (!(err instanceof WorkerPoolInitializationError))
|
|
979
|
+
throw err;
|
|
980
|
+
// Every worker crashed during startup and the pool's bounded self-heal
|
|
981
|
+
// was exhausted. Fail fast (#1741) — there is no sequential parser to
|
|
982
|
+
// degrade to. `handleWorkerStartupFailure` always throws, so
|
|
983
|
+
// `missResults` stays definitely assigned for the parked round below.
|
|
984
|
+
handleWorkerStartupFailure(err);
|
|
985
|
+
}
|
|
986
|
+
pendingRound = { entries: started.entries, missResults };
|
|
732
987
|
};
|
|
733
988
|
for (let chunkIdx = 0; chunkIdx < numChunks; chunkIdx++) {
|
|
734
989
|
if (heapProbeEveryN > 0 && chunkIdx > 0 && chunkIdx % heapProbeEveryN === 0) {
|
|
@@ -808,17 +1063,14 @@ export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, t
|
|
|
808
1063
|
parsedFileStorePath !== undefined &&
|
|
809
1064
|
durableExpectedPaths !== undefined &&
|
|
810
1065
|
(await durableChunkHasShards(parsedFileStorePath, chunkHash, durableExpectedPaths));
|
|
1066
|
+
// Set by whichever branch queues this chunk; drives the close below.
|
|
1067
|
+
let roundIsFull = false;
|
|
811
1068
|
if (cachedRaw && cachedRaw.length > 0 && (durableHit || parsedFileStorePath === undefined)) {
|
|
812
1069
|
// Cache hit: replay cached worker output. Finalize any parked worker
|
|
813
1070
|
// chunk FIRST so deferred aggregation stays in chunk order, then merge
|
|
814
1071
|
// + apply this hit inline (no worker dispatch to overlap).
|
|
815
|
-
if (pendingWorkerChunk) {
|
|
816
|
-
await finalizeWorkerChunk(pendingWorkerChunk);
|
|
817
|
-
pendingWorkerChunk = null;
|
|
818
|
-
}
|
|
819
1072
|
chunkCacheHits++;
|
|
820
1073
|
parseCacheHitFileCount += chunkFiles.length;
|
|
821
|
-
const chunkWorkerData = mergeChunkResults(graph, symbolTable, cachedRaw, exportedTypeMap);
|
|
822
1074
|
if (isDev) {
|
|
823
1075
|
logger.info(`📦 parse-cache HIT: chunk ${chunkIdx + 1}/${numChunks} (${chunkFiles.length} files, ${chunkHash?.slice(0, 8) ?? 'unknown'})`);
|
|
824
1076
|
}
|
|
@@ -830,97 +1082,55 @@ export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, t
|
|
|
830
1082
|
// takes 70-95 so the UI advances through the (potentially long)
|
|
831
1083
|
// resolution stages instead of holding at 82 (M2 from PR #1693
|
|
832
1084
|
// review).
|
|
833
|
-
percent: Math.round(20 + ((
|
|
1085
|
+
percent: Math.round(20 + ((queuedFilesSoFar + cachedFiles) / totalParseable) * 50),
|
|
834
1086
|
message: `Parsing chunk ${chunkIdx + 1}/${numChunks} (cache)...`,
|
|
835
1087
|
stats: {
|
|
836
|
-
filesProcessed:
|
|
1088
|
+
filesProcessed: queuedFilesSoFar + cachedFiles,
|
|
837
1089
|
totalFiles: totalParseable,
|
|
838
1090
|
nodesCreated: graph.nodeCount,
|
|
839
1091
|
},
|
|
840
1092
|
});
|
|
841
1093
|
// The durable gate already snapshotted warm `.v8` shards into the
|
|
842
|
-
// run-scoped store for scope resolution.
|
|
843
|
-
|
|
1094
|
+
// run-scoped store for scope resolution. Queue into the round so this
|
|
1095
|
+
// hit still finalizes in `chunkIdx` order relative to its neighbours.
|
|
1096
|
+
roundEntries.push({
|
|
1097
|
+
kind: 'hit',
|
|
1098
|
+
chunkIdx,
|
|
1099
|
+
fileCount: chunkFiles.length,
|
|
1100
|
+
chunkStartMs,
|
|
1101
|
+
cachedRaw,
|
|
1102
|
+
});
|
|
1103
|
+
roundIsFull = roundBudget.addChunk(chunkFiles.map((file) => file.content));
|
|
1104
|
+
queuedFilesSoFar += chunkFiles.length;
|
|
844
1105
|
}
|
|
845
1106
|
else {
|
|
846
|
-
// Cache miss:
|
|
847
|
-
//
|
|
1107
|
+
// Cache miss: queue for the round's single dispatch; the raw results
|
|
1108
|
+
// are stored under the chunk hash when the round drains.
|
|
848
1109
|
chunkCacheMisses++;
|
|
849
1110
|
reparsedFileCount += chunkFiles.length;
|
|
850
|
-
|
|
851
|
-
|
|
852
|
-
|
|
853
|
-
}
|
|
854
|
-
catch (err) {
|
|
855
|
-
// The durable store is an optimization — degrade like the restore
|
|
856
|
-
// path does instead of failing the analyze. Workers recreate the
|
|
857
|
-
// directory on write, so at worst the old generation lingers.
|
|
858
|
-
logger.warn({ err, chunkHash: chunkHash.slice(0, 8) }, 'parsedfile-cache: could not reset durable chunk generation; continuing');
|
|
859
|
-
}
|
|
860
|
-
}
|
|
861
|
-
const progressForChunk = (current, _total, filePath) => {
|
|
862
|
-
const globalCurrent = filesParsedSoFar + current;
|
|
863
|
-
// Parse phase covers 20-70 (M2). Deferred extraction handles 70-95.
|
|
864
|
-
const parsingProgress = 20 + (globalCurrent / totalParseable) * 50;
|
|
865
|
-
onProgress({
|
|
866
|
-
phase: 'parsing',
|
|
867
|
-
percent: Math.round(parsingProgress),
|
|
868
|
-
message: `Parsing chunk ${chunkIdx + 1}/${numChunks}...`,
|
|
869
|
-
detail: filePath,
|
|
870
|
-
stats: {
|
|
871
|
-
filesProcessed: globalCurrent,
|
|
872
|
-
totalFiles: totalParseable,
|
|
873
|
-
nodesCreated: graph.nodeCount,
|
|
874
|
-
},
|
|
875
|
-
});
|
|
876
|
-
};
|
|
877
|
-
const activeWorkerPool = getOrCreateWorkerPool();
|
|
878
|
-
// Worker path — PIPELINE: kick off this chunk's dispatch, merge the
|
|
879
|
-
// PREVIOUS chunk while these workers parse, then park this chunk for
|
|
880
|
-
// the next iteration to merge (overlapping its parse). The deferred
|
|
881
|
-
// merge + parse-cache write-guard + aggregation all run in
|
|
882
|
-
// `finalizeWorkerChunk`, in chunk order. The pool is the sole parse
|
|
883
|
-
// path — `getOrCreateWorkerPool` returns a pool or throws.
|
|
884
|
-
const dispatchPromise = dispatchChunkParse(chunkFiles, activeWorkerPool, progressForChunk, undefined, chunkHash ?? undefined);
|
|
885
|
-
// Mark handled so a rejection during the overlap drain below isn't
|
|
886
|
-
// flagged as unhandled; the `await` re-throws it for real handling.
|
|
887
|
-
dispatchPromise.catch(() => { });
|
|
888
|
-
if (pendingWorkerChunk) {
|
|
889
|
-
await finalizeWorkerChunk(pendingWorkerChunk);
|
|
890
|
-
pendingWorkerChunk = null;
|
|
891
|
-
}
|
|
892
|
-
let chunkResults;
|
|
893
|
-
try {
|
|
894
|
-
chunkResults = await dispatchPromise;
|
|
895
|
-
}
|
|
896
|
-
catch (err) {
|
|
897
|
-
if (!(err instanceof WorkerPoolInitializationError))
|
|
898
|
-
throw err;
|
|
899
|
-
// Every worker crashed during startup and the pool's bounded
|
|
900
|
-
// self-heal was exhausted. Fail fast (#1741) — there is no sequential
|
|
901
|
-
// parser to degrade to. `handleWorkerStartupFailure` always throws, so
|
|
902
|
-
// `chunkResults` stays definitely assigned for the parked chunk below.
|
|
903
|
-
handleWorkerStartupFailure(err);
|
|
904
|
-
}
|
|
905
|
-
pendingWorkerChunk = {
|
|
906
|
-
rawResults: chunkResults,
|
|
907
|
-
chunkIdx,
|
|
908
|
-
chunkHash,
|
|
909
|
-
chunkFiles,
|
|
910
|
-
chunkStartMs,
|
|
911
|
-
};
|
|
1111
|
+
roundEntries.push({ kind: 'miss', chunkIdx, chunkHash, chunkFiles, chunkStartMs });
|
|
1112
|
+
roundIsFull = roundBudget.addChunk(chunkFiles.map((file) => file.content));
|
|
1113
|
+
queuedFilesSoFar += chunkFiles.length;
|
|
912
1114
|
}
|
|
1115
|
+
// One cap, on what the main thread is holding. That bounds the worker
|
|
1116
|
+
// round too, since a round's dispatched bytes are a subset of its
|
|
1117
|
+
// buffered bytes.
|
|
1118
|
+
if (roundIsFull)
|
|
1119
|
+
await closeRound();
|
|
913
1120
|
// (Per-chunk aggregation + parse-cache write + throughput log now run in
|
|
914
1121
|
// `applyChunkResults` / `finalizeWorkerChunk` — see the merge-pipelining
|
|
915
1122
|
// block above. Route/import/inheritance edges are emitted later: route
|
|
916
1123
|
// resolution in the single end-of-loop pass below, the rest by the
|
|
917
1124
|
// scope-resolution phase, RING4-2 #943.)
|
|
918
1125
|
}
|
|
919
|
-
// Drain the
|
|
920
|
-
// successor to overlap its merge with
|
|
921
|
-
if (
|
|
922
|
-
await
|
|
923
|
-
|
|
1126
|
+
// Drain the tail: close the partially-filled round, then drain the round
|
|
1127
|
+
// it parked — the last round has no successor to overlap its merge with.
|
|
1128
|
+
if (roundEntries.length > 0)
|
|
1129
|
+
await closeRound();
|
|
1130
|
+
if (pendingRound) {
|
|
1131
|
+
const last = pendingRound;
|
|
1132
|
+
pendingRound = null;
|
|
1133
|
+
await drainRound(last);
|
|
924
1134
|
}
|
|
925
1135
|
if (isDev && parseCache && (chunkCacheHits > 0 || chunkCacheMisses > 0)) {
|
|
926
1136
|
logger.info(`📦 parse-cache summary: ${chunkCacheHits} chunk hit(s), ${chunkCacheMisses} miss(es) across ${numChunks} chunk(s)`);
|
|
@@ -1332,6 +1542,7 @@ export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, t
|
|
|
1332
1542
|
// cache: when the file's ParsedFile is here, scope-resolution skips its own
|
|
1333
1543
|
// `extractParsedFile` call.
|
|
1334
1544
|
parsedFiles: allParsedFiles,
|
|
1545
|
+
contentLanguageByPath,
|
|
1335
1546
|
// Repo-wide, file-path-keyed constants, already through each provider's
|
|
1336
1547
|
// `prepareRouteConstants` hook. Empty when no provider harvests constants
|
|
1337
1548
|
// for the languages in this repo.
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The fold that decides when an open dispatch round closes.
|
|
3
|
+
*
|
|
4
|
+
* Extracted so the decision is a shared, inspectable unit rather than four
|
|
5
|
+
* loose statements inside `runChunkedParseAndResolve`. The parse loop is
|
|
6
|
+
* STREAMING — it reads chunk contents lazily, so it cannot know every chunk's
|
|
7
|
+
* size up front and cannot "plan" rounds ahead. That makes an accumulator, not
|
|
8
|
+
* a planner, the honest shape: feed it each chunk as it is queued and it tells
|
|
9
|
+
* you whether the round is now full.
|
|
10
|
+
*
|
|
11
|
+
* Being a real unit is what makes round cadence observable. Round boundaries
|
|
12
|
+
* are otherwise invisible from outside the parse phase: they change no graph
|
|
13
|
+
* output (that is the point of batching) and surface only in a log line, which
|
|
14
|
+
* is why `bench/parse-dispatch-rounds` measures this directly rather than
|
|
15
|
+
* inferring cadence from a full analyze.
|
|
16
|
+
*/
|
|
17
|
+
/** Bytes a file contributes to the open round's retained total. */
|
|
18
|
+
export declare const roundFileBytes: (content: string) => number;
|
|
19
|
+
export interface RoundBudget {
|
|
20
|
+
/**
|
|
21
|
+
* Add one queued chunk's files. Returns true when the round is now full and
|
|
22
|
+
* the caller should close it. Closing resets the accumulator.
|
|
23
|
+
*/
|
|
24
|
+
addChunk(contents: readonly string[]): boolean;
|
|
25
|
+
/** Bytes currently held by the open round. */
|
|
26
|
+
readonly bufferedBytes: number;
|
|
27
|
+
/** Reset without closing — used when the caller closes for another reason. */
|
|
28
|
+
reset(): void;
|
|
29
|
+
}
|
|
30
|
+
/**
|
|
31
|
+
* `budgetBytes` bounds what the main thread HOLDS, counting cache hits as well
|
|
32
|
+
* as misses. Counting only cache-missing bytes would bound just the work sent
|
|
33
|
+
* to workers, so a warm run — where nothing misses — would never reach the
|
|
34
|
+
* close condition and would buffer every chunk's cached output until the tail
|
|
35
|
+
* drain. That is the #2649 heap failure on a large repo.
|
|
36
|
+
*
|
|
37
|
+
* Measured in UTF-8 bytes, matching `estimateItemBytes` in the worker pool.
|
|
38
|
+
* `String.length` would return UTF-16 code units, undercounting non-ASCII
|
|
39
|
+
* source by up to 3x and letting a CJK-heavy repo hold well past its nominal
|
|
40
|
+
* budget before draining.
|
|
41
|
+
*/
|
|
42
|
+
export declare const createRoundBudget: (budgetBytes: number) => RoundBudget;
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The fold that decides when an open dispatch round closes.
|
|
3
|
+
*
|
|
4
|
+
* Extracted so the decision is a shared, inspectable unit rather than four
|
|
5
|
+
* loose statements inside `runChunkedParseAndResolve`. The parse loop is
|
|
6
|
+
* STREAMING — it reads chunk contents lazily, so it cannot know every chunk's
|
|
7
|
+
* size up front and cannot "plan" rounds ahead. That makes an accumulator, not
|
|
8
|
+
* a planner, the honest shape: feed it each chunk as it is queued and it tells
|
|
9
|
+
* you whether the round is now full.
|
|
10
|
+
*
|
|
11
|
+
* Being a real unit is what makes round cadence observable. Round boundaries
|
|
12
|
+
* are otherwise invisible from outside the parse phase: they change no graph
|
|
13
|
+
* output (that is the point of batching) and surface only in a log line, which
|
|
14
|
+
* is why `bench/parse-dispatch-rounds` measures this directly rather than
|
|
15
|
+
* inferring cadence from a full analyze.
|
|
16
|
+
*/
|
|
17
|
+
/** Bytes a file contributes to the open round's retained total. */
|
|
18
|
+
export const roundFileBytes = (content) => Buffer.byteLength(content, 'utf8');
|
|
19
|
+
/**
|
|
20
|
+
* `budgetBytes` bounds what the main thread HOLDS, counting cache hits as well
|
|
21
|
+
* as misses. Counting only cache-missing bytes would bound just the work sent
|
|
22
|
+
* to workers, so a warm run — where nothing misses — would never reach the
|
|
23
|
+
* close condition and would buffer every chunk's cached output until the tail
|
|
24
|
+
* drain. That is the #2649 heap failure on a large repo.
|
|
25
|
+
*
|
|
26
|
+
* Measured in UTF-8 bytes, matching `estimateItemBytes` in the worker pool.
|
|
27
|
+
* `String.length` would return UTF-16 code units, undercounting non-ASCII
|
|
28
|
+
* source by up to 3x and letting a CJK-heavy repo hold well past its nominal
|
|
29
|
+
* budget before draining.
|
|
30
|
+
*/
|
|
31
|
+
export const createRoundBudget = (budgetBytes) => {
|
|
32
|
+
let bufferedBytes = 0;
|
|
33
|
+
return {
|
|
34
|
+
addChunk(contents) {
|
|
35
|
+
for (const content of contents)
|
|
36
|
+
bufferedBytes += roundFileBytes(content);
|
|
37
|
+
if (bufferedBytes >= budgetBytes) {
|
|
38
|
+
bufferedBytes = 0;
|
|
39
|
+
return true;
|
|
40
|
+
}
|
|
41
|
+
return false;
|
|
42
|
+
},
|
|
43
|
+
get bufferedBytes() {
|
|
44
|
+
return bufferedBytes;
|
|
45
|
+
},
|
|
46
|
+
reset() {
|
|
47
|
+
bufferedBytes = 0;
|
|
48
|
+
},
|
|
49
|
+
};
|
|
50
|
+
};
|