gitnexus 1.6.11 → 1.6.12-rc.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. package/README.md +1 -1
  2. package/dist/cli/analyze-config.d.ts +2 -2
  3. package/dist/cli/analyze-config.js +10 -33
  4. package/dist/cli/optional-grammars.d.ts +7 -8
  5. package/dist/cli/optional-grammars.js +9 -19
  6. package/dist/core/git-ref.d.ts +22 -0
  7. package/dist/core/git-ref.js +114 -0
  8. package/dist/core/ingestion/call-extractors/zig-static-gating.d.ts +5 -8
  9. package/dist/core/ingestion/languages/zig/query.js +7 -8
  10. package/dist/core/ingestion/languages/zig/scope-resolver.js +2 -0
  11. package/dist/core/ingestion/languages/zig/workspace-static-gating.d.ts +8 -0
  12. package/dist/core/ingestion/languages/zig/workspace-static-gating.js +76 -0
  13. package/dist/core/ingestion/parsing-processor.d.ts +22 -6
  14. package/dist/core/ingestion/parsing-processor.js +38 -19
  15. package/dist/core/ingestion/pipeline-phases/parse-impl.js +291 -93
  16. package/dist/core/ingestion/pipeline-phases/parse-round-budget.d.ts +42 -0
  17. package/dist/core/ingestion/pipeline-phases/parse-round-budget.js +50 -0
  18. package/dist/core/ingestion/scope-resolution/contract/scope-resolver.d.ts +13 -0
  19. package/dist/core/ingestion/scope-resolution/pipeline/phase.js +1 -0
  20. package/dist/core/ingestion/scope-resolution/pipeline/run.js +5 -0
  21. package/dist/core/ingestion/workers/parse-worker.js +14 -6
  22. package/dist/core/ingestion/workers/worker-pool.d.ts +56 -2
  23. package/dist/core/ingestion/workers/worker-pool.js +117 -26
  24. package/dist/core/search/fts-indexes.js +13 -4
  25. package/dist/core/tree-sitter/parser-loader.js +4 -12
  26. package/dist/core/tree-sitter/vendored-grammars.d.ts +1 -1
  27. package/dist/core/tree-sitter/vendored-grammars.js +2 -1
  28. package/dist/mcp/local/local-backend.js +9 -3
  29. package/dist/server/analyze-job.d.ts +19 -1
  30. package/dist/server/analyze-job.js +14 -3
  31. package/dist/server/analyze-launch.d.ts +14 -0
  32. package/dist/server/analyze-launch.js +59 -14
  33. package/dist/server/analyze-worker-ipc.d.ts +6 -5
  34. package/dist/server/analyze-worker-ipc.js +4 -0
  35. package/dist/server/api.js +53 -6
  36. package/dist/server/git-clone.d.ts +38 -5
  37. package/dist/server/git-clone.js +216 -38
  38. package/dist/storage/parsedfile-store.js +53 -1
  39. package/dist/storage/v8-sidecar.d.ts +11 -0
  40. package/dist/storage/v8-sidecar.js +15 -7
  41. package/package.json +1 -5
  42. package/scripts/build-tree-sitter-grammars.cjs +2 -1
  43. package/vendor/tree-sitter-zig/LICENSE +21 -0
  44. package/vendor/tree-sitter-zig/README.md +25 -0
  45. package/vendor/tree-sitter-zig/binding.gyp +29 -0
  46. package/vendor/tree-sitter-zig/bindings/node/binding.cc +20 -0
  47. package/vendor/tree-sitter-zig/bindings/node/index.d.ts +28 -0
  48. package/vendor/tree-sitter-zig/bindings/node/index.js +11 -0
  49. package/vendor/tree-sitter-zig/grammar.js +900 -0
  50. package/vendor/tree-sitter-zig/package.json +18 -0
  51. package/vendor/tree-sitter-zig/prebuilds/SHA256SUMS +6 -0
  52. package/vendor/tree-sitter-zig/prebuilds/darwin-arm64/tree-sitter-zig.node +0 -0
  53. package/vendor/tree-sitter-zig/prebuilds/darwin-x64/tree-sitter-zig.node +0 -0
  54. package/vendor/tree-sitter-zig/prebuilds/linux-arm64/tree-sitter-zig.node +0 -0
  55. package/vendor/tree-sitter-zig/prebuilds/linux-x64/tree-sitter-zig.node +0 -0
  56. package/vendor/tree-sitter-zig/prebuilds/win32-arm64/tree-sitter-zig.node +0 -0
  57. package/vendor/tree-sitter-zig/prebuilds/win32-x64/tree-sitter-zig.node +0 -0
  58. package/vendor/tree-sitter-zig/src/grammar.json +6570 -0
  59. package/vendor/tree-sitter-zig/src/node-types.json +3178 -0
  60. package/vendor/tree-sitter-zig/src/parser.c +169829 -0
  61. package/vendor/tree-sitter-zig/src/tree_sitter/alloc.h +54 -0
  62. package/vendor/tree-sitter-zig/src/tree_sitter/array.h +291 -0
  63. package/vendor/tree-sitter-zig/src/tree_sitter/parser.h +266 -0
@@ -15,7 +15,7 @@
15
15
  * @module
16
16
  */
17
17
  import { BindingAccumulator, enrichExportedTypeMap, } from '../binding-accumulator.js';
18
- import { mergeChunkResults, dispatchChunkParse } from '../parsing-processor.js';
18
+ import { mergeChunkResults, dispatchChunkParseRound } from '../parsing-processor.js';
19
19
  import { fileContentHash, computeChunkHash, loadParseCacheChunk, persistParseCacheChunk, PARSE_CACHE_VERSION, packParseCacheChunks, } from '../../../storage/parse-cache.js';
20
20
  import { clearParsedFileStore, persistParsedFileChunk, loadParsedFilesForPaths, getDurableParsedFileDir, loadDurableParsedFileIndex, prepareDurableParsedFileChunk, durableChunkHasShards, } from '../../../storage/parsedfile-store.js';
21
21
  import { DEFAULT_PDG_MAX_FUNCTION_LINES } from '../cfg/collect.js';
@@ -28,7 +28,7 @@ import { parseSourceSafe } from '../../tree-sitter/safe-parse.js';
28
28
  import { getProvider, getProviderForFile, providers } from '../languages/index.js';
29
29
  import { SCOPE_RESOLVERS } from '../scope-resolution/pipeline/registry.js';
30
30
  import { DATA_ROUTE_TABLE_SOURCE } from '../route-extractors/data-route-table.js';
31
- import { createWorkerPool, workerPoolDisabledByEnv, resolveAutoPoolSize, WorkerPoolInitializationError, WorkerPoolDisabledError, } from '../workers/worker-pool.js';
31
+ import { createWorkerPool, workerPoolDisabledByEnv, resolveAutoPoolSize, envWorkerPoolSize, resolveHostParallelism, WorkerPoolInitializationError, WorkerPoolDisabledError, } from '../workers/worker-pool.js';
32
32
  import { normalizeExtractedRoutePath } from '../route-extractors/route-path.js';
33
33
  import { resolveOperands } from '../route-extractors/python-const-resolver.js';
34
34
  import { prepareRouteConstantsByProvider } from '../language-provider.js';
@@ -43,6 +43,8 @@ import { isVerboseIngestionEnabled } from '../utils/verbose.js';
43
43
  import { endTimer, isDeferredResolutionProfileEnabled, logDeferredProfile, startTimer, } from '../utils/deferred-resolution-profile.js';
44
44
  import { isDebugHeapEnabled, logHeapProbe } from '../utils/heap-probe.js';
45
45
  import { logger } from '../../logger.js';
46
+ import { mapConcurrent } from '../../../lib/utils.js';
47
+ import { createRoundBudget } from './parse-round-budget.js';
46
48
  // ── Constants ──────────────────────────────────────────────────────────────
47
49
  /**
48
50
  * Heap-scale guardrail constants (#2649). Measured on a Linux-kernel analyze:
@@ -130,8 +132,36 @@ const CHUNK_BYTES_PER_WORKER = DEFAULT_CHUNK_BYTE_BUDGET;
130
132
  * while the rest finish early). Drives the derived `subBatchMaxBytes`.
131
133
  */
132
134
  const TARGET_JOBS_PER_WORKER = 3;
135
+ /**
136
+ * Concurrent durable ParsedFile directory resets per round. Matches the file
137
+ * reader's `READ_CONCURRENCY`, because both compete for the same descriptors.
138
+ */
139
+ const DURABLE_RESET_CONCURRENCY = 32;
133
140
  /** Floor for a derived sub-batch so jobs don't shrink to per-file IPC churn. */
134
141
  const MIN_SUB_BATCH_BYTES = 256 * 1024;
142
+ /**
143
+ * Source bytes an open round may HOLD — cache hits and misses alike — before
144
+ * it is dispatched and drained.
145
+ *
146
+ * A `dispatch` is a barrier, so one round-trip per cache pack leaves most slots
147
+ * idle: packs are keyed by `(language, hash(path) % 128)` and routinely land far
148
+ * under {@link DEFAULT_CHUNK_BYTE_BUDGET} (this repo: 1285 packs where the byte
149
+ * budget alone needs 16, 549 of them holding a single file). Rounds batch packs
150
+ * into one `dispatchGroups` call without touching pack identity.
151
+ *
152
+ * This is the in-flight cap, the same role Piscina's `maxQueue` plays: bigger
153
+ * rounds remove more barriers but hold more file content and more un-merged
154
+ * worker output on the main thread at once. Defaulting to one chunk budget
155
+ * keeps in-flight source bytes at the magnitude the loop already prefetched
156
+ * (`parseChunkConcurrency`, 2 chunks ahead). Override via
157
+ * `GITNEXUS_PARSE_ROUND_BYTES`.
158
+ */
159
+ function resolveParseRoundByteBudget(options) {
160
+ const env = Number(process.env.GITNEXUS_PARSE_ROUND_BYTES);
161
+ if (Number.isFinite(env) && env > 0)
162
+ return env;
163
+ return resolveChunkByteBudget(options);
164
+ }
135
165
  function resolveChunkByteBudget(options) {
136
166
  const opt = options?.chunkByteBudget;
137
167
  if (typeof opt === 'number' && Number.isFinite(opt) && opt > 0)
@@ -391,11 +421,38 @@ export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, t
391
421
  // cores-based auto size is capped by source bytes / CHUNK_BYTES_PER_WORKER
392
422
  // so a tiny repo does not spawn a full idle pool. Cache pack membership
393
423
  // is independent of this number (#3088).
394
- const explicitPoolSize = options?.workerPoolSize;
424
+ // `--workers <N>` and `GITNEXUS_WORKER_POOL_SIZE` are both deliberate
425
+ // operator input, so both bypass the work-proportional cap below. Only the
426
+ // env path used to be clamped by it, which made the documented escape hatch
427
+ // silently do nothing: on a 30MB repo the cap resolves to 16, so an operator
428
+ // asking for 24 still got 16 with no warning, while `--workers 24` got 24.
429
+ const explicitPoolSize = options?.workerPoolSize ?? envWorkerPoolSize();
430
+ // Cores-based auto size, bounded by source bytes so a tiny repo does not
431
+ // spawn a full idle pool.
395
432
  const workProportionalCap = Math.max(1, Math.ceil(totalBytes / CHUNK_BYTES_PER_WORKER));
433
+ // An operator's number is honored, but never exceeds the number of files
434
+ // there are to parse — `GITNEXUS_WORKER_POOL_SIZE=100000` on a five-file repo
435
+ // should not become the literal thread count. This bounds `--workers` and the
436
+ // env var identically, keeping the parity above intact. Note it does NOT
437
+ // shrink an incremental re-analyze: `totalParseable` counts every parseable
438
+ // file in the scan, not the changed ones, so a warm run of a large repo still
439
+ // spawns the full requested pool.
396
440
  const effectivePoolSize = explicitPoolSize && explicitPoolSize > 0
397
- ? explicitPoolSize
441
+ ? Math.min(explicitPoolSize, Math.max(1, totalParseable))
398
442
  : Math.min(resolveAutoPoolSize(), workProportionalCap);
443
+ // Deliberate over-subscription is the operator's call, so this warns rather
444
+ // than caps — silently capping is what the override exists to stop. But an
445
+ // exported `GITNEXUS_WORKER_POOL_SIZE` applies to EVERY analyze in a
446
+ // long-lived caller (watch auto-sync, the MCP server), including small
447
+ // incremental ones, and that is easy to set once and forget.
448
+ if (explicitPoolSize && explicitPoolSize > 0) {
449
+ const hostParallelism = resolveHostParallelism();
450
+ if (effectivePoolSize > hostParallelism) {
451
+ logger.warn({ requested: explicitPoolSize, spawning: effectivePoolSize, hostParallelism }, `Worker pool size ${effectivePoolSize} exceeds this host's ${hostParallelism} usable core(s); ` +
452
+ `parsing is CPU-bound, so the extra workers add memory pressure without throughput. ` +
453
+ `This applies to every analyze while the override is set.`);
454
+ }
455
+ }
399
456
  // Cache packs: stable (language, hash(path) mod 128) buckets, then the
400
457
  // per-call byte budget inside each bucket (#3088). Pool size is used only
401
458
  // for worker count and sub-batch fan-out, not membership.
@@ -596,13 +653,43 @@ export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, t
596
653
  // the env can't change mid-run.
597
654
  const verboseThroughputLog = isDev || isVerboseIngestionEnabled();
598
655
  const heapProbeEveryN = isDebugHeapEnabled() ? 25 : 0;
599
- let pendingWorkerChunk = null;
656
+ /**
657
+ * Chunk hashes whose durable ParsedFile directory could not be reset. The
658
+ * old generation's shards are still on disk, so a warm hit would union
659
+ * stale shards with the new ones. Treated exactly like a quarantined chunk:
660
+ * skip the parse-cache write so the next run re-dispatches into a clean
661
+ * directory rather than trusting a generation we could not clear.
662
+ */
663
+ const durablePrepareFailures = new Set();
664
+ const roundByteBudget = resolveParseRoundByteBudget(options);
665
+ let roundEntries = [];
666
+ /**
667
+ * Bytes an open round is HOLDING, counting hits as well as misses.
668
+ *
669
+ * Counting only the cache-MISSING bytes would bound just what the workers
670
+ * are asked to do, so a warm run — where nothing misses — would never reach
671
+ * the close condition and would buffer every chunk's cached output until
672
+ * the tail drain. That is the #2649 heap failure on a large repo. Counting
673
+ * both keeps a hits-only run draining at the same cadence as a cold one;
674
+ * `startRound` already supports a round with no misses.
675
+ *
676
+ * Measured in UTF-8 bytes, matching `estimateItemBytes` in the worker pool,
677
+ * so the cap means the same thing here as it does for a job's payload.
678
+ */
679
+ const roundBudget = createRoundBudget(roundByteBudget);
680
+ /**
681
+ * Files QUEUED into rounds so far. `filesParsedSoFar` only advances when a
682
+ * round drains, so it is the right number for the throughput log but would
683
+ * pin a warm run's progress bar at the phase floor for the whole loop.
684
+ */
685
+ let queuedFilesSoFar = 0;
686
+ let pendingRound = null;
600
687
  // Apply one chunk's merged worker data: per-chunk aggregation into the
601
688
  // run-level accumulators + the throughput log. Shared by the cache-hit
602
689
  // (inline) and worker (deferred) paths. The `| null` guard is defensive —
603
690
  // every live caller passes real worker data now that sequential parsing
604
691
  // (which was the only path that passed null) is gone.
605
- const applyChunkResults = async (chunkWorkerData, chunkIdx, chunkFiles, chunkStartMs) => {
692
+ const applyChunkResults = async (chunkWorkerData, chunkIdx, fileCount, chunkStartMs) => {
606
693
  if (chunkWorkerData) {
607
694
  for (const filePath of chunkWorkerData.scopeExtractionFailures) {
608
695
  scopeExtractionFailures.add(filePath);
@@ -690,30 +777,35 @@ export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, t
690
777
  allORMQueries.push(item);
691
778
  }
692
779
  }
693
- filesParsedSoFar += chunkFiles.length;
780
+ filesParsedSoFar += fileCount;
694
781
  if (verboseThroughputLog && chunkStartMs !== null) {
695
782
  const elapsedMs = Date.now() - chunkStartMs;
696
- const filesPerSec = elapsedMs > 0 ? (chunkFiles.length * 1000) / elapsedMs : 0;
783
+ const filesPerSec = elapsedMs > 0 ? (fileCount * 1000) / elapsedMs : 0;
697
784
  const stats = workerPool?.getStats?.();
698
785
  const poolFrag = stats
699
786
  ? ` pool: ${stats.activeSlots}/${stats.size} active, ` +
700
787
  `${stats.quarantined} quarantined${stats.poolBroken ? ', BROKEN' : ''}`
701
788
  : ' (cache replay)';
702
- logger.info(`📊 chunk ${chunkIdx + 1}/${numChunks}: ${chunkFiles.length} files in ${elapsedMs}ms ` +
789
+ logger.info(`📊 chunk ${chunkIdx + 1}/${numChunks}: ${fileCount} files in ${elapsedMs}ms ` +
703
790
  `(${filesPerSec.toFixed(1)} files/s)${poolFrag}`);
704
791
  }
705
792
  };
706
793
  // Merge + finalize a parked worker chunk: graph merge (the overlapped
707
794
  // main-thread step) → parse-cache write-guard → run-level aggregation.
708
- const finalizeWorkerChunk = async (p) => {
709
- const chunkWorkerData = mergeChunkResults(graph, symbolTable, p.rawResults, exportedTypeMap);
795
+ const finalizeWorkerChunk = async (p, rawResults) => {
796
+ const chunkWorkerData = mergeChunkResults(graph, symbolTable, rawResults, exportedTypeMap);
710
797
  // Persist raw results for this chunk hash (skipping when any chunk file
711
798
  // was worker-quarantined, so the narrower rawResults isn't cached under
712
799
  // the full-chunk key — see the original inline note / U20.U2).
713
- if (parseCache && p.chunkHash && p.rawResults.length > 0) {
800
+ if (parseCache && p.chunkHash && rawResults.length > 0) {
714
801
  const quarantineSet = new Set(workerPool?.getQuarantinedPaths?.() ?? []);
715
802
  const chunkHadQuarantine = p.chunkFiles.some((f) => quarantineSet.has(f.path));
716
- if (chunkHadQuarantine) {
803
+ const durableGenerationStale = durablePrepareFailures.has(p.chunkHash);
804
+ if (durableGenerationStale) {
805
+ logger.warn({ chunkHash: p.chunkHash.slice(0, 8) }, 'parse-cache SKIP: durable generation for this chunk could not be reset, ' +
806
+ 'so its shards may be stale; next run will re-dispatch it');
807
+ }
808
+ else if (chunkHadQuarantine) {
717
809
  if (isDev) {
718
810
  const quarantinedInChunk = p.chunkFiles.filter((f) => quarantineSet.has(f.path)).length;
719
811
  logger.info(`📦 parse-cache SKIP: chunk ${p.chunkIdx + 1}/${numChunks} ` +
@@ -722,13 +814,164 @@ export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, t
722
814
  }
723
815
  }
724
816
  else {
725
- await persistParseCacheChunk(parseCache, p.chunkHash, p.rawResults);
817
+ await persistParseCacheChunk(parseCache, p.chunkHash, rawResults);
726
818
  if (isDev) {
727
819
  logger.info(`📦 parse-cache MISS+store: chunk ${p.chunkIdx + 1}/${numChunks} (${p.chunkFiles.length} files, ${p.chunkHash.slice(0, 8)})`);
728
820
  }
729
821
  }
730
822
  }
731
- await applyChunkResults(chunkWorkerData, p.chunkIdx, p.chunkFiles, p.chunkStartMs);
823
+ await applyChunkResults(chunkWorkerData, p.chunkIdx, p.chunkFiles.length, p.chunkStartMs);
824
+ };
825
+ /**
826
+ * Dispatch a round's cache misses as ONE pool round. Returns the parked
827
+ * round; the caller drains it after starting the next one so the workers
828
+ * parse round N+1 while the main thread merges round N (the same overlap
829
+ * the per-chunk loop had, at round granularity).
830
+ */
831
+ const startRound = async (entries) => {
832
+ if (entries.length === 0)
833
+ return null;
834
+ const misses = entries.filter((entry) => entry.kind === 'miss');
835
+ if (misses.length === 0) {
836
+ return { entries, results: Promise.resolve([]) };
837
+ }
838
+ // Each chunk resets its own directory, so these are independent and run
839
+ // concurrently: serially they would sit on the critical path this round
840
+ // exists to shorten, with the pool idle and the previous round's merge
841
+ // waiting, once per miss.
842
+ //
843
+ // BOUNDED, though. A round can hold hundreds of small packs, and each
844
+ // reset is a recursive rm + mkdir. Firing all of them at once competes
845
+ // for descriptors with the chunk prefetch this loop already has in
846
+ // flight, and `readFileContents` degrades a losing read SILENTLY by
847
+ // contract — a dropped file would vanish from the chunk, from the graph,
848
+ // and from the chunk hash, shipping a narrowed index with exit 0. Same
849
+ // helper and width the file reads use.
850
+ await mapConcurrent(misses, async (miss) => {
851
+ if (durableParsedFileDir === undefined || miss.chunkHash === null)
852
+ return;
853
+ try {
854
+ await prepareDurableParsedFileChunk(durableParsedFileDir, miss.chunkHash);
855
+ }
856
+ catch (err) {
857
+ // The durable store is an optimization — degrade like the restore
858
+ // path does instead of failing the analyze. Workers recreate the
859
+ // directory on write, so at worst the old generation lingers.
860
+ // Caught per chunk so one failure cannot abort the others.
861
+ durablePrepareFailures.add(miss.chunkHash);
862
+ logger.warn({ err, chunkHash: miss.chunkHash.slice(0, 8) }, 'parsedfile-cache: could not reset durable chunk generation; ' +
863
+ 'continuing without caching this chunk');
864
+ }
865
+ }, { concurrency: DURABLE_RESET_CONCURRENCY });
866
+ const roundFiles = misses.reduce((sum, miss) => sum + miss.chunkFiles.length, 0);
867
+ const firstIdx = misses[0].chunkIdx;
868
+ const lastIdx = misses[misses.length - 1].chunkIdx;
869
+ const progressForRound = (current, _total, filePath) => {
870
+ // Rounds queued before this one are already counted in
871
+ // `queuedFilesSoFar`; `current` is this round's own worker progress.
872
+ const globalCurrent = queuedFilesSoFar - roundFiles + current;
873
+ // Parse phase covers 20-70 (M2). Deferred extraction handles 70-95.
874
+ const parsingProgress = 20 + (globalCurrent / totalParseable) * 50;
875
+ onProgress({
876
+ phase: 'parsing',
877
+ percent: Math.round(parsingProgress),
878
+ message: firstIdx === lastIdx
879
+ ? `Parsing chunk ${firstIdx + 1}/${numChunks}...`
880
+ : `Parsing chunks ${firstIdx + 1}-${lastIdx + 1}/${numChunks}...`,
881
+ detail: filePath,
882
+ stats: {
883
+ filesProcessed: globalCurrent,
884
+ totalFiles: totalParseable,
885
+ nodesCreated: graph.nodeCount,
886
+ },
887
+ });
888
+ };
889
+ const activeWorkerPool = getOrCreateWorkerPool();
890
+ if (verboseThroughputLog) {
891
+ logger.info(`🚚 round: ${misses.length} chunk(s) ${firstIdx + 1}-${lastIdx + 1}/${numChunks}, ` +
892
+ `${roundFiles} files in one dispatch`);
893
+ }
894
+ const results = dispatchChunkParseRound(misses.map((miss) => ({
895
+ items: miss.chunkFiles,
896
+ chunkHash: miss.chunkHash ?? undefined,
897
+ })), activeWorkerPool, progressForRound);
898
+ // Mark handled so a rejection during the overlap drain below isn't
899
+ // flagged as unhandled; the `await` in drainRound re-throws it for real
900
+ // handling.
901
+ results.catch(() => { });
902
+ return { entries, results };
903
+ };
904
+ /**
905
+ * Merge + finalize every chunk of a parked round, in `chunkIdx` order.
906
+ * Takes RESOLVED worker output: the round's dispatch must already have
907
+ * settled before this runs, because the pool allows only one dispatch in
908
+ * flight at a time (see `closeRound`).
909
+ */
910
+ const drainRound = async (round) => {
911
+ const missResults = round.missResults;
912
+ const missCount = round.entries.filter((entry) => entry.kind === 'miss').length;
913
+ // `dispatchGroups` returns one array per input group. If that contract
914
+ // ever breaks, every later entry in this round would silently merge the
915
+ // wrong chunk's results and skip its cache write, with a clean exit.
916
+ if (missResults.length !== missCount) {
917
+ throw new Error(`Parse round result mismatch: ${missResults.length} result group(s) for ${missCount} dispatched chunk(s).`);
918
+ }
919
+ let missIdx = 0;
920
+ for (const entry of round.entries) {
921
+ if (entry.kind === 'hit') {
922
+ const chunkWorkerData = mergeChunkResults(graph, symbolTable, entry.cachedRaw, exportedTypeMap);
923
+ await applyChunkResults(chunkWorkerData, entry.chunkIdx, entry.fileCount, entry.chunkStartMs);
924
+ continue;
925
+ }
926
+ await finalizeWorkerChunk(entry, missResults[missIdx++]);
927
+ }
928
+ };
929
+ /**
930
+ * Close the accumulated round.
931
+ *
932
+ * `WorkerPool.dispatch`/`dispatchGroups` is NOT reentrant — concurrent
933
+ * calls race on the shared per-slot busy/in-flight state and wedge the
934
+ * pool until every worker idle-times out. So exactly one dispatch is in
935
+ * flight here: start this round, merge the PREVIOUS round (whose results
936
+ * are already resolved) while these workers run, then await this round and
937
+ * park it resolved for the next close to merge.
938
+ */
939
+ const closeRound = async () => {
940
+ const started = await startRound(roundEntries);
941
+ roundEntries = [];
942
+ roundBudget.reset();
943
+ const previous = pendingRound;
944
+ pendingRound = null;
945
+ if (previous) {
946
+ try {
947
+ await drainRound(previous);
948
+ }
949
+ catch (err) {
950
+ // The round started above is still on the workers. Unwinding now
951
+ // reaches this function's `finally`, which calls `terminate()` — and
952
+ // terminate kills busy workers outright, which is the #2432
953
+ // mid-N-API SIGABRT hazard. Let the in-flight round settle first so
954
+ // the pool is idle, then propagate the original failure.
955
+ await started?.results.catch(() => undefined);
956
+ throw err;
957
+ }
958
+ }
959
+ if (!started)
960
+ return;
961
+ let missResults;
962
+ try {
963
+ missResults = await started.results;
964
+ }
965
+ catch (err) {
966
+ if (!(err instanceof WorkerPoolInitializationError))
967
+ throw err;
968
+ // Every worker crashed during startup and the pool's bounded self-heal
969
+ // was exhausted. Fail fast (#1741) — there is no sequential parser to
970
+ // degrade to. `handleWorkerStartupFailure` always throws, so
971
+ // `missResults` stays definitely assigned for the parked round below.
972
+ handleWorkerStartupFailure(err);
973
+ }
974
+ pendingRound = { entries: started.entries, missResults };
732
975
  };
733
976
  for (let chunkIdx = 0; chunkIdx < numChunks; chunkIdx++) {
734
977
  if (heapProbeEveryN > 0 && chunkIdx > 0 && chunkIdx % heapProbeEveryN === 0) {
@@ -808,17 +1051,14 @@ export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, t
808
1051
  parsedFileStorePath !== undefined &&
809
1052
  durableExpectedPaths !== undefined &&
810
1053
  (await durableChunkHasShards(parsedFileStorePath, chunkHash, durableExpectedPaths));
1054
+ // Set by whichever branch queues this chunk; drives the close below.
1055
+ let roundIsFull = false;
811
1056
  if (cachedRaw && cachedRaw.length > 0 && (durableHit || parsedFileStorePath === undefined)) {
812
1057
  // Cache hit: replay cached worker output. Finalize any parked worker
813
1058
  // chunk FIRST so deferred aggregation stays in chunk order, then merge
814
1059
  // + apply this hit inline (no worker dispatch to overlap).
815
- if (pendingWorkerChunk) {
816
- await finalizeWorkerChunk(pendingWorkerChunk);
817
- pendingWorkerChunk = null;
818
- }
819
1060
  chunkCacheHits++;
820
1061
  parseCacheHitFileCount += chunkFiles.length;
821
- const chunkWorkerData = mergeChunkResults(graph, symbolTable, cachedRaw, exportedTypeMap);
822
1062
  if (isDev) {
823
1063
  logger.info(`📦 parse-cache HIT: chunk ${chunkIdx + 1}/${numChunks} (${chunkFiles.length} files, ${chunkHash?.slice(0, 8) ?? 'unknown'})`);
824
1064
  }
@@ -830,97 +1070,55 @@ export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, t
830
1070
  // takes 70-95 so the UI advances through the (potentially long)
831
1071
  // resolution stages instead of holding at 82 (M2 from PR #1693
832
1072
  // review).
833
- percent: Math.round(20 + ((filesParsedSoFar + cachedFiles) / totalParseable) * 50),
1073
+ percent: Math.round(20 + ((queuedFilesSoFar + cachedFiles) / totalParseable) * 50),
834
1074
  message: `Parsing chunk ${chunkIdx + 1}/${numChunks} (cache)...`,
835
1075
  stats: {
836
- filesProcessed: filesParsedSoFar + cachedFiles,
1076
+ filesProcessed: queuedFilesSoFar + cachedFiles,
837
1077
  totalFiles: totalParseable,
838
1078
  nodesCreated: graph.nodeCount,
839
1079
  },
840
1080
  });
841
1081
  // The durable gate already snapshotted warm `.v8` shards into the
842
- // run-scoped store for scope resolution.
843
- await applyChunkResults(chunkWorkerData, chunkIdx, chunkFiles, chunkStartMs);
1082
+ // run-scoped store for scope resolution. Queue into the round so this
1083
+ // hit still finalizes in `chunkIdx` order relative to its neighbours.
1084
+ roundEntries.push({
1085
+ kind: 'hit',
1086
+ chunkIdx,
1087
+ fileCount: chunkFiles.length,
1088
+ chunkStartMs,
1089
+ cachedRaw,
1090
+ });
1091
+ roundIsFull = roundBudget.addChunk(chunkFiles.map((file) => file.content));
1092
+ queuedFilesSoFar += chunkFiles.length;
844
1093
  }
845
1094
  else {
846
- // Cache miss: dispatch to workers, capture the raw results, store
847
- // them under the chunk hash for the next run.
1095
+ // Cache miss: queue for the round's single dispatch; the raw results
1096
+ // are stored under the chunk hash when the round drains.
848
1097
  chunkCacheMisses++;
849
1098
  reparsedFileCount += chunkFiles.length;
850
- if (durableParsedFileDir !== undefined && chunkHash !== null) {
851
- try {
852
- await prepareDurableParsedFileChunk(durableParsedFileDir, chunkHash);
853
- }
854
- catch (err) {
855
- // The durable store is an optimization — degrade like the restore
856
- // path does instead of failing the analyze. Workers recreate the
857
- // directory on write, so at worst the old generation lingers.
858
- logger.warn({ err, chunkHash: chunkHash.slice(0, 8) }, 'parsedfile-cache: could not reset durable chunk generation; continuing');
859
- }
860
- }
861
- const progressForChunk = (current, _total, filePath) => {
862
- const globalCurrent = filesParsedSoFar + current;
863
- // Parse phase covers 20-70 (M2). Deferred extraction handles 70-95.
864
- const parsingProgress = 20 + (globalCurrent / totalParseable) * 50;
865
- onProgress({
866
- phase: 'parsing',
867
- percent: Math.round(parsingProgress),
868
- message: `Parsing chunk ${chunkIdx + 1}/${numChunks}...`,
869
- detail: filePath,
870
- stats: {
871
- filesProcessed: globalCurrent,
872
- totalFiles: totalParseable,
873
- nodesCreated: graph.nodeCount,
874
- },
875
- });
876
- };
877
- const activeWorkerPool = getOrCreateWorkerPool();
878
- // Worker path — PIPELINE: kick off this chunk's dispatch, merge the
879
- // PREVIOUS chunk while these workers parse, then park this chunk for
880
- // the next iteration to merge (overlapping its parse). The deferred
881
- // merge + parse-cache write-guard + aggregation all run in
882
- // `finalizeWorkerChunk`, in chunk order. The pool is the sole parse
883
- // path — `getOrCreateWorkerPool` returns a pool or throws.
884
- const dispatchPromise = dispatchChunkParse(chunkFiles, activeWorkerPool, progressForChunk, undefined, chunkHash ?? undefined);
885
- // Mark handled so a rejection during the overlap drain below isn't
886
- // flagged as unhandled; the `await` re-throws it for real handling.
887
- dispatchPromise.catch(() => { });
888
- if (pendingWorkerChunk) {
889
- await finalizeWorkerChunk(pendingWorkerChunk);
890
- pendingWorkerChunk = null;
891
- }
892
- let chunkResults;
893
- try {
894
- chunkResults = await dispatchPromise;
895
- }
896
- catch (err) {
897
- if (!(err instanceof WorkerPoolInitializationError))
898
- throw err;
899
- // Every worker crashed during startup and the pool's bounded
900
- // self-heal was exhausted. Fail fast (#1741) — there is no sequential
901
- // parser to degrade to. `handleWorkerStartupFailure` always throws, so
902
- // `chunkResults` stays definitely assigned for the parked chunk below.
903
- handleWorkerStartupFailure(err);
904
- }
905
- pendingWorkerChunk = {
906
- rawResults: chunkResults,
907
- chunkIdx,
908
- chunkHash,
909
- chunkFiles,
910
- chunkStartMs,
911
- };
1099
+ roundEntries.push({ kind: 'miss', chunkIdx, chunkHash, chunkFiles, chunkStartMs });
1100
+ roundIsFull = roundBudget.addChunk(chunkFiles.map((file) => file.content));
1101
+ queuedFilesSoFar += chunkFiles.length;
912
1102
  }
1103
+ // One cap, on what the main thread is holding. That bounds the worker
1104
+ // round too, since a round's dispatched bytes are a subset of its
1105
+ // buffered bytes.
1106
+ if (roundIsFull)
1107
+ await closeRound();
913
1108
  // (Per-chunk aggregation + parse-cache write + throughput log now run in
914
1109
  // `applyChunkResults` / `finalizeWorkerChunk` — see the merge-pipelining
915
1110
  // block above. Route/import/inheritance edges are emitted later: route
916
1111
  // resolution in the single end-of-loop pass below, the rest by the
917
1112
  // scope-resolution phase, RING4-2 #943.)
918
1113
  }
919
- // Drain the final parked worker chunk — the last pipelined chunk has no
920
- // successor to overlap its merge with, so merge + finalize it here.
921
- if (pendingWorkerChunk) {
922
- await finalizeWorkerChunk(pendingWorkerChunk);
923
- pendingWorkerChunk = null;
1114
+ // Drain the tail: close the partially-filled round, then drain the round
1115
+ // it parked — the last round has no successor to overlap its merge with.
1116
+ if (roundEntries.length > 0)
1117
+ await closeRound();
1118
+ if (pendingRound) {
1119
+ const last = pendingRound;
1120
+ pendingRound = null;
1121
+ await drainRound(last);
924
1122
  }
925
1123
  if (isDev && parseCache && (chunkCacheHits > 0 || chunkCacheMisses > 0)) {
926
1124
  logger.info(`📦 parse-cache summary: ${chunkCacheHits} chunk hit(s), ${chunkCacheMisses} miss(es) across ${numChunks} chunk(s)`);
@@ -0,0 +1,42 @@
1
+ /**
2
+ * The fold that decides when an open dispatch round closes.
3
+ *
4
+ * Extracted so the decision is a shared, inspectable unit rather than four
5
+ * loose statements inside `runChunkedParseAndResolve`. The parse loop is
6
+ * STREAMING — it reads chunk contents lazily, so it cannot know every chunk's
7
+ * size up front and cannot "plan" rounds ahead. That makes an accumulator, not
8
+ * a planner, the honest shape: feed it each chunk as it is queued and it tells
9
+ * you whether the round is now full.
10
+ *
11
+ * Being a real unit is what makes round cadence observable. Round boundaries
12
+ * are otherwise invisible from outside the parse phase: they change no graph
13
+ * output (that is the point of batching) and surface only in a log line, which
14
+ * is why `bench/parse-dispatch-rounds` measures this directly rather than
15
+ * inferring cadence from a full analyze.
16
+ */
17
+ /** Bytes a file contributes to the open round's retained total. */
18
+ export declare const roundFileBytes: (content: string) => number;
19
+ export interface RoundBudget {
20
+ /**
21
+ * Add one queued chunk's files. Returns true when the round is now full and
22
+ * the caller should close it. Closing resets the accumulator.
23
+ */
24
+ addChunk(contents: readonly string[]): boolean;
25
+ /** Bytes currently held by the open round. */
26
+ readonly bufferedBytes: number;
27
+ /** Reset without closing — used when the caller closes for another reason. */
28
+ reset(): void;
29
+ }
30
+ /**
31
+ * `budgetBytes` bounds what the main thread HOLDS, counting cache hits as well
32
+ * as misses. Counting only cache-missing bytes would bound just the work sent
33
+ * to workers, so a warm run — where nothing misses — would never reach the
34
+ * close condition and would buffer every chunk's cached output until the tail
35
+ * drain. That is the #2649 heap failure on a large repo.
36
+ *
37
+ * Measured in UTF-8 bytes, matching `estimateItemBytes` in the worker pool.
38
+ * `String.length` would return UTF-16 code units, undercounting non-ASCII
39
+ * source by up to 3x and letting a CJK-heavy repo hold well past its nominal
40
+ * budget before draining.
41
+ */
42
+ export declare const createRoundBudget: (budgetBytes: number) => RoundBudget;
@@ -0,0 +1,50 @@
1
+ /**
2
+ * The fold that decides when an open dispatch round closes.
3
+ *
4
+ * Extracted so the decision is a shared, inspectable unit rather than four
5
+ * loose statements inside `runChunkedParseAndResolve`. The parse loop is
6
+ * STREAMING — it reads chunk contents lazily, so it cannot know every chunk's
7
+ * size up front and cannot "plan" rounds ahead. That makes an accumulator, not
8
+ * a planner, the honest shape: feed it each chunk as it is queued and it tells
9
+ * you whether the round is now full.
10
+ *
11
+ * Being a real unit is what makes round cadence observable. Round boundaries
12
+ * are otherwise invisible from outside the parse phase: they change no graph
13
+ * output (that is the point of batching) and surface only in a log line, which
14
+ * is why `bench/parse-dispatch-rounds` measures this directly rather than
15
+ * inferring cadence from a full analyze.
16
+ */
17
+ /** Bytes a file contributes to the open round's retained total. */
18
+ export const roundFileBytes = (content) => Buffer.byteLength(content, 'utf8');
19
+ /**
20
+ * `budgetBytes` bounds what the main thread HOLDS, counting cache hits as well
21
+ * as misses. Counting only cache-missing bytes would bound just the work sent
22
+ * to workers, so a warm run — where nothing misses — would never reach the
23
+ * close condition and would buffer every chunk's cached output until the tail
24
+ * drain. That is the #2649 heap failure on a large repo.
25
+ *
26
+ * Measured in UTF-8 bytes, matching `estimateItemBytes` in the worker pool.
27
+ * `String.length` would return UTF-16 code units, undercounting non-ASCII
28
+ * source by up to 3x and letting a CJK-heavy repo hold well past its nominal
29
+ * budget before draining.
30
+ */
31
+ export const createRoundBudget = (budgetBytes) => {
32
+ let bufferedBytes = 0;
33
+ return {
34
+ addChunk(contents) {
35
+ for (const content of contents)
36
+ bufferedBytes += roundFileBytes(content);
37
+ if (bufferedBytes >= budgetBytes) {
38
+ bufferedBytes = 0;
39
+ return true;
40
+ }
41
+ return false;
42
+ },
43
+ get bufferedBytes() {
44
+ return bufferedBytes;
45
+ },
46
+ reset() {
47
+ bufferedBytes = 0;
48
+ },
49
+ };
50
+ };
@@ -615,6 +615,19 @@ export interface ScopeResolver {
615
615
  readonly populateWorkspaceOwners?: (parsedFiles: readonly ParsedFile[], ctx: {
616
616
  readonly fileContents: ReadonlyMap<string, string>;
617
617
  }) => void;
618
+ /**
619
+ * Optional workspace-wide enrichment of extracted reference sites. Runs
620
+ * after all files have been extracted and before reference finalization.
621
+ * Use this when a per-file capture needs conservative facts from an
622
+ * imported sibling (for example a compile-time branch constant).
623
+ */
624
+ readonly populateWorkspaceReferences?: (parsedFiles: ParsedFile[], ctx: {
625
+ readonly fileContents: ReadonlyMap<string, string>;
626
+ readonly treeCache?: {
627
+ get(filePath: string): unknown;
628
+ };
629
+ readonly resolutionConfig?: unknown;
630
+ }) => void;
618
631
  /**
619
632
  * Recognize a `super(...)`-style receiver text. Python returns
620
633
  * `/^super\s*\(/.test(t)`. Java returns `t === 'super'`. C++ may
@@ -55,6 +55,7 @@ const NOOP_OUTPUT = Object.freeze({
55
55
  /** Select source files that must be materialized for one resolver pass. */
56
56
  export function selectScopeSourcePathsToRead(provider, primaryFilePaths, preExtractedByPath) {
57
57
  const hasPostExtractHooks = provider.populateWorkspaceOwners !== undefined ||
58
+ provider.populateWorkspaceReferences !== undefined ||
58
59
  provider.populateNamespaceSiblings !== undefined ||
59
60
  provider.populateRangeBindings !== undefined ||
60
61
  provider.emitPostResolutionEdges !== undefined;
@@ -312,6 +312,11 @@ export function runScopeResolution(input, provider) {
312
312
  }
313
313
  logHeapProbe('sr-extract-end', `lang=${provider.language} parsedFiles=${parsedFiles.length} preExtractedHits=${preExtractedHits} skipped=${filesSkipped}`);
314
314
  provider.populateWorkspaceOwners?.(parsedFiles, { fileContents: getFileContents() });
315
+ provider.populateWorkspaceReferences?.(parsedFiles, {
316
+ fileContents: getFileContents(),
317
+ treeCache,
318
+ resolutionConfig: input.resolutionConfig,
319
+ });
315
320
  // A callable-flow-only provider has no reason to build the whole-graph
316
321
  // lookup or finalize ordinary references when none of its files emitted a
317
322
  // callable fact. This keeps the opt-in path proportional to source scanning