gitnexus 1.6.11-rc.2 → 1.6.11-rc.21

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (158) hide show
  1. package/README.md +43 -1
  2. package/dist/_shared/impact-risk.d.ts +37 -0
  3. package/dist/_shared/impact-risk.d.ts.map +1 -0
  4. package/dist/_shared/impact-risk.js +92 -0
  5. package/dist/_shared/impact-risk.js.map +1 -0
  6. package/dist/_shared/index.d.ts +2 -0
  7. package/dist/_shared/index.d.ts.map +1 -1
  8. package/dist/_shared/index.js +2 -0
  9. package/dist/_shared/index.js.map +1 -1
  10. package/dist/cli/ai-context.js +4 -4
  11. package/dist/cli/analyze-config.d.ts +2 -0
  12. package/dist/cli/analyze-config.js +16 -0
  13. package/dist/cli/analyze-options.d.ts +4 -0
  14. package/dist/cli/analyze.d.ts +17 -0
  15. package/dist/cli/analyze.js +43 -7
  16. package/dist/cli/eval-server.js +22 -5
  17. package/dist/cli/group.js +7 -1
  18. package/dist/cli/help-i18n.js +2 -0
  19. package/dist/cli/i18n/en.d.ts +12 -2
  20. package/dist/cli/i18n/en.js +12 -2
  21. package/dist/cli/i18n/resources.d.ts +22 -2
  22. package/dist/cli/i18n/zh-CN.d.ts +10 -0
  23. package/dist/cli/i18n/zh-CN.js +12 -2
  24. package/dist/cli/index.js +11 -4
  25. package/dist/cli/status.js +91 -6
  26. package/dist/cli/watch-queue.d.ts +41 -0
  27. package/dist/cli/watch-queue.js +184 -0
  28. package/dist/cli/watch.d.ts +20 -0
  29. package/dist/cli/watch.js +372 -0
  30. package/dist/cli/wiki.js +15 -2
  31. package/dist/config/ignore-service.d.ts +11 -0
  32. package/dist/config/ignore-service.js +41 -4
  33. package/dist/config/repo-control-file.d.ts +3 -0
  34. package/dist/config/repo-control-file.js +115 -0
  35. package/dist/core/group/config-parser.js +20 -2
  36. package/dist/core/group/cross-impact.d.ts +2 -1
  37. package/dist/core/group/cross-impact.js +33 -2
  38. package/dist/core/group/extractors/fs-utils.d.ts +2 -0
  39. package/dist/core/group/extractors/fs-utils.js +86 -0
  40. package/dist/core/group/extractors/graphql-extractor.d.ts +17 -0
  41. package/dist/core/group/extractors/graphql-extractor.js +652 -0
  42. package/dist/core/group/extractors/java-workspace-extractor.js +244 -31
  43. package/dist/core/group/extractors/manifest-extractor.d.ts +1 -1
  44. package/dist/core/group/extractors/manifest-extractor.js +1 -1
  45. package/dist/core/group/matching.js +8 -1
  46. package/dist/core/group/service.js +5 -1
  47. package/dist/core/group/storage.js +1 -0
  48. package/dist/core/group/sync.d.ts +3 -1
  49. package/dist/core/group/sync.js +43 -5
  50. package/dist/core/group/types.d.ts +19 -5
  51. package/dist/core/incremental/derived-writeback.d.ts +36 -0
  52. package/dist/core/incremental/derived-writeback.js +68 -0
  53. package/dist/core/incremental/subgraph-extract.d.ts +6 -4
  54. package/dist/core/incremental/subgraph-extract.js +7 -5
  55. package/dist/core/index-content-drift.d.ts +54 -0
  56. package/dist/core/index-content-drift.js +127 -0
  57. package/dist/core/ingestion/filesystem-walker.d.ts +15 -5
  58. package/dist/core/ingestion/filesystem-walker.js +20 -3
  59. package/dist/core/ingestion/frameworks/spring/dynamic-lookups.d.ts +22 -0
  60. package/dist/core/ingestion/frameworks/spring/dynamic-lookups.js +124 -0
  61. package/dist/core/ingestion/language-provider.d.ts +43 -0
  62. package/dist/core/ingestion/languages/csharp/razor-view-components.d.ts +62 -0
  63. package/dist/core/ingestion/languages/csharp/razor-view-components.js +954 -0
  64. package/dist/core/ingestion/languages/csharp/resolution-config.d.ts +3 -0
  65. package/dist/core/ingestion/languages/csharp/resolution-config.js +6 -1
  66. package/dist/core/ingestion/languages/csharp/scope-resolver.js +8 -0
  67. package/dist/core/ingestion/languages/java/capture-side-channel.d.ts +4 -0
  68. package/dist/core/ingestion/languages/java/capture-side-channel.js +16 -0
  69. package/dist/core/ingestion/languages/java/captures.js +14 -1
  70. package/dist/core/ingestion/languages/java/lombok-synthesizer.d.ts +42 -0
  71. package/dist/core/ingestion/languages/java/lombok-synthesizer.js +439 -0
  72. package/dist/core/ingestion/languages/java/scope-resolver.js +2 -0
  73. package/dist/core/ingestion/languages/java/spring-dynamic-lookup.d.ts +8 -0
  74. package/dist/core/ingestion/languages/java/spring-dynamic-lookup.js +62 -0
  75. package/dist/core/ingestion/languages/java.js +2 -0
  76. package/dist/core/ingestion/languages/jvm/accessor-synthesis.d.ts +99 -0
  77. package/dist/core/ingestion/languages/jvm/accessor-synthesis.js +173 -0
  78. package/dist/core/ingestion/languages/jvm/beanspec.d.ts +17 -0
  79. package/dist/core/ingestion/languages/jvm/beanspec.js +42 -0
  80. package/dist/core/ingestion/languages/kotlin/capture-side-channel.d.ts +5 -0
  81. package/dist/core/ingestion/languages/kotlin/capture-side-channel.js +16 -0
  82. package/dist/core/ingestion/languages/kotlin/captures.js +14 -1
  83. package/dist/core/ingestion/languages/kotlin/lombok-synthesizer.d.ts +26 -0
  84. package/dist/core/ingestion/languages/kotlin/lombok-synthesizer.js +417 -0
  85. package/dist/core/ingestion/languages/kotlin/scope-resolver.js +2 -0
  86. package/dist/core/ingestion/languages/kotlin/spring-dynamic-lookup.d.ts +8 -0
  87. package/dist/core/ingestion/languages/kotlin/spring-dynamic-lookup.js +77 -0
  88. package/dist/core/ingestion/languages/kotlin.js +2 -0
  89. package/dist/core/ingestion/pipeline-phases/di.js +47 -14
  90. package/dist/core/ingestion/pipeline-phases/parse-impl.d.ts +3 -1
  91. package/dist/core/ingestion/pipeline-phases/parse-impl.js +57 -86
  92. package/dist/core/ingestion/pipeline-phases/parse.d.ts +2 -0
  93. package/dist/core/ingestion/pipeline-phases/runner.d.ts +4 -1
  94. package/dist/core/ingestion/pipeline-phases/runner.js +34 -14
  95. package/dist/core/ingestion/pipeline-phases/scan.js +25 -13
  96. package/dist/core/ingestion/pipeline.d.ts +6 -0
  97. package/dist/core/ingestion/pipeline.js +44 -12
  98. package/dist/core/ingestion/scope-extractor.js +1 -0
  99. package/dist/core/ingestion/utils/symbol-labels.d.ts +2 -2
  100. package/dist/core/ingestion/utils/symbol-labels.js +2 -2
  101. package/dist/core/ingestion/workers/parse-worker.js +28 -2
  102. package/dist/core/lbug/lbug-adapter.d.ts +59 -0
  103. package/dist/core/lbug/lbug-adapter.js +154 -1
  104. package/dist/core/run-analyze.d.ts +17 -0
  105. package/dist/core/run-analyze.js +231 -23
  106. package/dist/core/search/fts-indexes.d.ts +19 -1
  107. package/dist/core/search/fts-indexes.js +28 -1
  108. package/dist/core/wiki/generator.js +8 -0
  109. package/dist/core/wiki/grok-client.d.ts +21 -0
  110. package/dist/core/wiki/grok-client.js +287 -0
  111. package/dist/core/wiki/llm-client.d.ts +1 -1
  112. package/dist/core/wiki/llm-client.js +5 -2
  113. package/dist/core/wiki/local-cli-client.d.ts +11 -0
  114. package/dist/core/wiki/local-cli-client.js +22 -9
  115. package/dist/mcp/local/local-backend.d.ts +24 -7
  116. package/dist/mcp/local/local-backend.js +190 -81
  117. package/dist/mcp/local/pdg-impact.d.ts +8 -4
  118. package/dist/mcp/local/pdg-impact.js +7 -2
  119. package/dist/mcp/repository-policy.d.ts +5 -1
  120. package/dist/mcp/repository-policy.js +48 -4
  121. package/dist/mcp/resources.js +2 -1
  122. package/dist/mcp/server.js +6 -5
  123. package/dist/mcp/tools.js +31 -19
  124. package/dist/server/api.js +23 -64
  125. package/dist/server/grep-params.d.ts +18 -0
  126. package/dist/server/grep-params.js +83 -0
  127. package/dist/server/grep-scan.d.ts +23 -0
  128. package/dist/server/grep-scan.js +107 -0
  129. package/dist/server/grep-worker.d.ts +1 -0
  130. package/dist/server/grep-worker.js +12 -0
  131. package/dist/server/mcp-http.d.ts +8 -0
  132. package/dist/server/mcp-http.js +16 -1
  133. package/dist/storage/file-hash.d.ts +5 -0
  134. package/dist/storage/file-hash.js +16 -6
  135. package/dist/storage/fs-atomic.d.ts +24 -0
  136. package/dist/storage/fs-atomic.js +86 -2
  137. package/dist/storage/git.d.ts +19 -8
  138. package/dist/storage/git.js +83 -24
  139. package/dist/storage/gitnexus-managed-paths.d.ts +36 -0
  140. package/dist/storage/gitnexus-managed-paths.js +46 -0
  141. package/dist/storage/parse-cache.d.ts +22 -4
  142. package/dist/storage/parse-cache.js +106 -29
  143. package/dist/storage/parsedfile-store.d.ts +34 -60
  144. package/dist/storage/parsedfile-store.js +177 -171
  145. package/dist/storage/repo-manager.d.ts +11 -1
  146. package/dist/storage/repo-manager.js +23 -2
  147. package/dist/storage/repo-meta.d.ts +12 -0
  148. package/dist/storage/v8-sidecar.d.ts +48 -0
  149. package/dist/storage/v8-sidecar.js +347 -0
  150. package/dist/types/pipeline.d.ts +14 -0
  151. package/package.json +4 -1
  152. package/scripts/cross-platform-shard.ts +4 -2
  153. package/scripts/cross-platform-tests.ts +7 -1
  154. package/skills/gitnexus-cli.md +11 -3
  155. package/skills/gitnexus-impact-analysis.md +9 -0
  156. package/web/assets/{agent-Dr4l5EOp.js → agent-CFqT4hjR.js} +108 -104
  157. package/web/assets/{index-2zdvEdzg.js → index-BMIniRtX.js} +3 -3
  158. package/web/index.html +1 -1
@@ -2,7 +2,7 @@
2
2
  * Parse implementation — chunked parse + resolve loop.
3
3
  *
4
4
  * This is the core parsing engine of the ingestion pipeline. It reads
5
- * source files in byte-budget chunks (~20MB each), parses via the worker
5
+ * source files in stable hash-bucket packs (~2MB each by default), parses via the worker
6
6
  * pool (the sole parse path — there is no sequential fallback), and emits
7
7
  * route CALLS edges. Import,
8
8
  * call, and inheritance resolution are owned by the scope-resolution
@@ -16,8 +16,8 @@
16
16
  */
17
17
  import { BindingAccumulator, enrichExportedTypeMap, } from '../binding-accumulator.js';
18
18
  import { mergeChunkResults, dispatchChunkParse } from '../parsing-processor.js';
19
- import { fileContentHash, computeChunkHash, loadParseCacheChunk, persistParseCacheChunk, PARSE_CACHE_VERSION, } from '../../../storage/parse-cache.js';
20
- import { clearParsedFileStore, persistParsedFileChunk, loadParsedFilesForPaths, getDurableParsedFileDir, loadDurableParsedFileIndex, prepareDurableParsedFileChunk, restoreDurableParsedFileShard, } from '../../../storage/parsedfile-store.js';
19
+ import { fileContentHash, computeChunkHash, loadParseCacheChunk, persistParseCacheChunk, PARSE_CACHE_VERSION, packParseCacheChunks, } from '../../../storage/parse-cache.js';
20
+ import { clearParsedFileStore, persistParsedFileChunk, loadParsedFilesForPaths, getDurableParsedFileDir, loadDurableParsedFileIndex, prepareDurableParsedFileChunk, durableChunkHasShards, } from '../../../storage/parsedfile-store.js';
21
21
  import { DEFAULT_PDG_MAX_FUNCTION_LINES } from '../cfg/collect.js';
22
22
  import { processRoutesFromExtracted, resolveRouteHandlerSymbols, buildExportedTypeMapFromGraph, } from '../call-processor.js';
23
23
  import { createSemanticModel } from '../model/index.js';
@@ -111,38 +111,18 @@ export function heapPressureRemedy(heapLimitBytes) {
111
111
  return (`This machine is at its memory ceiling: exclude generated or vendored directories ` +
112
112
  `via .gitnexusignore, or analyze on a machine with more memory.`);
113
113
  }
114
- /** Max bytes of source content to load per parse chunk.
114
+ /** Max bytes of source content to load per parse cache pack.
115
115
  *
116
- * Memory bound for the worker pool dispatch + a granularity knob for
117
- * the parse cache. A single file change invalidates only its enclosing
118
- * chunk, so smaller budgets → finer-grained invalidation.
119
- *
120
- * Override via GITNEXUS_CHUNK_BYTE_BUDGET (bytes) — the default of 2MB
121
- * gives a useful invalidation floor (~1/N chunks on a multi-MB repo)
122
- * while keeping worker dispatch overhead under 5% on cold runs.
123
- */
124
- /**
125
- * Built-in chunk byte budget when neither `PipelineOptions.chunkByteBudget`
126
- * nor `GITNEXUS_CHUNK_BYTE_BUDGET` is set. Tuned to give a useful
127
- * cache-invalidation floor (~1/N chunks on a multi-MB repo) while keeping
128
- * worker dispatch overhead under 5% on cold runs. Resolution happens at
129
- * call time inside `runChunkedParseAndResolve` (U14 from PR #1693 review)
130
- * — previously this was a module-load IIFE, which froze the env value at
131
- * import time and meant per-call option threading silently no-op'd.
116
+ * Granularity knob for the parse cache: a single file change invalidates only
117
+ * its enclosing pack. Override via GITNEXUS_CHUNK_BYTE_BUDGET. Resolution
118
+ * happens at call time (U14 from PR #1693) — not at module load.
132
119
  */
133
120
  const DEFAULT_CHUNK_BYTE_BUDGET = 2 * 1024 * 1024;
134
121
  /**
135
- * Per-worker share of a chunk's byte budget when auto-scaling (#worker-idle).
136
- *
137
- * A chunk is a single `WorkerPool.dispatch` unit; the pool fans a chunk's files
138
- * into sub-batch jobs and assigns them to idle workers (`wakeIdleSlots`). When
139
- * the chunk budget (2 MB) was far below the 8 MB sub-batch cap, every chunk
140
- * produced exactly ONE job → ONE busy worker while the other N-1 sat idle. To
141
- * keep all workers fed, the auto chunk budget now scales as
142
- * `poolSize × CHUNK_BYTES_PER_WORKER`, so each dispatch carries enough work to
143
- * fan across the whole pool. Sequential / explicit-budget runs are unaffected.
122
+ * Byte unit for auto pool sizing (one worker per this much source). Same
123
+ * magnitude as the default cache pack, but not a membership input (#3088).
144
124
  */
145
- const CHUNK_BYTES_PER_WORKER = 2 * 1024 * 1024;
125
+ const CHUNK_BYTES_PER_WORKER = DEFAULT_CHUNK_BYTE_BUDGET;
146
126
  /**
147
127
  * Target jobs-per-worker per dispatch. More jobs than workers gives the pool's
148
128
  * idle-slot assignment room to load-balance (a slow job doesn't strand a worker
@@ -151,16 +131,14 @@ const CHUNK_BYTES_PER_WORKER = 2 * 1024 * 1024;
151
131
  const TARGET_JOBS_PER_WORKER = 3;
152
132
  /** Floor for a derived sub-batch so jobs don't shrink to per-file IPC churn. */
153
133
  const MIN_SUB_BATCH_BYTES = 256 * 1024;
154
- function resolveChunkByteBudget(options, effectivePoolSize = 1) {
134
+ function resolveChunkByteBudget(options) {
155
135
  const opt = options?.chunkByteBudget;
156
136
  if (typeof opt === 'number' && Number.isFinite(opt) && opt > 0)
157
137
  return opt;
158
138
  const env = Number(process.env.GITNEXUS_CHUNK_BYTE_BUDGET);
159
139
  if (Number.isFinite(env) && env > 0)
160
140
  return env;
161
- // Auto: size each chunk so a dispatch can fan across the whole pool. A
162
- // single-worker (tiny-repo) run keeps the original 2 MB invalidation floor.
163
- return Math.max(DEFAULT_CHUNK_BYTE_BUDGET, effectivePoolSize * CHUNK_BYTES_PER_WORKER);
141
+ return DEFAULT_CHUNK_BYTE_BUDGET;
164
142
  }
165
143
  /**
166
144
  * Whole-repo, cross-file route extraction (main thread).
@@ -367,14 +345,6 @@ export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, t
367
345
  }
368
346
  }
369
347
  const unavailableScopeLanguageFiles = [...skippedByLang.values()].reduce((total, count) => total + count, 0);
370
- // Sort parseableScanned alphabetically for stable chunk membership
371
- // across runs (Finding 4). Without this, filesystem-scan order can
372
- // shift between runs (notably on macOS APFS where directory entry
373
- // order can change after modifications) — different files in the
374
- // same chunk → different chunk hash → cache miss even when no file
375
- // content changed. The cache also becomes platform-specific: a
376
- // Linux-built cache misses on macOS for the same repo.
377
- parseableScanned.sort((a, b) => (a.path < b.path ? -1 : a.path > b.path ? 1 : 0));
378
348
  const totalParseable = parseableScanned.length;
379
349
  const totalBytes = parseableScanned.reduce((sum, f) => sum + f.size, 0);
380
350
  if (totalParseable === 0) {
@@ -416,24 +386,24 @@ export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, t
416
386
  // runs. Resolving in the function body restores per-call configurability
417
387
  // and matches the pattern used by resolveAutoPoolSize and the U1
418
388
  // parseChunkConcurrency resolver.
419
- // Effective worker count, computed up-front so the chunk budget can scale to
420
- // keep the whole pool busy (#worker-idle). The pool is ALWAYS used (sequential
421
- // parsing was removed; the disabled channels threw above). Size it to the
422
- // work: an explicit `--workers <N>` pins the size; otherwise the cores-based
423
- // auto size is capped by the repo's worth of work (~one worker per
424
- // CHUNK_BYTES_PER_WORKER of source) so a tiny repo spawns ~1 worker instead of
425
- // a full pool, replacing the job the deleted small-repo threshold used to do.
426
- // KTD-3 of the remove-sequential plan; the cap formula is intentionally coarse
427
- // (tuning deferred).
389
+ // Effective worker count: explicit `--workers <N>` pins it; otherwise
390
+ // cores-based auto size is capped by source bytes / CHUNK_BYTES_PER_WORKER
391
+ // so a tiny repo does not spawn a full idle pool. Cache pack membership
392
+ // is independent of this number (#3088).
428
393
  const explicitPoolSize = options?.workerPoolSize;
429
394
  const workProportionalCap = Math.max(1, Math.ceil(totalBytes / CHUNK_BYTES_PER_WORKER));
430
395
  const effectivePoolSize = explicitPoolSize && explicitPoolSize > 0
431
396
  ? explicitPoolSize
432
397
  : Math.min(resolveAutoPoolSize(), workProportionalCap);
433
- const chunkByteBudget = resolveChunkByteBudget(options, effectivePoolSize);
434
- // Sub-batch size so each chunk fans into ~`TARGET_JOBS_PER_WORKER` jobs per
435
- // worker, giving the pool's idle-slot assignment room to load-balance. An
436
- // explicit `GITNEXUS_WORKER_SUB_BATCH_MAX_BYTES` operator override wins.
398
+ // Cache packs: stable (language, hash(path) mod 128) buckets, then the
399
+ // per-call byte budget inside each bucket (#3088). Pool size is used only
400
+ // for worker count and sub-batch fan-out, not membership.
401
+ const chunkByteBudget = resolveChunkByteBudget(options);
402
+ // Sub-batch size so a 2 MiB pack fans into ~TARGET_JOBS_PER_WORKER jobs
403
+ // per worker, floored at MIN_SUB_BATCH_BYTES (256 KiB) so an 8-worker
404
+ // pool still gets ~8 jobs from one pack instead of one idle-heavy job
405
+ // (#worker-idle). Do not derive this from pool×2 MiB while dispatching a
406
+ // 2 MiB pack. An explicit GITNEXUS_WORKER_SUB_BATCH_MAX_BYTES wins.
437
407
  const subBatchEnv = Number(process.env.GITNEXUS_WORKER_SUB_BATCH_MAX_BYTES);
438
408
  const dispatchSubBatchMaxBytes = Number.isFinite(subBatchEnv) && subBatchEnv > 0
439
409
  ? subBatchEnv
@@ -451,20 +421,11 @@ export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, t
451
421
  logger.warn(`Large repository: analyzing ${parseableScanned.length} files needs roughly ${Math.round(projectedHeapNeedBytes / 1024 / 1024 / 1024)}GB of memory, ` +
452
422
  `but Node is limited to ${Math.round(heapLimitBytes / 1024 / 1024 / 1024)}GB — analyze may stop early. ${heapPressureRemedy(heapLimitBytes)}`);
453
423
  }
454
- const chunks = [];
455
- let currentChunk = [];
456
- let currentBytes = 0;
457
- for (const file of parseableScanned) {
458
- if (currentChunk.length > 0 && currentBytes + file.size > chunkByteBudget) {
459
- chunks.push(currentChunk);
460
- currentChunk = [];
461
- currentBytes = 0;
462
- }
463
- currentChunk.push(file.path);
464
- currentBytes += file.size;
465
- }
466
- if (currentChunk.length > 0)
467
- chunks.push(currentChunk);
424
+ const chunks = packParseCacheChunks(parseableScanned.map((file) => ({
425
+ path: file.path,
426
+ size: file.size,
427
+ language: getLanguageFromFilename(file.path) ?? 'unknown',
428
+ })), chunkByteBudget);
468
429
  const numChunks = chunks.length;
469
430
  if (isDev) {
470
431
  const totalMB = parseableScanned.reduce((s, f) => s + f.size, 0) / (1024 * 1024);
@@ -586,15 +547,16 @@ export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, t
586
547
  // a sibling of the run-scoped store, NOT cleared per run. Workers write a
587
548
  // shard per chunk hash; on a warm parse-cache hit we restore the chunk's
588
549
  // shards into the run-scoped store so scope-resolution streams them without
589
- // re-parsing. `durableHitKeys` is the prior run's index, version-gated by
590
- // PARSE_CACHE_VERSION (a mismatch ⇒ empty ⇒ every chunk re-dispatches, which
591
- // repopulates the durable store — never the main-thread extract fallback).
550
+ // re-parsing. `durableHitEntries` is the prior run's path-coverage index,
551
+ // version-gated by PARSE_CACHE_VERSION (a mismatch ⇒ empty ⇒ every chunk
552
+ // re-dispatches, which repopulates the durable store).
592
553
  const durableParsedFileDir = parsedFileStorePath !== undefined ? getDurableParsedFileDir(parsedFileStorePath) : undefined;
593
- const durableHitKeys = durableParsedFileDir !== undefined
554
+ const durableHitEntries = durableParsedFileDir !== undefined
594
555
  ? await loadDurableParsedFileIndex(durableParsedFileDir, PARSE_CACHE_VERSION)
595
- : new Set();
556
+ : new Map();
596
557
  let chunkCacheHits = 0;
597
558
  let chunkCacheMisses = 0;
559
+ let reparsedFileCount = 0;
598
560
  try {
599
561
  // U1 — bounded chunk concurrency (B1 from PR #1693 review): pre-fetch
600
562
  // chunk file contents up to `parseChunkConcurrency` chunks ahead of the
@@ -645,7 +607,11 @@ export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, t
645
607
  }
646
608
  if (chunkWorkerData.parsedFiles?.length) {
647
609
  if (parsedFileStorePath) {
648
- await persistParsedFileChunk(parsedFileStorePath, `chunk-${chunkIdx}`, chunkWorkerData.parsedFiles);
610
+ const wrote = await persistParsedFileChunk(parsedFileStorePath, `chunk-${chunkIdx}`, chunkWorkerData.parsedFiles);
611
+ if (!wrote) {
612
+ for (const item of chunkWorkerData.parsedFiles)
613
+ allParsedFiles.push(item);
614
+ }
649
615
  }
650
616
  else {
651
617
  for (const item of chunkWorkerData.parsedFiles)
@@ -832,7 +798,14 @@ export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, t
832
798
  // store was introduced, or a pruned/version-stale shard — fall through to
833
799
  // a worker re-dispatch to repopulate them. NEVER let scope-resolution
834
800
  // re-extract on the main thread (the #1983 OOM the durable store closes).
835
- const durableHit = chunkHash !== null && durableParsedFileDir !== undefined && durableHitKeys.has(chunkHash);
801
+ const durableExpectedPaths = chunkHash === null ? undefined : durableHitEntries.get(chunkHash);
802
+ const durableHit = cachedRaw !== undefined &&
803
+ cachedRaw.length > 0 &&
804
+ chunkHash !== null &&
805
+ durableParsedFileDir !== undefined &&
806
+ parsedFileStorePath !== undefined &&
807
+ durableExpectedPaths !== undefined &&
808
+ (await durableChunkHasShards(parsedFileStorePath, chunkHash, durableExpectedPaths));
836
809
  if (cachedRaw && cachedRaw.length > 0 && (durableHit || parsedFileStorePath === undefined)) {
837
810
  // Cache hit: replay cached worker output. Finalize any parked worker
838
811
  // chunk FIRST so deferred aggregation stays in chunk order, then merge
@@ -862,22 +835,15 @@ export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, t
862
835
  nodesCreated: graph.nodeCount,
863
836
  },
864
837
  });
865
- // Restore the chunk's durable ParsedFile shards into the run-scoped
866
- // store so scope-resolution finds full coverage with ZERO main-thread
867
- // re-parse. A verbatim byte copy — byte-identical to a cold run.
868
- if (durableHit && durableParsedFileDir && parsedFileStorePath && chunkHash) {
869
- const restored = await restoreDurableParsedFileShard(durableParsedFileDir, parsedFileStorePath, chunkHash);
870
- if (restored === 0) {
871
- logger.warn(`parsedfile-cache: durable shards missing for cached chunk ` +
872
- `${chunkHash.slice(0, 8)} — scope-resolution will re-extract these files`);
873
- }
874
- }
838
+ // The durable gate already snapshotted warm `.v8` shards into the
839
+ // run-scoped store for scope resolution.
875
840
  await applyChunkResults(chunkWorkerData, chunkIdx, chunkFiles, chunkStartMs);
876
841
  }
877
842
  else {
878
843
  // Cache miss: dispatch to workers, capture the raw results, store
879
844
  // them under the chunk hash for the next run.
880
845
  chunkCacheMisses++;
846
+ reparsedFileCount += chunkFiles.length;
881
847
  if (durableParsedFileDir !== undefined && chunkHash !== null) {
882
848
  try {
883
849
  await prepareDurableParsedFileChunk(durableParsedFileDir, chunkHash);
@@ -1336,6 +1302,11 @@ export async function runChunkedParseAndResolve(graph, scannedFiles, allPaths, t
1336
1302
  // no pool was needed: a warm all-cache-hit run replays cached worker output
1337
1303
  // without spawning workers, or there were no parseable files.
1338
1304
  usedWorkerPool: workerPool !== undefined,
1305
+ // Exact number of files sent through workers on parse-cache misses. A
1306
+ // changed file can invalidate its whole content-addressed chunk, so this
1307
+ // is intentionally measured at dispatch time rather than inferred from
1308
+ // the git/hash diff.
1309
+ reparsedFileCount,
1339
1310
  // Per-file ParsedFile artifacts produced by workers' calls to
1340
1311
  // `extractParsedFile`. Consumed by scope-resolution as a re-extraction
1341
1312
  // cache: when the file's ParsedFile is here, scope-resolution skips its own
@@ -59,6 +59,8 @@ export interface ParseOutput {
59
59
  * is no sequential parser; the pool is the sole parse path on a cache miss.
60
60
  */
61
61
  readonly usedWorkerPool: boolean;
62
+ /** Files actually dispatched to parser workers after parse-cache lookup. */
63
+ readonly reparsedFileCount: number;
62
64
  /**
63
65
  * Per-file `ParsedFile` artifacts produced by workers' calls to
64
66
  * `extractParsedFile`. Threaded through to `scopeResolutionPhase`
@@ -17,6 +17,9 @@ import type { PipelinePhase, PipelineContext, PhaseResult } from './types.js';
17
17
  *
18
18
  * @param phases All phases to execute (order doesn't matter — sorted internally)
19
19
  * @param ctx Shared pipeline context
20
+ * @param seed Results of phases that already ran against this same context,
21
+ * available to `phases` as dependencies (#3016 deferred derived
22
+ * phases). Included in the returned map.
20
23
  * @returns Map of phase name → PhaseResult (all completed phases)
21
24
  */
22
- export declare function runPipeline(phases: readonly PipelinePhase[], ctx: PipelineContext): Promise<ReadonlyMap<string, PhaseResult<unknown>>>;
25
+ export declare function runPipeline(phases: readonly PipelinePhase[], ctx: PipelineContext, seed?: ReadonlyMap<string, PhaseResult<unknown>>): Promise<ReadonlyMap<string, PhaseResult<unknown>>>;
@@ -13,22 +13,30 @@
13
13
  */
14
14
  import { isDev } from '../utils/env.js';
15
15
  import { logger } from '../../logger.js';
16
- /**
17
- * Validate that the phases form a valid dependency graph (no cycles, all deps present).
18
- * Returns phases in topological execution order.
19
- */
20
- function topologicalSort(phases) {
21
- const phaseMap = new Map();
16
+ function assertUniquePhaseNames(phases) {
17
+ const seen = new Set();
22
18
  for (const phase of phases) {
23
- if (phaseMap.has(phase.name)) {
19
+ if (seen.has(phase.name)) {
24
20
  throw new Error(`Duplicate phase name: '${phase.name}'`);
25
21
  }
26
- phaseMap.set(phase.name, phase);
22
+ seen.add(phase.name);
27
23
  }
24
+ }
25
+ /**
26
+ * Validate that the phases form a valid dependency graph (no cycles, all deps present).
27
+ * Returns phases in topological execution order.
28
+ *
29
+ * `satisfied` names phases whose results are already available (a deferred
30
+ * follow-up run over the same context, #3016). Their edges are dropped rather
31
+ * than validated, because they are resolved by definition.
32
+ */
33
+ function topologicalSort(phases, satisfied = new Set()) {
34
+ assertUniquePhaseNames(phases);
35
+ const phaseMap = new Map(phases.map((p) => [p.name, p]));
28
36
  // Validate all deps exist
29
37
  for (const phase of phases) {
30
38
  for (const dep of phase.deps) {
31
- if (!phaseMap.has(dep)) {
39
+ if (!phaseMap.has(dep) && !satisfied.has(dep)) {
32
40
  throw new Error(`Phase '${phase.name}' depends on '${dep}', which is not registered`);
33
41
  }
34
42
  }
@@ -37,8 +45,9 @@ function topologicalSort(phases) {
37
45
  const inDegree = new Map();
38
46
  const reverseDeps = new Map();
39
47
  for (const phase of phases) {
40
- inDegree.set(phase.name, phase.deps.length);
41
- for (const dep of phase.deps) {
48
+ const pendingDeps = phase.deps.filter((dep) => !satisfied.has(dep));
49
+ inDegree.set(phase.name, pendingDeps.length);
50
+ for (const dep of pendingDeps) {
42
51
  let rev = reverseDeps.get(dep);
43
52
  if (!rev) {
44
53
  rev = [];
@@ -125,12 +134,23 @@ function findCyclePath(remaining, phaseMap) {
125
134
  *
126
135
  * @param phases All phases to execute (order doesn't matter — sorted internally)
127
136
  * @param ctx Shared pipeline context
137
+ * @param seed Results of phases that already ran against this same context,
138
+ * available to `phases` as dependencies (#3016 deferred derived
139
+ * phases). Included in the returned map.
128
140
  * @returns Map of phase name → PhaseResult (all completed phases)
129
141
  */
130
- export async function runPipeline(phases, ctx) {
142
+ export async function runPipeline(phases, ctx, seed) {
143
+ // A seeded phase has already run against this context; re-running it would
144
+ // apply its graph writes a second time. "Already ran" is the whole meaning of
145
+ // the seed, so honour it here rather than making every caller pre-filter.
146
+ const satisfied = new Set(seed?.keys() ?? []);
131
147
  let sorted;
132
148
  try {
133
- sorted = topologicalSort(phases);
149
+ // Duplicate names must be rejected on the caller-supplied list *before*
150
+ // seed-filtering. Filtering first would drop a seeded duplicate and let
151
+ // `topologicalSort` see a unique name (#3102).
152
+ assertUniquePhaseNames(phases);
153
+ sorted = topologicalSort(phases.filter((p) => !satisfied.has(p.name)), satisfied);
134
154
  }
135
155
  catch (err) {
136
156
  // Emit a terminal 'error' progress event for graph-validation failures
@@ -152,7 +172,7 @@ export async function runPipeline(phases, ctx) {
152
172
  }
153
173
  throw err;
154
174
  }
155
- const results = new Map();
175
+ const results = new Map(seed);
156
176
  for (const phase of sorted) {
157
177
  const start = Date.now();
158
178
  if (isDev) {
@@ -19,20 +19,32 @@ export const scanPhase = {
19
19
  percent: 0,
20
20
  message: 'Scanning repository...',
21
21
  });
22
- const scannedFiles = await walkRepositoryPaths(ctx.repoPath, (current, total, filePath) => {
23
- const scanProgress = Math.round((current / total) * 15);
24
- ctx.onProgress({
25
- phase: 'extracting',
26
- percent: scanProgress,
27
- message: 'Scanning repository...',
28
- detail: filePath,
29
- stats: {
30
- filesProcessed: current,
31
- totalFiles: total,
32
- nodesCreated: ctx.graph.nodeCount,
33
- },
22
+ let scannedFiles;
23
+ try {
24
+ scannedFiles = await walkRepositoryPaths(ctx.repoPath, (current, total, filePath) => {
25
+ const scanProgress = Math.round((current / total) * 15);
26
+ ctx.onProgress({
27
+ phase: 'extracting',
28
+ percent: scanProgress,
29
+ message: 'Scanning repository...',
30
+ detail: filePath,
31
+ stats: {
32
+ filesProcessed: current,
33
+ totalFiles: total,
34
+ nodesCreated: ctx.graph.nodeCount,
35
+ },
36
+ });
34
37
  });
35
- });
38
+ }
39
+ catch (err) {
40
+ // Missing roots throw so status cannot treat an empty glob as "every
41
+ // covered file was deleted". The pipeline still reports an empty scan
42
+ // for a path that is not a directory, matching analyze of a bad cwd.
43
+ if (err instanceof Error && err.message.startsWith('walkRepositoryPaths:')) {
44
+ return { scannedFiles: [], allPaths: [], totalFiles: 0 };
45
+ }
46
+ throw err;
47
+ }
36
48
  const totalFiles = scannedFiles.length;
37
49
  const allPaths = scannedFiles.map((f) => f.path);
38
50
  ctx.onProgress({
@@ -25,6 +25,12 @@ export interface PipelineOptions {
25
25
  * to retain those nodes under `skipGraphPhases`.
26
26
  */
27
27
  skipGraphPhases?: boolean;
28
+ /**
29
+ * Skip only Leiden community detection and process/flow extraction (#3016).
30
+ * MRO/DI still run. Used on warm incremental analyze so persisted
31
+ * Community/Process rows can be kept instead of wipe+rewrite.
32
+ */
33
+ skipDerivedGraphPhases?: boolean;
28
34
  /** Per-advice Spring AOP candidate inspection cap. `0` disables this cap. */
29
35
  springAopMaxCandidateInspectionsPerAdvice?: number;
30
36
  /** Aggregate Spring AOP candidate inspection cap for one analysis. `0` disables this cap. */
@@ -61,8 +61,12 @@ export function buildPhaseList(options) {
61
61
  .register(mroPhase, { enabledWhen: (o) => !o.skipGraphPhases })
62
62
  .register(springAopInheritancePhase, { enabledWhen: (o) => !o.skipGraphPhases })
63
63
  .register(diPhase, { enabledWhen: (o) => !o.skipGraphPhases })
64
- .register(communitiesPhase, { enabledWhen: (o) => !o.skipGraphPhases })
65
- .register(processesPhase, { enabledWhen: (o) => !o.skipGraphPhases })
64
+ .register(communitiesPhase, {
65
+ enabledWhen: (o) => !o.skipGraphPhases && o.skipDerivedGraphPhases !== true,
66
+ })
67
+ .register(processesPhase, {
68
+ enabledWhen: (o) => !o.skipGraphPhases && o.skipDerivedGraphPhases !== true,
69
+ })
66
70
  // Normalize a missing options object once here so phase predicates above
67
71
  // take a required PipelineOptions and need no `?.` guard (#2080 review S1).
68
72
  .build(options ?? {}));
@@ -91,17 +95,18 @@ export const runPipelineFromRepo = async (repoPath, onProgress, options) => {
91
95
  graphEmitSink = new GraphEmitSink(graph, options.graphEmitCsvDir);
92
96
  }
93
97
  const phases = buildPhaseList(options);
98
+ const ctx = {
99
+ repoPath,
100
+ graph: graphEmitSink ?? graph,
101
+ onProgress,
102
+ options,
103
+ pipelineStart,
104
+ graphEmit: graphEmitSink,
105
+ };
94
106
  let graphEmitManifest;
95
107
  let results;
96
108
  try {
97
- results = await runPipeline(phases, {
98
- repoPath,
99
- graph: graphEmitSink ?? graph,
100
- onProgress,
101
- options,
102
- pipelineStart,
103
- graphEmit: graphEmitSink,
104
- });
109
+ results = await runPipeline(phases, ctx);
105
110
  graphEmitManifest = graphEmitSink?.finalize();
106
111
  }
107
112
  finally {
@@ -109,7 +114,7 @@ export const runPipelineFromRepo = async (repoPath, onProgress, options) => {
109
114
  graphEmitSink?.close();
110
115
  }
111
116
  // Extract final results for the PipelineResult contract
112
- const { totalFiles, usedWorkerPool, unavailableScopeLanguageFiles } = getPhaseOutput(results, 'parse');
117
+ const { totalFiles, usedWorkerPool, reparsedFileCount, unavailableScopeLanguageFiles } = getPhaseOutput(results, 'parse');
113
118
  let communityResult;
114
119
  let processResult;
115
120
  const scopeResolutionOutput = getPhaseOutput(results, 'scopeResolution');
@@ -140,7 +145,7 @@ export const runPipelineFromRepo = async (repoPath, onProgress, options) => {
140
145
  nodesCreated: graph.nodeCount,
141
146
  },
142
147
  });
143
- return {
148
+ const result = {
144
149
  // The RAW graph, deliberately — NOT `graphEmitSink`. Phases above received
145
150
  // the sink so their reads are complete, but `loadGraphToLbug` feeds this to
146
151
  // `streamAllCSVsToDisk`, and the sink's complete iterator would then emit
@@ -156,9 +161,36 @@ export const runPipelineFromRepo = async (repoPath, onProgress, options) => {
156
161
  resolutionOutcomes,
157
162
  undecidedSatisfaction,
158
163
  usedWorkerPool,
164
+ reparsedFileCount,
159
165
  scopeExtractionFailures,
160
166
  unavailableScopeLanguageFiles,
161
167
  pdgEmitManifest,
162
168
  propertyInference,
163
169
  };
170
+ // #3016: hand back a way to run the derived phases `skipDerivedGraphPhases`
171
+ // held back. Which phases those are is answered by re-asking the registry
172
+ // with only that flag cleared — the one form of the question that stays
173
+ // correct when a different predicate (`skipGraphPhases`) also disables them,
174
+ // since then they are absent for a reason a deferred run cannot fix and the
175
+ // filter yields nothing. The sink guard mirrors the `graph` note above: a
176
+ // streaming run is a full rebuild, which never sets the skip flag, so an
177
+ // active sink here means the two got combined by mistake — and deferred
178
+ // phases writing into a finalized sink would emit past its manifest.
179
+ const deferredDerivedPhases = options?.skipDerivedGraphPhases === true && graphEmitSink === undefined
180
+ ? buildPhaseList({ ...options, skipDerivedGraphPhases: false }).filter((p) => (p.name === 'communities' || p.name === 'processes') && !results.has(p.name))
181
+ : [];
182
+ if (deferredDerivedPhases.length > 0) {
183
+ result.runDeferredDerivedPhases = async () => {
184
+ const derived = await runPipeline(deferredDerivedPhases, ctx, results);
185
+ // Presence-checked for the same reason as the block above: a phase the
186
+ // registry filtered out is absent, and `getPhaseOutput` throws on absent.
187
+ if (derived.has('communities')) {
188
+ result.communityResult = getPhaseOutput(derived, 'communities').communityResult;
189
+ }
190
+ if (derived.has('processes')) {
191
+ result.processResult = getPhaseOutput(derived, 'processes').processResult;
192
+ }
193
+ };
194
+ }
195
+ return result;
164
196
  };
@@ -1450,6 +1450,7 @@ const KNOWN_SUB_TAGS = new Set([
1450
1450
  '@scope.lexical-names',
1451
1451
  '@declaration.name',
1452
1452
  '@declaration.qualified_name',
1453
+ '@declaration.is-synthetic',
1453
1454
  '@import.name',
1454
1455
  '@import.source',
1455
1456
  '@import.alias',
@@ -12,8 +12,8 @@ import type { NodeLabel } from '../../../_shared/index.js';
12
12
  * Single source of truth so the set can't silently drift the way the inline copy
13
13
  * did in #2379.
14
14
  *
15
- * NOTE: `group/extractors/manifest-extractor.ts`'s `CUSTOM_CONTRACT_RESOLVE_QUERY`
16
- * carries a near-identical hand-list that is intentionally a SUBSET — it excludes
15
+ * NOTE: group extractor queries in `manifest-extractor.ts` and `graphql-extractor.ts`
16
+ * carry near-identical hand-lists that are intentionally SUBSETS — they exclude
17
17
  * `Namespace`, `Variable`, `Module`. Unifying the two needs a contract-resolution
18
18
  * behavior check (would widen which nodes resolve as contract symbols), so it is
19
19
  * deliberately left separate for now.
@@ -11,8 +11,8 @@
11
11
  * Single source of truth so the set can't silently drift the way the inline copy
12
12
  * did in #2379.
13
13
  *
14
- * NOTE: `group/extractors/manifest-extractor.ts`'s `CUSTOM_CONTRACT_RESOLVE_QUERY`
15
- * carries a near-identical hand-list that is intentionally a SUBSET — it excludes
14
+ * NOTE: group extractor queries in `manifest-extractor.ts` and `graphql-extractor.ts`
15
+ * carry near-identical hand-lists that are intentionally SUBSETS — they exclude
16
16
  * `Namespace`, `Variable`, `Module`. Unifying the two needs a contract-resolution
17
17
  * behavior check (would widen which nodes resolve as contract symbols), so it is
18
18
  * deliberately left separate for now.
@@ -1025,6 +1025,10 @@ const processFileGroup = (files, language, queryString, result, onFileProcessed)
1025
1025
  continue;
1026
1026
  }
1027
1027
  const provider = getProvider(language);
1028
+ // Owner map for provider.synthesizeStructureMembers: type-declaration AST
1029
+ // node id → graph node id for classes THIS file's capture loop materialized.
1030
+ // Keyed by in-memory AST identity (never persisted); filled below.
1031
+ const classOwnersByNodeId = new Map();
1028
1032
  // #2687: ONE pass over `matches` yields both suppression sets — the
1029
1033
  // definition-name claims by rank (callable > Property > value), so the dedup
1030
1034
  // below cannot depend on tree-sitter's match order, and the concrete-typedef
@@ -2248,6 +2252,14 @@ const processFileGroup = (files, language, queryString, result, onFileProcessed)
2248
2252
  ? { annotations: methodProps.annotations }
2249
2253
  : {}),
2250
2254
  });
2255
+ // Class-like definitions register their AST node id → graph node id for
2256
+ // provider.synthesizeStructureMembers. The definition node is the same
2257
+ // type-declaration AST node that the provider-specific planner receives.
2258
+ if (isClassLikeLabel &&
2259
+ definitionNode &&
2260
+ provider.classExtractor?.isTypeDeclaration(definitionNode)) {
2261
+ classOwnersByNodeId.set(definitionNode.id, nodeId);
2262
+ }
2251
2263
  // Object-literal callables remain file definitions as well as members of
2252
2264
  // their exported binding. Class members still use HAS_METHOD alone.
2253
2265
  const isTopLevelObjectCallable = objectLiteralBindingInfo?.ownerName !== undefined &&
@@ -2337,6 +2349,18 @@ const processFileGroup = (files, language, queryString, result, onFileProcessed)
2337
2349
  if (springTypes.length > 0)
2338
2350
  (result.springTypes ??= []).push(...springTypes);
2339
2351
  }
2352
+ if (provider.synthesizeStructureMembers) {
2353
+ const synthetic = provider.synthesizeStructureMembers(tree, file.path, classOwnersByNodeId);
2354
+ for (const node of synthetic.nodes) {
2355
+ result.nodes.push(node);
2356
+ }
2357
+ for (const sym of synthetic.symbols) {
2358
+ result.symbols.push(sym);
2359
+ }
2360
+ for (const rel of synthetic.relationships) {
2361
+ result.relationships.push(rel);
2362
+ }
2363
+ }
2340
2364
  // Vue: emit CALLS edges for components used in <template>
2341
2365
  if (language === SupportedLanguages.Vue) {
2342
2366
  const templateComponents = extractTemplateComponents(file.content);
@@ -2469,8 +2493,10 @@ parentPort.on('message', (msg) => {
2469
2493
  persistDurableParsedFileShardSync(DURABLE_PARSED_FILE_STORAGE_PATH, msg.chunkHash, threadId, seq, accumulated.parsedFiles);
2470
2494
  }
2471
2495
  if (PARSED_FILE_STORE_STORAGE_PATH) {
2472
- persistParsedFileShardSync(PARSED_FILE_STORE_STORAGE_PATH, `w${threadId}-${seq}`, accumulated.parsedFiles);
2473
- accumulated.parsedFiles = [];
2496
+ const wrote = persistParsedFileShardSync(PARSED_FILE_STORE_STORAGE_PATH, `w${threadId}-${seq}`, accumulated.parsedFiles);
2497
+ if (wrote) {
2498
+ accumulated.parsedFiles = [];
2499
+ }
2474
2500
  }
2475
2501
  }
2476
2502
  postResultCloneSafe(accumulated);