gitnexus 1.6.11-rc.2 → 1.6.11-rc.21

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (158) hide show
  1. package/README.md +43 -1
  2. package/dist/_shared/impact-risk.d.ts +37 -0
  3. package/dist/_shared/impact-risk.d.ts.map +1 -0
  4. package/dist/_shared/impact-risk.js +92 -0
  5. package/dist/_shared/impact-risk.js.map +1 -0
  6. package/dist/_shared/index.d.ts +2 -0
  7. package/dist/_shared/index.d.ts.map +1 -1
  8. package/dist/_shared/index.js +2 -0
  9. package/dist/_shared/index.js.map +1 -1
  10. package/dist/cli/ai-context.js +4 -4
  11. package/dist/cli/analyze-config.d.ts +2 -0
  12. package/dist/cli/analyze-config.js +16 -0
  13. package/dist/cli/analyze-options.d.ts +4 -0
  14. package/dist/cli/analyze.d.ts +17 -0
  15. package/dist/cli/analyze.js +43 -7
  16. package/dist/cli/eval-server.js +22 -5
  17. package/dist/cli/group.js +7 -1
  18. package/dist/cli/help-i18n.js +2 -0
  19. package/dist/cli/i18n/en.d.ts +12 -2
  20. package/dist/cli/i18n/en.js +12 -2
  21. package/dist/cli/i18n/resources.d.ts +22 -2
  22. package/dist/cli/i18n/zh-CN.d.ts +10 -0
  23. package/dist/cli/i18n/zh-CN.js +12 -2
  24. package/dist/cli/index.js +11 -4
  25. package/dist/cli/status.js +91 -6
  26. package/dist/cli/watch-queue.d.ts +41 -0
  27. package/dist/cli/watch-queue.js +184 -0
  28. package/dist/cli/watch.d.ts +20 -0
  29. package/dist/cli/watch.js +372 -0
  30. package/dist/cli/wiki.js +15 -2
  31. package/dist/config/ignore-service.d.ts +11 -0
  32. package/dist/config/ignore-service.js +41 -4
  33. package/dist/config/repo-control-file.d.ts +3 -0
  34. package/dist/config/repo-control-file.js +115 -0
  35. package/dist/core/group/config-parser.js +20 -2
  36. package/dist/core/group/cross-impact.d.ts +2 -1
  37. package/dist/core/group/cross-impact.js +33 -2
  38. package/dist/core/group/extractors/fs-utils.d.ts +2 -0
  39. package/dist/core/group/extractors/fs-utils.js +86 -0
  40. package/dist/core/group/extractors/graphql-extractor.d.ts +17 -0
  41. package/dist/core/group/extractors/graphql-extractor.js +652 -0
  42. package/dist/core/group/extractors/java-workspace-extractor.js +244 -31
  43. package/dist/core/group/extractors/manifest-extractor.d.ts +1 -1
  44. package/dist/core/group/extractors/manifest-extractor.js +1 -1
  45. package/dist/core/group/matching.js +8 -1
  46. package/dist/core/group/service.js +5 -1
  47. package/dist/core/group/storage.js +1 -0
  48. package/dist/core/group/sync.d.ts +3 -1
  49. package/dist/core/group/sync.js +43 -5
  50. package/dist/core/group/types.d.ts +19 -5
  51. package/dist/core/incremental/derived-writeback.d.ts +36 -0
  52. package/dist/core/incremental/derived-writeback.js +68 -0
  53. package/dist/core/incremental/subgraph-extract.d.ts +6 -4
  54. package/dist/core/incremental/subgraph-extract.js +7 -5
  55. package/dist/core/index-content-drift.d.ts +54 -0
  56. package/dist/core/index-content-drift.js +127 -0
  57. package/dist/core/ingestion/filesystem-walker.d.ts +15 -5
  58. package/dist/core/ingestion/filesystem-walker.js +20 -3
  59. package/dist/core/ingestion/frameworks/spring/dynamic-lookups.d.ts +22 -0
  60. package/dist/core/ingestion/frameworks/spring/dynamic-lookups.js +124 -0
  61. package/dist/core/ingestion/language-provider.d.ts +43 -0
  62. package/dist/core/ingestion/languages/csharp/razor-view-components.d.ts +62 -0
  63. package/dist/core/ingestion/languages/csharp/razor-view-components.js +954 -0
  64. package/dist/core/ingestion/languages/csharp/resolution-config.d.ts +3 -0
  65. package/dist/core/ingestion/languages/csharp/resolution-config.js +6 -1
  66. package/dist/core/ingestion/languages/csharp/scope-resolver.js +8 -0
  67. package/dist/core/ingestion/languages/java/capture-side-channel.d.ts +4 -0
  68. package/dist/core/ingestion/languages/java/capture-side-channel.js +16 -0
  69. package/dist/core/ingestion/languages/java/captures.js +14 -1
  70. package/dist/core/ingestion/languages/java/lombok-synthesizer.d.ts +42 -0
  71. package/dist/core/ingestion/languages/java/lombok-synthesizer.js +439 -0
  72. package/dist/core/ingestion/languages/java/scope-resolver.js +2 -0
  73. package/dist/core/ingestion/languages/java/spring-dynamic-lookup.d.ts +8 -0
  74. package/dist/core/ingestion/languages/java/spring-dynamic-lookup.js +62 -0
  75. package/dist/core/ingestion/languages/java.js +2 -0
  76. package/dist/core/ingestion/languages/jvm/accessor-synthesis.d.ts +99 -0
  77. package/dist/core/ingestion/languages/jvm/accessor-synthesis.js +173 -0
  78. package/dist/core/ingestion/languages/jvm/beanspec.d.ts +17 -0
  79. package/dist/core/ingestion/languages/jvm/beanspec.js +42 -0
  80. package/dist/core/ingestion/languages/kotlin/capture-side-channel.d.ts +5 -0
  81. package/dist/core/ingestion/languages/kotlin/capture-side-channel.js +16 -0
  82. package/dist/core/ingestion/languages/kotlin/captures.js +14 -1
  83. package/dist/core/ingestion/languages/kotlin/lombok-synthesizer.d.ts +26 -0
  84. package/dist/core/ingestion/languages/kotlin/lombok-synthesizer.js +417 -0
  85. package/dist/core/ingestion/languages/kotlin/scope-resolver.js +2 -0
  86. package/dist/core/ingestion/languages/kotlin/spring-dynamic-lookup.d.ts +8 -0
  87. package/dist/core/ingestion/languages/kotlin/spring-dynamic-lookup.js +77 -0
  88. package/dist/core/ingestion/languages/kotlin.js +2 -0
  89. package/dist/core/ingestion/pipeline-phases/di.js +47 -14
  90. package/dist/core/ingestion/pipeline-phases/parse-impl.d.ts +3 -1
  91. package/dist/core/ingestion/pipeline-phases/parse-impl.js +57 -86
  92. package/dist/core/ingestion/pipeline-phases/parse.d.ts +2 -0
  93. package/dist/core/ingestion/pipeline-phases/runner.d.ts +4 -1
  94. package/dist/core/ingestion/pipeline-phases/runner.js +34 -14
  95. package/dist/core/ingestion/pipeline-phases/scan.js +25 -13
  96. package/dist/core/ingestion/pipeline.d.ts +6 -0
  97. package/dist/core/ingestion/pipeline.js +44 -12
  98. package/dist/core/ingestion/scope-extractor.js +1 -0
  99. package/dist/core/ingestion/utils/symbol-labels.d.ts +2 -2
  100. package/dist/core/ingestion/utils/symbol-labels.js +2 -2
  101. package/dist/core/ingestion/workers/parse-worker.js +28 -2
  102. package/dist/core/lbug/lbug-adapter.d.ts +59 -0
  103. package/dist/core/lbug/lbug-adapter.js +154 -1
  104. package/dist/core/run-analyze.d.ts +17 -0
  105. package/dist/core/run-analyze.js +231 -23
  106. package/dist/core/search/fts-indexes.d.ts +19 -1
  107. package/dist/core/search/fts-indexes.js +28 -1
  108. package/dist/core/wiki/generator.js +8 -0
  109. package/dist/core/wiki/grok-client.d.ts +21 -0
  110. package/dist/core/wiki/grok-client.js +287 -0
  111. package/dist/core/wiki/llm-client.d.ts +1 -1
  112. package/dist/core/wiki/llm-client.js +5 -2
  113. package/dist/core/wiki/local-cli-client.d.ts +11 -0
  114. package/dist/core/wiki/local-cli-client.js +22 -9
  115. package/dist/mcp/local/local-backend.d.ts +24 -7
  116. package/dist/mcp/local/local-backend.js +190 -81
  117. package/dist/mcp/local/pdg-impact.d.ts +8 -4
  118. package/dist/mcp/local/pdg-impact.js +7 -2
  119. package/dist/mcp/repository-policy.d.ts +5 -1
  120. package/dist/mcp/repository-policy.js +48 -4
  121. package/dist/mcp/resources.js +2 -1
  122. package/dist/mcp/server.js +6 -5
  123. package/dist/mcp/tools.js +31 -19
  124. package/dist/server/api.js +23 -64
  125. package/dist/server/grep-params.d.ts +18 -0
  126. package/dist/server/grep-params.js +83 -0
  127. package/dist/server/grep-scan.d.ts +23 -0
  128. package/dist/server/grep-scan.js +107 -0
  129. package/dist/server/grep-worker.d.ts +1 -0
  130. package/dist/server/grep-worker.js +12 -0
  131. package/dist/server/mcp-http.d.ts +8 -0
  132. package/dist/server/mcp-http.js +16 -1
  133. package/dist/storage/file-hash.d.ts +5 -0
  134. package/dist/storage/file-hash.js +16 -6
  135. package/dist/storage/fs-atomic.d.ts +24 -0
  136. package/dist/storage/fs-atomic.js +86 -2
  137. package/dist/storage/git.d.ts +19 -8
  138. package/dist/storage/git.js +83 -24
  139. package/dist/storage/gitnexus-managed-paths.d.ts +36 -0
  140. package/dist/storage/gitnexus-managed-paths.js +46 -0
  141. package/dist/storage/parse-cache.d.ts +22 -4
  142. package/dist/storage/parse-cache.js +106 -29
  143. package/dist/storage/parsedfile-store.d.ts +34 -60
  144. package/dist/storage/parsedfile-store.js +177 -171
  145. package/dist/storage/repo-manager.d.ts +11 -1
  146. package/dist/storage/repo-manager.js +23 -2
  147. package/dist/storage/repo-meta.d.ts +12 -0
  148. package/dist/storage/v8-sidecar.d.ts +48 -0
  149. package/dist/storage/v8-sidecar.js +347 -0
  150. package/dist/types/pipeline.d.ts +14 -0
  151. package/package.json +4 -1
  152. package/scripts/cross-platform-shard.ts +4 -2
  153. package/scripts/cross-platform-tests.ts +7 -1
  154. package/skills/gitnexus-cli.md +11 -3
  155. package/skills/gitnexus-impact-analysis.md +9 -0
  156. package/web/assets/{agent-Dr4l5EOp.js → agent-CFqT4hjR.js} +108 -104
  157. package/web/assets/{index-2zdvEdzg.js → index-BMIniRtX.js} +3 -3
  158. package/web/index.html +1 -1
@@ -24,11 +24,11 @@
24
24
  *
25
25
  * ## Shape
26
26
  *
27
- * `<storagePath>/parsedfile-store/<shardId>.json` — one shard per parse chunk,
28
- * a JSON array of `ParsedFile` serialized with the same `mapReplacer` the parse
29
- * cache uses (Scope.bindings / Scope.typeBindings are `Map`s). The store is
30
- * cleared at the start of each parse and after scope-resolution consumes it, so
31
- * it never lingers and never goes stale across runs.
27
+ * `<storagePath>/parsedfile-store/<shardId>.v8` — one shard per parse chunk,
28
+ * a V8 envelope of `ParsedFile[]` (Scope.bindings / Scope.typeBindings stay
29
+ * `Map`s). The store is cleared at the start of each parse and after
30
+ * scope-resolution consumes it, so it never lingers and never goes stale
31
+ * across runs.
32
32
  *
33
33
  * ## Durable sibling store (`parsedfile-cache/`, warm-cache coverage)
34
34
  *
@@ -41,19 +41,21 @@
41
41
  * that gap we ALSO write the worker's ParsedFiles to a second, CONTENT-ADDRESSED
42
42
  * store keyed by the parse chunk hash (`getDurableParsedFileDir`), which mirrors
43
43
  * the parse cache's lifecycle (persists across runs, pruned by `usedKeys`,
44
- * version-tied via `PARSE_CACHE_VERSION`). On a warm hit the chunk's durable
45
- * shards are byte-COPIED into the run-scoped store (no re-parse, no
46
- * re-serialize → byte-identical), so scope-resolution streams them exactly as
47
- * on a cold run. Content-addressing makes stale reuse impossible: a changed
48
- * file changes its chunk hash, which misses BOTH stores and re-dispatches.
44
+ * version-tied via `PARSE_CACHE_VERSION`). On a warm hit the chunk's immutable
45
+ * durable shards are hardlinked (or atomically copied) into the run store after
46
+ * their envelope metadata proves complete coverage. That pins a stable snapshot
47
+ * before workers are skipped, even when another branch refreshes the shared
48
+ * durable directory concurrently.
49
49
  */
50
- import { promises as fs, mkdirSync, writeFileSync } from 'node:fs';
50
+ import { promises as fs, mkdirSync } from 'node:fs';
51
51
  import path from 'node:path';
52
52
  import v8 from 'node:v8';
53
53
  import vm from 'node:vm';
54
54
  import { isValidReceiverChain } from '../core/ingestion/utils/receiver-chain-codec.js';
55
55
  import { logger } from '../core/logger.js';
56
- import { mapReplacer, mapReviver } from './parse-cache.js';
56
+ import { mapReviver } from './parse-cache.js';
57
+ import { linkOrCopyFile } from './fs-atomic.js';
58
+ import { inspectV8Cache, tryLoadV8Cache, writeV8CacheFile, writeV8CacheFileSync, } from './v8-sidecar.js';
57
59
  const STORE_DIRNAME = 'parsedfile-store';
58
60
  const DURABLE_DIRNAME = 'parsedfile-cache';
59
61
  const DURABLE_INDEX_FILENAME = 'index.json';
@@ -148,32 +150,28 @@ export const getParsedFileStoreDir = (storagePath) => path.join(storagePath, STO
148
150
  export const clearParsedFileStore = async (storagePath) => {
149
151
  await fs.rm(getParsedFileStoreDir(storagePath), { recursive: true, force: true });
150
152
  };
153
+ const isV8ShardName = (name) => name.endsWith('.v8') && !name.includes('.v8.');
154
+ const shardPath = (storagePath, shardId) => path.join(getParsedFileStoreDir(storagePath), `${shardId}.v8`);
155
+ const shardFilePaths = (parsedFiles) => parsedFiles.map((pf) => pf.filePath);
156
+ const LOAD_YIELD_EVERY_SHARDS = 128;
151
157
  /**
152
- * Single source of truth for a shard's bytes. Returns `null` for an empty
153
- * chunk (caller writes nothing). Both the async (`persistParsedFileChunk`) and
154
- * sync (`persistParsedFileShardSync`) writers go through this so the two paths
155
- * are guaranteed byte-identical — the shards must round-trip through the same
156
- * `mapReviver`, and matching bytes by having both authors type the same
157
- * `mapReplacer` call would be a coincidence, not a guarantee.
158
+ * Test seam for #3086. Production always calls {@link forceGc}; unit tests
159
+ * replace `run` to count cadence without requiring `--expose-gc`.
158
160
  */
159
- const serializeParsedFileShard = (parsedFiles) => {
160
- if (parsedFiles.length === 0)
161
- return null;
162
- return JSON.stringify(parsedFiles, mapReplacer);
161
+ export const parsedFileLoadGc = {
162
+ run: forceGc,
163
+ /** V8 envelope bytes visited between GCs (#3086). Tests may lower this. */
164
+ byteBudget: 128 * 1024 * 1024,
163
165
  };
164
- const shardPath = (storagePath, shardId) => path.join(getParsedFileStoreDir(storagePath), `${shardId}.json`);
165
166
  /**
166
- * Write one parse chunk's `ParsedFile[]` to the store as a single shard (async).
167
- * No-op for an empty chunk. `shardId` must be unique within a run. Used by the
168
- * main-thread no-store-disabled fallback and any non-worker writer; the worker
169
- * store path uses {@link persistParsedFileShardSync}.
167
+ * Write one parse chunk's `ParsedFile[]` to the store as a single `.v8` shard.
168
+ * No-op for an empty chunk. `shardId` must be unique within a run.
170
169
  */
171
170
  export const persistParsedFileChunk = async (storagePath, shardId, parsedFiles) => {
172
- const payload = serializeParsedFileShard(parsedFiles);
173
- if (payload === null)
174
- return;
171
+ if (parsedFiles.length === 0)
172
+ return true;
175
173
  await fs.mkdir(getParsedFileStoreDir(storagePath), { recursive: true });
176
- await fs.writeFile(shardPath(storagePath, shardId), payload, 'utf-8');
174
+ return writeV8CacheFile(shardPath(storagePath, shardId), parsedFiles, shardFilePaths(parsedFiles));
177
175
  };
178
176
  // Per-process set of store dirs we've already `mkdir`ed, so the sync worker
179
177
  // writer (called once per job, many times into the same dir) doesn't issue a
@@ -181,130 +179,106 @@ export const persistParsedFileChunk = async (storagePath, shardId, parsedFiles)
181
179
  const createdStoreDirs = new Set();
182
180
  /**
183
181
  * Synchronous shard writer for use INSIDE a parse worker (#1983 parallel
184
- * serialization). The worker is a dedicated thread, so a blocking write there
185
- * protects the main thread, and a sync write avoids threading `async`/`await`
186
- * through the synchronous per-file extract loop. Produces byte-identical shards
187
- * to {@link persistParsedFileChunk} via the shared {@link serializeParsedFileShard}.
188
- * No-op for an empty chunk. `shardId` must be globally unique for the run (the
189
- * worker uses `w<threadId>-<seq>`); a duplicate would silently overwrite.
182
+ * serialization). Returns false on write failure so the worker can keep
183
+ * ParsedFiles in the result instead of dropping them.
190
184
  */
191
185
  export const persistParsedFileShardSync = (storagePath, shardId, parsedFiles) => {
192
- const payload = serializeParsedFileShard(parsedFiles);
193
- if (payload === null)
194
- return;
186
+ if (parsedFiles.length === 0)
187
+ return true;
195
188
  const dir = getParsedFileStoreDir(storagePath);
196
189
  if (!createdStoreDirs.has(dir)) {
197
190
  mkdirSync(dir, { recursive: true });
198
191
  createdStoreDirs.add(dir);
199
192
  }
200
- writeFileSync(shardPath(storagePath, shardId), payload, 'utf-8');
193
+ return writeV8CacheFileSync(shardPath(storagePath, shardId), parsedFiles, shardFilePaths(parsedFiles));
201
194
  };
202
- /**
203
- * Stream the store and return the `ParsedFile`s whose `filePath` is in
204
- * `wantPaths`, keyed by path. Loads one shard at a time and retains only the
205
- * matching entries, so peak heap is bounded by (matched set) + (one shard)
206
- * rather than the whole store. Returns an empty map when the store is absent
207
- * (e.g. tests, or a run with no worker pool) — callers fall back to a fresh
208
- * extract for the missing files.
209
- */
210
- export const loadParsedFilesForPaths = async (storagePath, wantPaths) => {
211
- const out = new Map();
212
- if (wantPaths.size === 0)
213
- return out;
214
- const dir = getParsedFileStoreDir(storagePath);
215
- let shards;
195
+ const listV8Shards = async (dir) => {
216
196
  try {
217
- shards = (await fs.readdir(dir)).filter((f) => f.endsWith('.json'));
197
+ return (await fs.readdir(dir)).filter(isV8ShardName).map((name) => path.join(dir, name));
218
198
  }
219
199
  catch {
220
- return out; // store absent
200
+ return [];
221
201
  }
222
- // Shared interning pool for this load — deduplicates strings ACROSS shards
223
- // (one `int` / one repeated filePath for the whole language), which is where
224
- // most of the saving comes from. Dropped when this function returns.
202
+ };
203
+ export const loadParsedFilesForPaths = async (storagePath, wantPaths) => {
204
+ const out = new Map();
205
+ if (wantPaths.size === 0)
206
+ return out;
207
+ const shardPaths = await listV8Shards(getParsedFileStoreDir(storagePath));
225
208
  const pool = new Map();
226
209
  let droppedSites = 0;
227
210
  let filesWithDroppedSites = 0;
228
211
  let droppedChains = 0;
229
212
  let rejectedFiles = 0;
230
- for (let i = 0; i < shards.length; i++) {
231
- // Per-shard def pool: a SymbolDefinition's three serialized copies live within
232
- // a single shard (one ParsedFile), so the dedup is shard-local. A cross-shard
233
- // pool would retain defs of files NOT in `wantPaths` (loaded-but-discarded
234
- // shards), reintroducing the leak; per-shard drops them with the shard.
235
- const defPool = new Map();
236
- const reviver = makeInterningReviver(pool, defPool);
237
- let parsed;
238
- try {
239
- const raw = await fs.readFile(path.join(dir, shards[i]), 'utf-8');
240
- parsed = JSON.parse(raw, reviver);
213
+ let bytesSinceGc = 0;
214
+ let shardsSinceYield = 0;
215
+ const maybeYieldAndGc = async (forceByteGc) => {
216
+ if (forceByteGc) {
217
+ parsedFileLoadGc.run();
218
+ bytesSinceGc = 0;
219
+ shardsSinceYield = 0;
220
+ await new Promise((resolve) => setImmediate(resolve));
221
+ return;
241
222
  }
242
- catch {
243
- continue; // skip a corrupt shard; missing files fall back to fresh extract
223
+ shardsSinceYield++;
224
+ if (shardsSinceYield >= LOAD_YIELD_EVERY_SHARDS) {
225
+ shardsSinceYield = 0;
226
+ await new Promise((resolve) => setImmediate(resolve));
244
227
  }
245
- if (!Array.isArray(parsed))
228
+ };
229
+ for (const shardFull of shardPaths) {
230
+ const loaded = await tryLoadV8Cache(shardFull, pool, wantPaths);
231
+ if (loaded === undefined) {
232
+ await maybeYieldAndGc(false);
246
233
  continue;
247
- for (const pf of parsed) {
248
- if (!pf || typeof pf.filePath !== 'string' || !wantPaths.has(pf.filePath))
249
- continue;
250
- const flow = sanitizeCallableFlowSites(pf.callableFlowSites);
251
- if (flow === undefined) {
252
- // non-array garbage → distrust the file, re-extract
253
- rejectedFiles++;
254
- continue;
255
- }
256
- const chains = sanitizeReceiverChains(pf.referenceSites);
257
- if (chains === undefined) {
258
- rejectedFiles++;
259
- continue;
260
- }
261
- if (flow.dropped === 0 && chains.dropped === 0) {
262
- out.set(pf.filePath, pf);
263
- }
264
- else {
265
- droppedSites += flow.dropped;
266
- droppedChains += chains.dropped;
267
- filesWithDroppedSites++;
268
- out.set(pf.filePath, {
269
- ...pf,
270
- ...(flow.dropped === 0 ? {} : { callableFlowSites: flow.sites }),
271
- ...(chains.dropped === 0 ? {} : { referenceSites: chains.sites }),
272
- });
273
- }
274
234
  }
275
- // Every few shards, reclaim the transient pre-intern parse churn before it
276
- // piles up against the heap limit (~5 GB avoidable on the kernel), and
277
- // yield so the GC + any pending I/O can run.
278
- if ((i & 7) === 7) {
279
- forceGc();
280
- await new Promise((resolve) => setImmediate(resolve));
235
+ if (loaded.kind === 'skip') {
236
+ bytesSinceGc += loaded.bytes;
237
+ await maybeYieldAndGc(bytesSinceGc >= parsedFileLoadGc.byteBudget);
238
+ continue;
281
239
  }
240
+ bytesSinceGc += loaded.bytes;
241
+ const parsed = Array.isArray(loaded.value) ? loaded.value : undefined;
242
+ const crossedBudget = bytesSinceGc >= parsedFileLoadGc.byteBudget;
243
+ if (Array.isArray(parsed)) {
244
+ for (const pf of parsed) {
245
+ if (!pf || typeof pf.filePath !== 'string' || !wantPaths.has(pf.filePath))
246
+ continue;
247
+ const flow = sanitizeCallableFlowSites(pf.callableFlowSites);
248
+ if (flow === undefined) {
249
+ rejectedFiles++;
250
+ continue;
251
+ }
252
+ const chains = sanitizeReceiverChains(pf.referenceSites);
253
+ if (chains === undefined) {
254
+ rejectedFiles++;
255
+ continue;
256
+ }
257
+ if (flow.dropped === 0 && chains.dropped === 0) {
258
+ out.set(pf.filePath, pf);
259
+ }
260
+ else {
261
+ droppedSites += flow.dropped;
262
+ droppedChains += chains.dropped;
263
+ filesWithDroppedSites++;
264
+ out.set(pf.filePath, {
265
+ ...pf,
266
+ ...(flow.dropped === 0 ? {} : { callableFlowSites: flow.sites }),
267
+ ...(chains.dropped === 0 ? {} : { referenceSites: chains.sites }),
268
+ });
269
+ }
270
+ }
271
+ }
272
+ await maybeYieldAndGc(crossedBudget);
282
273
  }
283
274
  if (droppedSites > 0 || droppedChains > 0) {
284
- // Facts for the dropped sites are omitted this run (the file itself is
285
- // retained, so no re-extract happens) — surface it so a recurring drop
286
- // on every warm load is observable rather than silent (#2522 review).
287
275
  logger.warn({ droppedSites, droppedChains, files: filesWithDroppedSites }, 'parsedfile-store: dropped malformed/over-bound sites at load; files retained without those facts');
288
276
  }
289
277
  if (rejectedFiles > 0) {
290
- // The other half of the same defect. A rejected file silently falls back to
291
- // a fresh extract EVERY load, so a writer that keeps minting what this
292
- // reader keeps refusing is a permanent warm-cache miss that costs real time
293
- // and says nothing about why.
294
278
  logger.warn({ rejectedFiles }, 'parsedfile-store: rejected shard entries at load (untrusted shape); those files re-extract every run');
295
279
  }
296
280
  return out;
297
281
  };
298
- /**
299
- * Treat the durable ParsedFile store as an untrusted serialization boundary.
300
- * Sanitation is per-SITE, not per-file: one malformed or over-bound fact drops
301
- * only itself (counted, logged by the caller), so a legitimately pathological
302
- * source file cannot push its whole ParsedFile into a permanent, silent
303
- * warm-cache-miss reparse loop (#2522 review). Only a non-array field —
304
- * i.e. garbage that says the serialization itself is untrustworthy — rejects
305
- * the file, and `undefined` (never emitted / no facts) passes through.
306
- * Returns `undefined` for the reject-file case.
307
- */
308
282
  function sanitizeCallableFlowSites(value) {
309
283
  if (value === undefined)
310
284
  return { sites: undefined, dropped: 0 };
@@ -486,9 +460,6 @@ export const prepareDurableParsedFileChunk = async (durableDir, chunkHash) => {
486
460
  await fs.rm(dir, { recursive: true, force: true });
487
461
  await fs.mkdir(dir, { recursive: true });
488
462
  };
489
- // Per-process set of durable chunk subdirs already `mkdir`ed (mirrors
490
- // `createdStoreDirs`) so the worker doesn't `mkdirSync` on every shard.
491
- const createdDurableDirs = new Set();
492
463
  /**
493
464
  * Synchronous durable-shard writer for use INSIDE a parse worker, alongside
494
465
  * {@link persistParsedFileShardSync}. Writes the SAME bytes to a content-addressed
@@ -498,63 +469,85 @@ const createdDurableDirs = new Set();
498
469
  * uniqueness that makes the run-scoped `w<tid>-<seq>` name safe, prefixed by
499
470
  * content. No-op for an empty chunk.
500
471
  */
472
+ const createdDurableDirs = new Set();
501
473
  export const persistDurableParsedFileShardSync = (durableDir, chunkHash, threadId, shardSeq, parsedFiles) => {
502
- const payload = serializeParsedFileShard(parsedFiles);
503
- if (payload === null)
504
- return;
474
+ if (parsedFiles.length === 0)
475
+ return true;
505
476
  const dir = durableChunkDir(durableDir, chunkHash);
506
477
  if (!createdDurableDirs.has(dir)) {
507
478
  mkdirSync(dir, { recursive: true });
508
479
  createdDurableDirs.add(dir);
509
480
  }
510
- writeFileSync(path.join(dir, `${chunkHash}-w${threadId}-${shardSeq}.json`), payload, 'utf-8');
481
+ const dest = path.join(dir, `${chunkHash}-w${threadId}-${shardSeq}.v8`);
482
+ return writeV8CacheFileSync(dest, parsedFiles, shardFilePaths(parsedFiles));
511
483
  };
512
484
  /**
513
- * Restore a cached chunk's durable shards into the run-scoped store on a warm
514
- * hit. A verbatim byte copy (no parse, no re-serialize), so the restored
515
- * ParsedFiles are byte-identical to a cold run and `loadParsedFilesForPaths`
516
- * (which keys on `filePath`, not shard name) gives scope-resolution full
517
- * coverage. The durable shard names already carry the chunk hash, so they never
518
- * collide with the worker's run-scoped `w<tid>-<seq>` shards. Returns the number
519
- * of shards restored (0 ⇒ no durable coverage for this chunk; caller treats it
520
- * as a miss).
485
+ * Validate and snapshot one durable chunk into the run store. Every envelope
486
+ * must be runtime-compatible and integrity-valid, and together they must match
487
+ * the path coverage recorded when the durable index was published. Linking
488
+ * before returning pins the inodes against concurrent branch-cache rotation.
521
489
  */
522
- export const restoreDurableParsedFileShard = async (durableDir, runStoragePath, chunkHash) => {
523
- const src = durableChunkDir(durableDir, chunkHash);
524
- let shards;
490
+ export const durableChunkHasShards = async (runStoragePath, chunkHash, expectedPaths) => {
491
+ const sourceDir = durableChunkDir(getDurableParsedFileDir(runStoragePath), chunkHash);
492
+ const shards = await listV8Shards(sourceDir);
493
+ if (shards.length === 0 || expectedPaths.size === 0)
494
+ return false;
495
+ const runDir = getParsedFileStoreDir(runStoragePath);
525
496
  try {
526
- shards = (await fs.readdir(src)).filter((f) => f.endsWith('.json'));
497
+ await fs.mkdir(runDir, { recursive: true });
527
498
  }
528
499
  catch {
529
- return 0; // no durable shards for this chunk
500
+ return false;
530
501
  }
531
- if (shards.length === 0)
532
- return 0;
533
- const dst = getParsedFileStoreDir(runStoragePath);
534
- await fs.mkdir(dst, { recursive: true });
535
- for (const name of shards) {
536
- await fs.copyFile(path.join(src, name), path.join(dst, name));
502
+ const restored = [];
503
+ const covered = new Set();
504
+ const rollback = async () => {
505
+ await Promise.all(restored.map((filePath) => fs.rm(filePath, { force: true }).catch(() => { })));
506
+ return false;
507
+ };
508
+ for (const sourcePath of shards) {
509
+ const name = path.basename(sourcePath);
510
+ const destinationPath = path.join(runDir, name);
511
+ try {
512
+ await linkOrCopyFile(sourcePath, destinationPath);
513
+ restored.push(destinationPath);
514
+ }
515
+ catch {
516
+ return rollback();
517
+ }
518
+ const inspected = await inspectV8Cache(destinationPath);
519
+ if (!inspected)
520
+ return rollback();
521
+ for (const filePath of inspected.paths) {
522
+ if (!expectedPaths.has(filePath))
523
+ return rollback();
524
+ covered.add(filePath);
525
+ }
537
526
  }
538
- return shards.length;
527
+ if (covered.size !== expectedPaths.size)
528
+ return rollback();
529
+ return true;
539
530
  };
540
- /**
541
- * Read the durable index and return the set of chunk hashes it vouches for,
542
- * gated on `expectedVersion` (`PARSE_CACHE_VERSION`). A version mismatch or a
543
- * missing/corrupt index returns the empty set — the caller then treats every
544
- * chunk as a durable miss and re-dispatches workers (NEVER the main-thread
545
- * `extractParsedFile` fallback), which rewrites the durable store under the new
546
- * version. Mirrors `loadParseCache`'s version-invalidation contract.
547
- */
548
531
  export const loadDurableParsedFileIndex = async (durableDir, expectedVersion) => {
549
532
  try {
550
533
  const raw = await fs.readFile(path.join(durableDir, DURABLE_INDEX_FILENAME), 'utf-8');
551
534
  const idx = JSON.parse(raw);
552
- if (idx?.version !== expectedVersion || !Array.isArray(idx.keys))
553
- return new Set();
554
- return new Set(idx.keys);
535
+ if (!isRecord(idx) || idx.version !== expectedVersion || !isRecord(idx.entries)) {
536
+ return new Map();
537
+ }
538
+ const entries = new Map();
539
+ for (const [key, paths] of Object.entries(idx.entries)) {
540
+ if (!Array.isArray(paths) ||
541
+ paths.length === 0 ||
542
+ paths.some((filePath) => typeof filePath !== 'string')) {
543
+ return new Map();
544
+ }
545
+ entries.set(key, new Set(paths));
546
+ }
547
+ return entries;
555
548
  }
556
549
  catch {
557
- return new Set();
550
+ return new Map();
558
551
  }
559
552
  };
560
553
  /**
@@ -562,9 +555,9 @@ export const loadDurableParsedFileIndex = async (durableDir, expectedVersion) =>
562
555
  * be the parse cache's surviving on-disk keys (so the two stores stay coherent:
563
556
  * a chunk is "cached" iff BOTH its parse-cache shard and its durable shards
564
557
  * exist; a quarantined chunk — no parse-cache shard — drops its durable subdir
565
- * here and re-dispatches next run). Only subdirs with ≥1 shard are indexed
566
- * (mirrors `saveParseCache`'s written-keys discipline — never vouch for a chunk
567
- * hash with no backing shard). The index write is tmp+rename atomic.
558
+ * here and re-dispatches next run). Only chunks whose envelopes all validate
559
+ * are indexed, together with their exact persisted path coverage (never vouch
560
+ * for a missing/corrupt shard). The index write is tmp+rename atomic.
568
561
  */
569
562
  export const pruneAndSaveDurableParsedFileStore = async (durableDir, version, keepKeys) => {
570
563
  let entries;
@@ -574,17 +567,30 @@ export const pruneAndSaveDurableParsedFileStore = async (durableDir, version, ke
574
567
  catch {
575
568
  return; // nothing written this run
576
569
  }
577
- const survivors = [];
570
+ const survivors = {};
578
571
  for (const name of entries) {
579
572
  if (name === DURABLE_INDEX_FILENAME)
580
573
  continue;
581
574
  const full = path.join(durableDir, name);
582
575
  if (keepKeys.has(name)) {
583
576
  try {
584
- const shards = (await fs.readdir(full)).filter((f) => f.endsWith('.json'));
577
+ const shards = await listV8Shards(full);
585
578
  if (shards.length > 0) {
586
- survivors.push(name);
587
- continue;
579
+ const covered = new Set();
580
+ let valid = true;
581
+ for (const shard of shards) {
582
+ const inspected = await inspectV8Cache(shard);
583
+ if (!inspected) {
584
+ valid = false;
585
+ break;
586
+ }
587
+ for (const filePath of inspected.paths)
588
+ covered.add(filePath);
589
+ }
590
+ if (valid && covered.size > 0) {
591
+ survivors[name] = [...covered].sort();
592
+ continue;
593
+ }
588
594
  }
589
595
  }
590
596
  catch {
@@ -593,7 +599,7 @@ export const pruneAndSaveDurableParsedFileStore = async (durableDir, version, ke
593
599
  }
594
600
  await fs.rm(full, { recursive: true, force: true });
595
601
  }
596
- const idx = { version, keys: survivors };
602
+ const idx = { version, entries: survivors };
597
603
  const tmp = path.join(durableDir, `${DURABLE_INDEX_FILENAME}.tmp`);
598
604
  await fs.mkdir(durableDir, { recursive: true });
599
605
  await fs.writeFile(tmp, JSON.stringify(idx), 'utf-8');
@@ -475,6 +475,15 @@ export declare const assertSafeStoragePath: (entry: RegistryEntry) => void;
475
475
  * `GITNEXUS_HOME`.
476
476
  */
477
477
  export declare const resolveRegistryEntry: (entries: RegistryEntry[], target: string) => RegistryEntry;
478
+ /**
479
+ * Name-only registry match (the name tier of {@link resolveRegistryEntry},
480
+ * without path matching). Used by `group.yaml` member *values*, which are
481
+ * registry aliases, not filesystem paths.
482
+ *
483
+ * Zero matches → `undefined` (caller treats as missing). One match → that
484
+ * entry. Two or more → {@link RegistryAmbiguousTargetError}.
485
+ */
486
+ export declare const findRegistryEntryByName: (entries: RegistryEntry[], name: string) => RegistryEntry | undefined;
478
487
  /**
479
488
  * List all registered repos from the global registry.
480
489
  *
@@ -494,11 +503,12 @@ export interface CLIConfig {
494
503
  apiKey?: string;
495
504
  model?: string;
496
505
  baseUrl?: string;
497
- provider?: 'openai' | 'openrouter' | 'azure' | 'custom' | 'cursor' | 'claude' | 'codex' | 'opencode' | 'minimax';
506
+ provider?: 'openai' | 'openrouter' | 'azure' | 'custom' | 'cursor' | 'claude' | 'codex' | 'opencode' | 'grok' | 'minimax';
498
507
  cursorModel?: string;
499
508
  claudeModel?: string;
500
509
  codexModel?: string;
501
510
  opencodeModel?: string;
511
+ grokModel?: string;
502
512
  /** Azure api-version query param (e.g. '2024-10-21'). Only used when provider is 'azure'. */
503
513
  apiVersion?: string;
504
514
  /** Set true when the deployment is a reasoning model (o1, o3, o4-mini). Auto-detected for OpenAI; must be set for Azure deployments. */
@@ -771,7 +771,10 @@ const registerRepoUnlocked = async (repoPath, meta, opts) => {
771
771
  // expands macOS /var → /private/var and Windows 8.3 → long-name,
772
772
  // falling back to `path.resolve` when the path doesn't exist.
773
773
  const canonicalInput = canonicalizePath(repoPath);
774
- const entries = await readRegistry();
774
+ // Mutating writes must not treat an unreadable/truncated registry as empty
775
+ // (#3094): lenient `readRegistry()` returns `[]` on parse failure and would
776
+ // replace the machine-wide file with only this entry. ENOENT stays empty.
777
+ const entries = await readRegistryStrict();
775
778
  const existingIdx = entries.findIndex((e) => {
776
779
  // Canonicalise the STORED entry too so pre-canonicalisation
777
780
  // registries (written by older versions, or paths passed in a
@@ -881,7 +884,7 @@ const registerRepoUnlocked = async (repoPath, meta, opts) => {
881
884
  // R9): re-derive THIS run's delta against the FRESHEST snapshot so a
882
885
  // concurrent change to the OTHER axis (a branch upsert vs a primary refresh)
883
886
  // survives instead of being clobbered by a stale entry-time view.
884
- const fresh = await readRegistry();
887
+ const fresh = await readRegistryStrict();
885
888
  const freshIdx = fresh.findIndex((e) => {
886
889
  const a = canonicalizePath(e.path);
887
890
  return registryPathEquals(a, canonicalInput);
@@ -1301,6 +1304,24 @@ export const resolveRegistryEntry = (entries, target) => {
1301
1304
  const availableNames = entries.map((e) => (nameCounts.get(e.name.toLowerCase()) ?? 0) > 1 ? `${e.name} (${e.path})` : e.name);
1302
1305
  throw new RegistryNotFoundError(target, availableNames);
1303
1306
  };
1307
+ /**
1308
+ * Name-only registry match (the name tier of {@link resolveRegistryEntry},
1309
+ * without path matching). Used by `group.yaml` member *values*, which are
1310
+ * registry aliases, not filesystem paths.
1311
+ *
1312
+ * Zero matches → `undefined` (caller treats as missing). One match → that
1313
+ * entry. Two or more → {@link RegistryAmbiguousTargetError}.
1314
+ */
1315
+ export const findRegistryEntryByName = (entries, name) => {
1316
+ const targetLower = name.toLowerCase();
1317
+ const nameMatches = entries.filter((e) => e.name.toLowerCase() === targetLower);
1318
+ if (nameMatches.length === 1)
1319
+ return nameMatches[0];
1320
+ if (nameMatches.length > 1) {
1321
+ throw new RegistryAmbiguousTargetError(name, nameMatches);
1322
+ }
1323
+ return undefined;
1324
+ };
1304
1325
  /**
1305
1326
  * List all registered repos from the global registry.
1306
1327
  *
@@ -279,6 +279,18 @@ export interface RepoMeta {
279
279
  * Map keys are repo-relative paths.
280
280
  */
281
281
  fileHashes?: Record<string, string>;
282
+ /**
283
+ * Coverage policy used when `fileHashes` was recorded. `status` replays it
284
+ * so analyze-time `--max-file-size` / `GITNEXUS_MAX_FILE_SIZE` cannot make
285
+ * a later default-cap walk drop a file the index actually covers.
286
+ * `dirtyPaths` are covered files that were dirty vs HEAD at that moment —
287
+ * status must re-hash those even after Git becomes clean (indexed-dirty then
288
+ * restore). Absent on indexes written before this field.
289
+ */
290
+ indexCoverage?: {
291
+ maxFileSizeBytes: number;
292
+ dirtyPaths?: string[];
293
+ };
282
294
  /**
283
295
  * Set when a run finished but the persisted edge count came back far short
284
296
  * of what the pipeline produced — the B2 "refresh reports SUCCESS while the
@@ -0,0 +1,48 @@
1
+ /** Envelope version — independent of PARSE_CACHE_VERSION / SCHEMA_BUMP. */
2
+ export declare const V8_CACHE_FORMAT = 5;
3
+ /**
4
+ * Collapse duplicate strings in a live deserialized graph into `pool`, mutating
5
+ * in place so object identity (shared `SymbolDefinition`s, Maps) is preserved.
6
+ * Required after `v8.deserialize` of ParsedFile shards: V8 does not recreate
7
+ * the JSON reviver's cross-shard string intern, and skipping it regresses
8
+ * retained heap (~+59% measured vs interned JSON).
9
+ */
10
+ export declare const internGraphStrings: (root: unknown, pool: Map<string, string>) => unknown;
11
+ /** NDJSON listing cannot encode paths that themselves contain CR/LF/NUL. */
12
+ export declare const encodeCachePathListing: (paths: readonly string[]) => Buffer;
13
+ /**
14
+ * Parse a counted NDJSON path listing. Returns `null` when it must not be
15
+ * trusted to skip the payload: missing trailing newline, CR/NUL, or a count
16
+ * that does not match the remaining lines.
17
+ */
18
+ export declare const parseCachePathListing: (raw: Buffer) => string[] | null;
19
+ export type V8CacheHit = {
20
+ kind: 'hit';
21
+ value: unknown;
22
+ bytes: number;
23
+ };
24
+ export type V8CacheSkip = {
25
+ kind: 'skip';
26
+ bytes: number;
27
+ };
28
+ export type V8CacheLoad = V8CacheHit | V8CacheSkip;
29
+ export type V8CacheInspection = {
30
+ paths: readonly string[];
31
+ };
32
+ /**
33
+ * Validate the immutable envelope metadata needed by the durable ParsedFile
34
+ * warm-hit gate without deserializing its payload. Atomic publication means a
35
+ * runtime-compatible envelope whose exact file length and counted path listing
36
+ * validate is a stable snapshot candidate; malformed/truncated envelopes miss.
37
+ */
38
+ export declare const inspectV8Cache: (filePath: string) => Promise<V8CacheInspection | undefined>;
39
+ /**
40
+ * Load a cache file. When `wantPaths` is set and a digest-verified non-empty
41
+ * path listing has no intersection, returns `{ kind: 'skip' }` without
42
+ * deserializing. An authentic listing that does not parse deserializes (fail
43
+ * closed). Envelope/runtime/digest failure returns undefined (miss).
44
+ */
45
+ export declare const tryLoadV8Cache: (filePath: string, internPool?: Map<string, string>, wantPaths?: ReadonlySet<string>) => Promise<V8CacheLoad | undefined>;
46
+ export declare const writeV8CacheFile: (filePath: string, graph: unknown, paths?: readonly string[]) => Promise<boolean>;
47
+ export declare const writeV8CacheFileSync: (filePath: string, graph: unknown, paths?: readonly string[]) => boolean;
48
+ export declare const copyV8CacheIfPresent: (srcPath: string, dstPath: string) => Promise<boolean>;