knodin 0.7.5 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. package/README.md +18 -3
  2. package/benchmarks/competitors/SYNTHESIS.md +66 -0
  3. package/dist/bin/cli.js +371 -66
  4. package/dist/bin/launcher.js +16 -1
  5. package/dist/src/agent-integration.js +82 -16
  6. package/dist/src/artifact-refresh.js +2 -1
  7. package/dist/src/cli-args.js +19 -1
  8. package/dist/src/cli-model.js +28 -2
  9. package/dist/src/codeflow-replay.js +2 -1
  10. package/dist/src/compare.js +39 -0
  11. package/dist/src/competitive-constraints.js +2 -1
  12. package/dist/src/competitive-runner.js +4 -4
  13. package/dist/src/context-export.js +3 -2
  14. package/dist/src/context.js +1 -1
  15. package/dist/src/deterministic-random.js +34 -0
  16. package/dist/src/diagnostics-write-helper.js +473 -0
  17. package/dist/src/diagnostics.js +1160 -133
  18. package/dist/src/doctor.js +3 -1
  19. package/dist/src/engine/ann-hnsw.js +2 -12
  20. package/dist/src/engine/file-walker.js +8 -2
  21. package/dist/src/engine/git-history.js +12 -12
  22. package/dist/src/engine/index.js +1174 -313
  23. package/dist/src/engine/sarif-import.js +341 -0
  24. package/dist/src/engine/scip-import.js +28 -13
  25. package/dist/src/engine/source-policy.js +16 -0
  26. package/dist/src/engine/state-paths.js +175 -0
  27. package/dist/src/execution-profile.js +15 -10
  28. package/dist/src/failure-diagnosis.js +7 -1
  29. package/dist/src/graph-layout.js +173 -0
  30. package/dist/src/index-activity.js +2 -1
  31. package/dist/src/init.js +86 -45
  32. package/dist/src/lifecycle-health.js +41 -9
  33. package/dist/src/mcp-graph-worker.js +69 -0
  34. package/dist/src/mcp-reliability.js +154 -0
  35. package/dist/src/mcp-worker-supervisor.js +350 -0
  36. package/dist/src/mirror.js +290 -0
  37. package/dist/src/node-runtime.js +157 -0
  38. package/dist/src/output-compression.js +2 -1
  39. package/dist/src/output-telemetry.js +16 -11
  40. package/dist/src/progressive-evidence.js +30 -26
  41. package/dist/src/pure-compression-cli.js +4 -3
  42. package/dist/src/relationship-adapters.js +15 -8
  43. package/dist/src/release-preflight.js +13 -10
  44. package/dist/src/repair-lease.js +85 -0
  45. package/dist/src/repository-init-process.js +13 -9
  46. package/dist/src/repository-management.js +34 -4
  47. package/dist/src/response-budget.js +8 -6
  48. package/dist/src/server.js +80 -35
  49. package/dist/src/structural-fast-path.js +16 -10
  50. package/dist/src/structural-snapshot.js +6 -2
  51. package/dist/src/system-config.js +25 -2
  52. package/dist/src/tools/knodin-tools.js +142 -31
  53. package/dist/src/update-ceremony.js +9 -5
  54. package/dist/src/update-trust.js +5 -4
  55. package/dist/src/visualization.js +372 -19
  56. package/dist/src/worktree-lifecycle.js +5 -2
  57. package/docs/BEHAVIORAL-CONTRACT.md +72 -0
  58. package/docs/CLI.md +20 -1
  59. package/docs/COMPARISON.md +403 -0
  60. package/docs/COMPETITIVE-LANDSCAPE-2026-08.md +267 -0
  61. package/docs/DIAGNOSTICS.md +46 -11
  62. package/docs/HANDOFF.md +180 -0
  63. package/docs/INSTALLATION.md +21 -2
  64. package/docs/MCP.md +59 -8
  65. package/docs/PT-ACCESS-RECOMMENDATION.md +5 -7
  66. package/docs/REPOSITORIES-AND-WORKTREES.md +18 -6
  67. package/docs/SCIP-IMPORT.md +5 -0
  68. package/docs/TOKEN-OPTIMIZER-SCORECARD.md +79 -0
  69. package/docs/releases/0.5.1.md +4 -4
  70. package/docs/releases/0.8.0.md +74 -0
  71. package/docs/releases/0.8.2.md +34 -0
  72. package/package.json +17 -4
  73. package/roadmap/competitive-roadmap.md +3801 -0
  74. package/schemas/release-attestation-v1.schema.json +1 -1
  75. package/schemas/support-bundle-v2.schema.json +212 -0
@@ -0,0 +1,341 @@
1
+ /**
2
+ * Bounded SARIF import.
3
+ *
4
+ * SARIF (OASIS "Static Analysis Results Interchange Format") is what most
5
+ * analyzers can emit — Salesforce Code Analyzer, ESLint, Semgrep, CodeQL, PMD,
6
+ * Trivy. Importing the format rather than any one tool's native output means one
7
+ * reader covers all of them.
8
+ *
9
+ * Findings are NOT graph facts. knodin resolves references from source; a
10
+ * finding is somebody else's judgement about a location. It is stored in its own
11
+ * table, tagged with the tool that produced it, and never merged into symbols or
12
+ * references. What the graph adds is the join: a finding on a hub matters more
13
+ * than the same finding on a leaf, and only the graph knows which is which.
14
+ *
15
+ * Everything here is bounded and fail-closed, matching the SCIP importer: a file
16
+ * over the byte ceiling, a malformed document, or a finding count over the cap
17
+ * is refused rather than partially applied.
18
+ */
19
+ import fs from "node:fs";
20
+ import path from "node:path";
21
+ import { compareBytes } from "../compare.js";
22
+ // Calibrated against real output rather than guessed: one Salesforce Code
23
+ // Analyzer run over a mid-sized org (pmd + regex + retire-js) produced a 97 MB
24
+ // SARIF log with ~75k findings. A 64 MB ceiling would have refused it.
25
+ export const SARIF_DEFAULT_LIMITS = {
26
+ maxBytes: 256 * 1024 * 1024,
27
+ maxFindings: 250_000,
28
+ timeoutMs: 60_000,
29
+ };
30
+ /**
31
+ * The ceiling knodin cannot raise. `JSON.parse` needs the document as a single
32
+ * string and V8 caps string length at 2^29 - 24 bytes on 64-bit, so a log past
33
+ * this size cannot be read whole regardless of the configured limit. Kept
34
+ * slightly under the hard maximum to leave room for decoding overhead.
35
+ */
36
+ const MAX_PARSEABLE_BYTES = 512 * 1024 * 1024;
37
+ /** Byte counts appear in operator-facing errors, so render them readably. */
38
+ function formatBytes(value) {
39
+ if (value < 1024)
40
+ return `${value} bytes`;
41
+ const units = ["KiB", "MiB", "GiB"];
42
+ let scaled = value / 1024;
43
+ let unit = 0;
44
+ while (scaled >= 1024 && unit < units.length - 1) {
45
+ scaled /= 1024;
46
+ unit++;
47
+ }
48
+ return `${scaled.toFixed(1)} ${units[unit]}`;
49
+ }
50
+ /** SARIF level is optional; analyzers often carry severity on the rule instead. */
51
+ function normalizeLevel(value) {
52
+ if (value === "error" || value === "warning" || value === "note" || value === "none")
53
+ return value;
54
+ // SARIF's own default when level is absent and no rule default applies.
55
+ return "warning";
56
+ }
57
+ /**
58
+ * Resolves a SARIF artifact location to a repository-relative POSIX path.
59
+ *
60
+ * SARIF permits `file:///abs/path`, a bare absolute path, or a path relative to
61
+ * some base URI. Anything landing outside the repository is rejected: a findings
62
+ * file is external input, and a `../` escape must not attach a finding to a path
63
+ * the caller never indexed.
64
+ */
65
+ function resolveArtifactPath(repoRoot, uri, uriBase) {
66
+ if (!uri)
67
+ return null;
68
+ let raw = uri;
69
+ if (raw.startsWith("file://")) {
70
+ try {
71
+ raw = decodeURIComponent(new URL(raw).pathname);
72
+ }
73
+ catch {
74
+ return null;
75
+ }
76
+ }
77
+ else {
78
+ try {
79
+ raw = decodeURIComponent(raw);
80
+ }
81
+ catch {
82
+ // Not percent-encoded; use as-is.
83
+ }
84
+ }
85
+ // A relative uri may be anchored by an originalUriBaseId value.
86
+ const anchor = uriBase && path.isAbsolute(uriBase) ? uriBase : repoRoot;
87
+ const candidate = path.isAbsolute(raw) ? raw : path.resolve(anchor, raw);
88
+ // Compare real paths. The analyzer may have recorded a route through a
89
+ // symlink — /var vs /private/var on macOS, or a symlinked checkout — and a
90
+ // textual prefix check would reject those valid locations. Resolving also
91
+ // means a symlink pointing outside the repository is still caught, because
92
+ // the real path is what gets compared.
93
+ let resolved = path.resolve(candidate);
94
+ try {
95
+ resolved = fs.realpathSync(resolved);
96
+ }
97
+ catch {
98
+ // Not on disk: the analyzer may report a file since deleted. Keep the
99
+ // lexically resolved path so containment is still enforced below.
100
+ }
101
+ if (resolved !== repoRoot && !resolved.startsWith(`${repoRoot}${path.sep}`))
102
+ return null;
103
+ const relative = path.relative(repoRoot, resolved);
104
+ if (!relative || relative.startsWith(".."))
105
+ return null;
106
+ return relative.split(path.sep).join("/");
107
+ }
108
+ /** Collects `originalUriBaseId` -> absolute path, used to anchor relative uris. */
109
+ function readUriBases(run) {
110
+ const bases = new Map();
111
+ const raw = run.originalUriBaseIds;
112
+ if (!raw || typeof raw !== "object")
113
+ return bases;
114
+ for (const [id, entry] of Object.entries(raw)) {
115
+ const uri = entry?.uri;
116
+ if (typeof uri !== "string")
117
+ continue;
118
+ let value = uri;
119
+ if (value.startsWith("file://")) {
120
+ try {
121
+ value = decodeURIComponent(new URL(value).pathname);
122
+ }
123
+ catch {
124
+ continue;
125
+ }
126
+ }
127
+ bases.set(id, value);
128
+ }
129
+ return bases;
130
+ }
131
+ /** Rule metadata by id, so a finding can inherit a helpUri it does not carry. */
132
+ function readRuleHelp(run) {
133
+ const help = new Map();
134
+ const driver = run.tool?.driver;
135
+ const rules = driver?.rules;
136
+ if (!Array.isArray(rules))
137
+ return help;
138
+ for (const rule of rules) {
139
+ const id = rule.id;
140
+ const uri = rule.helpUri;
141
+ if (typeof id === "string" && typeof uri === "string")
142
+ help.set(id, uri);
143
+ }
144
+ return help;
145
+ }
146
+ /** Stable order so repeated imports of an unchanged log agree exactly. */
147
+ function compareFindings(left, right) {
148
+ if (left.filePath !== right.filePath)
149
+ return compareBytes(left.filePath, right.filePath);
150
+ if (left.startLine !== right.startLine)
151
+ return left.startLine - right.startLine;
152
+ return compareBytes(left.ruleId, right.ruleId);
153
+ }
154
+ /**
155
+ * Builds one finding from a SARIF result, or null when the result cannot be
156
+ * placed inside the repository. The caller counts those rather than guessing
157
+ * at a path.
158
+ */
159
+ function readResult(result, context) {
160
+ const location = Array.isArray(result.locations) ? result.locations[0] : undefined;
161
+ const physical = location
162
+ ?.physicalLocation;
163
+ const artifact = physical?.artifactLocation;
164
+ const uri = typeof artifact?.uri === "string" ? artifact.uri : "";
165
+ const baseId = typeof artifact?.uriBaseId === "string" ? artifact.uriBaseId : undefined;
166
+ // Spec says uriBaseId is a key into originalUriBaseIds. Salesforce Code
167
+ // Analyzer instead puts the absolute workspace path there directly and
168
+ // omits originalUriBaseIds entirely, so accept either form.
169
+ const base = baseId ? (context.uriBases.get(baseId) ?? baseId) : undefined;
170
+ const filePath = resolveArtifactPath(context.repoRoot, uri, base);
171
+ if (!filePath)
172
+ return null;
173
+ const region = (physical?.region ?? {});
174
+ const startLine = Number.isInteger(region.startLine) ? region.startLine : 1;
175
+ const startColumn = Number.isInteger(region.startColumn) ? region.startColumn : 1;
176
+ const ruleIdRaw = result.ruleId;
177
+ const ruleId = typeof ruleIdRaw === "string" && ruleIdRaw ? ruleIdRaw : "unknown-rule";
178
+ const messageText = result.message?.text;
179
+ const helpUri = context.ruleHelp.get(ruleId);
180
+ return {
181
+ ruleId,
182
+ tool: context.tool,
183
+ level: normalizeLevel(result.level),
184
+ message: typeof messageText === "string" ? messageText : "",
185
+ filePath,
186
+ startLine,
187
+ startColumn,
188
+ // SARIF omits end positions for single-point results; collapse to start.
189
+ endLine: Number.isInteger(region.endLine) ? region.endLine : startLine,
190
+ endColumn: Number.isInteger(region.endColumn) ? region.endColumn : startColumn,
191
+ ...(helpUri ? { helpUri } : {}),
192
+ };
193
+ }
194
+ /**
195
+ * Validates the input against the size ceilings and parses it, returning the
196
+ * SARIF runs. Every refusal names the flag and a value that would admit the
197
+ * input, because a ceiling the operator cannot move is just a failure.
198
+ */
199
+ function readRuns(displayPath, inputPath, size, maxBytes) {
200
+ if (size > maxBytes)
201
+ throw new Error(`knodin index --sarif: ${displayPath} is ${formatBytes(size)}, over the ` +
202
+ `${formatBytes(maxBytes)} ceiling. Raise it with ` +
203
+ `--sarif-max-bytes ${size}, or narrow the analysis that produced the log.`);
204
+ // A ceiling knodin cannot raise. JSON.parse needs the whole document as one
205
+ // string, and V8 refuses strings beyond roughly 512 MiB, so a larger log
206
+ // fails inside readFileSync no matter what --sarif-max-bytes says. Saying so
207
+ // is more useful than letting "Invalid string length" surface.
208
+ if (size > MAX_PARSEABLE_BYTES)
209
+ throw new Error(`knodin index --sarif: ${displayPath} is ${formatBytes(size)}, beyond the ` +
210
+ `${formatBytes(MAX_PARSEABLE_BYTES)} limit this runtime can parse as one document. ` +
211
+ `Raising --sarif-max-bytes will not help. Split the analysis by rule selector or ` +
212
+ `workspace and import each log, which is additive across tools.`);
213
+ let document;
214
+ try {
215
+ document = JSON.parse(fs.readFileSync(inputPath, "utf8"));
216
+ }
217
+ catch (error) {
218
+ const detail = error instanceof Error ? error.message : String(error);
219
+ // A resource failure is not a malformed document, and reporting it as one
220
+ // sends the reader to inspect a file that is probably fine.
221
+ if (error instanceof RangeError || /heap|memory|string length/i.test(detail))
222
+ throw new Error(`knodin index --sarif: ran out of memory reading ${displayPath} ` +
223
+ `(${formatBytes(size)}). The document is held whole while parsing, so this is a ` +
224
+ `runtime limit rather than a knodin one. Retry with a larger heap, for example ` +
225
+ `NODE_OPTIONS=--max-old-space-size=8192, or split the analysis into several logs.`);
226
+ throw new Error(`knodin index --sarif: ${displayPath} is not valid JSON (${detail})`);
227
+ }
228
+ const runs = document?.runs;
229
+ if (!Array.isArray(runs))
230
+ throw new Error("knodin index --sarif: input has no runs array; not a SARIF log");
231
+ return runs;
232
+ }
233
+ /** Applies the defaults for any ceiling the caller left unset. */
234
+ function resolveLimits(options) {
235
+ return {
236
+ maxBytes: options.maxBytes ?? SARIF_DEFAULT_LIMITS.maxBytes,
237
+ maxFindings: options.maxFindings ?? SARIF_DEFAULT_LIMITS.maxFindings,
238
+ timeoutMs: options.timeoutMs ?? SARIF_DEFAULT_LIMITS.timeoutMs,
239
+ };
240
+ }
241
+ /**
242
+ * Reads one SARIF run into findings, or null when the value is not a run that
243
+ * carries results at all.
244
+ *
245
+ * `skipped` counts results whose location could not be placed inside the
246
+ * repository; the caller reports that total rather than discarding it.
247
+ *
248
+ * `budget.carried` is how many findings earlier runs already contributed, so
249
+ * the ceiling can be enforced against the running total from inside the loop.
250
+ * It is checked on every finding rather than once per run because the ceiling
251
+ * exists to stop the work, not to report on it afterwards: a SARIF log commonly
252
+ * holds exactly one run, so a per-run check would only fire once the entire log
253
+ * was already accumulated — the case the ceiling is there to prevent.
254
+ */
255
+ function readRun(rawRun, repoRoot, budget) {
256
+ if (!rawRun || typeof rawRun !== "object")
257
+ return null;
258
+ const run = rawRun;
259
+ const results = run.results;
260
+ if (!Array.isArray(results))
261
+ return null;
262
+ const driverName = run.tool?.driver?.name;
263
+ const tool = typeof driverName === "string" && driverName ? driverName : "unknown";
264
+ const context = {
265
+ repoRoot,
266
+ tool,
267
+ uriBases: readUriBases(run),
268
+ ruleHelp: readRuleHelp(run),
269
+ };
270
+ const findings = [];
271
+ let skipped = 0;
272
+ for (const rawResult of results) {
273
+ if (!rawResult || typeof rawResult !== "object")
274
+ continue;
275
+ const finding = readResult(rawResult, context);
276
+ // A finding we cannot place inside the repository is counted, not guessed at.
277
+ if (!finding) {
278
+ skipped++;
279
+ continue;
280
+ }
281
+ findings.push(finding);
282
+ if (budget.carried + findings.length > budget.maxFindings)
283
+ throw new Error(`knodin index --sarif: more than ${budget.maxFindings.toLocaleString()} findings. ` +
284
+ `Raise it with --sarif-max-findings, or narrow the analysis that produced the log.`);
285
+ }
286
+ return { tool, findings, skipped };
287
+ }
288
+ /**
289
+ * Reads and validates a SARIF log, returning findings resolved against the
290
+ * repository. Throws rather than returning partial facts.
291
+ */
292
+ export function readSarifLog(repoPath, options) {
293
+ const started = Date.now();
294
+ const limits = resolveLimits(options);
295
+ const repoRoot = fs.realpathSync(path.resolve(repoPath));
296
+ const inputPath = path.resolve(options.path);
297
+ const stat = fs.statSync(inputPath, { throwIfNoEntry: false });
298
+ if (!stat?.isFile())
299
+ throw new Error(`knodin index --sarif: ${options.path} is not a file`);
300
+ const runs = readRuns(options.path, inputPath, stat.size, limits.maxBytes);
301
+ const findings = [];
302
+ const tools = new Set();
303
+ const rules = new Set();
304
+ const files = new Set();
305
+ let skippedUnresolved = 0;
306
+ for (const rawRun of runs) {
307
+ if (Date.now() - started > limits.timeoutMs)
308
+ throw new Error(`knodin index --sarif: exceeded the ${limits.timeoutMs}ms read budget. ` +
309
+ `Raise it with --sarif-timeout-ms.`);
310
+ // readRun enforces the findings ceiling against this running total as it
311
+ // reads, so it throws on the offending finding rather than after the run.
312
+ const run = readRun(rawRun, repoRoot, {
313
+ carried: findings.length,
314
+ maxFindings: limits.maxFindings,
315
+ });
316
+ if (!run)
317
+ continue;
318
+ skippedUnresolved += run.skipped;
319
+ // A tool is only recorded once it has placed a finding, so a run that
320
+ // resolved nothing does not advertise coverage it did not deliver.
321
+ if (run.findings.length === 0)
322
+ continue;
323
+ tools.add(run.tool);
324
+ for (const finding of run.findings) {
325
+ findings.push(finding);
326
+ rules.add(finding.ruleId);
327
+ files.add(finding.filePath);
328
+ }
329
+ }
330
+ findings.sort(compareFindings);
331
+ return {
332
+ inputPath,
333
+ bytes: stat.size,
334
+ tools: [...tools].sort(compareBytes),
335
+ rules: rules.size,
336
+ files: files.size,
337
+ findings,
338
+ skippedUnresolved,
339
+ elapsedMs: Date.now() - started,
340
+ };
341
+ }
@@ -1,5 +1,6 @@
1
1
  import fs from "node:fs";
2
2
  import path from "node:path";
3
+ import { compareBytes } from "../compare.js";
3
4
  export const SCIP_DEFAULT_LIMITS = {
4
5
  maxBytes: 64 * 1024 * 1024,
5
6
  maxFiles: 10_000,
@@ -16,7 +17,7 @@ function fail(message) {
16
17
  }
17
18
  function checkDeadline() {
18
19
  if (performance.now() > activeDeadline)
19
- fail(`import exceeded ${activeTimeoutMs} ms limit`);
20
+ fail(`import exceeded the ${activeTimeoutMs} ms limit. Raise it with --scip-timeout-ms.`);
20
21
  }
21
22
  function readVarint(bytes, offset) {
22
23
  let value = 0;
@@ -245,7 +246,8 @@ function parseMetadata(index) {
245
246
  return [];
246
247
  const name = strings(toolInfo, 1)[0];
247
248
  const version = strings(toolInfo, 2)[0];
248
- return name ? [`${name}${version ? `@${version}` : ""}`] : [];
249
+ const suffix = version ? `@${version}` : "";
250
+ return name ? [`${name}${suffix}`] : [];
249
251
  }
250
252
  export function readScipIndex(repoPath, options) {
251
253
  const started = performance.now();
@@ -272,15 +274,21 @@ export function readScipIndex(repoPath, options) {
272
274
  }
273
275
  if (!inputStat.isFile() || inputStat.isSymbolicLink())
274
276
  fail("input must be a regular repository file");
277
+ // A ceiling nobody can raise is just a failure, so name the flag and a value
278
+ // that would admit this file.
275
279
  if (inputStat.size > limits.maxBytes)
276
- fail(`input exceeds ${limits.maxBytes} byte limit`);
280
+ fail(`input is ${inputStat.size.toLocaleString()} bytes, over the ` +
281
+ `${limits.maxBytes.toLocaleString()} byte limit. Raise it with ` +
282
+ `--scip-max-bytes ${inputStat.size}, or narrow the indexer that produced this file.`);
277
283
  const bytes = fs.readFileSync(inputPath);
278
284
  const index = new Uint8Array(bytes.buffer, bytes.byteOffset, bytes.byteLength);
279
285
  const documents = [];
280
286
  let parsedFacts = 0;
281
287
  const documentCount = countLengthDelimitedFields(index, 2, limits.maxFiles);
282
288
  if (documentCount > limits.maxFiles)
283
- fail(`index exceeds ${limits.maxFiles} file limit`);
289
+ fail(`index covers ${documentCount.toLocaleString()} files, over the ` +
290
+ `${limits.maxFiles.toLocaleString()} file limit. Raise it with ` +
291
+ `--scip-max-files ${documentCount}, or narrow the indexer that produced this file.`);
284
292
  for (const message of messages(index, 2)) {
285
293
  const filePath = safeRelativePath(resolvedRepo, strings(message, 1)[0] ?? "");
286
294
  const sourceStat = fs.statSync(path.join(resolvedRepo, filePath));
@@ -289,18 +297,21 @@ export function readScipIndex(repoPath, options) {
289
297
  const occurrenceCount = countLengthDelimitedFields(message, 2, limits.maxFacts - parsedFacts);
290
298
  parsedFacts += occurrenceCount;
291
299
  if (parsedFacts > limits.maxFacts)
292
- fail(`index exceeds ${limits.maxFacts} fact limit`);
300
+ fail(`index exceeds the ${limits.maxFacts.toLocaleString()} fact limit. ` +
301
+ `Raise it with --scip-max-facts, or narrow the indexer that produced this file.`);
293
302
  const symbolCount = countLengthDelimitedFields(message, 3, limits.maxFacts - parsedFacts);
294
303
  parsedFacts += symbolCount;
295
304
  if (parsedFacts > limits.maxFacts)
296
- fail(`index exceeds ${limits.maxFacts} fact limit`);
305
+ fail(`index exceeds the ${limits.maxFacts.toLocaleString()} fact limit. ` +
306
+ `Raise it with --scip-max-facts, or narrow the indexer that produced this file.`);
297
307
  const occurrenceMessages = messages(message, 2);
298
308
  const symbolMessages = messages(message, 3);
299
309
  for (const symbolMessage of symbolMessages) {
300
310
  const remainingRelationships = Math.floor((limits.maxFacts - parsedFacts) / 4);
301
311
  parsedFacts += countLengthDelimitedFields(symbolMessage, 4, remainingRelationships) * 4;
302
312
  if (parsedFacts > limits.maxFacts)
303
- fail(`index exceeds ${limits.maxFacts} fact limit`);
313
+ fail(`index exceeds the ${limits.maxFacts.toLocaleString()} fact limit. ` +
314
+ `Raise it with --scip-max-facts, or narrow the indexer that produced this file.`);
304
315
  }
305
316
  documents.push({
306
317
  filePath,
@@ -389,17 +400,21 @@ export function readScipIndex(repoPath, options) {
389
400
  fail(`index exceeds ${limits.maxFacts} fact limit`);
390
401
  if (performance.now() - started > limits.timeoutMs)
391
402
  fail(`import exceeded ${limits.timeoutMs} ms limit`);
403
+ // Byte order: imported facts are persisted and compared across runs, so their
404
+ // order has to be a function of the bytes rather than of host collation.
405
+ symbols.sort((a, b) => compareBytes(a.filePath, b.filePath) || a.startLine - b.startLine);
406
+ references.sort((a, b) => compareBytes(a.callerFile, b.callerFile) ||
407
+ a.line - b.line ||
408
+ compareBytes(a.calleeSymbol, b.calleeSymbol));
392
409
  const result = {
393
410
  inputPath: inputRelative.replaceAll("\\", "/"),
394
411
  bytes: inputStat.size,
395
412
  files: documents.length,
396
413
  facts: factCount,
397
- languages: [...new Set(documents.map((document) => document.language))].sort(),
398
- indexers: parseMetadata(index).sort(),
399
- symbols: symbols.sort((a, b) => a.filePath.localeCompare(b.filePath) || a.startLine - b.startLine),
400
- references: references.sort((a, b) => a.callerFile.localeCompare(b.callerFile) ||
401
- a.line - b.line ||
402
- a.calleeSymbol.localeCompare(b.calleeSymbol)),
414
+ languages: [...new Set(documents.map((document) => document.language))].sort(compareBytes),
415
+ indexers: parseMetadata(index).sort(compareBytes),
416
+ symbols,
417
+ references,
403
418
  elapsedMs: Math.round((performance.now() - started) * 1000) / 1000,
404
419
  };
405
420
  activeDeadline = Number.POSITIVE_INFINITY;
@@ -37,6 +37,22 @@ const SOURCE_EXTENSIONS = new Set([
37
37
  ".hcl",
38
38
  ".dockerfile",
39
39
  ".lsif",
40
+ // Documentation participates as a link-only graph: doc-to-doc and doc-to-code
41
+ // edges, no symbols. Without it, "which docs describe this subsystem" and
42
+ // "which docs point at code that no longer exists" are unanswerable.
43
+ ".md",
44
+ ".mdx",
45
+ // CI pipeline definitions participate as link-only template references.
46
+ ".yml",
47
+ ".yaml",
48
+ // The .NET build graph: project-to-project references across services.
49
+ ".csproj",
50
+ ".sln",
51
+ ".props",
52
+ ".targets",
53
+ // Razor views/components: the C# members in their @code/@functions blocks.
54
+ ".cshtml",
55
+ ".razor",
40
56
  ]);
41
57
  function isSalesforceBundleMarkup(normalized) {
42
58
  return (/^force-app\/main\/default\/lwc\/([A-Za-z][A-Za-z0-9_]*)\/\1\.(?:html|css)$/.test(normalized) ||
@@ -0,0 +1,175 @@
1
+ import fs from "node:fs";
2
+ import os from "node:os";
3
+ import path from "node:path";
4
+ /**
5
+ * Resolves where a repository's knodin state (graph database, wiki, hook
6
+ * scaffolding) lives.
7
+ *
8
+ * A normal checkout keeps its state in-tree at `<repo>/.knodin`, unchanged from
9
+ * the original hardcoded behaviour. A *mirror* — a read-only shallow clone of a
10
+ * repository the user never checked out themselves — keeps its state outside the
11
+ * clone entirely, so that a refetch can discard and replace the source tree
12
+ * without destroying a graph that took minutes to build.
13
+ *
14
+ * Mirror layout, rooted at {@link mirrorRoot}:
15
+ *
16
+ * ```
17
+ * <root>/registry.json the mirror index
18
+ * <root>/.staging/<id> in-progress clones, renamed into place on success
19
+ * <root>/<id>/source the shallow clone
20
+ * <root>/<id>/state db.sqlite, wiki/, and friends
21
+ * <root>/<id>/README marker: local edits here are discarded on refresh
22
+ * ```
23
+ *
24
+ * Every consumer must route through {@link resolveStateDir} /
25
+ * {@link resolveDbPath} rather than rebuilding `path.join(repo, ".knodin")`
26
+ * inline, or mirrors silently read from the wrong location.
27
+ */
28
+ /** Directory name holding the shallow clone inside a mirror entry. */
29
+ export const MIRROR_SOURCE_DIR = "source";
30
+ /** Directory name holding graph state inside a mirror entry. */
31
+ export const MIRROR_STATE_DIR = "state";
32
+ /** In-tree state directory name for a normal checkout. */
33
+ export const IN_TREE_STATE_DIR = ".knodin";
34
+ const REGISTRY_VERSION = 1;
35
+ const EMPTY_REGISTRY = { version: REGISTRY_VERSION, mirrors: [] };
36
+ /**
37
+ * Root for all mirror storage. Defaults to `~/.knodin/mirrors` rather than an
38
+ * XDG data directory: a mirror costs minutes of indexing to rebuild, so it needs
39
+ * to sit somewhere the user will actually notice and can audit with
40
+ * `knodin remote list`. `KNODIN_MIRROR_ROOT` overrides it.
41
+ */
42
+ export function mirrorRoot() {
43
+ const override = process.env.KNODIN_MIRROR_ROOT;
44
+ return override ? path.resolve(override) : path.join(os.homedir(), ".knodin", "mirrors");
45
+ }
46
+ /** Path to the mirror registry document. */
47
+ export function mirrorRegistryPath() {
48
+ return path.join(mirrorRoot(), "registry.json");
49
+ }
50
+ /** Root for in-progress clones, renamed into place only once complete. */
51
+ export function mirrorStagingRoot() {
52
+ return path.join(mirrorRoot(), ".staging");
53
+ }
54
+ /** Entry root for one mirror: the parent of both `source/` and `state/`. */
55
+ export function mirrorEntryPath(identity) {
56
+ return path.join(mirrorRoot(), identity);
57
+ }
58
+ /** Absolute path to a mirror's shallow clone. */
59
+ export function mirrorSourcePath(identity) {
60
+ return path.join(mirrorEntryPath(identity), MIRROR_SOURCE_DIR);
61
+ }
62
+ /** Absolute path to a mirror's graph state directory. */
63
+ export function mirrorStatePath(identity) {
64
+ return path.join(mirrorEntryPath(identity), MIRROR_STATE_DIR);
65
+ }
66
+ let cache = null;
67
+ function isMirrorRecord(value) {
68
+ if (typeof value !== "object" || value === null)
69
+ return false;
70
+ const record = value;
71
+ return (typeof record.identity === "string" &&
72
+ typeof record.url === "string" &&
73
+ typeof record.path === "string" &&
74
+ typeof record.fetchedAt === "string" &&
75
+ typeof record.sha === "string");
76
+ }
77
+ function indexByPath(registry) {
78
+ const byPath = new Map();
79
+ for (const record of registry.mirrors)
80
+ byPath.set(path.resolve(record.path), record);
81
+ return byPath;
82
+ }
83
+ /**
84
+ * Reads the mirror registry, memoized against the file's mtime and size so a
85
+ * mirror added by a concurrent `knodin remote add` is picked up without paying a
86
+ * JSON parse on every path resolution.
87
+ *
88
+ * The memo key is (mtimeMs, size), which is a heuristic rather than a guarantee:
89
+ * an out-of-process write landing in the same millisecond AND producing the same
90
+ * byte length would not invalidate it. Accepted deliberately — that requires a
91
+ * concurrent edit of identical length within one millisecond, the in-process
92
+ * writer drops the memo explicitly, and the cost of being wrong is one stale
93
+ * read, not corruption. Recorded so it reads as a decision, not an oversight.
94
+ *
95
+ * A missing or corrupt registry reads as empty. That is deliberate too: it
96
+ * degrades to "no mirrors configured", which routes every repository to its
97
+ * in-tree state — the pre-existing behaviour — rather than failing a query
98
+ * outright.
99
+ */
100
+ export function readMirrorRegistry() {
101
+ const file = mirrorRegistryPath();
102
+ let stat;
103
+ try {
104
+ stat = fs.statSync(file);
105
+ }
106
+ catch {
107
+ cache = null;
108
+ return EMPTY_REGISTRY;
109
+ }
110
+ if (cache?.mtimeMs === stat.mtimeMs && cache?.size === stat.size)
111
+ return cache.registry;
112
+ try {
113
+ const parsed = JSON.parse(fs.readFileSync(file, "utf8"));
114
+ const mirrors = typeof parsed === "object" &&
115
+ parsed !== null &&
116
+ Array.isArray(parsed.mirrors)
117
+ ? parsed.mirrors.filter(isMirrorRecord)
118
+ : [];
119
+ const registry = { version: REGISTRY_VERSION, mirrors };
120
+ cache = { mtimeMs: stat.mtimeMs, size: stat.size, registry, byPath: indexByPath(registry) };
121
+ return registry;
122
+ }
123
+ catch {
124
+ cache = null;
125
+ return EMPTY_REGISTRY;
126
+ }
127
+ }
128
+ /** Persists the registry and drops the memo so the next read reloads it. */
129
+ export function writeMirrorRegistry(registry) {
130
+ const file = mirrorRegistryPath();
131
+ fs.mkdirSync(path.dirname(file), { recursive: true });
132
+ const document = { version: REGISTRY_VERSION, mirrors: registry.mirrors };
133
+ fs.writeFileSync(file, `${JSON.stringify(document, null, 2)}\n`);
134
+ cache = null;
135
+ }
136
+ /** Clears the in-process registry memo. Test seam. */
137
+ export function __resetMirrorRegistryCache() {
138
+ cache = null;
139
+ }
140
+ /**
141
+ * Returns the mirror record whose clone is at `repoPath`, if any. Matching is on
142
+ * the resolved path, so a caller passing a relative or unnormalized path still
143
+ * resolves correctly.
144
+ */
145
+ export function lookupMirror(repoPath) {
146
+ readMirrorRegistry();
147
+ return cache?.byPath.get(path.resolve(repoPath));
148
+ }
149
+ /** True when `repoPath` is a registered read-only mirror clone. */
150
+ export function isMirror(repoPath) {
151
+ return lookupMirror(repoPath) !== undefined;
152
+ }
153
+ /**
154
+ * Directory holding knodin state for `repoPath`: in-tree `.knodin` for a normal
155
+ * checkout, an out-of-tree sibling of the clone for a mirror.
156
+ */
157
+ export function resolveStateDir(repoPath) {
158
+ const resolved = path.resolve(repoPath);
159
+ const mirror = lookupMirror(resolved);
160
+ if (mirror)
161
+ return mirrorStatePath(mirror.identity);
162
+ return path.join(resolved, IN_TREE_STATE_DIR);
163
+ }
164
+ /** Graph database path for `repoPath`. */
165
+ export function resolveDbPath(repoPath) {
166
+ return path.join(resolveStateDir(repoPath), "db.sqlite");
167
+ }
168
+ /**
169
+ * Whether opening an index for `repoPath` may write to the repository itself —
170
+ * false for mirrors, whose source tree is replaced wholesale on refresh and must
171
+ * stay byte-identical to the remote in the meantime.
172
+ */
173
+ export function mayWriteToRepository(repoPath) {
174
+ return !isMirror(repoPath);
175
+ }