knodin 0.7.5 → 0.8.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +18 -3
- package/benchmarks/competitors/SYNTHESIS.md +66 -0
- package/dist/bin/cli.js +371 -66
- package/dist/bin/launcher.js +16 -1
- package/dist/src/agent-integration.js +82 -16
- package/dist/src/artifact-refresh.js +2 -1
- package/dist/src/cli-args.js +19 -1
- package/dist/src/cli-model.js +28 -2
- package/dist/src/codeflow-replay.js +2 -1
- package/dist/src/compare.js +39 -0
- package/dist/src/competitive-constraints.js +2 -1
- package/dist/src/competitive-runner.js +4 -4
- package/dist/src/context-export.js +3 -2
- package/dist/src/context.js +1 -1
- package/dist/src/deterministic-random.js +34 -0
- package/dist/src/diagnostics-write-helper.js +473 -0
- package/dist/src/diagnostics.js +1160 -133
- package/dist/src/doctor.js +3 -1
- package/dist/src/engine/ann-hnsw.js +2 -12
- package/dist/src/engine/file-walker.js +8 -2
- package/dist/src/engine/git-history.js +12 -12
- package/dist/src/engine/index.js +1174 -313
- package/dist/src/engine/sarif-import.js +341 -0
- package/dist/src/engine/scip-import.js +28 -13
- package/dist/src/engine/source-policy.js +16 -0
- package/dist/src/engine/state-paths.js +175 -0
- package/dist/src/execution-profile.js +15 -10
- package/dist/src/failure-diagnosis.js +7 -1
- package/dist/src/graph-layout.js +173 -0
- package/dist/src/index-activity.js +2 -1
- package/dist/src/init.js +86 -45
- package/dist/src/lifecycle-health.js +41 -9
- package/dist/src/mcp-graph-worker.js +69 -0
- package/dist/src/mcp-reliability.js +154 -0
- package/dist/src/mcp-worker-supervisor.js +350 -0
- package/dist/src/mirror.js +290 -0
- package/dist/src/node-runtime.js +157 -0
- package/dist/src/output-compression.js +2 -1
- package/dist/src/output-telemetry.js +16 -11
- package/dist/src/progressive-evidence.js +30 -26
- package/dist/src/pure-compression-cli.js +4 -3
- package/dist/src/relationship-adapters.js +15 -8
- package/dist/src/release-preflight.js +13 -10
- package/dist/src/repair-lease.js +85 -0
- package/dist/src/repository-init-process.js +13 -9
- package/dist/src/repository-management.js +34 -4
- package/dist/src/response-budget.js +8 -6
- package/dist/src/server.js +80 -35
- package/dist/src/structural-fast-path.js +16 -10
- package/dist/src/structural-snapshot.js +6 -2
- package/dist/src/system-config.js +25 -2
- package/dist/src/tools/knodin-tools.js +142 -31
- package/dist/src/update-ceremony.js +9 -5
- package/dist/src/update-trust.js +5 -4
- package/dist/src/visualization.js +372 -19
- package/dist/src/worktree-lifecycle.js +5 -2
- package/docs/BEHAVIORAL-CONTRACT.md +72 -0
- package/docs/CLI.md +20 -1
- package/docs/COMPARISON.md +403 -0
- package/docs/COMPETITIVE-LANDSCAPE-2026-08.md +267 -0
- package/docs/DIAGNOSTICS.md +46 -11
- package/docs/HANDOFF.md +180 -0
- package/docs/INSTALLATION.md +21 -2
- package/docs/MCP.md +59 -8
- package/docs/PT-ACCESS-RECOMMENDATION.md +5 -7
- package/docs/REPOSITORIES-AND-WORKTREES.md +18 -6
- package/docs/SCIP-IMPORT.md +5 -0
- package/docs/TOKEN-OPTIMIZER-SCORECARD.md +79 -0
- package/docs/releases/0.5.1.md +4 -4
- package/docs/releases/0.8.0.md +74 -0
- package/docs/releases/0.8.2.md +34 -0
- package/package.json +17 -4
- package/roadmap/competitive-roadmap.md +3801 -0
- package/schemas/release-attestation-v1.schema.json +1 -1
- package/schemas/support-bundle-v2.schema.json +212 -0
|
@@ -0,0 +1,341 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Bounded SARIF import.
|
|
3
|
+
*
|
|
4
|
+
* SARIF (OASIS "Static Analysis Results Interchange Format") is what most
|
|
5
|
+
* analyzers can emit — Salesforce Code Analyzer, ESLint, Semgrep, CodeQL, PMD,
|
|
6
|
+
* Trivy. Importing the format rather than any one tool's native output means one
|
|
7
|
+
* reader covers all of them.
|
|
8
|
+
*
|
|
9
|
+
* Findings are NOT graph facts. knodin resolves references from source; a
|
|
10
|
+
* finding is somebody else's judgement about a location. It is stored in its own
|
|
11
|
+
* table, tagged with the tool that produced it, and never merged into symbols or
|
|
12
|
+
* references. What the graph adds is the join: a finding on a hub matters more
|
|
13
|
+
* than the same finding on a leaf, and only the graph knows which is which.
|
|
14
|
+
*
|
|
15
|
+
* Everything here is bounded and fail-closed, matching the SCIP importer: a file
|
|
16
|
+
* over the byte ceiling, a malformed document, or a finding count over the cap
|
|
17
|
+
* is refused rather than partially applied.
|
|
18
|
+
*/
|
|
19
|
+
import fs from "node:fs";
|
|
20
|
+
import path from "node:path";
|
|
21
|
+
import { compareBytes } from "../compare.js";
|
|
22
|
+
// Calibrated against real output rather than guessed: one Salesforce Code
|
|
23
|
+
// Analyzer run over a mid-sized org (pmd + regex + retire-js) produced a 97 MB
|
|
24
|
+
// SARIF log with ~75k findings. A 64 MB ceiling would have refused it.
|
|
25
|
+
export const SARIF_DEFAULT_LIMITS = {
|
|
26
|
+
maxBytes: 256 * 1024 * 1024,
|
|
27
|
+
maxFindings: 250_000,
|
|
28
|
+
timeoutMs: 60_000,
|
|
29
|
+
};
|
|
30
|
+
/**
|
|
31
|
+
* The ceiling knodin cannot raise. `JSON.parse` needs the document as a single
|
|
32
|
+
* string and V8 caps string length at 2^29 - 24 bytes on 64-bit, so a log past
|
|
33
|
+
* this size cannot be read whole regardless of the configured limit. Kept
|
|
34
|
+
* slightly under the hard maximum to leave room for decoding overhead.
|
|
35
|
+
*/
|
|
36
|
+
const MAX_PARSEABLE_BYTES = 512 * 1024 * 1024;
|
|
37
|
+
/** Byte counts appear in operator-facing errors, so render them readably. */
|
|
38
|
+
function formatBytes(value) {
|
|
39
|
+
if (value < 1024)
|
|
40
|
+
return `${value} bytes`;
|
|
41
|
+
const units = ["KiB", "MiB", "GiB"];
|
|
42
|
+
let scaled = value / 1024;
|
|
43
|
+
let unit = 0;
|
|
44
|
+
while (scaled >= 1024 && unit < units.length - 1) {
|
|
45
|
+
scaled /= 1024;
|
|
46
|
+
unit++;
|
|
47
|
+
}
|
|
48
|
+
return `${scaled.toFixed(1)} ${units[unit]}`;
|
|
49
|
+
}
|
|
50
|
+
/** SARIF level is optional; analyzers often carry severity on the rule instead. */
|
|
51
|
+
function normalizeLevel(value) {
|
|
52
|
+
if (value === "error" || value === "warning" || value === "note" || value === "none")
|
|
53
|
+
return value;
|
|
54
|
+
// SARIF's own default when level is absent and no rule default applies.
|
|
55
|
+
return "warning";
|
|
56
|
+
}
|
|
57
|
+
/**
|
|
58
|
+
* Resolves a SARIF artifact location to a repository-relative POSIX path.
|
|
59
|
+
*
|
|
60
|
+
* SARIF permits `file:///abs/path`, a bare absolute path, or a path relative to
|
|
61
|
+
* some base URI. Anything landing outside the repository is rejected: a findings
|
|
62
|
+
* file is external input, and a `../` escape must not attach a finding to a path
|
|
63
|
+
* the caller never indexed.
|
|
64
|
+
*/
|
|
65
|
+
function resolveArtifactPath(repoRoot, uri, uriBase) {
|
|
66
|
+
if (!uri)
|
|
67
|
+
return null;
|
|
68
|
+
let raw = uri;
|
|
69
|
+
if (raw.startsWith("file://")) {
|
|
70
|
+
try {
|
|
71
|
+
raw = decodeURIComponent(new URL(raw).pathname);
|
|
72
|
+
}
|
|
73
|
+
catch {
|
|
74
|
+
return null;
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
else {
|
|
78
|
+
try {
|
|
79
|
+
raw = decodeURIComponent(raw);
|
|
80
|
+
}
|
|
81
|
+
catch {
|
|
82
|
+
// Not percent-encoded; use as-is.
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
// A relative uri may be anchored by an originalUriBaseId value.
|
|
86
|
+
const anchor = uriBase && path.isAbsolute(uriBase) ? uriBase : repoRoot;
|
|
87
|
+
const candidate = path.isAbsolute(raw) ? raw : path.resolve(anchor, raw);
|
|
88
|
+
// Compare real paths. The analyzer may have recorded a route through a
|
|
89
|
+
// symlink — /var vs /private/var on macOS, or a symlinked checkout — and a
|
|
90
|
+
// textual prefix check would reject those valid locations. Resolving also
|
|
91
|
+
// means a symlink pointing outside the repository is still caught, because
|
|
92
|
+
// the real path is what gets compared.
|
|
93
|
+
let resolved = path.resolve(candidate);
|
|
94
|
+
try {
|
|
95
|
+
resolved = fs.realpathSync(resolved);
|
|
96
|
+
}
|
|
97
|
+
catch {
|
|
98
|
+
// Not on disk: the analyzer may report a file since deleted. Keep the
|
|
99
|
+
// lexically resolved path so containment is still enforced below.
|
|
100
|
+
}
|
|
101
|
+
if (resolved !== repoRoot && !resolved.startsWith(`${repoRoot}${path.sep}`))
|
|
102
|
+
return null;
|
|
103
|
+
const relative = path.relative(repoRoot, resolved);
|
|
104
|
+
if (!relative || relative.startsWith(".."))
|
|
105
|
+
return null;
|
|
106
|
+
return relative.split(path.sep).join("/");
|
|
107
|
+
}
|
|
108
|
+
/** Collects `originalUriBaseId` -> absolute path, used to anchor relative uris. */
|
|
109
|
+
function readUriBases(run) {
|
|
110
|
+
const bases = new Map();
|
|
111
|
+
const raw = run.originalUriBaseIds;
|
|
112
|
+
if (!raw || typeof raw !== "object")
|
|
113
|
+
return bases;
|
|
114
|
+
for (const [id, entry] of Object.entries(raw)) {
|
|
115
|
+
const uri = entry?.uri;
|
|
116
|
+
if (typeof uri !== "string")
|
|
117
|
+
continue;
|
|
118
|
+
let value = uri;
|
|
119
|
+
if (value.startsWith("file://")) {
|
|
120
|
+
try {
|
|
121
|
+
value = decodeURIComponent(new URL(value).pathname);
|
|
122
|
+
}
|
|
123
|
+
catch {
|
|
124
|
+
continue;
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
bases.set(id, value);
|
|
128
|
+
}
|
|
129
|
+
return bases;
|
|
130
|
+
}
|
|
131
|
+
/** Rule metadata by id, so a finding can inherit a helpUri it does not carry. */
|
|
132
|
+
function readRuleHelp(run) {
|
|
133
|
+
const help = new Map();
|
|
134
|
+
const driver = run.tool?.driver;
|
|
135
|
+
const rules = driver?.rules;
|
|
136
|
+
if (!Array.isArray(rules))
|
|
137
|
+
return help;
|
|
138
|
+
for (const rule of rules) {
|
|
139
|
+
const id = rule.id;
|
|
140
|
+
const uri = rule.helpUri;
|
|
141
|
+
if (typeof id === "string" && typeof uri === "string")
|
|
142
|
+
help.set(id, uri);
|
|
143
|
+
}
|
|
144
|
+
return help;
|
|
145
|
+
}
|
|
146
|
+
/** Stable order so repeated imports of an unchanged log agree exactly. */
|
|
147
|
+
function compareFindings(left, right) {
|
|
148
|
+
if (left.filePath !== right.filePath)
|
|
149
|
+
return compareBytes(left.filePath, right.filePath);
|
|
150
|
+
if (left.startLine !== right.startLine)
|
|
151
|
+
return left.startLine - right.startLine;
|
|
152
|
+
return compareBytes(left.ruleId, right.ruleId);
|
|
153
|
+
}
|
|
154
|
+
/**
|
|
155
|
+
* Builds one finding from a SARIF result, or null when the result cannot be
|
|
156
|
+
* placed inside the repository. The caller counts those rather than guessing
|
|
157
|
+
* at a path.
|
|
158
|
+
*/
|
|
159
|
+
function readResult(result, context) {
|
|
160
|
+
const location = Array.isArray(result.locations) ? result.locations[0] : undefined;
|
|
161
|
+
const physical = location
|
|
162
|
+
?.physicalLocation;
|
|
163
|
+
const artifact = physical?.artifactLocation;
|
|
164
|
+
const uri = typeof artifact?.uri === "string" ? artifact.uri : "";
|
|
165
|
+
const baseId = typeof artifact?.uriBaseId === "string" ? artifact.uriBaseId : undefined;
|
|
166
|
+
// Spec says uriBaseId is a key into originalUriBaseIds. Salesforce Code
|
|
167
|
+
// Analyzer instead puts the absolute workspace path there directly and
|
|
168
|
+
// omits originalUriBaseIds entirely, so accept either form.
|
|
169
|
+
const base = baseId ? (context.uriBases.get(baseId) ?? baseId) : undefined;
|
|
170
|
+
const filePath = resolveArtifactPath(context.repoRoot, uri, base);
|
|
171
|
+
if (!filePath)
|
|
172
|
+
return null;
|
|
173
|
+
const region = (physical?.region ?? {});
|
|
174
|
+
const startLine = Number.isInteger(region.startLine) ? region.startLine : 1;
|
|
175
|
+
const startColumn = Number.isInteger(region.startColumn) ? region.startColumn : 1;
|
|
176
|
+
const ruleIdRaw = result.ruleId;
|
|
177
|
+
const ruleId = typeof ruleIdRaw === "string" && ruleIdRaw ? ruleIdRaw : "unknown-rule";
|
|
178
|
+
const messageText = result.message?.text;
|
|
179
|
+
const helpUri = context.ruleHelp.get(ruleId);
|
|
180
|
+
return {
|
|
181
|
+
ruleId,
|
|
182
|
+
tool: context.tool,
|
|
183
|
+
level: normalizeLevel(result.level),
|
|
184
|
+
message: typeof messageText === "string" ? messageText : "",
|
|
185
|
+
filePath,
|
|
186
|
+
startLine,
|
|
187
|
+
startColumn,
|
|
188
|
+
// SARIF omits end positions for single-point results; collapse to start.
|
|
189
|
+
endLine: Number.isInteger(region.endLine) ? region.endLine : startLine,
|
|
190
|
+
endColumn: Number.isInteger(region.endColumn) ? region.endColumn : startColumn,
|
|
191
|
+
...(helpUri ? { helpUri } : {}),
|
|
192
|
+
};
|
|
193
|
+
}
|
|
194
|
+
/**
|
|
195
|
+
* Validates the input against the size ceilings and parses it, returning the
|
|
196
|
+
* SARIF runs. Every refusal names the flag and a value that would admit the
|
|
197
|
+
* input, because a ceiling the operator cannot move is just a failure.
|
|
198
|
+
*/
|
|
199
|
+
function readRuns(displayPath, inputPath, size, maxBytes) {
|
|
200
|
+
if (size > maxBytes)
|
|
201
|
+
throw new Error(`knodin index --sarif: ${displayPath} is ${formatBytes(size)}, over the ` +
|
|
202
|
+
`${formatBytes(maxBytes)} ceiling. Raise it with ` +
|
|
203
|
+
`--sarif-max-bytes ${size}, or narrow the analysis that produced the log.`);
|
|
204
|
+
// A ceiling knodin cannot raise. JSON.parse needs the whole document as one
|
|
205
|
+
// string, and V8 refuses strings beyond roughly 512 MiB, so a larger log
|
|
206
|
+
// fails inside readFileSync no matter what --sarif-max-bytes says. Saying so
|
|
207
|
+
// is more useful than letting "Invalid string length" surface.
|
|
208
|
+
if (size > MAX_PARSEABLE_BYTES)
|
|
209
|
+
throw new Error(`knodin index --sarif: ${displayPath} is ${formatBytes(size)}, beyond the ` +
|
|
210
|
+
`${formatBytes(MAX_PARSEABLE_BYTES)} limit this runtime can parse as one document. ` +
|
|
211
|
+
`Raising --sarif-max-bytes will not help. Split the analysis by rule selector or ` +
|
|
212
|
+
`workspace and import each log, which is additive across tools.`);
|
|
213
|
+
let document;
|
|
214
|
+
try {
|
|
215
|
+
document = JSON.parse(fs.readFileSync(inputPath, "utf8"));
|
|
216
|
+
}
|
|
217
|
+
catch (error) {
|
|
218
|
+
const detail = error instanceof Error ? error.message : String(error);
|
|
219
|
+
// A resource failure is not a malformed document, and reporting it as one
|
|
220
|
+
// sends the reader to inspect a file that is probably fine.
|
|
221
|
+
if (error instanceof RangeError || /heap|memory|string length/i.test(detail))
|
|
222
|
+
throw new Error(`knodin index --sarif: ran out of memory reading ${displayPath} ` +
|
|
223
|
+
`(${formatBytes(size)}). The document is held whole while parsing, so this is a ` +
|
|
224
|
+
`runtime limit rather than a knodin one. Retry with a larger heap, for example ` +
|
|
225
|
+
`NODE_OPTIONS=--max-old-space-size=8192, or split the analysis into several logs.`);
|
|
226
|
+
throw new Error(`knodin index --sarif: ${displayPath} is not valid JSON (${detail})`);
|
|
227
|
+
}
|
|
228
|
+
const runs = document?.runs;
|
|
229
|
+
if (!Array.isArray(runs))
|
|
230
|
+
throw new Error("knodin index --sarif: input has no runs array; not a SARIF log");
|
|
231
|
+
return runs;
|
|
232
|
+
}
|
|
233
|
+
/** Applies the defaults for any ceiling the caller left unset. */
|
|
234
|
+
function resolveLimits(options) {
|
|
235
|
+
return {
|
|
236
|
+
maxBytes: options.maxBytes ?? SARIF_DEFAULT_LIMITS.maxBytes,
|
|
237
|
+
maxFindings: options.maxFindings ?? SARIF_DEFAULT_LIMITS.maxFindings,
|
|
238
|
+
timeoutMs: options.timeoutMs ?? SARIF_DEFAULT_LIMITS.timeoutMs,
|
|
239
|
+
};
|
|
240
|
+
}
|
|
241
|
+
/**
|
|
242
|
+
* Reads one SARIF run into findings, or null when the value is not a run that
|
|
243
|
+
* carries results at all.
|
|
244
|
+
*
|
|
245
|
+
* `skipped` counts results whose location could not be placed inside the
|
|
246
|
+
* repository; the caller reports that total rather than discarding it.
|
|
247
|
+
*
|
|
248
|
+
* `budget.carried` is how many findings earlier runs already contributed, so
|
|
249
|
+
* the ceiling can be enforced against the running total from inside the loop.
|
|
250
|
+
* It is checked on every finding rather than once per run because the ceiling
|
|
251
|
+
* exists to stop the work, not to report on it afterwards: a SARIF log commonly
|
|
252
|
+
* holds exactly one run, so a per-run check would only fire once the entire log
|
|
253
|
+
* was already accumulated — the case the ceiling is there to prevent.
|
|
254
|
+
*/
|
|
255
|
+
function readRun(rawRun, repoRoot, budget) {
|
|
256
|
+
if (!rawRun || typeof rawRun !== "object")
|
|
257
|
+
return null;
|
|
258
|
+
const run = rawRun;
|
|
259
|
+
const results = run.results;
|
|
260
|
+
if (!Array.isArray(results))
|
|
261
|
+
return null;
|
|
262
|
+
const driverName = run.tool?.driver?.name;
|
|
263
|
+
const tool = typeof driverName === "string" && driverName ? driverName : "unknown";
|
|
264
|
+
const context = {
|
|
265
|
+
repoRoot,
|
|
266
|
+
tool,
|
|
267
|
+
uriBases: readUriBases(run),
|
|
268
|
+
ruleHelp: readRuleHelp(run),
|
|
269
|
+
};
|
|
270
|
+
const findings = [];
|
|
271
|
+
let skipped = 0;
|
|
272
|
+
for (const rawResult of results) {
|
|
273
|
+
if (!rawResult || typeof rawResult !== "object")
|
|
274
|
+
continue;
|
|
275
|
+
const finding = readResult(rawResult, context);
|
|
276
|
+
// A finding we cannot place inside the repository is counted, not guessed at.
|
|
277
|
+
if (!finding) {
|
|
278
|
+
skipped++;
|
|
279
|
+
continue;
|
|
280
|
+
}
|
|
281
|
+
findings.push(finding);
|
|
282
|
+
if (budget.carried + findings.length > budget.maxFindings)
|
|
283
|
+
throw new Error(`knodin index --sarif: more than ${budget.maxFindings.toLocaleString()} findings. ` +
|
|
284
|
+
`Raise it with --sarif-max-findings, or narrow the analysis that produced the log.`);
|
|
285
|
+
}
|
|
286
|
+
return { tool, findings, skipped };
|
|
287
|
+
}
|
|
288
|
+
/**
|
|
289
|
+
* Reads and validates a SARIF log, returning findings resolved against the
|
|
290
|
+
* repository. Throws rather than returning partial facts.
|
|
291
|
+
*/
|
|
292
|
+
export function readSarifLog(repoPath, options) {
|
|
293
|
+
const started = Date.now();
|
|
294
|
+
const limits = resolveLimits(options);
|
|
295
|
+
const repoRoot = fs.realpathSync(path.resolve(repoPath));
|
|
296
|
+
const inputPath = path.resolve(options.path);
|
|
297
|
+
const stat = fs.statSync(inputPath, { throwIfNoEntry: false });
|
|
298
|
+
if (!stat?.isFile())
|
|
299
|
+
throw new Error(`knodin index --sarif: ${options.path} is not a file`);
|
|
300
|
+
const runs = readRuns(options.path, inputPath, stat.size, limits.maxBytes);
|
|
301
|
+
const findings = [];
|
|
302
|
+
const tools = new Set();
|
|
303
|
+
const rules = new Set();
|
|
304
|
+
const files = new Set();
|
|
305
|
+
let skippedUnresolved = 0;
|
|
306
|
+
for (const rawRun of runs) {
|
|
307
|
+
if (Date.now() - started > limits.timeoutMs)
|
|
308
|
+
throw new Error(`knodin index --sarif: exceeded the ${limits.timeoutMs}ms read budget. ` +
|
|
309
|
+
`Raise it with --sarif-timeout-ms.`);
|
|
310
|
+
// readRun enforces the findings ceiling against this running total as it
|
|
311
|
+
// reads, so it throws on the offending finding rather than after the run.
|
|
312
|
+
const run = readRun(rawRun, repoRoot, {
|
|
313
|
+
carried: findings.length,
|
|
314
|
+
maxFindings: limits.maxFindings,
|
|
315
|
+
});
|
|
316
|
+
if (!run)
|
|
317
|
+
continue;
|
|
318
|
+
skippedUnresolved += run.skipped;
|
|
319
|
+
// A tool is only recorded once it has placed a finding, so a run that
|
|
320
|
+
// resolved nothing does not advertise coverage it did not deliver.
|
|
321
|
+
if (run.findings.length === 0)
|
|
322
|
+
continue;
|
|
323
|
+
tools.add(run.tool);
|
|
324
|
+
for (const finding of run.findings) {
|
|
325
|
+
findings.push(finding);
|
|
326
|
+
rules.add(finding.ruleId);
|
|
327
|
+
files.add(finding.filePath);
|
|
328
|
+
}
|
|
329
|
+
}
|
|
330
|
+
findings.sort(compareFindings);
|
|
331
|
+
return {
|
|
332
|
+
inputPath,
|
|
333
|
+
bytes: stat.size,
|
|
334
|
+
tools: [...tools].sort(compareBytes),
|
|
335
|
+
rules: rules.size,
|
|
336
|
+
files: files.size,
|
|
337
|
+
findings,
|
|
338
|
+
skippedUnresolved,
|
|
339
|
+
elapsedMs: Date.now() - started,
|
|
340
|
+
};
|
|
341
|
+
}
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import fs from "node:fs";
|
|
2
2
|
import path from "node:path";
|
|
3
|
+
import { compareBytes } from "../compare.js";
|
|
3
4
|
export const SCIP_DEFAULT_LIMITS = {
|
|
4
5
|
maxBytes: 64 * 1024 * 1024,
|
|
5
6
|
maxFiles: 10_000,
|
|
@@ -16,7 +17,7 @@ function fail(message) {
|
|
|
16
17
|
}
|
|
17
18
|
function checkDeadline() {
|
|
18
19
|
if (performance.now() > activeDeadline)
|
|
19
|
-
fail(`import exceeded ${activeTimeoutMs} ms limit
|
|
20
|
+
fail(`import exceeded the ${activeTimeoutMs} ms limit. Raise it with --scip-timeout-ms.`);
|
|
20
21
|
}
|
|
21
22
|
function readVarint(bytes, offset) {
|
|
22
23
|
let value = 0;
|
|
@@ -245,7 +246,8 @@ function parseMetadata(index) {
|
|
|
245
246
|
return [];
|
|
246
247
|
const name = strings(toolInfo, 1)[0];
|
|
247
248
|
const version = strings(toolInfo, 2)[0];
|
|
248
|
-
|
|
249
|
+
const suffix = version ? `@${version}` : "";
|
|
250
|
+
return name ? [`${name}${suffix}`] : [];
|
|
249
251
|
}
|
|
250
252
|
export function readScipIndex(repoPath, options) {
|
|
251
253
|
const started = performance.now();
|
|
@@ -272,15 +274,21 @@ export function readScipIndex(repoPath, options) {
|
|
|
272
274
|
}
|
|
273
275
|
if (!inputStat.isFile() || inputStat.isSymbolicLink())
|
|
274
276
|
fail("input must be a regular repository file");
|
|
277
|
+
// A ceiling nobody can raise is just a failure, so name the flag and a value
|
|
278
|
+
// that would admit this file.
|
|
275
279
|
if (inputStat.size > limits.maxBytes)
|
|
276
|
-
fail(`input
|
|
280
|
+
fail(`input is ${inputStat.size.toLocaleString()} bytes, over the ` +
|
|
281
|
+
`${limits.maxBytes.toLocaleString()} byte limit. Raise it with ` +
|
|
282
|
+
`--scip-max-bytes ${inputStat.size}, or narrow the indexer that produced this file.`);
|
|
277
283
|
const bytes = fs.readFileSync(inputPath);
|
|
278
284
|
const index = new Uint8Array(bytes.buffer, bytes.byteOffset, bytes.byteLength);
|
|
279
285
|
const documents = [];
|
|
280
286
|
let parsedFacts = 0;
|
|
281
287
|
const documentCount = countLengthDelimitedFields(index, 2, limits.maxFiles);
|
|
282
288
|
if (documentCount > limits.maxFiles)
|
|
283
|
-
fail(`index
|
|
289
|
+
fail(`index covers ${documentCount.toLocaleString()} files, over the ` +
|
|
290
|
+
`${limits.maxFiles.toLocaleString()} file limit. Raise it with ` +
|
|
291
|
+
`--scip-max-files ${documentCount}, or narrow the indexer that produced this file.`);
|
|
284
292
|
for (const message of messages(index, 2)) {
|
|
285
293
|
const filePath = safeRelativePath(resolvedRepo, strings(message, 1)[0] ?? "");
|
|
286
294
|
const sourceStat = fs.statSync(path.join(resolvedRepo, filePath));
|
|
@@ -289,18 +297,21 @@ export function readScipIndex(repoPath, options) {
|
|
|
289
297
|
const occurrenceCount = countLengthDelimitedFields(message, 2, limits.maxFacts - parsedFacts);
|
|
290
298
|
parsedFacts += occurrenceCount;
|
|
291
299
|
if (parsedFacts > limits.maxFacts)
|
|
292
|
-
fail(`index exceeds ${limits.maxFacts} fact limit`
|
|
300
|
+
fail(`index exceeds the ${limits.maxFacts.toLocaleString()} fact limit. ` +
|
|
301
|
+
`Raise it with --scip-max-facts, or narrow the indexer that produced this file.`);
|
|
293
302
|
const symbolCount = countLengthDelimitedFields(message, 3, limits.maxFacts - parsedFacts);
|
|
294
303
|
parsedFacts += symbolCount;
|
|
295
304
|
if (parsedFacts > limits.maxFacts)
|
|
296
|
-
fail(`index exceeds ${limits.maxFacts} fact limit`
|
|
305
|
+
fail(`index exceeds the ${limits.maxFacts.toLocaleString()} fact limit. ` +
|
|
306
|
+
`Raise it with --scip-max-facts, or narrow the indexer that produced this file.`);
|
|
297
307
|
const occurrenceMessages = messages(message, 2);
|
|
298
308
|
const symbolMessages = messages(message, 3);
|
|
299
309
|
for (const symbolMessage of symbolMessages) {
|
|
300
310
|
const remainingRelationships = Math.floor((limits.maxFacts - parsedFacts) / 4);
|
|
301
311
|
parsedFacts += countLengthDelimitedFields(symbolMessage, 4, remainingRelationships) * 4;
|
|
302
312
|
if (parsedFacts > limits.maxFacts)
|
|
303
|
-
fail(`index exceeds ${limits.maxFacts} fact limit`
|
|
313
|
+
fail(`index exceeds the ${limits.maxFacts.toLocaleString()} fact limit. ` +
|
|
314
|
+
`Raise it with --scip-max-facts, or narrow the indexer that produced this file.`);
|
|
304
315
|
}
|
|
305
316
|
documents.push({
|
|
306
317
|
filePath,
|
|
@@ -389,17 +400,21 @@ export function readScipIndex(repoPath, options) {
|
|
|
389
400
|
fail(`index exceeds ${limits.maxFacts} fact limit`);
|
|
390
401
|
if (performance.now() - started > limits.timeoutMs)
|
|
391
402
|
fail(`import exceeded ${limits.timeoutMs} ms limit`);
|
|
403
|
+
// Byte order: imported facts are persisted and compared across runs, so their
|
|
404
|
+
// order has to be a function of the bytes rather than of host collation.
|
|
405
|
+
symbols.sort((a, b) => compareBytes(a.filePath, b.filePath) || a.startLine - b.startLine);
|
|
406
|
+
references.sort((a, b) => compareBytes(a.callerFile, b.callerFile) ||
|
|
407
|
+
a.line - b.line ||
|
|
408
|
+
compareBytes(a.calleeSymbol, b.calleeSymbol));
|
|
392
409
|
const result = {
|
|
393
410
|
inputPath: inputRelative.replaceAll("\\", "/"),
|
|
394
411
|
bytes: inputStat.size,
|
|
395
412
|
files: documents.length,
|
|
396
413
|
facts: factCount,
|
|
397
|
-
languages: [...new Set(documents.map((document) => document.language))].sort(),
|
|
398
|
-
indexers: parseMetadata(index).sort(),
|
|
399
|
-
symbols
|
|
400
|
-
references
|
|
401
|
-
a.line - b.line ||
|
|
402
|
-
a.calleeSymbol.localeCompare(b.calleeSymbol)),
|
|
414
|
+
languages: [...new Set(documents.map((document) => document.language))].sort(compareBytes),
|
|
415
|
+
indexers: parseMetadata(index).sort(compareBytes),
|
|
416
|
+
symbols,
|
|
417
|
+
references,
|
|
403
418
|
elapsedMs: Math.round((performance.now() - started) * 1000) / 1000,
|
|
404
419
|
};
|
|
405
420
|
activeDeadline = Number.POSITIVE_INFINITY;
|
|
@@ -37,6 +37,22 @@ const SOURCE_EXTENSIONS = new Set([
|
|
|
37
37
|
".hcl",
|
|
38
38
|
".dockerfile",
|
|
39
39
|
".lsif",
|
|
40
|
+
// Documentation participates as a link-only graph: doc-to-doc and doc-to-code
|
|
41
|
+
// edges, no symbols. Without it, "which docs describe this subsystem" and
|
|
42
|
+
// "which docs point at code that no longer exists" are unanswerable.
|
|
43
|
+
".md",
|
|
44
|
+
".mdx",
|
|
45
|
+
// CI pipeline definitions participate as link-only template references.
|
|
46
|
+
".yml",
|
|
47
|
+
".yaml",
|
|
48
|
+
// The .NET build graph: project-to-project references across services.
|
|
49
|
+
".csproj",
|
|
50
|
+
".sln",
|
|
51
|
+
".props",
|
|
52
|
+
".targets",
|
|
53
|
+
// Razor views/components: the C# members in their @code/@functions blocks.
|
|
54
|
+
".cshtml",
|
|
55
|
+
".razor",
|
|
40
56
|
]);
|
|
41
57
|
function isSalesforceBundleMarkup(normalized) {
|
|
42
58
|
return (/^force-app\/main\/default\/lwc\/([A-Za-z][A-Za-z0-9_]*)\/\1\.(?:html|css)$/.test(normalized) ||
|
|
@@ -0,0 +1,175 @@
|
|
|
1
|
+
import fs from "node:fs";
|
|
2
|
+
import os from "node:os";
|
|
3
|
+
import path from "node:path";
|
|
4
|
+
/**
|
|
5
|
+
* Resolves where a repository's knodin state (graph database, wiki, hook
|
|
6
|
+
* scaffolding) lives.
|
|
7
|
+
*
|
|
8
|
+
* A normal checkout keeps its state in-tree at `<repo>/.knodin`, unchanged from
|
|
9
|
+
* the original hardcoded behaviour. A *mirror* — a read-only shallow clone of a
|
|
10
|
+
* repository the user never checked out themselves — keeps its state outside the
|
|
11
|
+
* clone entirely, so that a refetch can discard and replace the source tree
|
|
12
|
+
* without destroying a graph that took minutes to build.
|
|
13
|
+
*
|
|
14
|
+
* Mirror layout, rooted at {@link mirrorRoot}:
|
|
15
|
+
*
|
|
16
|
+
* ```
|
|
17
|
+
* <root>/registry.json the mirror index
|
|
18
|
+
* <root>/.staging/<id> in-progress clones, renamed into place on success
|
|
19
|
+
* <root>/<id>/source the shallow clone
|
|
20
|
+
* <root>/<id>/state db.sqlite, wiki/, and friends
|
|
21
|
+
* <root>/<id>/README marker: local edits here are discarded on refresh
|
|
22
|
+
* ```
|
|
23
|
+
*
|
|
24
|
+
* Every consumer must route through {@link resolveStateDir} /
|
|
25
|
+
* {@link resolveDbPath} rather than rebuilding `path.join(repo, ".knodin")`
|
|
26
|
+
* inline, or mirrors silently read from the wrong location.
|
|
27
|
+
*/
|
|
28
|
+
/** Directory name holding the shallow clone inside a mirror entry. */
|
|
29
|
+
export const MIRROR_SOURCE_DIR = "source";
|
|
30
|
+
/** Directory name holding graph state inside a mirror entry. */
|
|
31
|
+
export const MIRROR_STATE_DIR = "state";
|
|
32
|
+
/** In-tree state directory name for a normal checkout. */
|
|
33
|
+
export const IN_TREE_STATE_DIR = ".knodin";
|
|
34
|
+
const REGISTRY_VERSION = 1;
|
|
35
|
+
const EMPTY_REGISTRY = { version: REGISTRY_VERSION, mirrors: [] };
|
|
36
|
+
/**
|
|
37
|
+
* Root for all mirror storage. Defaults to `~/.knodin/mirrors` rather than an
|
|
38
|
+
* XDG data directory: a mirror costs minutes of indexing to rebuild, so it needs
|
|
39
|
+
* to sit somewhere the user will actually notice and can audit with
|
|
40
|
+
* `knodin remote list`. `KNODIN_MIRROR_ROOT` overrides it.
|
|
41
|
+
*/
|
|
42
|
+
export function mirrorRoot() {
|
|
43
|
+
const override = process.env.KNODIN_MIRROR_ROOT;
|
|
44
|
+
return override ? path.resolve(override) : path.join(os.homedir(), ".knodin", "mirrors");
|
|
45
|
+
}
|
|
46
|
+
/** Path to the mirror registry document. */
|
|
47
|
+
export function mirrorRegistryPath() {
|
|
48
|
+
return path.join(mirrorRoot(), "registry.json");
|
|
49
|
+
}
|
|
50
|
+
/** Root for in-progress clones, renamed into place only once complete. */
|
|
51
|
+
export function mirrorStagingRoot() {
|
|
52
|
+
return path.join(mirrorRoot(), ".staging");
|
|
53
|
+
}
|
|
54
|
+
/** Entry root for one mirror: the parent of both `source/` and `state/`. */
|
|
55
|
+
export function mirrorEntryPath(identity) {
|
|
56
|
+
return path.join(mirrorRoot(), identity);
|
|
57
|
+
}
|
|
58
|
+
/** Absolute path to a mirror's shallow clone. */
|
|
59
|
+
export function mirrorSourcePath(identity) {
|
|
60
|
+
return path.join(mirrorEntryPath(identity), MIRROR_SOURCE_DIR);
|
|
61
|
+
}
|
|
62
|
+
/** Absolute path to a mirror's graph state directory. */
|
|
63
|
+
export function mirrorStatePath(identity) {
|
|
64
|
+
return path.join(mirrorEntryPath(identity), MIRROR_STATE_DIR);
|
|
65
|
+
}
|
|
66
|
+
let cache = null;
|
|
67
|
+
function isMirrorRecord(value) {
|
|
68
|
+
if (typeof value !== "object" || value === null)
|
|
69
|
+
return false;
|
|
70
|
+
const record = value;
|
|
71
|
+
return (typeof record.identity === "string" &&
|
|
72
|
+
typeof record.url === "string" &&
|
|
73
|
+
typeof record.path === "string" &&
|
|
74
|
+
typeof record.fetchedAt === "string" &&
|
|
75
|
+
typeof record.sha === "string");
|
|
76
|
+
}
|
|
77
|
+
function indexByPath(registry) {
|
|
78
|
+
const byPath = new Map();
|
|
79
|
+
for (const record of registry.mirrors)
|
|
80
|
+
byPath.set(path.resolve(record.path), record);
|
|
81
|
+
return byPath;
|
|
82
|
+
}
|
|
83
|
+
/**
|
|
84
|
+
* Reads the mirror registry, memoized against the file's mtime and size so a
|
|
85
|
+
* mirror added by a concurrent `knodin remote add` is picked up without paying a
|
|
86
|
+
* JSON parse on every path resolution.
|
|
87
|
+
*
|
|
88
|
+
* The memo key is (mtimeMs, size), which is a heuristic rather than a guarantee:
|
|
89
|
+
* an out-of-process write landing in the same millisecond AND producing the same
|
|
90
|
+
* byte length would not invalidate it. Accepted deliberately — that requires a
|
|
91
|
+
* concurrent edit of identical length within one millisecond, the in-process
|
|
92
|
+
* writer drops the memo explicitly, and the cost of being wrong is one stale
|
|
93
|
+
* read, not corruption. Recorded so it reads as a decision, not an oversight.
|
|
94
|
+
*
|
|
95
|
+
* A missing or corrupt registry reads as empty. That is deliberate too: it
|
|
96
|
+
* degrades to "no mirrors configured", which routes every repository to its
|
|
97
|
+
* in-tree state — the pre-existing behaviour — rather than failing a query
|
|
98
|
+
* outright.
|
|
99
|
+
*/
|
|
100
|
+
export function readMirrorRegistry() {
|
|
101
|
+
const file = mirrorRegistryPath();
|
|
102
|
+
let stat;
|
|
103
|
+
try {
|
|
104
|
+
stat = fs.statSync(file);
|
|
105
|
+
}
|
|
106
|
+
catch {
|
|
107
|
+
cache = null;
|
|
108
|
+
return EMPTY_REGISTRY;
|
|
109
|
+
}
|
|
110
|
+
if (cache?.mtimeMs === stat.mtimeMs && cache?.size === stat.size)
|
|
111
|
+
return cache.registry;
|
|
112
|
+
try {
|
|
113
|
+
const parsed = JSON.parse(fs.readFileSync(file, "utf8"));
|
|
114
|
+
const mirrors = typeof parsed === "object" &&
|
|
115
|
+
parsed !== null &&
|
|
116
|
+
Array.isArray(parsed.mirrors)
|
|
117
|
+
? parsed.mirrors.filter(isMirrorRecord)
|
|
118
|
+
: [];
|
|
119
|
+
const registry = { version: REGISTRY_VERSION, mirrors };
|
|
120
|
+
cache = { mtimeMs: stat.mtimeMs, size: stat.size, registry, byPath: indexByPath(registry) };
|
|
121
|
+
return registry;
|
|
122
|
+
}
|
|
123
|
+
catch {
|
|
124
|
+
cache = null;
|
|
125
|
+
return EMPTY_REGISTRY;
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
/** Persists the registry and drops the memo so the next read reloads it. */
|
|
129
|
+
export function writeMirrorRegistry(registry) {
|
|
130
|
+
const file = mirrorRegistryPath();
|
|
131
|
+
fs.mkdirSync(path.dirname(file), { recursive: true });
|
|
132
|
+
const document = { version: REGISTRY_VERSION, mirrors: registry.mirrors };
|
|
133
|
+
fs.writeFileSync(file, `${JSON.stringify(document, null, 2)}\n`);
|
|
134
|
+
cache = null;
|
|
135
|
+
}
|
|
136
|
+
/** Clears the in-process registry memo. Test seam. */
|
|
137
|
+
export function __resetMirrorRegistryCache() {
|
|
138
|
+
cache = null;
|
|
139
|
+
}
|
|
140
|
+
/**
|
|
141
|
+
* Returns the mirror record whose clone is at `repoPath`, if any. Matching is on
|
|
142
|
+
* the resolved path, so a caller passing a relative or unnormalized path still
|
|
143
|
+
* resolves correctly.
|
|
144
|
+
*/
|
|
145
|
+
export function lookupMirror(repoPath) {
|
|
146
|
+
readMirrorRegistry();
|
|
147
|
+
return cache?.byPath.get(path.resolve(repoPath));
|
|
148
|
+
}
|
|
149
|
+
/** True when `repoPath` is a registered read-only mirror clone. */
|
|
150
|
+
export function isMirror(repoPath) {
|
|
151
|
+
return lookupMirror(repoPath) !== undefined;
|
|
152
|
+
}
|
|
153
|
+
/**
|
|
154
|
+
* Directory holding knodin state for `repoPath`: in-tree `.knodin` for a normal
|
|
155
|
+
* checkout, an out-of-tree sibling of the clone for a mirror.
|
|
156
|
+
*/
|
|
157
|
+
export function resolveStateDir(repoPath) {
|
|
158
|
+
const resolved = path.resolve(repoPath);
|
|
159
|
+
const mirror = lookupMirror(resolved);
|
|
160
|
+
if (mirror)
|
|
161
|
+
return mirrorStatePath(mirror.identity);
|
|
162
|
+
return path.join(resolved, IN_TREE_STATE_DIR);
|
|
163
|
+
}
|
|
164
|
+
/** Graph database path for `repoPath`. */
|
|
165
|
+
export function resolveDbPath(repoPath) {
|
|
166
|
+
return path.join(resolveStateDir(repoPath), "db.sqlite");
|
|
167
|
+
}
|
|
168
|
+
/**
|
|
169
|
+
* Whether opening an index for `repoPath` may write to the repository itself —
|
|
170
|
+
* false for mirrors, whose source tree is replaced wholesale on refresh and must
|
|
171
|
+
* stay byte-identical to the remote in the meantime.
|
|
172
|
+
*/
|
|
173
|
+
export function mayWriteToRepository(repoPath) {
|
|
174
|
+
return !isMirror(repoPath);
|
|
175
|
+
}
|