knodin 0.9.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/bin/cli.js +35 -7
- package/dist/fixtures/update-verification/fixture.json +1 -1
- package/dist/src/cli-model.js +14 -0
- package/dist/src/engine/index.js +919 -355
- package/dist/src/engine/parse-pool-resources.js +96 -0
- package/dist/src/engine/parse-pool.js +352 -0
- package/dist/src/engine/parse-protocol.js +1 -0
- package/dist/src/engine/parse-worker.js +51 -0
- package/dist/src/engine/reflink-copy.js +23 -0
- package/dist/src/engine/seal-command.js +116 -0
- package/dist/src/engine/seal.js +270 -0
- package/dist/src/engine/sealed-open.js +116 -0
- package/dist/src/engine/sealed-query.js +49 -0
- package/dist/src/init.js +29 -4
- package/dist/src/shared-index/selection.js +1 -1
- package/dist/src/tools/knodin-tools.js +16 -3
- package/dist/src/update-executor.js +29 -10
- package/dist/src/worktree-seed.js +172 -0
- package/docs/SHARED-INDEX-CONTRACT.md +9 -0
- package/docs/releases/0.10.0.md +127 -0
- package/package.json +2 -1
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
import os from "node:os";
|
|
2
|
+
/**
|
|
3
|
+
* Per-worker resident memory allowance.
|
|
4
|
+
*
|
|
5
|
+
* Measured rather than guessed: a worker's fixed cost is ~136 MB — roughly
|
|
6
|
+
* 64 MB of Node baseline, 22 MB to import the engine, and up to 50 MB of
|
|
7
|
+
* tree-sitter grammars if it ends up touching every language. The rest is
|
|
8
|
+
* headroom for parse trees, which scale with file size. Kept deliberately
|
|
9
|
+
* above the measurement so the pool under-spawns rather than over-spawns:
|
|
10
|
+
* too few workers is a slower index, too many is the user's machine thrashing.
|
|
11
|
+
*/
|
|
12
|
+
export const PER_WORKER_MEMORY_ESTIMATE_BYTES = 192 * 1024 * 1024;
|
|
13
|
+
/**
|
|
14
|
+
* Fraction of TOTAL system memory the pool may claim.
|
|
15
|
+
*
|
|
16
|
+
* This deliberately does NOT use `os.freemem()`, which an earlier version did
|
|
17
|
+
* on the reasoning that a machine already under load should be respected.
|
|
18
|
+
* Measured, that reasoning does not survive contact with macOS: on a 24 GB
|
|
19
|
+
* host, `os.freemem()` reported **133 MB**, because the OS keeps nearly all
|
|
20
|
+
* RAM in cache and compressed pages rather than "free". The pool computed a
|
|
21
|
+
* memory ceiling of zero and refused to spawn on an 11-core machine — a
|
|
22
|
+
* feature that was enabled and silently never used.
|
|
23
|
+
*
|
|
24
|
+
* Node exposes no portable "available" figure, so total memory with a
|
|
25
|
+
* conservative fraction is the honest signal. Real memory pressure is handled
|
|
26
|
+
* by the hard cap and the per-worker allowance above, not by a number that
|
|
27
|
+
* means something different on every platform.
|
|
28
|
+
*/
|
|
29
|
+
export const MEMORY_HEADROOM_FRACTION = 0.25;
|
|
30
|
+
/**
|
|
31
|
+
* Absolute worker ceiling regardless of how large the host is. Guards against
|
|
32
|
+
* pathological spawn counts on very-high-core machines, where the writer
|
|
33
|
+
* becomes the bottleneck long before the parsers do and extra workers buy
|
|
34
|
+
* nothing but memory.
|
|
35
|
+
*/
|
|
36
|
+
export const PARSE_POOL_HARD_CAP = 16;
|
|
37
|
+
/** Read a positive-integer env override, matching `mcp-reliability.ts`'s idiom. */
|
|
38
|
+
function positiveIntegerEnv(raw) {
|
|
39
|
+
const parsed = Number.parseInt(raw ?? "", 10);
|
|
40
|
+
return Number.isSafeInteger(parsed) && parsed > 0 ? parsed : null;
|
|
41
|
+
}
|
|
42
|
+
/**
|
|
43
|
+
* Decide the parse-pool size from injected machine facts.
|
|
44
|
+
*
|
|
45
|
+
* Precedence, highest first:
|
|
46
|
+
* 1. `KNODIN_DISABLE_PARSE_WORKERS=1` -- absolute kill switch, always 0.
|
|
47
|
+
* 2. `KNODIN_PARSE_POOL_MAX_WORKERS=n` -- an explicit ceiling, still clamped
|
|
48
|
+
* by the hard cap but NOT by the derived CPU/memory bounds, so an operator
|
|
49
|
+
* debugging a machine can pin the value without fighting the heuristic.
|
|
50
|
+
* 3. The smaller of the CPU and memory ceilings.
|
|
51
|
+
*/
|
|
52
|
+
export function resolveParsePoolSize(inputs) {
|
|
53
|
+
const env = inputs.env ?? process.env;
|
|
54
|
+
// One core is reserved for the main thread, which is not an idle consumer
|
|
55
|
+
// here: it performs the synchronous, blocking SQLite writes plus the two
|
|
56
|
+
// whole-graph post-passes, and it must stay responsive enough to keep the
|
|
57
|
+
// workers fed.
|
|
58
|
+
const cpuBound = Math.max(1, Math.floor(inputs.cpuCount) - 1);
|
|
59
|
+
const usableMemory = Math.max(0, inputs.totalMemoryBytes) * MEMORY_HEADROOM_FRACTION;
|
|
60
|
+
const memoryBound = Math.floor(usableMemory / PER_WORKER_MEMORY_ESTIMATE_BYTES);
|
|
61
|
+
if (env.KNODIN_DISABLE_PARSE_WORKERS === "1")
|
|
62
|
+
return { workers: 0, reason: "disabled-by-env", cpuBound, memoryBound };
|
|
63
|
+
const override = positiveIntegerEnv(env.KNODIN_PARSE_POOL_MAX_WORKERS);
|
|
64
|
+
if (override !== null)
|
|
65
|
+
return {
|
|
66
|
+
workers: Math.min(override, PARSE_POOL_HARD_CAP),
|
|
67
|
+
reason: "explicit-override",
|
|
68
|
+
cpuBound,
|
|
69
|
+
memoryBound,
|
|
70
|
+
};
|
|
71
|
+
// A single-core host has no core to spare once the writer is accounted for,
|
|
72
|
+
// so the pool would be one worker handing results to an idle main thread --
|
|
73
|
+
// strictly worse than doing the work inline. Report it distinctly from a
|
|
74
|
+
// memory refusal: the remedies are completely different.
|
|
75
|
+
if (cpuBound < 2)
|
|
76
|
+
return { workers: 0, reason: "single-core", cpuBound, memoryBound };
|
|
77
|
+
// Too little memory for even one worker's fixed overhead. Sequential
|
|
78
|
+
// is not a degraded outcome here, it is the correct one.
|
|
79
|
+
if (memoryBound < 1)
|
|
80
|
+
return { workers: 0, reason: "memory-bound", cpuBound, memoryBound };
|
|
81
|
+
const workers = Math.min(cpuBound, memoryBound, PARSE_POOL_HARD_CAP);
|
|
82
|
+
return {
|
|
83
|
+
workers,
|
|
84
|
+
reason: memoryBound < cpuBound ? "memory-bound" : "cpu-bound",
|
|
85
|
+
cpuBound,
|
|
86
|
+
memoryBound,
|
|
87
|
+
};
|
|
88
|
+
}
|
|
89
|
+
/** Sample the live machine. Separated from the policy so the policy stays testable. */
|
|
90
|
+
export function currentParsePoolSizing(env = process.env) {
|
|
91
|
+
return resolveParsePoolSize({
|
|
92
|
+
cpuCount: os.cpus().length,
|
|
93
|
+
totalMemoryBytes: os.totalmem(),
|
|
94
|
+
env,
|
|
95
|
+
});
|
|
96
|
+
}
|
|
@@ -0,0 +1,352 @@
|
|
|
1
|
+
import path from "node:path";
|
|
2
|
+
import { fileURLToPath } from "node:url";
|
|
3
|
+
import { Worker } from "node:worker_threads";
|
|
4
|
+
import { currentParsePoolSizing } from "./parse-pool-resources.js";
|
|
5
|
+
/**
|
|
6
|
+
* On by default; `KNODIN_DISABLE_PARSE_WORKERS=1` is the kill switch.
|
|
7
|
+
*
|
|
8
|
+
* Defaulted on only after the gate it was waiting for actually passed, and
|
|
9
|
+
* after the economics changed enough to justify it. When the pool was first
|
|
10
|
+
* wired, `parse_extract` was 10.9% of a full index and the pool bought 1.16x —
|
|
11
|
+
* not worth worker-thread complexity. Fixing the real bottleneck
|
|
12
|
+
* (EASFDC-8421) took a full index from 60.5s to 21.3s and left parsing at
|
|
13
|
+
* 31.6%, which raised the pool's own contribution to a measured 1.57x. The
|
|
14
|
+
* combined figure against the original baseline is 4.7x.
|
|
15
|
+
*
|
|
16
|
+
* Enabling it is safe rather than merely faster: sizing refuses to spawn on
|
|
17
|
+
* machines that cannot afford workers and returns null BEFORE any Worker is
|
|
18
|
+
* constructed, a worker crash retries once and then hands the file back for
|
|
19
|
+
* inline indexing, an exhausted pool drains its queue rather than stalling,
|
|
20
|
+
* and the graph is proven set-equivalent to the sequential build. Every one of
|
|
21
|
+
* those has a test.
|
|
22
|
+
*
|
|
23
|
+
* `KNODIN_ENABLE_PARSE_WORKERS` is still honoured so anyone who set it during
|
|
24
|
+
* the opt-in period keeps working, but it is now redundant.
|
|
25
|
+
*/
|
|
26
|
+
export function parseWorkersEnabled(env = process.env) {
|
|
27
|
+
return env.KNODIN_DISABLE_PARSE_WORKERS !== "1";
|
|
28
|
+
}
|
|
29
|
+
/**
|
|
30
|
+
* Grammar identity for affinity purposes. Grammars are cached in module-level
|
|
31
|
+
* state inside the engine, so each worker holds a copy of every language it
|
|
32
|
+
* sees: measured at 7.4 MB for a TypeScript-only worker against ~50 MB for one
|
|
33
|
+
* that eventually touches every grammar. Keeping a worker on one extension is
|
|
34
|
+
* therefore a real memory lever, not a micro-optimization.
|
|
35
|
+
*/
|
|
36
|
+
function languageKey(relativePath) {
|
|
37
|
+
return path.extname(relativePath).toLowerCase() || "<none>";
|
|
38
|
+
}
|
|
39
|
+
function defaultWorkerEntry() {
|
|
40
|
+
const extension = import.meta.url.endsWith(".ts") ? "ts" : "js";
|
|
41
|
+
return fileURLToPath(new URL(`./parse-worker.${extension}`, import.meta.url));
|
|
42
|
+
}
|
|
43
|
+
/**
|
|
44
|
+
* When running from TypeScript sources, a worker needs its own TS loader.
|
|
45
|
+
*
|
|
46
|
+
* A worker thread does NOT inherit the host's transform pipeline. Under a
|
|
47
|
+
* plain `tsx` process it happens to work because tsx registers a real Node
|
|
48
|
+
* loader; under Vitest it does not, and the worker dies with "Cannot find
|
|
49
|
+
* module .../index.js" while resolving this module's own `./index.js` import.
|
|
50
|
+
* The pool then silently falls back to the sequential path — correct results,
|
|
51
|
+
* but no parallelism, and no way to tell from the outside.
|
|
52
|
+
*
|
|
53
|
+
* Mirrors what `src/__tests__/support/node-subprocess.ts` already does for the
|
|
54
|
+
* child-process worker. Returns undefined for a compiled `.js` entry, where
|
|
55
|
+
* the import resolves natively and tsx may not be installed at all.
|
|
56
|
+
*/
|
|
57
|
+
function workerExecArgv(entry) {
|
|
58
|
+
if (!entry.endsWith(".ts"))
|
|
59
|
+
return undefined;
|
|
60
|
+
try {
|
|
61
|
+
return ["--import", fileURLToPath(import.meta.resolve("tsx"))];
|
|
62
|
+
}
|
|
63
|
+
catch {
|
|
64
|
+
return undefined;
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
function realWorkerFactory() {
|
|
68
|
+
const entry = defaultWorkerEntry();
|
|
69
|
+
const execArgv = workerExecArgv(entry);
|
|
70
|
+
const worker = new Worker(entry, execArgv ? { execArgv } : undefined);
|
|
71
|
+
worker.unref();
|
|
72
|
+
return {
|
|
73
|
+
postMessage: (task) => worker.postMessage(task),
|
|
74
|
+
onMessage: (listener) => worker.on("message", listener),
|
|
75
|
+
onError: (listener) => worker.on("error", listener),
|
|
76
|
+
onExit: (listener) => worker.on("exit", listener),
|
|
77
|
+
terminate: async () => {
|
|
78
|
+
await worker.terminate();
|
|
79
|
+
},
|
|
80
|
+
};
|
|
81
|
+
}
|
|
82
|
+
/**
|
|
83
|
+
* One `run()` call's state machine.
|
|
84
|
+
*
|
|
85
|
+
* Deliberately a class of flat methods rather than nested closures: the same
|
|
86
|
+
* logic written inline nests worker callbacks inside a promise executor inside
|
|
87
|
+
* a method, which is both hard to follow and a real reviewability problem
|
|
88
|
+
* (Sonar S2004 flagged five separate spots). Methods keep every step at one
|
|
89
|
+
* level and let the slot be passed explicitly.
|
|
90
|
+
*/
|
|
91
|
+
class PoolRun {
|
|
92
|
+
slots;
|
|
93
|
+
factory;
|
|
94
|
+
onOutcome;
|
|
95
|
+
nextId;
|
|
96
|
+
buckets = new Map();
|
|
97
|
+
remaining;
|
|
98
|
+
settled = false;
|
|
99
|
+
draining = false;
|
|
100
|
+
resolve;
|
|
101
|
+
reject;
|
|
102
|
+
constructor(slots, factory, tasks, onOutcome, nextId) {
|
|
103
|
+
this.slots = slots;
|
|
104
|
+
this.factory = factory;
|
|
105
|
+
this.onOutcome = onOutcome;
|
|
106
|
+
this.nextId = nextId;
|
|
107
|
+
for (const task of tasks) {
|
|
108
|
+
const key = languageKey(task.relativePath);
|
|
109
|
+
const bucket = this.buckets.get(key);
|
|
110
|
+
if (bucket)
|
|
111
|
+
bucket.push(task);
|
|
112
|
+
else
|
|
113
|
+
this.buckets.set(key, [task]);
|
|
114
|
+
}
|
|
115
|
+
this.remaining = tasks.length;
|
|
116
|
+
}
|
|
117
|
+
start() {
|
|
118
|
+
return new Promise((resolve, reject) => {
|
|
119
|
+
this.resolve = resolve;
|
|
120
|
+
this.reject = reject;
|
|
121
|
+
for (const slot of this.slots)
|
|
122
|
+
this.attach(slot);
|
|
123
|
+
for (const slot of this.slots)
|
|
124
|
+
this.pump(slot);
|
|
125
|
+
});
|
|
126
|
+
}
|
|
127
|
+
/** Prefer the slot's own language; fall back to the largest bucket. */
|
|
128
|
+
takeNext(slot) {
|
|
129
|
+
if (slot.affinity) {
|
|
130
|
+
const own = this.buckets.get(slot.affinity);
|
|
131
|
+
if (own && own.length > 0)
|
|
132
|
+
return own.pop() ?? null;
|
|
133
|
+
}
|
|
134
|
+
// Preferring affinity without enforcing it keeps a single-language
|
|
135
|
+
// repository from leaving most of the pool idle.
|
|
136
|
+
let bestKey = null;
|
|
137
|
+
let bestLength = 0;
|
|
138
|
+
for (const [key, bucket] of this.buckets) {
|
|
139
|
+
if (bucket.length > bestLength) {
|
|
140
|
+
bestKey = key;
|
|
141
|
+
bestLength = bucket.length;
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
if (!bestKey)
|
|
145
|
+
return null;
|
|
146
|
+
slot.affinity = bestKey;
|
|
147
|
+
return this.buckets.get(bestKey)?.pop() ?? null;
|
|
148
|
+
}
|
|
149
|
+
attach(slot) {
|
|
150
|
+
slot.handle.onMessage((response) => this.onResponse(slot, response));
|
|
151
|
+
slot.handle.onError((error) => this.onSlotFailure(slot, error.message));
|
|
152
|
+
slot.handle.onExit((code) => {
|
|
153
|
+
if (slot.inFlight)
|
|
154
|
+
this.onSlotFailure(slot, `worker exited with code ${code}`);
|
|
155
|
+
});
|
|
156
|
+
}
|
|
157
|
+
onResponse(slot, response) {
|
|
158
|
+
if (!slot.inFlight || response.id !== slot.inFlight.id)
|
|
159
|
+
return;
|
|
160
|
+
switch (response.kind) {
|
|
161
|
+
case "result":
|
|
162
|
+
this.finishOne(slot, { kind: "result", result: response.result });
|
|
163
|
+
return;
|
|
164
|
+
case "unparsed":
|
|
165
|
+
this.finishOne(slot, { kind: "unparsed" });
|
|
166
|
+
return;
|
|
167
|
+
case "skipped":
|
|
168
|
+
this.finishOne(slot, { kind: "skipped" });
|
|
169
|
+
return;
|
|
170
|
+
case "special":
|
|
171
|
+
this.finishOne(slot, { kind: "fallback", reason: "needs main-thread indexer" });
|
|
172
|
+
return;
|
|
173
|
+
default:
|
|
174
|
+
this.finishOne(slot, { kind: "fallback", reason: response.message });
|
|
175
|
+
return;
|
|
176
|
+
}
|
|
177
|
+
}
|
|
178
|
+
pump(slot) {
|
|
179
|
+
if (this.settled || slot.dead || slot.inFlight)
|
|
180
|
+
return;
|
|
181
|
+
const task = this.takeNext(slot);
|
|
182
|
+
if (!task)
|
|
183
|
+
return;
|
|
184
|
+
const id = this.nextId();
|
|
185
|
+
slot.inFlight = { task, id, attempt: 0 };
|
|
186
|
+
this.dispatch(slot, task, id);
|
|
187
|
+
}
|
|
188
|
+
dispatch(slot, task, id) {
|
|
189
|
+
try {
|
|
190
|
+
slot.handle.postMessage({
|
|
191
|
+
id,
|
|
192
|
+
absolutePath: task.absolutePath,
|
|
193
|
+
relativePath: task.relativePath,
|
|
194
|
+
repoPath: task.repoPath,
|
|
195
|
+
});
|
|
196
|
+
}
|
|
197
|
+
catch (error) {
|
|
198
|
+
this.onSlotFailure(slot, error instanceof Error ? error.message : String(error));
|
|
199
|
+
}
|
|
200
|
+
}
|
|
201
|
+
replaceSlot(slot) {
|
|
202
|
+
try {
|
|
203
|
+
slot.handle = this.factory();
|
|
204
|
+
slot.dead = false;
|
|
205
|
+
this.attach(slot);
|
|
206
|
+
}
|
|
207
|
+
catch {
|
|
208
|
+
slot.dead = true;
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
onSlotFailure(slot, reason) {
|
|
212
|
+
const flight = slot.inFlight;
|
|
213
|
+
slot.dead = true;
|
|
214
|
+
void slot.handle.terminate().catch(() => { });
|
|
215
|
+
if (!flight) {
|
|
216
|
+
// Died idle. Bring a replacement up so the pool keeps its width.
|
|
217
|
+
this.replaceSlot(slot);
|
|
218
|
+
if (!slot.dead)
|
|
219
|
+
this.pump(slot);
|
|
220
|
+
// A slot that dies while idle delivers no outcome, so nothing else
|
|
221
|
+
// would notice the pool has run out of workers.
|
|
222
|
+
else if (!this.anyLive())
|
|
223
|
+
void this.drainRemaining(reason);
|
|
224
|
+
return;
|
|
225
|
+
}
|
|
226
|
+
if (flight.attempt === 0) {
|
|
227
|
+
// One retry on a fresh worker: a crash is far more often about this
|
|
228
|
+
// particular file than about the worker.
|
|
229
|
+
this.replaceSlot(slot);
|
|
230
|
+
if (!slot.dead) {
|
|
231
|
+
const id = this.nextId();
|
|
232
|
+
slot.inFlight = { ...flight, attempt: 1, id };
|
|
233
|
+
this.dispatch(slot, flight.task, id);
|
|
234
|
+
return;
|
|
235
|
+
}
|
|
236
|
+
}
|
|
237
|
+
// Retried already, or no replacement available: give the file back so
|
|
238
|
+
// the caller indexes it inline. Never dropped.
|
|
239
|
+
this.finishOne(slot, { kind: "fallback", reason });
|
|
240
|
+
if (!slot.dead)
|
|
241
|
+
this.pump(slot);
|
|
242
|
+
}
|
|
243
|
+
finishOne(slot, outcome) {
|
|
244
|
+
const flight = slot.inFlight;
|
|
245
|
+
if (!flight)
|
|
246
|
+
return;
|
|
247
|
+
slot.inFlight = null;
|
|
248
|
+
this.remaining--;
|
|
249
|
+
void this.deliver(slot, flight.task, outcome);
|
|
250
|
+
}
|
|
251
|
+
async deliver(slot, task, outcome) {
|
|
252
|
+
try {
|
|
253
|
+
await this.onOutcome(task, outcome);
|
|
254
|
+
}
|
|
255
|
+
catch (error) {
|
|
256
|
+
this.fail(error instanceof Error ? error : new Error(String(error)));
|
|
257
|
+
return;
|
|
258
|
+
}
|
|
259
|
+
if (this.remaining === 0) {
|
|
260
|
+
this.finish();
|
|
261
|
+
return;
|
|
262
|
+
}
|
|
263
|
+
this.pump(slot);
|
|
264
|
+
if (!this.anyLive())
|
|
265
|
+
await this.drainRemaining("no parse worker available");
|
|
266
|
+
}
|
|
267
|
+
anyLive() {
|
|
268
|
+
return this.slots.some((slot) => !slot.dead);
|
|
269
|
+
}
|
|
270
|
+
/**
|
|
271
|
+
* Every worker is gone and could not be replaced. Hand back every task that
|
|
272
|
+
* has not been dispatched so the caller indexes them inline.
|
|
273
|
+
*
|
|
274
|
+
* Without this the run simply stops: queued work has no worker to claim it,
|
|
275
|
+
* nothing ever decrements `remaining`, and `run()` never settles. A pool
|
|
276
|
+
* that hangs is far worse than one that declines to help, and "never the
|
|
277
|
+
* reason a run fails" has to include "never the reason a run stalls".
|
|
278
|
+
*/
|
|
279
|
+
async drainRemaining(reason) {
|
|
280
|
+
if (this.draining || this.settled)
|
|
281
|
+
return;
|
|
282
|
+
this.draining = true;
|
|
283
|
+
const pending = [];
|
|
284
|
+
for (const bucket of this.buckets.values())
|
|
285
|
+
pending.push(...bucket.splice(0));
|
|
286
|
+
for (const task of pending) {
|
|
287
|
+
this.remaining--;
|
|
288
|
+
try {
|
|
289
|
+
await this.onOutcome(task, { kind: "fallback", reason });
|
|
290
|
+
}
|
|
291
|
+
catch (error) {
|
|
292
|
+
this.fail(error instanceof Error ? error : new Error(String(error)));
|
|
293
|
+
return;
|
|
294
|
+
}
|
|
295
|
+
}
|
|
296
|
+
this.draining = false;
|
|
297
|
+
if (this.remaining === 0)
|
|
298
|
+
this.finish();
|
|
299
|
+
}
|
|
300
|
+
finish() {
|
|
301
|
+
if (this.settled)
|
|
302
|
+
return;
|
|
303
|
+
this.settled = true;
|
|
304
|
+
this.resolve();
|
|
305
|
+
}
|
|
306
|
+
fail(error) {
|
|
307
|
+
if (this.settled)
|
|
308
|
+
return;
|
|
309
|
+
this.settled = true;
|
|
310
|
+
this.reject(error);
|
|
311
|
+
}
|
|
312
|
+
}
|
|
313
|
+
/**
|
|
314
|
+
* Builds a pool, or returns `null` when the machine says not to.
|
|
315
|
+
*
|
|
316
|
+
* `null` is a normal outcome meaning "use the sequential path" — the decision
|
|
317
|
+
* is made here, before any worker exists, so an unsuitable machine never
|
|
318
|
+
* discovers the problem partway through an index.
|
|
319
|
+
*/
|
|
320
|
+
export function createParsePool(options) {
|
|
321
|
+
const env = options?.env ?? process.env;
|
|
322
|
+
const sizing = options?.sizing ?? currentParsePoolSizing(env);
|
|
323
|
+
if (sizing.workers < 1)
|
|
324
|
+
return null;
|
|
325
|
+
const factory = options?.workerFactory ?? realWorkerFactory;
|
|
326
|
+
const slots = [];
|
|
327
|
+
try {
|
|
328
|
+
for (let i = 0; i < sizing.workers; i++)
|
|
329
|
+
slots.push({ handle: factory(), affinity: null, inFlight: null, dead: false });
|
|
330
|
+
}
|
|
331
|
+
catch {
|
|
332
|
+
// Spawning failed outright. Tear down whatever came up and let the
|
|
333
|
+
// caller stay sequential rather than run a half-sized pool.
|
|
334
|
+
for (const slot of slots)
|
|
335
|
+
void slot.handle.terminate().catch(() => { });
|
|
336
|
+
return null;
|
|
337
|
+
}
|
|
338
|
+
let nextId = 1;
|
|
339
|
+
const allocateId = () => nextId++;
|
|
340
|
+
return {
|
|
341
|
+
size: slots.length,
|
|
342
|
+
sizing,
|
|
343
|
+
async run(tasks, onOutcome) {
|
|
344
|
+
if (tasks.length === 0)
|
|
345
|
+
return;
|
|
346
|
+
await new PoolRun(slots, factory, tasks, onOutcome, allocateId).start();
|
|
347
|
+
},
|
|
348
|
+
async close() {
|
|
349
|
+
await Promise.all(slots.map((slot) => slot.handle.terminate().catch(() => { })));
|
|
350
|
+
},
|
|
351
|
+
};
|
|
352
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
import { parentPort } from "node:worker_threads";
|
|
2
|
+
import { extractGenericFileForIndex } from "./index.js";
|
|
3
|
+
/**
|
|
4
|
+
* Parse-pool worker entry.
|
|
5
|
+
*
|
|
6
|
+
* Owns read + parse + extract for one file at a time and posts back plain
|
|
7
|
+
* data. It never touches SQLite: the database handle is a native, non-
|
|
8
|
+
* transferable `node:sqlite` object, and exactly one writer on the main
|
|
9
|
+
* thread remains the invariant this whole design is built around.
|
|
10
|
+
*
|
|
11
|
+
* Verified during Step 4 that a worker inherits the tsx loader by default, so
|
|
12
|
+
* importing the TypeScript engine here works in development and against the
|
|
13
|
+
* compiled output alike without forwarding `execArgv`.
|
|
14
|
+
*/
|
|
15
|
+
const port = parentPort;
|
|
16
|
+
if (!port)
|
|
17
|
+
throw new Error("knodin parse worker started without a parent port");
|
|
18
|
+
port.on("message", (task) => {
|
|
19
|
+
void (async () => {
|
|
20
|
+
let response;
|
|
21
|
+
try {
|
|
22
|
+
const extraction = await extractGenericFileForIndex(task.absolutePath, task.relativePath, task.repoPath);
|
|
23
|
+
switch (extraction.kind) {
|
|
24
|
+
case "result":
|
|
25
|
+
response = { id: task.id, kind: "result", result: extraction.result };
|
|
26
|
+
break;
|
|
27
|
+
case "unparsed":
|
|
28
|
+
response = { id: task.id, kind: "unparsed" };
|
|
29
|
+
break;
|
|
30
|
+
case "skipped":
|
|
31
|
+
response = { id: task.id, kind: "skipped" };
|
|
32
|
+
break;
|
|
33
|
+
default:
|
|
34
|
+
// `special` carries a live tree; hand the file back instead.
|
|
35
|
+
response = { id: task.id, kind: "special" };
|
|
36
|
+
break;
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
catch (error) {
|
|
40
|
+
// Serialize by hand. A thrown Error clones without its message on some
|
|
41
|
+
// paths, and losing the message turns a diagnosable failure into a
|
|
42
|
+
// mystery at exactly the moment the pool decides whether to retry.
|
|
43
|
+
response = {
|
|
44
|
+
id: task.id,
|
|
45
|
+
kind: "error",
|
|
46
|
+
message: error instanceof Error ? error.message : String(error),
|
|
47
|
+
};
|
|
48
|
+
}
|
|
49
|
+
port.postMessage(response);
|
|
50
|
+
})();
|
|
51
|
+
});
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
import fs from "node:fs";
|
|
2
|
+
/**
|
|
3
|
+
* Copy `source` to `destination`, preferring an OS-level copy-on-write clone
|
|
4
|
+
* (`clonefile()` on APFS, `FICLONE` on Btrfs/XFS) so the two files share disk
|
|
5
|
+
* blocks until one of them is later mutated. Falls back to an ordinary byte
|
|
6
|
+
* copy on any filesystem/platform that doesn't support cloning.
|
|
7
|
+
*/
|
|
8
|
+
export function reflinkCopyFile(source, destination) {
|
|
9
|
+
try {
|
|
10
|
+
fs.copyFileSync(source, destination, fs.constants.COPYFILE_FICLONE_FORCE);
|
|
11
|
+
return { reflinked: true };
|
|
12
|
+
}
|
|
13
|
+
catch {
|
|
14
|
+
try {
|
|
15
|
+
fs.rmSync(destination, { force: true });
|
|
16
|
+
}
|
|
17
|
+
catch {
|
|
18
|
+
// Best-effort cleanup of a partial clone attempt; the plain copy below still succeeds or throws.
|
|
19
|
+
}
|
|
20
|
+
fs.copyFileSync(source, destination);
|
|
21
|
+
return { reflinked: false };
|
|
22
|
+
}
|
|
23
|
+
}
|
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
import { execFileSync } from "node:child_process";
|
|
2
|
+
import fs from "node:fs";
|
|
3
|
+
import path from "node:path";
|
|
4
|
+
import { createEngine } from "./index.js";
|
|
5
|
+
import { sealIndex } from "./seal.js";
|
|
6
|
+
import { resolveDbPath } from "./state-paths.js";
|
|
7
|
+
/**
|
|
8
|
+
* Orchestrates `knodin seal`: gather live health and git identity, then hand
|
|
9
|
+
* both to the pure sealing transform.
|
|
10
|
+
*
|
|
11
|
+
* Kept out of `seal.ts` so that module stays a leaf — the sealed READ path
|
|
12
|
+
* lives inside the engine, and a seal module that imported the engine back
|
|
13
|
+
* would make that circular.
|
|
14
|
+
*/
|
|
15
|
+
/**
|
|
16
|
+
* Profiles are 58% of the SalesforceCI payload in 187 files. They are
|
|
17
|
+
* permission matrices; no design conversation reads one verbatim, and their
|
|
18
|
+
* structure is already captured as graph edges. Excluding them is what makes a
|
|
19
|
+
* 1Gi resident ceiling reachable — but it is recorded as a NAMED rule so a
|
|
20
|
+
* query about a profile answers "excluded by policy" rather than empty.
|
|
21
|
+
*/
|
|
22
|
+
export const DEFAULT_SOURCE_EXCLUSIONS = [
|
|
23
|
+
{ rule: "profiles-excluded", test: /\.profile-meta\.xml$|(^|\/)profiles\//i },
|
|
24
|
+
];
|
|
25
|
+
function git(repoPath, args) {
|
|
26
|
+
try {
|
|
27
|
+
return execFileSync("git", args, {
|
|
28
|
+
cwd: repoPath,
|
|
29
|
+
encoding: "utf8",
|
|
30
|
+
stdio: ["ignore", "pipe", "ignore"],
|
|
31
|
+
}).trim();
|
|
32
|
+
}
|
|
33
|
+
catch {
|
|
34
|
+
return null;
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
/**
|
|
38
|
+
* Reduce an origin remote to canonical lowercase `owner/name`.
|
|
39
|
+
*
|
|
40
|
+
* The raw URL is NOT usable as identity, and the SSH form is the dangerous
|
|
41
|
+
* one rather than the obviously-broken one: `git@host:owner/name.git` splits
|
|
42
|
+
* into exactly two segments, so a consumer validating "must be owner/name"
|
|
43
|
+
* accepts it and then keys on `git@host:owner` and `name.git`. It passes the
|
|
44
|
+
* shape check while being semantically garbage, which is worse than the HTTPS
|
|
45
|
+
* form that simply fails.
|
|
46
|
+
*
|
|
47
|
+
* Returns null rather than a guess when the URL does not yield a clean pair —
|
|
48
|
+
* an identity a consumer cannot key on should be absent, not approximated.
|
|
49
|
+
*/
|
|
50
|
+
export function canonicalRepositoryIdentity(remoteUrl) {
|
|
51
|
+
if (!remoteUrl)
|
|
52
|
+
return null;
|
|
53
|
+
const withoutSuffix = remoteUrl.trim().replace(/\.git$/i, "");
|
|
54
|
+
// Strip scheme, or the `git@host:` / `host:` prefix of the SCP-like form.
|
|
55
|
+
// The host may carry an SSH config alias (`github.com-company`), which is a
|
|
56
|
+
// local routing detail and must not reach identity.
|
|
57
|
+
const withoutHost = withoutSuffix
|
|
58
|
+
.replace(/^[a-z][a-z0-9+.-]*:\/\//i, "")
|
|
59
|
+
.replace(/^[^/]*@/, "")
|
|
60
|
+
.replace(/^[^/:]+[:/]/, "");
|
|
61
|
+
const segments = withoutHost.split("/").filter(Boolean);
|
|
62
|
+
if (segments.length < 2)
|
|
63
|
+
return null;
|
|
64
|
+
// Take the LAST two: nested group paths (GitLab subgroups) still reduce to
|
|
65
|
+
// the owner/name pair a consumer keys on.
|
|
66
|
+
return segments.slice(-2).join("/").toLowerCase();
|
|
67
|
+
}
|
|
68
|
+
export async function runSeal(options) {
|
|
69
|
+
// Resolve symlinks, not just relative segments. The engine canonicalises the
|
|
70
|
+
// repository path when it decides where state lives, so a caller passing a
|
|
71
|
+
// symlinked path (every macOS temp directory: /var -> /private/var) would
|
|
72
|
+
// have sealing look for the database somewhere the index never wrote it and
|
|
73
|
+
// report `database-missing` for a repository that is perfectly well indexed.
|
|
74
|
+
const requested = path.resolve(options.repoPath);
|
|
75
|
+
const repoPath = fs.existsSync(requested) ? fs.realpathSync(requested) : requested;
|
|
76
|
+
const engine = createEngine({ watcher: "disabled" });
|
|
77
|
+
try {
|
|
78
|
+
const health = await engine.status(repoPath, { audit: "deep" });
|
|
79
|
+
const remoteUrl = git(repoPath, ["remote", "get-url", "origin"]);
|
|
80
|
+
const commit = git(repoPath, ["rev-parse", "HEAD"]);
|
|
81
|
+
// `--symbolic-full-name` so a detached HEAD reports null rather than the
|
|
82
|
+
// literal string "HEAD", which would read as a real ref downstream.
|
|
83
|
+
const ref = git(repoPath, ["rev-parse", "--symbolic-full-name", "HEAD"]);
|
|
84
|
+
const excluded = options.includeExcluded
|
|
85
|
+
? undefined
|
|
86
|
+
: (relativePath) => DEFAULT_SOURCE_EXCLUSIONS.some((entry) => entry.test.test(relativePath));
|
|
87
|
+
return sealIndex({
|
|
88
|
+
repoPath,
|
|
89
|
+
sourceDatabasePath: resolveDbPath(repoPath),
|
|
90
|
+
outputPath: path.resolve(options.outputPath),
|
|
91
|
+
stripEmbeddings: options.stripEmbeddings,
|
|
92
|
+
excludeSource: excluded,
|
|
93
|
+
health: {
|
|
94
|
+
status: health.status,
|
|
95
|
+
pendingPaths: health.freshness.workingTree.pendingPaths ?? 0,
|
|
96
|
+
freshness: health.freshness.state,
|
|
97
|
+
},
|
|
98
|
+
// Canonical owner/name from the remote. `status` reports the local
|
|
99
|
+
// checkout path, which is not identity a consumer can key on — two
|
|
100
|
+
// machines cloning the same repository would attest two different
|
|
101
|
+
// "repositories". The raw URL is kept separately for provenance.
|
|
102
|
+
identity: canonicalRepositoryIdentity(remoteUrl) ?? health.repo,
|
|
103
|
+
remoteUrl,
|
|
104
|
+
commit,
|
|
105
|
+
ref: ref === "HEAD" ? null : ref,
|
|
106
|
+
// Files the graph knows about that are GONE from disk. Files that were
|
|
107
|
+
// MODIFIED do not appear here — those are caught per-file by the drift
|
|
108
|
+
// check inside `sealIndex`, which compares against `index_state`
|
|
109
|
+
// directly rather than trusting an aggregate.
|
|
110
|
+
dirtyPaths: health.missing.files,
|
|
111
|
+
});
|
|
112
|
+
}
|
|
113
|
+
finally {
|
|
114
|
+
await engine.close();
|
|
115
|
+
}
|
|
116
|
+
}
|