knodin 0.9.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,96 @@
1
+ import os from "node:os";
2
+ /**
3
+ * Per-worker resident memory allowance.
4
+ *
5
+ * Measured rather than guessed: a worker's fixed cost is ~136 MB — roughly
6
+ * 64 MB of Node baseline, 22 MB to import the engine, and up to 50 MB of
7
+ * tree-sitter grammars if it ends up touching every language. The rest is
8
+ * headroom for parse trees, which scale with file size. Kept deliberately
9
+ * above the measurement so the pool under-spawns rather than over-spawns:
10
+ * too few workers is a slower index, too many is the user's machine thrashing.
11
+ */
12
+ export const PER_WORKER_MEMORY_ESTIMATE_BYTES = 192 * 1024 * 1024;
13
+ /**
14
+ * Fraction of TOTAL system memory the pool may claim.
15
+ *
16
+ * This deliberately does NOT use `os.freemem()`, which an earlier version did
17
+ * on the reasoning that a machine already under load should be respected.
18
+ * Measured, that reasoning does not survive contact with macOS: on a 24 GB
19
+ * host, `os.freemem()` reported **133 MB**, because the OS keeps nearly all
20
+ * RAM in cache and compressed pages rather than "free". The pool computed a
21
+ * memory ceiling of zero and refused to spawn on an 11-core machine — a
22
+ * feature that was enabled and silently never used.
23
+ *
24
+ * Node exposes no portable "available" figure, so total memory with a
25
+ * conservative fraction is the honest signal. Real memory pressure is handled
26
+ * by the hard cap and the per-worker allowance above, not by a number that
27
+ * means something different on every platform.
28
+ */
29
+ export const MEMORY_HEADROOM_FRACTION = 0.25;
30
+ /**
31
+ * Absolute worker ceiling regardless of how large the host is. Guards against
32
+ * pathological spawn counts on very-high-core machines, where the writer
33
+ * becomes the bottleneck long before the parsers do and extra workers buy
34
+ * nothing but memory.
35
+ */
36
+ export const PARSE_POOL_HARD_CAP = 16;
37
+ /** Read a positive-integer env override, matching `mcp-reliability.ts`'s idiom. */
38
+ function positiveIntegerEnv(raw) {
39
+ const parsed = Number.parseInt(raw ?? "", 10);
40
+ return Number.isSafeInteger(parsed) && parsed > 0 ? parsed : null;
41
+ }
42
+ /**
43
+ * Decide the parse-pool size from injected machine facts.
44
+ *
45
+ * Precedence, highest first:
46
+ * 1. `KNODIN_DISABLE_PARSE_WORKERS=1` -- absolute kill switch, always 0.
47
+ * 2. `KNODIN_PARSE_POOL_MAX_WORKERS=n` -- an explicit ceiling, still clamped
48
+ * by the hard cap but NOT by the derived CPU/memory bounds, so an operator
49
+ * debugging a machine can pin the value without fighting the heuristic.
50
+ * 3. The smaller of the CPU and memory ceilings.
51
+ */
52
+ export function resolveParsePoolSize(inputs) {
53
+ const env = inputs.env ?? process.env;
54
+ // One core is reserved for the main thread, which is not an idle consumer
55
+ // here: it performs the synchronous, blocking SQLite writes plus the two
56
+ // whole-graph post-passes, and it must stay responsive enough to keep the
57
+ // workers fed.
58
+ const cpuBound = Math.max(1, Math.floor(inputs.cpuCount) - 1);
59
+ const usableMemory = Math.max(0, inputs.totalMemoryBytes) * MEMORY_HEADROOM_FRACTION;
60
+ const memoryBound = Math.floor(usableMemory / PER_WORKER_MEMORY_ESTIMATE_BYTES);
61
+ if (env.KNODIN_DISABLE_PARSE_WORKERS === "1")
62
+ return { workers: 0, reason: "disabled-by-env", cpuBound, memoryBound };
63
+ const override = positiveIntegerEnv(env.KNODIN_PARSE_POOL_MAX_WORKERS);
64
+ if (override !== null)
65
+ return {
66
+ workers: Math.min(override, PARSE_POOL_HARD_CAP),
67
+ reason: "explicit-override",
68
+ cpuBound,
69
+ memoryBound,
70
+ };
71
+ // A single-core host has no core to spare once the writer is accounted for,
72
+ // so the pool would be one worker handing results to an idle main thread --
73
+ // strictly worse than doing the work inline. Report it distinctly from a
74
+ // memory refusal: the remedies are completely different.
75
+ if (cpuBound < 2)
76
+ return { workers: 0, reason: "single-core", cpuBound, memoryBound };
77
+ // Too little memory for even one worker's fixed overhead. Sequential
78
+ // is not a degraded outcome here, it is the correct one.
79
+ if (memoryBound < 1)
80
+ return { workers: 0, reason: "memory-bound", cpuBound, memoryBound };
81
+ const workers = Math.min(cpuBound, memoryBound, PARSE_POOL_HARD_CAP);
82
+ return {
83
+ workers,
84
+ reason: memoryBound < cpuBound ? "memory-bound" : "cpu-bound",
85
+ cpuBound,
86
+ memoryBound,
87
+ };
88
+ }
89
+ /** Sample the live machine. Separated from the policy so the policy stays testable. */
90
+ export function currentParsePoolSizing(env = process.env) {
91
+ return resolveParsePoolSize({
92
+ cpuCount: os.cpus().length,
93
+ totalMemoryBytes: os.totalmem(),
94
+ env,
95
+ });
96
+ }
@@ -0,0 +1,352 @@
1
+ import path from "node:path";
2
+ import { fileURLToPath } from "node:url";
3
+ import { Worker } from "node:worker_threads";
4
+ import { currentParsePoolSizing } from "./parse-pool-resources.js";
5
+ /**
6
+ * On by default; `KNODIN_DISABLE_PARSE_WORKERS=1` is the kill switch.
7
+ *
8
+ * Defaulted on only after the gate it was waiting for actually passed, and
9
+ * after the economics changed enough to justify it. When the pool was first
10
+ * wired, `parse_extract` was 10.9% of a full index and the pool bought 1.16x —
11
+ * not worth worker-thread complexity. Fixing the real bottleneck
12
+ * (EASFDC-8421) took a full index from 60.5s to 21.3s and left parsing at
13
+ * 31.6%, which raised the pool's own contribution to a measured 1.57x. The
14
+ * combined figure against the original baseline is 4.7x.
15
+ *
16
+ * Enabling it is safe rather than merely faster: sizing refuses to spawn on
17
+ * machines that cannot afford workers and returns null BEFORE any Worker is
18
+ * constructed, a worker crash retries once and then hands the file back for
19
+ * inline indexing, an exhausted pool drains its queue rather than stalling,
20
+ * and the graph is proven set-equivalent to the sequential build. Every one of
21
+ * those has a test.
22
+ *
23
+ * `KNODIN_ENABLE_PARSE_WORKERS` is still honoured so anyone who set it during
24
+ * the opt-in period keeps working, but it is now redundant.
25
+ */
26
+ export function parseWorkersEnabled(env = process.env) {
27
+ return env.KNODIN_DISABLE_PARSE_WORKERS !== "1";
28
+ }
29
+ /**
30
+ * Grammar identity for affinity purposes. Grammars are cached in module-level
31
+ * state inside the engine, so each worker holds a copy of every language it
32
+ * sees: measured at 7.4 MB for a TypeScript-only worker against ~50 MB for one
33
+ * that eventually touches every grammar. Keeping a worker on one extension is
34
+ * therefore a real memory lever, not a micro-optimization.
35
+ */
36
+ function languageKey(relativePath) {
37
+ return path.extname(relativePath).toLowerCase() || "<none>";
38
+ }
39
+ function defaultWorkerEntry() {
40
+ const extension = import.meta.url.endsWith(".ts") ? "ts" : "js";
41
+ return fileURLToPath(new URL(`./parse-worker.${extension}`, import.meta.url));
42
+ }
43
+ /**
44
+ * When running from TypeScript sources, a worker needs its own TS loader.
45
+ *
46
+ * A worker thread does NOT inherit the host's transform pipeline. Under a
47
+ * plain `tsx` process it happens to work because tsx registers a real Node
48
+ * loader; under Vitest it does not, and the worker dies with "Cannot find
49
+ * module .../index.js" while resolving this module's own `./index.js` import.
50
+ * The pool then silently falls back to the sequential path — correct results,
51
+ * but no parallelism, and no way to tell from the outside.
52
+ *
53
+ * Mirrors what `src/__tests__/support/node-subprocess.ts` already does for the
54
+ * child-process worker. Returns undefined for a compiled `.js` entry, where
55
+ * the import resolves natively and tsx may not be installed at all.
56
+ */
57
+ function workerExecArgv(entry) {
58
+ if (!entry.endsWith(".ts"))
59
+ return undefined;
60
+ try {
61
+ return ["--import", fileURLToPath(import.meta.resolve("tsx"))];
62
+ }
63
+ catch {
64
+ return undefined;
65
+ }
66
+ }
67
+ function realWorkerFactory() {
68
+ const entry = defaultWorkerEntry();
69
+ const execArgv = workerExecArgv(entry);
70
+ const worker = new Worker(entry, execArgv ? { execArgv } : undefined);
71
+ worker.unref();
72
+ return {
73
+ postMessage: (task) => worker.postMessage(task),
74
+ onMessage: (listener) => worker.on("message", listener),
75
+ onError: (listener) => worker.on("error", listener),
76
+ onExit: (listener) => worker.on("exit", listener),
77
+ terminate: async () => {
78
+ await worker.terminate();
79
+ },
80
+ };
81
+ }
82
+ /**
83
+ * One `run()` call's state machine.
84
+ *
85
+ * Deliberately a class of flat methods rather than nested closures: the same
86
+ * logic written inline nests worker callbacks inside a promise executor inside
87
+ * a method, which is both hard to follow and a real reviewability problem
88
+ * (Sonar S2004 flagged five separate spots). Methods keep every step at one
89
+ * level and let the slot be passed explicitly.
90
+ */
91
+ class PoolRun {
92
+ slots;
93
+ factory;
94
+ onOutcome;
95
+ nextId;
96
+ buckets = new Map();
97
+ remaining;
98
+ settled = false;
99
+ draining = false;
100
+ resolve;
101
+ reject;
102
+ constructor(slots, factory, tasks, onOutcome, nextId) {
103
+ this.slots = slots;
104
+ this.factory = factory;
105
+ this.onOutcome = onOutcome;
106
+ this.nextId = nextId;
107
+ for (const task of tasks) {
108
+ const key = languageKey(task.relativePath);
109
+ const bucket = this.buckets.get(key);
110
+ if (bucket)
111
+ bucket.push(task);
112
+ else
113
+ this.buckets.set(key, [task]);
114
+ }
115
+ this.remaining = tasks.length;
116
+ }
117
+ start() {
118
+ return new Promise((resolve, reject) => {
119
+ this.resolve = resolve;
120
+ this.reject = reject;
121
+ for (const slot of this.slots)
122
+ this.attach(slot);
123
+ for (const slot of this.slots)
124
+ this.pump(slot);
125
+ });
126
+ }
127
+ /** Prefer the slot's own language; fall back to the largest bucket. */
128
+ takeNext(slot) {
129
+ if (slot.affinity) {
130
+ const own = this.buckets.get(slot.affinity);
131
+ if (own && own.length > 0)
132
+ return own.pop() ?? null;
133
+ }
134
+ // Preferring affinity without enforcing it keeps a single-language
135
+ // repository from leaving most of the pool idle.
136
+ let bestKey = null;
137
+ let bestLength = 0;
138
+ for (const [key, bucket] of this.buckets) {
139
+ if (bucket.length > bestLength) {
140
+ bestKey = key;
141
+ bestLength = bucket.length;
142
+ }
143
+ }
144
+ if (!bestKey)
145
+ return null;
146
+ slot.affinity = bestKey;
147
+ return this.buckets.get(bestKey)?.pop() ?? null;
148
+ }
149
+ attach(slot) {
150
+ slot.handle.onMessage((response) => this.onResponse(slot, response));
151
+ slot.handle.onError((error) => this.onSlotFailure(slot, error.message));
152
+ slot.handle.onExit((code) => {
153
+ if (slot.inFlight)
154
+ this.onSlotFailure(slot, `worker exited with code ${code}`);
155
+ });
156
+ }
157
+ onResponse(slot, response) {
158
+ if (!slot.inFlight || response.id !== slot.inFlight.id)
159
+ return;
160
+ switch (response.kind) {
161
+ case "result":
162
+ this.finishOne(slot, { kind: "result", result: response.result });
163
+ return;
164
+ case "unparsed":
165
+ this.finishOne(slot, { kind: "unparsed" });
166
+ return;
167
+ case "skipped":
168
+ this.finishOne(slot, { kind: "skipped" });
169
+ return;
170
+ case "special":
171
+ this.finishOne(slot, { kind: "fallback", reason: "needs main-thread indexer" });
172
+ return;
173
+ default:
174
+ this.finishOne(slot, { kind: "fallback", reason: response.message });
175
+ return;
176
+ }
177
+ }
178
+ pump(slot) {
179
+ if (this.settled || slot.dead || slot.inFlight)
180
+ return;
181
+ const task = this.takeNext(slot);
182
+ if (!task)
183
+ return;
184
+ const id = this.nextId();
185
+ slot.inFlight = { task, id, attempt: 0 };
186
+ this.dispatch(slot, task, id);
187
+ }
188
+ dispatch(slot, task, id) {
189
+ try {
190
+ slot.handle.postMessage({
191
+ id,
192
+ absolutePath: task.absolutePath,
193
+ relativePath: task.relativePath,
194
+ repoPath: task.repoPath,
195
+ });
196
+ }
197
+ catch (error) {
198
+ this.onSlotFailure(slot, error instanceof Error ? error.message : String(error));
199
+ }
200
+ }
201
+ replaceSlot(slot) {
202
+ try {
203
+ slot.handle = this.factory();
204
+ slot.dead = false;
205
+ this.attach(slot);
206
+ }
207
+ catch {
208
+ slot.dead = true;
209
+ }
210
+ }
211
+ onSlotFailure(slot, reason) {
212
+ const flight = slot.inFlight;
213
+ slot.dead = true;
214
+ void slot.handle.terminate().catch(() => { });
215
+ if (!flight) {
216
+ // Died idle. Bring a replacement up so the pool keeps its width.
217
+ this.replaceSlot(slot);
218
+ if (!slot.dead)
219
+ this.pump(slot);
220
+ // A slot that dies while idle delivers no outcome, so nothing else
221
+ // would notice the pool has run out of workers.
222
+ else if (!this.anyLive())
223
+ void this.drainRemaining(reason);
224
+ return;
225
+ }
226
+ if (flight.attempt === 0) {
227
+ // One retry on a fresh worker: a crash is far more often about this
228
+ // particular file than about the worker.
229
+ this.replaceSlot(slot);
230
+ if (!slot.dead) {
231
+ const id = this.nextId();
232
+ slot.inFlight = { ...flight, attempt: 1, id };
233
+ this.dispatch(slot, flight.task, id);
234
+ return;
235
+ }
236
+ }
237
+ // Retried already, or no replacement available: give the file back so
238
+ // the caller indexes it inline. Never dropped.
239
+ this.finishOne(slot, { kind: "fallback", reason });
240
+ if (!slot.dead)
241
+ this.pump(slot);
242
+ }
243
+ finishOne(slot, outcome) {
244
+ const flight = slot.inFlight;
245
+ if (!flight)
246
+ return;
247
+ slot.inFlight = null;
248
+ this.remaining--;
249
+ void this.deliver(slot, flight.task, outcome);
250
+ }
251
+ async deliver(slot, task, outcome) {
252
+ try {
253
+ await this.onOutcome(task, outcome);
254
+ }
255
+ catch (error) {
256
+ this.fail(error instanceof Error ? error : new Error(String(error)));
257
+ return;
258
+ }
259
+ if (this.remaining === 0) {
260
+ this.finish();
261
+ return;
262
+ }
263
+ this.pump(slot);
264
+ if (!this.anyLive())
265
+ await this.drainRemaining("no parse worker available");
266
+ }
267
+ anyLive() {
268
+ return this.slots.some((slot) => !slot.dead);
269
+ }
270
+ /**
271
+ * Every worker is gone and could not be replaced. Hand back every task that
272
+ * has not been dispatched so the caller indexes them inline.
273
+ *
274
+ * Without this the run simply stops: queued work has no worker to claim it,
275
+ * nothing ever decrements `remaining`, and `run()` never settles. A pool
276
+ * that hangs is far worse than one that declines to help, and "never the
277
+ * reason a run fails" has to include "never the reason a run stalls".
278
+ */
279
+ async drainRemaining(reason) {
280
+ if (this.draining || this.settled)
281
+ return;
282
+ this.draining = true;
283
+ const pending = [];
284
+ for (const bucket of this.buckets.values())
285
+ pending.push(...bucket.splice(0));
286
+ for (const task of pending) {
287
+ this.remaining--;
288
+ try {
289
+ await this.onOutcome(task, { kind: "fallback", reason });
290
+ }
291
+ catch (error) {
292
+ this.fail(error instanceof Error ? error : new Error(String(error)));
293
+ return;
294
+ }
295
+ }
296
+ this.draining = false;
297
+ if (this.remaining === 0)
298
+ this.finish();
299
+ }
300
+ finish() {
301
+ if (this.settled)
302
+ return;
303
+ this.settled = true;
304
+ this.resolve();
305
+ }
306
+ fail(error) {
307
+ if (this.settled)
308
+ return;
309
+ this.settled = true;
310
+ this.reject(error);
311
+ }
312
+ }
313
+ /**
314
+ * Builds a pool, or returns `null` when the machine says not to.
315
+ *
316
+ * `null` is a normal outcome meaning "use the sequential path" — the decision
317
+ * is made here, before any worker exists, so an unsuitable machine never
318
+ * discovers the problem partway through an index.
319
+ */
320
+ export function createParsePool(options) {
321
+ const env = options?.env ?? process.env;
322
+ const sizing = options?.sizing ?? currentParsePoolSizing(env);
323
+ if (sizing.workers < 1)
324
+ return null;
325
+ const factory = options?.workerFactory ?? realWorkerFactory;
326
+ const slots = [];
327
+ try {
328
+ for (let i = 0; i < sizing.workers; i++)
329
+ slots.push({ handle: factory(), affinity: null, inFlight: null, dead: false });
330
+ }
331
+ catch {
332
+ // Spawning failed outright. Tear down whatever came up and let the
333
+ // caller stay sequential rather than run a half-sized pool.
334
+ for (const slot of slots)
335
+ void slot.handle.terminate().catch(() => { });
336
+ return null;
337
+ }
338
+ let nextId = 1;
339
+ const allocateId = () => nextId++;
340
+ return {
341
+ size: slots.length,
342
+ sizing,
343
+ async run(tasks, onOutcome) {
344
+ if (tasks.length === 0)
345
+ return;
346
+ await new PoolRun(slots, factory, tasks, onOutcome, allocateId).start();
347
+ },
348
+ async close() {
349
+ await Promise.all(slots.map((slot) => slot.handle.terminate().catch(() => { })));
350
+ },
351
+ };
352
+ }
@@ -0,0 +1 @@
1
+ export {};
@@ -0,0 +1,51 @@
1
+ import { parentPort } from "node:worker_threads";
2
+ import { extractGenericFileForIndex } from "./index.js";
3
+ /**
4
+ * Parse-pool worker entry.
5
+ *
6
+ * Owns read + parse + extract for one file at a time and posts back plain
7
+ * data. It never touches SQLite: the database handle is a native, non-
8
+ * transferable `node:sqlite` object, and exactly one writer on the main
9
+ * thread remains the invariant this whole design is built around.
10
+ *
11
+ * Verified during Step 4 that a worker inherits the tsx loader by default, so
12
+ * importing the TypeScript engine here works in development and against the
13
+ * compiled output alike without forwarding `execArgv`.
14
+ */
15
+ const port = parentPort;
16
+ if (!port)
17
+ throw new Error("knodin parse worker started without a parent port");
18
+ port.on("message", (task) => {
19
+ void (async () => {
20
+ let response;
21
+ try {
22
+ const extraction = await extractGenericFileForIndex(task.absolutePath, task.relativePath, task.repoPath);
23
+ switch (extraction.kind) {
24
+ case "result":
25
+ response = { id: task.id, kind: "result", result: extraction.result };
26
+ break;
27
+ case "unparsed":
28
+ response = { id: task.id, kind: "unparsed" };
29
+ break;
30
+ case "skipped":
31
+ response = { id: task.id, kind: "skipped" };
32
+ break;
33
+ default:
34
+ // `special` carries a live tree; hand the file back instead.
35
+ response = { id: task.id, kind: "special" };
36
+ break;
37
+ }
38
+ }
39
+ catch (error) {
40
+ // Serialize by hand. A thrown Error clones without its message on some
41
+ // paths, and losing the message turns a diagnosable failure into a
42
+ // mystery at exactly the moment the pool decides whether to retry.
43
+ response = {
44
+ id: task.id,
45
+ kind: "error",
46
+ message: error instanceof Error ? error.message : String(error),
47
+ };
48
+ }
49
+ port.postMessage(response);
50
+ })();
51
+ });
@@ -0,0 +1,23 @@
1
+ import fs from "node:fs";
2
+ /**
3
+ * Copy `source` to `destination`, preferring an OS-level copy-on-write clone
4
+ * (`clonefile()` on APFS, `FICLONE` on Btrfs/XFS) so the two files share disk
5
+ * blocks until one of them is later mutated. Falls back to an ordinary byte
6
+ * copy on any filesystem/platform that doesn't support cloning.
7
+ */
8
+ export function reflinkCopyFile(source, destination) {
9
+ try {
10
+ fs.copyFileSync(source, destination, fs.constants.COPYFILE_FICLONE_FORCE);
11
+ return { reflinked: true };
12
+ }
13
+ catch {
14
+ try {
15
+ fs.rmSync(destination, { force: true });
16
+ }
17
+ catch {
18
+ // Best-effort cleanup of a partial clone attempt; the plain copy below still succeeds or throws.
19
+ }
20
+ fs.copyFileSync(source, destination);
21
+ return { reflinked: false };
22
+ }
23
+ }
@@ -0,0 +1,116 @@
1
+ import { execFileSync } from "node:child_process";
2
+ import fs from "node:fs";
3
+ import path from "node:path";
4
+ import { createEngine } from "./index.js";
5
+ import { sealIndex } from "./seal.js";
6
+ import { resolveDbPath } from "./state-paths.js";
7
+ /**
8
+ * Orchestrates `knodin seal`: gather live health and git identity, then hand
9
+ * both to the pure sealing transform.
10
+ *
11
+ * Kept out of `seal.ts` so that module stays a leaf — the sealed READ path
12
+ * lives inside the engine, and a seal module that imported the engine back
13
+ * would make that circular.
14
+ */
15
+ /**
16
+ * Profiles are 58% of the SalesforceCI payload in 187 files. They are
17
+ * permission matrices; no design conversation reads one verbatim, and their
18
+ * structure is already captured as graph edges. Excluding them is what makes a
19
+ * 1Gi resident ceiling reachable — but it is recorded as a NAMED rule so a
20
+ * query about a profile answers "excluded by policy" rather than empty.
21
+ */
22
+ export const DEFAULT_SOURCE_EXCLUSIONS = [
23
+ { rule: "profiles-excluded", test: /\.profile-meta\.xml$|(^|\/)profiles\//i },
24
+ ];
25
+ function git(repoPath, args) {
26
+ try {
27
+ return execFileSync("git", args, {
28
+ cwd: repoPath,
29
+ encoding: "utf8",
30
+ stdio: ["ignore", "pipe", "ignore"],
31
+ }).trim();
32
+ }
33
+ catch {
34
+ return null;
35
+ }
36
+ }
37
+ /**
38
+ * Reduce an origin remote to canonical lowercase `owner/name`.
39
+ *
40
+ * The raw URL is NOT usable as identity, and the SSH form is the dangerous
41
+ * one rather than the obviously-broken one: `git@host:owner/name.git` splits
42
+ * into exactly two segments, so a consumer validating "must be owner/name"
43
+ * accepts it and then keys on `git@host:owner` and `name.git`. It passes the
44
+ * shape check while being semantically garbage, which is worse than the HTTPS
45
+ * form that simply fails.
46
+ *
47
+ * Returns null rather than a guess when the URL does not yield a clean pair —
48
+ * an identity a consumer cannot key on should be absent, not approximated.
49
+ */
50
+ export function canonicalRepositoryIdentity(remoteUrl) {
51
+ if (!remoteUrl)
52
+ return null;
53
+ const withoutSuffix = remoteUrl.trim().replace(/\.git$/i, "");
54
+ // Strip scheme, or the `git@host:` / `host:` prefix of the SCP-like form.
55
+ // The host may carry an SSH config alias (`github.com-company`), which is a
56
+ // local routing detail and must not reach identity.
57
+ const withoutHost = withoutSuffix
58
+ .replace(/^[a-z][a-z0-9+.-]*:\/\//i, "")
59
+ .replace(/^[^/]*@/, "")
60
+ .replace(/^[^/:]+[:/]/, "");
61
+ const segments = withoutHost.split("/").filter(Boolean);
62
+ if (segments.length < 2)
63
+ return null;
64
+ // Take the LAST two: nested group paths (GitLab subgroups) still reduce to
65
+ // the owner/name pair a consumer keys on.
66
+ return segments.slice(-2).join("/").toLowerCase();
67
+ }
68
+ export async function runSeal(options) {
69
+ // Resolve symlinks, not just relative segments. The engine canonicalises the
70
+ // repository path when it decides where state lives, so a caller passing a
71
+ // symlinked path (every macOS temp directory: /var -> /private/var) would
72
+ // have sealing look for the database somewhere the index never wrote it and
73
+ // report `database-missing` for a repository that is perfectly well indexed.
74
+ const requested = path.resolve(options.repoPath);
75
+ const repoPath = fs.existsSync(requested) ? fs.realpathSync(requested) : requested;
76
+ const engine = createEngine({ watcher: "disabled" });
77
+ try {
78
+ const health = await engine.status(repoPath, { audit: "deep" });
79
+ const remoteUrl = git(repoPath, ["remote", "get-url", "origin"]);
80
+ const commit = git(repoPath, ["rev-parse", "HEAD"]);
81
+ // `--symbolic-full-name` so a detached HEAD reports null rather than the
82
+ // literal string "HEAD", which would read as a real ref downstream.
83
+ const ref = git(repoPath, ["rev-parse", "--symbolic-full-name", "HEAD"]);
84
+ const excluded = options.includeExcluded
85
+ ? undefined
86
+ : (relativePath) => DEFAULT_SOURCE_EXCLUSIONS.some((entry) => entry.test.test(relativePath));
87
+ return sealIndex({
88
+ repoPath,
89
+ sourceDatabasePath: resolveDbPath(repoPath),
90
+ outputPath: path.resolve(options.outputPath),
91
+ stripEmbeddings: options.stripEmbeddings,
92
+ excludeSource: excluded,
93
+ health: {
94
+ status: health.status,
95
+ pendingPaths: health.freshness.workingTree.pendingPaths ?? 0,
96
+ freshness: health.freshness.state,
97
+ },
98
+ // Canonical owner/name from the remote. `status` reports the local
99
+ // checkout path, which is not identity a consumer can key on — two
100
+ // machines cloning the same repository would attest two different
101
+ // "repositories". The raw URL is kept separately for provenance.
102
+ identity: canonicalRepositoryIdentity(remoteUrl) ?? health.repo,
103
+ remoteUrl,
104
+ commit,
105
+ ref: ref === "HEAD" ? null : ref,
106
+ // Files the graph knows about that are GONE from disk. Files that were
107
+ // MODIFIED do not appear here — those are caught per-file by the drift
108
+ // check inside `sealIndex`, which compares against `index_state`
109
+ // directly rather than trusting an aggregate.
110
+ dirtyPaths: health.missing.files,
111
+ });
112
+ }
113
+ finally {
114
+ await engine.close();
115
+ }
116
+ }