knodin 0.9.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,172 @@
1
+ import childProcess from "node:child_process";
2
+ import fs from "node:fs";
3
+ import path from "node:path";
4
+ import { Database } from "./engine/sqlite.js";
5
+ import { resolveDbPath } from "./engine/state-paths.js";
6
+ import { deriveSharedOverlay } from "./shared-index/overlay.js";
7
+ import { commitRelation } from "./shared-index/selection.js";
8
+ import { KNODIN_VERSION } from "./version.js";
9
+ import { inspectWorktrees } from "./worktree-lifecycle.js";
10
+ const OPT_OUT_ENV = "KNODIN_DISABLE_WORKTREE_SEED";
11
+ function readHead(repoPath) {
12
+ try {
13
+ return childProcess
14
+ .execFileSync("git", ["rev-parse", "HEAD"], {
15
+ cwd: repoPath,
16
+ encoding: "utf8",
17
+ stdio: ["ignore", "pipe", "ignore"],
18
+ })
19
+ .trim();
20
+ }
21
+ catch {
22
+ return null;
23
+ }
24
+ }
25
+ /** Opens `dbPath` read-only and confirms schema + build compatibility with this process. */
26
+ function siblingIsCompatible(dbPath, schemaVersion) {
27
+ let db = null;
28
+ try {
29
+ db = new Database(dbPath, { readonly: true });
30
+ const version = db.query("PRAGMA user_version").get()?.user_version ?? 0;
31
+ if (version !== schemaVersion)
32
+ return null;
33
+ const versionRow = db
34
+ .query("SELECT value FROM meta WHERE key = ?")
35
+ .get("knodinVersion");
36
+ if (!versionRow || versionRow.value !== KNODIN_VERSION)
37
+ return null;
38
+ const headRow = db
39
+ .query("SELECT value FROM meta WHERE key = ?")
40
+ .get("lastIndexedHead");
41
+ return headRow?.value || null;
42
+ }
43
+ catch {
44
+ return null;
45
+ }
46
+ finally {
47
+ db?.close();
48
+ }
49
+ }
50
+ /**
51
+ * Finds the cheapest-to-reconcile already-indexed sibling worktree of `repoPath`,
52
+ * or `null` when no eligible sibling exists (single-worktree repo, no healthy
53
+ * siblings, or none pass the schema/build compatibility gate).
54
+ */
55
+ export async function selectSeedSibling(repoPath, schemaVersion, status) {
56
+ const resolved = path.resolve(repoPath);
57
+ const head = readHead(resolved);
58
+ if (!head)
59
+ return null;
60
+ let worktrees;
61
+ try {
62
+ worktrees = (await inspectWorktrees(resolved, status)).worktrees;
63
+ }
64
+ catch {
65
+ return null;
66
+ }
67
+ if (worktrees.length < 2)
68
+ return null;
69
+ const eligible = worktrees.filter((worktree) => path.resolve(worktree.path) !== resolved &&
70
+ worktree.graphInitialized &&
71
+ worktree.freshness !== "removed");
72
+ const ranked = eligible
73
+ .map((worktree) => {
74
+ const dbPath = resolveDbPath(worktree.path);
75
+ const indexedHead = siblingIsCompatible(dbPath, schemaVersion);
76
+ if (!indexedHead)
77
+ return null;
78
+ const relation = commitRelation(resolved, head, indexedHead);
79
+ const distance = relation ? relation.distance : Number.MAX_SAFE_INTEGER;
80
+ return { worktree, dbPath, indexedHead, distance };
81
+ })
82
+ .filter((entry) => entry !== null)
83
+ .sort((left, right) => left.distance - right.distance);
84
+ const best = ranked[0];
85
+ if (!best)
86
+ return null;
87
+ return {
88
+ siblingPath: best.worktree.path,
89
+ siblingDbPath: best.dbPath,
90
+ siblingIndexedHead: best.indexedHead,
91
+ };
92
+ }
93
+ /**
94
+ * Seeds a brand-new worktree's graph database from an already-indexed sibling's
95
+ * clean baseline (OS-level copy-on-write clone where supported) and reconciles it
96
+ * to this worktree's actual HEAD/working tree, instead of building from scratch.
97
+ */
98
+ export async function seedWorktreeIndex(repoPath, engine, schemaVersion) {
99
+ const resolved = path.resolve(repoPath);
100
+ const sibling = await selectSeedSibling(resolved, schemaVersion, (p) => engine.status(p));
101
+ if (!sibling)
102
+ return { seeded: false, reason: "no eligible sibling worktree" };
103
+ let candidate;
104
+ try {
105
+ candidate = await engine.createCandidate(resolved, {
106
+ sourceDatabasePath: sibling.siblingDbPath,
107
+ sourceLabel: sibling.siblingPath,
108
+ });
109
+ const overlay = deriveSharedOverlay(resolved, sibling.siblingIndexedHead);
110
+ const reconciliation = await engine.reconcileCandidate(candidate, resolved, overlay.changedPaths);
111
+ const audit = await engine.auditCandidate(candidate, { mode: "deep" });
112
+ if (!audit.healthy) {
113
+ return { seeded: false, reason: audit.issues.join(",") || "candidate audit failed" };
114
+ }
115
+ await engine.promoteCandidate(candidate);
116
+ const promoted = candidate;
117
+ candidate = undefined;
118
+ const health = await engine.status(resolved, { audit: "deep", ignoreActiveOperation: true });
119
+ return {
120
+ seeded: true,
121
+ reflinked: promoted.reflinked ?? false,
122
+ seededFrom: sibling.siblingPath,
123
+ reconciledPaths: reconciliation.reconciled,
124
+ status: health,
125
+ };
126
+ }
127
+ catch (error) {
128
+ return { seeded: false, reason: error instanceof Error ? error.message : "seed failed" };
129
+ }
130
+ finally {
131
+ if (candidate) {
132
+ try {
133
+ await engine.discardCandidate(candidate);
134
+ }
135
+ catch {
136
+ // Best-effort cleanup; a fresh candidate directory is harmless to leave for the next attempt.
137
+ }
138
+ }
139
+ }
140
+ }
141
+ function toIndexResult(status, reconciledPaths) {
142
+ return {
143
+ indexed: reconciledPaths,
144
+ unchanged: [],
145
+ verification: {
146
+ status: status.status,
147
+ issueCount: status.missing.files.length + status.missing.records.length,
148
+ missing: status.missing,
149
+ verifiedAt: status.verification.verifiedAt,
150
+ },
151
+ };
152
+ }
153
+ /**
154
+ * Drop-in replacement for `(repo, options) => engine.index(repo, undefined, false, options)`
155
+ * used by every `knodin init`-family command: for a brand-new worktree, tries to
156
+ * seed-and-reconcile from a sibling before falling back to a full from-scratch
157
+ * index. For a worktree that already has a database (repair/update path),
158
+ * behaves exactly like the wrapped full index call.
159
+ */
160
+ export function indexOrSeed(engine, schemaVersion) {
161
+ const fullIndex = (target, options) => engine.index(target, undefined, false, options);
162
+ return async (target, options) => {
163
+ if (fs.existsSync(resolveDbPath(target)))
164
+ return fullIndex(target, options);
165
+ if (process.env[OPT_OUT_ENV] === "1")
166
+ return fullIndex(target, options);
167
+ const seed = await seedWorktreeIndex(target, engine, schemaVersion);
168
+ if (seed.seeded)
169
+ return toIndexResult(seed.status, seed.reconciledPaths);
170
+ return fullIndex(target, options);
171
+ };
172
+ }
@@ -114,6 +114,15 @@ the signed output size and digest, rejects trailing compressed data, validates
114
114
  the SQLite header and internal commit/schema/fingerprint, reconciles final Git
115
115
  and filesystem truth, deep-audits, and only then promotes.
116
116
 
117
+ A brand-new worktree's distinct database may itself be populated by cloning an
118
+ already-indexed sibling worktree's baseline (an OS-level copy-on-write file
119
+ clone where the filesystem supports it, falling back to a plain copy
120
+ otherwise) instead of a from-scratch build — see `knodin init`'s seed-and-
121
+ reconcile path in `src/worktree-seed.ts`. That clone is reconciled to the new
122
+ worktree's actual state through this same overlay-diff machinery before
123
+ promotion, so the "distinct, writable" guarantee above is unchanged; only how
124
+ the initial bytes are produced differs.
125
+
117
126
  ## Operator surfaces
118
127
 
119
128
  - `knodin shared status` is network-free. `--probe` is the explicit S3/auth
@@ -0,0 +1,127 @@
1
+ # knodin 0.10.0
2
+
3
+ This release makes a full index roughly 2.8x faster, turns parallel parsing on
4
+ by default, adds sealed artifacts so a repository can be queried with no
5
+ checkout at all, and makes unreadable source distinguishable from source that
6
+ does not exist. It also makes `knodin init` fast on a brand-new git worktree by
7
+ seeding its graph from an already-indexed sibling.
8
+
9
+ ## Sealed artifacts: query a repository with no checkout
10
+
11
+ An ordinary `.knodin/db.sqlite` stores `filePath` plus line ranges and reads the
12
+ bytes off a live working tree, so it cannot answer anything on a machine that
13
+ has no clone. `knodin seal` embeds those bytes and stamps an attestation
14
+ describing exactly which tree they came from.
15
+
16
+ - `knodin seal --output <path.sqlite>` produces a portable artifact. Sealing is
17
+ a post-index transform, never index-time work: an ordinary user with a
18
+ checkout should not pay for source already on disk, so opting in is
19
+ structural rather than a matter of flag discipline. The live database is
20
+ copied, never mutated.
21
+ - `knodin sealed <artifact> [symbol]` inspects or queries an artifact, and the
22
+ same capability is available as a `sealed` operation on the MCP gateway. Both
23
+ are read-only; creating an artifact stays CLI-only because it writes to disk.
24
+ - The attestation carries schema version, knodin version, canonical
25
+ `owner/name` repository identity, the exact commit, **and which ref** was
26
+ sealed — a citation cannot be checked without knowing whether it came from
27
+ `main` or from whatever production is running.
28
+ - Refusals are machine-readable codes rather than prose, because a caller
29
+ branches on them: `database-missing`, `index-unhealthy`, `working-tree-dirty`,
30
+ `commit-divergent`, `no-commit`, `source-drift`.
31
+ - `source-drift` refuses the whole seal when a file changed between indexing
32
+ and sealing. Embedded bytes that disagree with recorded line ranges make
33
+ `explain` return the wrong function's body with full confidence and no
34
+ omission — a false positive that looks correct, which is worse than any
35
+ absence. Two checks run: `mtimeMs`/`size` against `index_state`, and a bounds
36
+ check that recorded line ranges fit the embedded bytes, which holds without
37
+ consulting a clock at all.
38
+ - **Known residual, stated rather than implied:** a same-size edit that also
39
+ preserves line count and lands inside the mtime comparison window is still
40
+ invisible. Closing it needs a content hash in `index_state`, which forces a
41
+ full reindex and is tracked separately.
42
+ - Uncovered content reads as absent, never as an empty string. An empty string
43
+ is indistinguishable from an empty file and downstream reads as "this symbol
44
+ has no body" rather than "this was not covered".
45
+ - Size: profiles are excluded by default under the named rule
46
+ `profiles-excluded` (58% of one Salesforce payload in 187 files, already
47
+ captured as graph edges); blobs are zstd-compressed **inside** the database,
48
+ since SQLite does not transparently compress and a consumer holds the
49
+ decompressed file resident; storage is content-addressed so duplicate files
50
+ cost one blob; and embeddings are stripped unless `--keep-embeddings` is
51
+ passed, with semantic search then reporting itself unavailable rather than
52
+ returning zero results.
53
+ - Measured on this repository: 832 files, 6.05 MB of source compressing to
54
+ 2.05 MB (2.95x), a 17 MB artifact, `integrity_check: ok`.
55
+ - Opening an artifact requires `mode: "sealed"`. Without it the engine
56
+ reconciles the graph against an empty mount, prunes every symbol, and answers
57
+ "not found" for content the artifact is carrying.
58
+
59
+ ## A full index is about 2.8x faster
60
+
61
+ - Scoping an FTS trigger to the columns it actually mirrors, and removing a
62
+ whole-body string build from symbol identity, took a full index of this
63
+ repository from 60.5s to 21.3s.
64
+ - Salesforce metadata is now symbolized by kind rather than as one
65
+ undifferentiated `salesforce-metadata` symbol carrying identical boilerplate,
66
+ so FTS and embeddings over it mean something for the first time.
67
+
68
+ ## Parallel parsing is on by default
69
+
70
+ - The parse pool now runs unless `KNODIN_DISABLE_PARSE_WORKERS=1` is set. Pool
71
+ size is derived from total memory and core count, and an unsuitable host
72
+ silently keeps the previous sequential behaviour rather than discovering the
73
+ problem partway through an index.
74
+ - Files that need a main-thread indexer are kept off the pool. `.page`,
75
+ `.component`, `.clp`, `.ws`, `.xml`, `.sql`, `.pkb`, `.pks` and `.prisma`
76
+ classify as generic but hand off to indexers that write to the database
77
+ themselves, and routing them to a worker made a Visualforce controller edge
78
+ resolve on some runs and not others.
79
+
80
+ ## Unreadable source is distinguishable from missing source
81
+
82
+ - Source reads report an availability state instead of collapsing every failure
83
+ into `""`. A file that is gone, a file that could not be read this time, and a
84
+ symbol that does not exist are now three different answers.
85
+ - `dead_code` no longer reports a symbol as dead because its file could not be
86
+ read, and a missing file no longer aborts an entire `map` or `explain`.
87
+
88
+ ## Faster `knodin init` on a new worktree
89
+
90
+ - A brand-new worktree's graph database is seeded from the cheapest-to-
91
+ reconcile already-indexed sibling worktree of the same repository, cloned
92
+ via an OS-level copy-on-write file clone where the filesystem supports it
93
+ (plain-copy fallback otherwise), then reconciled to the new worktree's
94
+ actual committed/staged/working/untracked state through the existing
95
+ candidate allocate/reconcile/audit/promote pipeline before atomic
96
+ promotion.
97
+ - Seeding is opportunistic: no eligible sibling, a schema/build-compatibility
98
+ mismatch, or any candidate-audit failure falls straight back to a full
99
+ from-scratch index, with no error surfaced to the caller. Every git
100
+ lifecycle operation (pull, reset, merge, rebase, checkout, stash, clean)
101
+ on a seeded worktree is handled by the same incremental reconciliation and
102
+ hook machinery as any other worktree.
103
+ - Measured on this repository (810 files, 5,086 symbols): `knodin init` on a
104
+ new worktree with an eligible sibling completed in roughly 5 seconds versus
105
+ 30-40 seconds for a from-scratch build.
106
+ - Fixes a latent bug in the candidate reconciliation path: a candidate
107
+ populated from a foreign source (a downloaded shared-index artifact, or now
108
+ another worktree's database) kept `index_state` mtimes from wherever that
109
+ source was built, so every file outside the explicit changed-path set could
110
+ trip the deep audit's drift check once source and target were genuinely
111
+ different checkouts. `reconcileCandidate` now refreshes that bookkeeping
112
+ (a stat, not a re-index) for known files outside the changed-path set,
113
+ which also benefits the existing shared-index restore path.
114
+ - Set `KNODIN_DISABLE_WORKTREE_SEED=1` to always force a full from-scratch
115
+ index.
116
+
117
+ No existing worktree's database is touched by that change; it only affects the
118
+ code path for a worktree whose `.knodin/db.sqlite` does not yet exist.
119
+
120
+ ## Also in this release
121
+
122
+ - Five benchmark runners no longer call `.replace("Z", "Z")`, a no-op that
123
+ CodeQL correctly flagged as `js/identity-replacement`. Output filenames are
124
+ byte-identical.
125
+ - Fixed an `update-executor` test that failed roughly one run in three: a
126
+ version probe raced a 10-second timeout, and the probe timeout is now
127
+ injectable so the suite does not depend on process-spawn latency.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "knodin",
3
- "version": "0.9.0",
3
+ "version": "0.10.0",
4
4
  "knodin": {
5
5
  "compatibility": "breaking"
6
6
  },
@@ -71,6 +71,7 @@
71
71
  "docs/releases/0.8.6.md",
72
72
  "docs/releases/0.8.7.md",
73
73
  "docs/releases/0.9.0.md",
74
+ "docs/releases/0.10.0.md",
74
75
  "docs/assets/knodin-favicon.svg",
75
76
  "docs/SYSTEMS-AND-RELATIONSHIPS.md",
76
77
  "docs/TELEMETRY.md",