knodin 0.9.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/bin/cli.js +35 -7
- package/dist/fixtures/update-verification/fixture.json +1 -1
- package/dist/src/cli-model.js +14 -0
- package/dist/src/engine/index.js +919 -355
- package/dist/src/engine/parse-pool-resources.js +96 -0
- package/dist/src/engine/parse-pool.js +352 -0
- package/dist/src/engine/parse-protocol.js +1 -0
- package/dist/src/engine/parse-worker.js +51 -0
- package/dist/src/engine/reflink-copy.js +23 -0
- package/dist/src/engine/seal-command.js +116 -0
- package/dist/src/engine/seal.js +270 -0
- package/dist/src/engine/sealed-open.js +116 -0
- package/dist/src/engine/sealed-query.js +49 -0
- package/dist/src/init.js +29 -4
- package/dist/src/shared-index/selection.js +1 -1
- package/dist/src/tools/knodin-tools.js +16 -3
- package/dist/src/update-executor.js +29 -10
- package/dist/src/worktree-seed.js +172 -0
- package/docs/SHARED-INDEX-CONTRACT.md +9 -0
- package/docs/releases/0.10.0.md +127 -0
- package/package.json +2 -1
package/dist/src/engine/index.js
CHANGED
|
@@ -32,10 +32,16 @@ import { allocateCandidate, assertCandidate, discardCandidateFiles, promoteCandi
|
|
|
32
32
|
import { computeSimilarity, generateEmbedding, generateEmbeddings, } from "./embeddings.js";
|
|
33
33
|
import { walkRepoFiles } from "./file-walker.js";
|
|
34
34
|
import { clearGitHistorySignalCache, collectGitHistorySignals, } from "./git-history.js";
|
|
35
|
+
// Runtime import, but not a cycle: parse-pool imports only TYPES from here,
|
|
36
|
+
// which erase at compile time. parse-worker's runtime import of this module
|
|
37
|
+
// resolves inside the worker thread, never in this one.
|
|
38
|
+
import { createParsePool, parseWorkersEnabled } from "./parse-pool.js";
|
|
35
39
|
import { beginPerfPhase, measurePerfPhase, measurePerfPhaseSync } from "./perf.js";
|
|
36
40
|
import { isIndexablePath, makeWatchIgnorePredicate } from "./prune.js";
|
|
41
|
+
import { reflinkCopyFile } from "./reflink-copy.js";
|
|
37
42
|
import { readSarifLog, SARIF_DEFAULT_LIMITS, } from "./sarif-import.js";
|
|
38
43
|
import { readScipIndex, SCIP_DEFAULT_LIMITS, } from "./scip-import.js";
|
|
44
|
+
import { listSealedFiles, readSealedSource } from "./seal.js";
|
|
39
45
|
import { isIndexableSourcePath } from "./source-policy.js";
|
|
40
46
|
import { Database } from "./sqlite.js";
|
|
41
47
|
import { lookupMirror, mayWriteToRepository, resolveDbPath, resolveStateDir, } from "./state-paths.js";
|
|
@@ -501,10 +507,13 @@ function persistSymbolIdentities(db, repoPath, filePaths) {
|
|
|
501
507
|
}
|
|
502
508
|
sourceLines.set(row.filePath, lines);
|
|
503
509
|
}
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
|
|
510
|
+
// Only the symbol's FIRST line is used. This used to slice the whole
|
|
511
|
+
// body, join it into one string and split that back apart to take
|
|
512
|
+
// element zero — building the entire declaration text of every symbol
|
|
513
|
+
// to read one line of it, which cost 2.9ms per symbol on a real
|
|
514
|
+
// corpus. `lines` comes from `split(/\r?\n/)`, so no element contains
|
|
515
|
+
// a newline and indexing is exactly equivalent.
|
|
516
|
+
const declaration = (lines[Math.max(0, row.startLine - 1)] ?? "")
|
|
508
517
|
// `(?<!\s)` pins the match to the start of a whitespace run; without it
|
|
509
518
|
// every position inside the run rescans it, which is quadratic.
|
|
510
519
|
//
|
|
@@ -784,8 +793,8 @@ function freshnessMechanismFor(repoPath, policy) {
|
|
|
784
793
|
}
|
|
785
794
|
// Bumped whenever any repo is (re)indexed, so cached community-detection results
|
|
786
795
|
// below can be invalidated cheaply instead of recomputed on every call.
|
|
787
|
-
export const KNODIN_SCHEMA_VERSION =
|
|
788
|
-
const LAST_REBUILD_SCHEMA_VERSION =
|
|
796
|
+
export const KNODIN_SCHEMA_VERSION = 24;
|
|
797
|
+
const LAST_REBUILD_SCHEMA_VERSION = 24;
|
|
789
798
|
let indexGeneration = 0;
|
|
790
799
|
const resourceReachabilityCache = new Map();
|
|
791
800
|
const RESOURCE_REACHABILITY_CACHE_LIMIT = 32;
|
|
@@ -1126,6 +1135,19 @@ export function getSearchCacheStats() {
|
|
|
1126
1135
|
communityLookupCacheEntries: searchCommunityLookupCache.size,
|
|
1127
1136
|
};
|
|
1128
1137
|
}
|
|
1138
|
+
/**
|
|
1139
|
+
* How many repositories still have a live engine owner in this process.
|
|
1140
|
+
*
|
|
1141
|
+
* The process-global caches are released only when this reaches zero (see
|
|
1142
|
+
* `close()`), so a test asserting that those caches emptied is really
|
|
1143
|
+
* asserting that it held the last engine. The suite shares one process per
|
|
1144
|
+
* shard (`isolate: false`), where that is a property of the whole shard rather
|
|
1145
|
+
* than of any single spec — hence exposing it, so such an assertion can state
|
|
1146
|
+
* its precondition instead of depending on test ordering.
|
|
1147
|
+
*/
|
|
1148
|
+
export function getOpenRepositoryEngineCount() {
|
|
1149
|
+
return repositoryEngineOwners.size;
|
|
1150
|
+
}
|
|
1129
1151
|
/** Reset C46 work counters without changing cached production state. */
|
|
1130
1152
|
export function resetSearchCacheStats() {
|
|
1131
1153
|
for (const key of Object.keys(searchCacheStats)) {
|
|
@@ -1286,12 +1308,8 @@ function computeFlows(db, resolvedRepoPath) {
|
|
|
1286
1308
|
const isExported = (relFile, name) => {
|
|
1287
1309
|
let src = sourceCache.get(relFile);
|
|
1288
1310
|
if (src === undefined) {
|
|
1289
|
-
|
|
1290
|
-
|
|
1291
|
-
}
|
|
1292
|
-
catch {
|
|
1293
|
-
src = null;
|
|
1294
|
-
}
|
|
1311
|
+
const read = readWholeSource(resolvedRepoPath, relFile);
|
|
1312
|
+
src = read.availability.state === "available" ? read.text : null;
|
|
1295
1313
|
sourceCache.set(relFile, src);
|
|
1296
1314
|
}
|
|
1297
1315
|
if (!src)
|
|
@@ -3181,15 +3199,18 @@ function resolveImportPath(sourceFile, importSource, repoPath) {
|
|
|
3181
3199
|
// Extensionless specifier: try appending each known source extension.
|
|
3182
3200
|
for (const e of RESOLVE_EXTS)
|
|
3183
3201
|
candidates.push(absolutePath + e);
|
|
3202
|
+
// Probed through the resolver, not the filesystem: a sealed repository has
|
|
3203
|
+
// no working tree, and `fs.existsSync` there would fail every candidate and
|
|
3204
|
+
// silently report the module unresolvable rather than erroring.
|
|
3184
3205
|
for (const c of candidates) {
|
|
3185
|
-
if (
|
|
3206
|
+
if (sourceFileExists(repoPath, path.relative(repoPath, c))) {
|
|
3186
3207
|
return path.relative(repoPath, c);
|
|
3187
3208
|
}
|
|
3188
3209
|
}
|
|
3189
3210
|
// Directory import -> its index file.
|
|
3190
3211
|
for (const e of RESOLVE_EXTS) {
|
|
3191
3212
|
const p = path.join(absolutePath, `index${e}`);
|
|
3192
|
-
if (
|
|
3213
|
+
if (sourceFileExists(repoPath, path.relative(repoPath, p))) {
|
|
3193
3214
|
return path.relative(repoPath, p);
|
|
3194
3215
|
}
|
|
3195
3216
|
}
|
|
@@ -3213,11 +3234,12 @@ function getPackageMap(repoPath) {
|
|
|
3213
3234
|
return cached.value;
|
|
3214
3235
|
const map = new Map();
|
|
3215
3236
|
let cacheable = true;
|
|
3216
|
-
for (const rel of
|
|
3217
|
-
accept: (file) => path.posix.basename(file) === "package.json",
|
|
3218
|
-
})) {
|
|
3237
|
+
for (const rel of listSourceFiles(repoPath, (file) => path.posix.basename(file) === "package.json")) {
|
|
3219
3238
|
try {
|
|
3220
|
-
const
|
|
3239
|
+
const manifest = readWholeSource(repoPath, rel);
|
|
3240
|
+
if (manifest.availability.state !== "available")
|
|
3241
|
+
continue;
|
|
3242
|
+
const pkg = JSON.parse(manifest.text);
|
|
3221
3243
|
if (pkg && typeof pkg.name === "string") {
|
|
3222
3244
|
const dir = path.posix.dirname(rel);
|
|
3223
3245
|
map.set(pkg.name, {
|
|
@@ -3250,7 +3272,7 @@ function resolvePackageEntry(repoPath, dir, pkg) {
|
|
|
3250
3272
|
}
|
|
3251
3273
|
for (const rel of ["src/index.ts", "src/index.tsx", "src/index.js", "index.ts", "index.js"]) {
|
|
3252
3274
|
const p = path.join(dir, rel);
|
|
3253
|
-
if (
|
|
3275
|
+
if (sourceFileExists(repoPath, p))
|
|
3254
3276
|
return p;
|
|
3255
3277
|
}
|
|
3256
3278
|
return null;
|
|
@@ -3396,10 +3418,10 @@ function findClassFilePath(className, repoPath) {
|
|
|
3396
3418
|
* LWC claim, so the LWC path uses this stricter resolver instead.
|
|
3397
3419
|
*/
|
|
3398
3420
|
function findUniqueApexClassFilePath(className, repoPath) {
|
|
3399
|
-
|
|
3400
|
-
|
|
3401
|
-
|
|
3402
|
-
});
|
|
3421
|
+
// Enumerated through the resolver so a sealed repository answers from its
|
|
3422
|
+
// coverage. Still bounded to two matches, because "unique" is the whole
|
|
3423
|
+
// contract: a second match must resolve to null rather than a first hit.
|
|
3424
|
+
const matches = listSourceFiles(repoPath, (file) => path.posix.basename(file) === `${className}.cls`).slice(0, 2);
|
|
3403
3425
|
return matches.length === 1 ? matches[0] : null;
|
|
3404
3426
|
}
|
|
3405
3427
|
/**
|
|
@@ -3440,7 +3462,7 @@ function experienceSiteFile(sitePath, category, value, repoPath) {
|
|
|
3440
3462
|
return null;
|
|
3441
3463
|
const extension = category === "assets" ? "" : ".json";
|
|
3442
3464
|
const target = `${sitePath}/${category}/${name}${extension}`;
|
|
3443
|
-
return
|
|
3465
|
+
return sourceFileExists(repoPath, target) ? target : null;
|
|
3444
3466
|
}
|
|
3445
3467
|
function findExperienceComponentFile(value, repoPath) {
|
|
3446
3468
|
if (typeof value !== "string")
|
|
@@ -3451,7 +3473,7 @@ function findExperienceComponentFile(value, repoPath) {
|
|
|
3451
3473
|
return (findSalesforceLwcComponentFile(component, repoPath) ??
|
|
3452
3474
|
(() => {
|
|
3453
3475
|
const candidate = `force-app/main/default/aura/${component}/${component}.cmp`;
|
|
3454
|
-
return
|
|
3476
|
+
return sourceFileExists(repoPath, candidate) ? candidate : null;
|
|
3455
3477
|
})());
|
|
3456
3478
|
}
|
|
3457
3479
|
/**
|
|
@@ -3489,7 +3511,7 @@ function auraBundleFile(relativePath) {
|
|
|
3489
3511
|
function findLwcComponentFile(fromFile, componentName, repoPath) {
|
|
3490
3512
|
const lwcRoot = path.dirname(path.dirname(fromFile));
|
|
3491
3513
|
const candidate = path.join(lwcRoot, componentName, `${componentName}.js`);
|
|
3492
|
-
return
|
|
3514
|
+
return sourceFileExists(repoPath, candidate) ? candidate : null;
|
|
3493
3515
|
}
|
|
3494
3516
|
function findApexMethodFilePath(className, methodName, repoPath) {
|
|
3495
3517
|
const classFile = findUniqueApexClassFilePath(className, repoPath);
|
|
@@ -3551,13 +3573,10 @@ function findUniqueInvocableApexMethod(className, methodName, repoPath) {
|
|
|
3551
3573
|
const file = findUniqueApexClassFilePath(className, repoPath);
|
|
3552
3574
|
if (!file)
|
|
3553
3575
|
return null;
|
|
3554
|
-
|
|
3555
|
-
|
|
3556
|
-
source = fs.readFileSync(path.join(repoPath, file), "utf8");
|
|
3557
|
-
}
|
|
3558
|
-
catch {
|
|
3576
|
+
const read = readWholeSource(repoPath, file);
|
|
3577
|
+
if (read.availability.state !== "available")
|
|
3559
3578
|
return null;
|
|
3560
|
-
|
|
3579
|
+
const source = read.text;
|
|
3561
3580
|
const matches = [];
|
|
3562
3581
|
for (const match of source.matchAll(/@InvocableMethod\b[\s\S]{0,300}?\b(?:global|public)\s+static\s(?:\s|[^({;]*?(?<!\s))\s+([a-z]\w*)\s*\(/gi)) {
|
|
3563
3582
|
if (!isExecutablePosition(source, match.index ?? 0) || match[1] !== methodName)
|
|
@@ -3597,7 +3616,7 @@ function findSalesforceLwcComponentFile(componentName, repoPath) {
|
|
|
3597
3616
|
if (!/^[A-Za-z]\w*$/.test(componentName))
|
|
3598
3617
|
return null;
|
|
3599
3618
|
const candidate = `force-app/main/default/lwc/${componentName}/${componentName}.js`;
|
|
3600
|
-
return
|
|
3619
|
+
return sourceFileExists(repoPath, candidate) ? candidate : null;
|
|
3601
3620
|
}
|
|
3602
3621
|
function apexMethodIsDeclared(content, methodName) {
|
|
3603
3622
|
const escaped = methodName.replace(/[.*+?^${}()|[\]\\]/g, String.raw `\$&`);
|
|
@@ -3753,6 +3772,71 @@ function extractApexPlatformDependencies(content, repoPath) {
|
|
|
3753
3772
|
return dependencies;
|
|
3754
3773
|
}
|
|
3755
3774
|
/** Index C20's static, deployable Salesforce DX metadata without interpreting formulas or runtime values. */
|
|
3775
|
+
/** Symbol kinds emitted for Salesforce DX metadata, one per metadata type. */
|
|
3776
|
+
export const SALESFORCE_METADATA_KINDS = [
|
|
3777
|
+
"salesforce_object",
|
|
3778
|
+
"salesforce_field",
|
|
3779
|
+
"salesforce_record_type",
|
|
3780
|
+
"salesforce_permission_set",
|
|
3781
|
+
"salesforce_flow",
|
|
3782
|
+
"salesforce_workflow",
|
|
3783
|
+
"salesforce_approval",
|
|
3784
|
+
];
|
|
3785
|
+
const SALESFORCE_KIND_BY_TYPE = {
|
|
3786
|
+
object: "salesforce_object",
|
|
3787
|
+
field: "salesforce_field",
|
|
3788
|
+
recordType: "salesforce_record_type",
|
|
3789
|
+
permissionSet: "salesforce_permission_set",
|
|
3790
|
+
flow: "salesforce_flow",
|
|
3791
|
+
workflow: "salesforce_workflow",
|
|
3792
|
+
approval: "salesforce_approval",
|
|
3793
|
+
};
|
|
3794
|
+
/**
|
|
3795
|
+
* Give a metadata symbol a kind and a summary that describe it.
|
|
3796
|
+
*
|
|
3797
|
+
* Every Salesforce symbol previously shared one kind and one identical
|
|
3798
|
+
* boilerplate summary, which made 36,763 of them indistinguishable to both FTS
|
|
3799
|
+
* (which indexes name, kind and summary) and to embeddings. A field's type,
|
|
3800
|
+
* label and lookup target are sitting in the file and were simply discarded.
|
|
3801
|
+
*/
|
|
3802
|
+
function describeSalesforceMetadata(metadata, content) {
|
|
3803
|
+
const kind = SALESFORCE_KIND_BY_TYPE[metadata.type] ?? "salesforce_metadata";
|
|
3804
|
+
const first = (tag) => staticXmlTagValues(content, tag)[0]?.value?.trim() || undefined;
|
|
3805
|
+
if (metadata.type === "field") {
|
|
3806
|
+
const fieldType = first("type");
|
|
3807
|
+
const referenceTo = first("referenceTo");
|
|
3808
|
+
const label = first("label");
|
|
3809
|
+
const parts = [
|
|
3810
|
+
`Salesforce field ${metadata.object}.${metadata.field}`,
|
|
3811
|
+
fieldType ? (referenceTo ? `${fieldType}(${referenceTo})` : fieldType) : undefined,
|
|
3812
|
+
label ? `"${label}"` : undefined,
|
|
3813
|
+
].filter(Boolean);
|
|
3814
|
+
return { kind, summary: parts.join(" — ") };
|
|
3815
|
+
}
|
|
3816
|
+
if (metadata.type === "object") {
|
|
3817
|
+
const label = first("label");
|
|
3818
|
+
// A `__e` suffix is the Platform Event marker, and callers ask about
|
|
3819
|
+
// those specifically ("which Platform Event triggers lack a retry guard").
|
|
3820
|
+
const platformEvent = metadata.object.endsWith("__e");
|
|
3821
|
+
const parts = [
|
|
3822
|
+
`Salesforce ${platformEvent ? "Platform Event" : "object"} ${metadata.object}`,
|
|
3823
|
+
label ? `"${label}"` : undefined,
|
|
3824
|
+
].filter(Boolean);
|
|
3825
|
+
return { kind, summary: parts.join(" — ") };
|
|
3826
|
+
}
|
|
3827
|
+
const label = first("label") ?? first("masterLabel");
|
|
3828
|
+
const name = metadata.type === "flow"
|
|
3829
|
+
? metadata.name
|
|
3830
|
+
: metadata.type === "permissionSet"
|
|
3831
|
+
? metadata.name
|
|
3832
|
+
: metadata.type === "recordType"
|
|
3833
|
+
? `${metadata.object}.${metadata.recordType}`
|
|
3834
|
+
: metadata.type === "approval"
|
|
3835
|
+
? `${metadata.object}.${metadata.name}`
|
|
3836
|
+
: metadata.object;
|
|
3837
|
+
const parts = [`Salesforce ${metadata.type} ${name}`, label ? `"${label}"` : undefined].filter(Boolean);
|
|
3838
|
+
return { kind, summary: parts.join(" — ") };
|
|
3839
|
+
}
|
|
3756
3840
|
async function indexSalesforceMetadataFile(content, relativePath, repoPath, db, metadata) {
|
|
3757
3841
|
const dependencies = [];
|
|
3758
3842
|
const add = (to, kind, sourceEvidence, methodReference) => {
|
|
@@ -3773,9 +3857,17 @@ async function indexSalesforceMetadataFile(content, relativePath, repoPath, db,
|
|
|
3773
3857
|
case "object":
|
|
3774
3858
|
add(metadata.object, "salesforce_metadata_object", "Salesforce DX CustomObject file path");
|
|
3775
3859
|
break;
|
|
3776
|
-
case "field":
|
|
3860
|
+
case "field": {
|
|
3777
3861
|
add(`${metadata.object}.${metadata.field}`, "salesforce_metadata_field", "Salesforce DX CustomField file path");
|
|
3862
|
+
// A Lookup/MasterDetail field names the object it points at. Emitting
|
|
3863
|
+
// it as an edge is what turns "which objects are near the 40-lookup
|
|
3864
|
+
// limit" into a graph query instead of a grep over raw XML.
|
|
3865
|
+
for (const entry of staticXmlTagValues(content, "referenceTo")) {
|
|
3866
|
+
if (staticSalesforceIdentifier(entry.value))
|
|
3867
|
+
add(entry.value, "salesforce_lookup_target", entry.evidence);
|
|
3868
|
+
}
|
|
3778
3869
|
break;
|
|
3870
|
+
}
|
|
3779
3871
|
case "recordType":
|
|
3780
3872
|
add(`${metadata.object}.${metadata.recordType}`, "salesforce_record_type", "Salesforce DX RecordType file path");
|
|
3781
3873
|
break;
|
|
@@ -3864,16 +3956,17 @@ async function indexSalesforceMetadataFile(content, relativePath, repoPath, db,
|
|
|
3864
3956
|
db.run("DELETE FROM dependencies WHERE fromFile = ?", [relativePath]);
|
|
3865
3957
|
db.run("DELETE FROM mcp_tools WHERE filePath = ?", [relativePath]);
|
|
3866
3958
|
const flowSymbol = metadata.type === "flow" ? metadata.name : path.basename(relativePath, ".xml");
|
|
3959
|
+
const descriptor = describeSalesforceMetadata(metadata, content);
|
|
3867
3960
|
db.run(`INSERT INTO symbols (name, kind, filePath, startLine, endLine, startCol, endCol, summary)
|
|
3868
3961
|
VALUES (?, ?, ?, ?, ?, ?, ?, ?)`, [
|
|
3869
3962
|
flowSymbol,
|
|
3870
|
-
|
|
3963
|
+
descriptor.kind,
|
|
3871
3964
|
relativePath,
|
|
3872
3965
|
1,
|
|
3873
3966
|
content.split("\n").length,
|
|
3874
3967
|
0,
|
|
3875
3968
|
0,
|
|
3876
|
-
|
|
3969
|
+
descriptor.summary,
|
|
3877
3970
|
]);
|
|
3878
3971
|
const insertRef = db.prepare(`INSERT INTO "references" (callerSymbol, callerFile, calleeSymbol, calleeFile, line, column, kind, confidence)
|
|
3879
3972
|
VALUES (?, ?, ?, ?, ?, ?, ?, ?)`);
|
|
@@ -3912,7 +4005,7 @@ async function refreshFlowsForApexFiles(db, repoPath, apexFiles) {
|
|
|
3912
4005
|
if (classNames.size === 0)
|
|
3913
4006
|
return [];
|
|
3914
4007
|
const flowFiles = db
|
|
3915
|
-
.query("SELECT DISTINCT filePath FROM symbols WHERE kind = '
|
|
4008
|
+
.query("SELECT DISTINCT filePath FROM symbols WHERE kind = 'salesforce_flow' AND filePath LIKE 'force-app/main/default/flows/%.flow-meta.xml' ORDER BY filePath")
|
|
3916
4009
|
.all();
|
|
3917
4010
|
const refreshed = [];
|
|
3918
4011
|
for (const { filePath } of flowFiles) {
|
|
@@ -5894,262 +5987,444 @@ export async function indexDbtManifestFile(content, _absolutePath, relativePath,
|
|
|
5894
5987
|
}
|
|
5895
5988
|
}
|
|
5896
5989
|
/**
|
|
5897
|
-
*
|
|
5990
|
+
* Builds a `FileIndexResult` from an already-parsed tree -- no DB access,
|
|
5991
|
+
* no `db` parameter. Factored out of `indexFile` so this same logic is
|
|
5992
|
+
* usable both from the current synchronous caller and, later, from a worker
|
|
5993
|
+
* thread that owns its own read+parse step (see the parallel-parsing plan).
|
|
5994
|
+
*/
|
|
5995
|
+
function buildFileIndexResult(tree, content, isPython, isApex, absolutePath, relativePath, repoPath, mcpRegistrations) {
|
|
5996
|
+
const { definitions, references, fileDependencies, tableDependencies, ormDependencies, importBindings, apiContracts, } = measurePerfPhaseSync("parse_extract", () => extractSymbolsAndReferences(tree.rootNode, isPython, isApex, absolutePath));
|
|
5997
|
+
const apexPlatformDependencies = isApex ? extractApexPlatformDependencies(content, repoPath) : [];
|
|
5998
|
+
// Resolve each callee to the file its definition lives in, so references
|
|
5999
|
+
// are file-qualified instead of name-only. Precedence: an imported name
|
|
6000
|
+
// resolves to its module's file; otherwise a name defined in THIS file
|
|
6001
|
+
// resolves to this file; otherwise unresolved (null) — external, dynamic,
|
|
6002
|
+
// or a member call — which consumers treat as best-effort.
|
|
6003
|
+
const localDefNames = new Set(definitions.map((d) => d.name));
|
|
6004
|
+
const resolveCalleeFile = (calleeName, explicitModule) => {
|
|
6005
|
+
// null means a member call on an ordinary value (`obj.run()`): do not
|
|
6006
|
+
// accidentally reuse an unrelated named import called `run`. undefined
|
|
6007
|
+
// means a bare call and is eligible for normal import-binding resolution.
|
|
6008
|
+
const mod = explicitModule === null ? undefined : (explicitModule ?? importBindings.get(calleeName));
|
|
6009
|
+
if (mod) {
|
|
6010
|
+
// Resolve the module (relative or workspace package alias) to a file,
|
|
6011
|
+
// then follow re-export barrels to the file that actually defines the
|
|
6012
|
+
// symbol (so `@pkg` -> index barrel -> definition file).
|
|
6013
|
+
const moduleFile = resolveModuleToFile(relativePath, mod, repoPath);
|
|
6014
|
+
if (moduleFile)
|
|
6015
|
+
return followReExports(repoPath, moduleFile, calleeName, 0);
|
|
6016
|
+
return null;
|
|
6017
|
+
}
|
|
6018
|
+
if (localDefNames.has(calleeName))
|
|
6019
|
+
return relativePath;
|
|
6020
|
+
return null;
|
|
6021
|
+
};
|
|
6022
|
+
const resolvedReferences = references.map((ref) => ({
|
|
6023
|
+
callerSymbol: ref.callerSymbol,
|
|
6024
|
+
calleeSymbol: ref.calleeSymbol,
|
|
6025
|
+
calleeFile: resolveCalleeFile(ref.calleeSymbol, ref.calleeModule),
|
|
6026
|
+
line: ref.line,
|
|
6027
|
+
column: ref.column,
|
|
6028
|
+
kind: ref.kind ?? "call",
|
|
6029
|
+
}));
|
|
6030
|
+
const dependencies = [];
|
|
6031
|
+
for (const depSource of fileDependencies) {
|
|
6032
|
+
const resolved = resolveImportPath(relativePath, depSource, repoPath);
|
|
6033
|
+
if (resolved)
|
|
6034
|
+
dependencies.push({ toFile: resolved, kind: "import" });
|
|
6035
|
+
}
|
|
6036
|
+
if (tableDependencies) {
|
|
6037
|
+
for (const tableName of tableDependencies)
|
|
6038
|
+
dependencies.push({ toFile: tableName, kind: "table" });
|
|
6039
|
+
}
|
|
6040
|
+
if (ormDependencies) {
|
|
6041
|
+
for (const dep of ormDependencies)
|
|
6042
|
+
dependencies.push({ toFile: dep.to, kind: dep.kind });
|
|
6043
|
+
}
|
|
6044
|
+
const dedupedApexPlatformDependencies = [];
|
|
6045
|
+
const seenApexPlatformDependencies = new Set();
|
|
6046
|
+
for (const dependency of apexPlatformDependencies) {
|
|
6047
|
+
const key = `${dependency.to}\0${dependency.kind}`;
|
|
6048
|
+
if (seenApexPlatformDependencies.has(key))
|
|
6049
|
+
continue;
|
|
6050
|
+
seenApexPlatformDependencies.add(key);
|
|
6051
|
+
dedupedApexPlatformDependencies.push(dependency);
|
|
6052
|
+
}
|
|
6053
|
+
return {
|
|
6054
|
+
relativePath,
|
|
6055
|
+
mcpRegistrations,
|
|
6056
|
+
definitions,
|
|
6057
|
+
references: resolvedReferences,
|
|
6058
|
+
apiContracts,
|
|
6059
|
+
dependencies,
|
|
6060
|
+
apexPlatformDependencies: dedupedApexPlatformDependencies,
|
|
6061
|
+
};
|
|
6062
|
+
}
|
|
6063
|
+
/**
|
|
6064
|
+
* Writes a single `FileIndexResult` inside its own transaction -- identical
|
|
6065
|
+
* shape to `indexFile`'s previous inline write, just parameterized on
|
|
6066
|
+
* already-extracted data instead of closure-captured locals.
|
|
6067
|
+
*/
|
|
6068
|
+
function writeFileIndexResult(db, result) {
|
|
6069
|
+
const finishWrite = beginPerfPhase("sqlite_write");
|
|
6070
|
+
db.run("BEGIN TRANSACTION;");
|
|
6071
|
+
try {
|
|
6072
|
+
writeFileIndexResultUnwrapped(db, result);
|
|
6073
|
+
db.run("COMMIT;");
|
|
6074
|
+
}
|
|
6075
|
+
catch (error) {
|
|
6076
|
+
db.run("ROLLBACK;");
|
|
6077
|
+
throw error;
|
|
6078
|
+
}
|
|
6079
|
+
finally {
|
|
6080
|
+
finishWrite();
|
|
6081
|
+
}
|
|
6082
|
+
}
|
|
6083
|
+
/** The actual delete+insert statements, run inside a transaction/savepoint the caller owns. */
|
|
6084
|
+
function writeFileIndexResultUnwrapped(db, result) {
|
|
6085
|
+
const { relativePath, mcpRegistrations, definitions, references, apiContracts, dependencies } = result;
|
|
6086
|
+
db.run('DELETE FROM "references" WHERE callerFile = ?', [relativePath]);
|
|
6087
|
+
deleteSymbolsForFile(db, relativePath);
|
|
6088
|
+
db.run("DELETE FROM dependencies WHERE fromFile = ?", [relativePath]);
|
|
6089
|
+
db.run("DELETE FROM mcp_tools WHERE filePath = ?", [relativePath]);
|
|
6090
|
+
db.run("DELETE FROM api_contracts WHERE filePath = ?", [relativePath]);
|
|
6091
|
+
const insertMcp = db.prepare(`
|
|
6092
|
+
INSERT INTO mcp_tools(name, description, schemaSymbol, handlerSymbol, filePath, line, confidence, associationKey)
|
|
6093
|
+
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
|
|
6094
|
+
`);
|
|
6095
|
+
for (const registration of mcpRegistrations) {
|
|
6096
|
+
insertMcp.run(registration.name, registration.description ?? null, registration.schemaSymbol ?? null, registration.handlerSymbol ?? null, relativePath, registration.line, registration.confidence, registration.associationKey ?? null);
|
|
6097
|
+
}
|
|
6098
|
+
insertMcp.finalize();
|
|
6099
|
+
const insertSym = db.prepare(`
|
|
6100
|
+
INSERT INTO symbols (name, kind, filePath, startLine, endLine, startCol, endCol, summary)
|
|
6101
|
+
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
|
|
6102
|
+
`);
|
|
6103
|
+
for (const def of definitions) {
|
|
6104
|
+
insertSym.run(def.name, def.kind, relativePath, def.startLine, def.endLine, def.startCol, def.endCol, def.summary);
|
|
6105
|
+
}
|
|
6106
|
+
insertSym.finalize();
|
|
6107
|
+
const insertRef = db.prepare(`
|
|
6108
|
+
INSERT INTO "references" (callerSymbol, callerFile, calleeSymbol, calleeFile, line, column, kind)
|
|
6109
|
+
VALUES (?, ?, ?, ?, ?, ?, ?)
|
|
6110
|
+
`);
|
|
6111
|
+
for (const ref of references) {
|
|
6112
|
+
insertRef.run(ref.callerSymbol, relativePath, ref.calleeSymbol, ref.calleeFile, ref.line, ref.column, ref.kind);
|
|
6113
|
+
}
|
|
6114
|
+
insertRef.finalize();
|
|
6115
|
+
const insertApiContract = db.prepare(`
|
|
6116
|
+
INSERT INTO api_contracts (kind, framework, method, path, origin, handler, middleware, responseFields, responseKnown, declaredResponseFields, accessedFields, unresolved, filePath, line, confidence, evidence)
|
|
6117
|
+
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
6118
|
+
`);
|
|
6119
|
+
for (const fact of apiContracts) {
|
|
6120
|
+
insertApiContract.run(fact.kind, fact.framework, fact.method, fact.path, fact.origin, fact.handler, JSON.stringify(fact.middleware), JSON.stringify(fact.responseFields), fact.responseKnown ? 1 : 0, JSON.stringify(fact.declaredResponseFields), JSON.stringify(fact.accessedFields), fact.unresolved ? 1 : 0, relativePath, fact.line, fact.confidence, fact.evidence);
|
|
6121
|
+
}
|
|
6122
|
+
insertApiContract.finalize();
|
|
6123
|
+
const insertDep = db.prepare(`
|
|
6124
|
+
INSERT INTO dependencies (fromFile, toFile, kind)
|
|
6125
|
+
VALUES (?, ?, ?)
|
|
6126
|
+
`);
|
|
6127
|
+
const insertApexPlatformDependency = db.prepare("INSERT INTO dependencies (fromFile, toFile, kind, confidence, sourceEvidence) VALUES (?, ?, ?, ?, ?)");
|
|
6128
|
+
for (const dep of dependencies) {
|
|
6129
|
+
insertDep.run(relativePath, dep.toFile, dep.kind);
|
|
6130
|
+
}
|
|
6131
|
+
insertDep.finalize();
|
|
6132
|
+
for (const dependency of result.apexPlatformDependencies) {
|
|
6133
|
+
insertApexPlatformDependency.run(relativePath, dependency.to, dependency.kind, 1, dependency.sourceEvidence);
|
|
6134
|
+
}
|
|
6135
|
+
insertApexPlatformDependency.finalize();
|
|
6136
|
+
}
|
|
6137
|
+
/**
|
|
6138
|
+
* Write a batch of `FileIndexResult`s in a single transaction, isolating
|
|
6139
|
+
* each file behind its own SAVEPOINT so one file's write failure rolls back
|
|
6140
|
+
* only that file and does not lose the rest of the batch -- the same
|
|
6141
|
+
* per-file error-isolation contract `indexFile`'s single-file transaction
|
|
6142
|
+
* always had, just amortized over one BEGIN/COMMIT per batch instead of per
|
|
6143
|
+
* file. Returns the files whose write failed, mirroring `indexFile`'s own
|
|
6144
|
+
* catch (caller tallies into `unparsedByExtension` the same way).
|
|
6145
|
+
*/
|
|
6146
|
+
export function writeFileIndexResultBatch(db, results) {
|
|
6147
|
+
if (results.length === 0)
|
|
6148
|
+
return [];
|
|
6149
|
+
const failures = [];
|
|
6150
|
+
const finishWrite = beginPerfPhase("sqlite_write");
|
|
6151
|
+
db.run("BEGIN TRANSACTION;");
|
|
6152
|
+
try {
|
|
6153
|
+
results.forEach((result, i) => {
|
|
6154
|
+
const savepoint = `f${i}`;
|
|
6155
|
+
db.run(`SAVEPOINT ${savepoint};`);
|
|
6156
|
+
try {
|
|
6157
|
+
writeFileIndexResultUnwrapped(db, result);
|
|
6158
|
+
db.run(`RELEASE ${savepoint};`);
|
|
6159
|
+
}
|
|
6160
|
+
catch (error) {
|
|
6161
|
+
db.run(`ROLLBACK TO ${savepoint};`);
|
|
6162
|
+
db.run(`RELEASE ${savepoint};`);
|
|
6163
|
+
failures.push({ relativePath: result.relativePath, error });
|
|
6164
|
+
}
|
|
6165
|
+
});
|
|
6166
|
+
db.run("COMMIT;");
|
|
6167
|
+
}
|
|
6168
|
+
catch (error) {
|
|
6169
|
+
db.run("ROLLBACK;");
|
|
6170
|
+
throw error;
|
|
6171
|
+
}
|
|
6172
|
+
finally {
|
|
6173
|
+
finishWrite();
|
|
6174
|
+
}
|
|
6175
|
+
return failures;
|
|
6176
|
+
}
|
|
6177
|
+
const INDEX_WRITE_BATCH_SIZE = (() => {
|
|
6178
|
+
const raw = Number(process.env.KNODIN_INDEX_WRITE_BATCH_SIZE);
|
|
6179
|
+
return Number.isFinite(raw) && raw > 0 ? Math.floor(raw) : 200;
|
|
6180
|
+
})();
|
|
6181
|
+
/**
|
|
6182
|
+
* Read, parse, and extract one file through the generic tree-sitter path.
|
|
6183
|
+
*
|
|
6184
|
+
* Deliberately free of any `Database` parameter so the whole read → parse →
|
|
6185
|
+
* extract sequence can run off the main thread. `indexFile` is the only
|
|
6186
|
+
* in-tree caller today; the parse pool becomes the second, and both must go
|
|
6187
|
+
* through here so routing can never drift between them.
|
|
6188
|
+
*/
|
|
6189
|
+
export async function extractGenericFileForIndex(absolutePath, relativePath, repoPath) {
|
|
6190
|
+
const lang = await measurePerfPhase("parser_wasm_init", () => getLanguageForFile(absolutePath));
|
|
6191
|
+
if (!lang)
|
|
6192
|
+
return { kind: "unparsed" };
|
|
6193
|
+
const content = measurePerfPhaseSync("read", () => fs.readFileSync(absolutePath, "utf-8"));
|
|
6194
|
+
// Skip minified / generated bundles (e.g. a `*.bundle.js`): they're not
|
|
6195
|
+
// readable source and would inject thousands of junk "symbols" that pollute
|
|
6196
|
+
// the graph, communities, and hubs. Heuristic: a very high average line
|
|
6197
|
+
// length (minified code packs everything onto a few enormous lines).
|
|
6198
|
+
if (isMinifiedSource(content))
|
|
6199
|
+
return { kind: "unparsed" };
|
|
6200
|
+
const isSql = absolutePath.endsWith(".sql") || absolutePath.endsWith(".pkb") || absolutePath.endsWith(".pks");
|
|
6201
|
+
// SQL: neutralize dollar-quoted bodies that crash the grammar's scanner.
|
|
6202
|
+
// The tree and every downstream indexer must see the SAME string, so pass
|
|
6203
|
+
// `parseContent` (not the raw file) forward.
|
|
6204
|
+
const parseContent = isSql ? sanitizeDollarQuotedSql(content) : content;
|
|
6205
|
+
const parser = new Parser();
|
|
6206
|
+
parser.setLanguage(lang);
|
|
6207
|
+
// A malformed/pathological input can make the wasm parser throw rather than
|
|
6208
|
+
// return an ERROR tree. Skip that one file instead of aborting the file's
|
|
6209
|
+
// index (and without dumping the wasm source into the log).
|
|
6210
|
+
let tree;
|
|
6211
|
+
try {
|
|
6212
|
+
tree = measurePerfPhaseSync("parse_extract", () => parser.parse(parseContent));
|
|
6213
|
+
}
|
|
6214
|
+
catch (e) {
|
|
6215
|
+
console.warn(`knodin: skipped ${relativePath} (parser error: ${e.message})`);
|
|
6216
|
+
return { kind: "skipped", reason: e.message };
|
|
6217
|
+
}
|
|
6218
|
+
if (!tree)
|
|
6219
|
+
return { kind: "skipped", reason: "parser returned no tree" };
|
|
6220
|
+
const mcpRegistrations = /\.[cm]?[jt]sx?$/.test(absolutePath) && /@modelcontextprotocol\/sdk/.test(content)
|
|
6221
|
+
? extractMcpToolRegistrations(content, relativePath, tree.rootNode)
|
|
6222
|
+
: [];
|
|
6223
|
+
const isPython = absolutePath.endsWith(".py");
|
|
6224
|
+
const isApex = absolutePath.endsWith(".cls") || absolutePath.endsWith(".trigger");
|
|
6225
|
+
const isVisualforce = absolutePath.endsWith(".page") || absolutePath.endsWith(".component");
|
|
6226
|
+
const isPrisma = absolutePath.endsWith(".prisma");
|
|
6227
|
+
const isWorkday = absolutePath.endsWith(".clp") ||
|
|
6228
|
+
absolutePath.endsWith(".ws") ||
|
|
6229
|
+
(absolutePath.endsWith(".xml") && isWorkdayStudioFile(content, absolutePath));
|
|
6230
|
+
const special = (route) => ({
|
|
6231
|
+
kind: "special",
|
|
6232
|
+
route,
|
|
6233
|
+
content,
|
|
6234
|
+
parseContent,
|
|
6235
|
+
tree: tree,
|
|
6236
|
+
});
|
|
6237
|
+
if (isWorkday)
|
|
6238
|
+
return special("workday");
|
|
6239
|
+
if (isVisualforce)
|
|
6240
|
+
return special("visualforce");
|
|
6241
|
+
if (isSql)
|
|
6242
|
+
return special("sql");
|
|
6243
|
+
if (isPrisma)
|
|
6244
|
+
return special("prisma");
|
|
6245
|
+
return {
|
|
6246
|
+
kind: "result",
|
|
6247
|
+
result: buildFileIndexResult(tree, content, isPython, isApex, absolutePath, relativePath, repoPath, mcpRegistrations),
|
|
6248
|
+
};
|
|
6249
|
+
}
|
|
6250
|
+
/**
|
|
6251
|
+
* The single source of truth for indexer routing.
|
|
6252
|
+
*
|
|
6253
|
+
* `indexFile` dispatches on this, and the parse pool decides worker
|
|
6254
|
+
* eligibility from it. Both must consult the SAME function: a second copy of
|
|
6255
|
+
* this ordering would drift the moment a file type is added, and the failure
|
|
6256
|
+
* mode is silent — a markdown file routed to a worker comes back "unparsed"
|
|
6257
|
+
* rather than reaching `indexMarkdownFile`, which reads as "this file has no
|
|
6258
|
+
* symbols" instead of as an error.
|
|
6259
|
+
*
|
|
6260
|
+
* Order is significant and matches the historical dispatch exactly.
|
|
6261
|
+
*/
|
|
6262
|
+
export function classifyIndexRoute(absolutePath, relativePath, options) {
|
|
6263
|
+
const baseName = path.basename(absolutePath).toLowerCase();
|
|
6264
|
+
if (!options?.skipDbt && baseName === "manifest.json")
|
|
6265
|
+
return { kind: "dbt-candidate" };
|
|
6266
|
+
if (absolutePath.endsWith(".lsif") || baseName === "dump.lsif")
|
|
6267
|
+
return { kind: "lsif" };
|
|
6268
|
+
if (absolutePath.endsWith(".tf") ||
|
|
6269
|
+
absolutePath.endsWith(".tfvars") ||
|
|
6270
|
+
absolutePath.endsWith(".hcl"))
|
|
6271
|
+
return { kind: "terraform" };
|
|
6272
|
+
if (absolutePath.endsWith(".dockerfile") || baseName === "dockerfile")
|
|
6273
|
+
return { kind: "dockerfile" };
|
|
6274
|
+
const salesforceExperience = salesforceExperienceFile(relativePath);
|
|
6275
|
+
if (salesforceExperience)
|
|
6276
|
+
return { kind: "salesforce-experience", file: salesforceExperience };
|
|
6277
|
+
const lwcBundle = lwcBundleFile(relativePath);
|
|
6278
|
+
if (lwcBundle)
|
|
6279
|
+
return { kind: "lwc-bundle", file: lwcBundle };
|
|
6280
|
+
const auraBundle = auraBundleFile(relativePath);
|
|
6281
|
+
if (auraBundle)
|
|
6282
|
+
return { kind: "aura-bundle", file: auraBundle };
|
|
6283
|
+
const salesforceMetadata = salesforceMetadataFile(relativePath);
|
|
6284
|
+
if (salesforceMetadata)
|
|
6285
|
+
return { kind: "salesforce-metadata", file: salesforceMetadata };
|
|
6286
|
+
if (relativePath.endsWith(".md") || relativePath.endsWith(".mdx"))
|
|
6287
|
+
return { kind: "markdown" };
|
|
6288
|
+
if (relativePath.endsWith(".yml") || relativePath.endsWith(".yaml"))
|
|
6289
|
+
return { kind: "yaml" };
|
|
6290
|
+
if (relativePath.endsWith(".csproj") ||
|
|
6291
|
+
relativePath.endsWith(".sln") ||
|
|
6292
|
+
relativePath.endsWith(".props") ||
|
|
6293
|
+
relativePath.endsWith(".targets"))
|
|
6294
|
+
return { kind: "dotnet-project" };
|
|
6295
|
+
if (relativePath.endsWith(".cshtml") || relativePath.endsWith(".razor"))
|
|
6296
|
+
return { kind: "razor" };
|
|
6297
|
+
return { kind: "generic" };
|
|
6298
|
+
}
|
|
6299
|
+
/**
|
|
6300
|
+
* Whether a file may be parsed off the main thread.
|
|
6301
|
+
*
|
|
6302
|
+
* Only the generic tree-sitter path qualifies: every other route calls an
|
|
6303
|
+
* indexer that writes to the database, and the database has exactly one
|
|
6304
|
+
* writer. `dbt-candidate` is excluded on purpose — it cannot be resolved
|
|
6305
|
+
* without reading the file, and `manifest.json` is rare enough that sending
|
|
6306
|
+
* it down the sequential path costs nothing.
|
|
6307
|
+
*/
|
|
6308
|
+
export function isWorkerEligibleFile(absolutePath, relativePath) {
|
|
6309
|
+
if (classifyIndexRoute(absolutePath, relativePath).kind !== "generic")
|
|
6310
|
+
return false;
|
|
6311
|
+
// `generic` is necessary but NOT sufficient. `extractGenericFileForIndex`
|
|
6312
|
+
// can still return a `special` route — visualforce, workday, sql, prisma —
|
|
6313
|
+
// each of which hands off to a main-thread indexer that opens its own
|
|
6314
|
+
// BEGIN/COMMIT against the single-writer database. Sending one to a worker
|
|
6315
|
+
// leaves it to the pool's fallback, which re-indexes it inline while pooled
|
|
6316
|
+
// results are still accumulating for a batched write, and the two
|
|
6317
|
+
// interleave nondeterministically: observed as a Visualforce controller
|
|
6318
|
+
// edge that resolved on some runs and not others, at roughly 1 in 3.
|
|
6319
|
+
//
|
|
6320
|
+
// Decided from the path alone, so it can be answered before reading the
|
|
6321
|
+
// file. `.xml` is excluded conservatively because Workday detection needs
|
|
6322
|
+
// the content; an XML file indexing inline costs nothing.
|
|
6323
|
+
return !SPECIAL_ROUTE_EXTENSIONS.some((extension) => absolutePath.endsWith(extension));
|
|
6324
|
+
}
|
|
6325
|
+
/**
|
|
6326
|
+
* Extensions that may reach a `special` route in `extractGenericFileForIndex`.
|
|
5898
6327
|
*
|
|
5899
|
-
*
|
|
5900
|
-
*
|
|
5901
|
-
*
|
|
5902
|
-
* otherwise: the function swallows all three cases, so a whole language silently
|
|
5903
|
-
* contributing nothing looks exactly like a language with nothing to say.
|
|
6328
|
+
* Kept adjacent to that function's own extension tests on purpose: if a new
|
|
6329
|
+
* special route is added there without a matching entry here, the file becomes
|
|
6330
|
+
* worker-eligible again and the failure is silent rather than loud.
|
|
5904
6331
|
*/
|
|
5905
|
-
|
|
6332
|
+
const SPECIAL_ROUTE_EXTENSIONS = [
|
|
6333
|
+
".page",
|
|
6334
|
+
".component",
|
|
6335
|
+
".clp",
|
|
6336
|
+
".ws",
|
|
6337
|
+
".xml",
|
|
6338
|
+
".sql",
|
|
6339
|
+
".pkb",
|
|
6340
|
+
".pks",
|
|
6341
|
+
".prisma",
|
|
6342
|
+
];
|
|
6343
|
+
async function indexFile(absolutePath, relativePath, repoPath, db, unparsed, collectResult) {
|
|
5906
6344
|
try {
|
|
5907
6345
|
invalidateScipFacts(db);
|
|
5908
|
-
const
|
|
5909
|
-
|
|
6346
|
+
const read = () => measurePerfPhaseSync("read", () => fs.readFileSync(absolutePath, "utf-8"));
|
|
6347
|
+
let route = classifyIndexRoute(absolutePath, relativePath);
|
|
6348
|
+
if (route.kind === "dbt-candidate") {
|
|
5910
6349
|
const content = fs.readFileSync(absolutePath, "utf-8");
|
|
5911
6350
|
if (isDbtManifest(content)) {
|
|
5912
6351
|
await indexDbtManifestFile(content, absolutePath, relativePath, repoPath, db);
|
|
5913
6352
|
return;
|
|
5914
6353
|
}
|
|
6354
|
+
// Not a dbt manifest after all. Re-classify without the dbt check so
|
|
6355
|
+
// the file still reaches whatever route it would otherwise have taken.
|
|
6356
|
+
route = classifyIndexRoute(absolutePath, relativePath, { skipDbt: true });
|
|
5915
6357
|
}
|
|
5916
|
-
|
|
5917
|
-
|
|
5918
|
-
|
|
5919
|
-
|
|
5920
|
-
|
|
5921
|
-
|
|
5922
|
-
|
|
5923
|
-
|
|
5924
|
-
|
|
5925
|
-
|
|
5926
|
-
|
|
5927
|
-
|
|
5928
|
-
|
|
5929
|
-
|
|
5930
|
-
|
|
5931
|
-
|
|
5932
|
-
|
|
5933
|
-
|
|
5934
|
-
|
|
5935
|
-
|
|
5936
|
-
|
|
5937
|
-
|
|
5938
|
-
|
|
5939
|
-
|
|
5940
|
-
|
|
5941
|
-
|
|
5942
|
-
|
|
5943
|
-
|
|
5944
|
-
|
|
5945
|
-
|
|
5946
|
-
|
|
5947
|
-
|
|
5948
|
-
|
|
5949
|
-
|
|
5950
|
-
|
|
5951
|
-
|
|
5952
|
-
const salesforceMetadata = salesforceMetadataFile(relativePath);
|
|
5953
|
-
if (salesforceMetadata) {
|
|
5954
|
-
const content = measurePerfPhaseSync("read", () => fs.readFileSync(absolutePath, "utf-8"));
|
|
5955
|
-
await indexSalesforceMetadataFile(content, relativePath, repoPath, db, salesforceMetadata);
|
|
5956
|
-
return;
|
|
5957
|
-
}
|
|
5958
|
-
if (relativePath.endsWith(".md") || relativePath.endsWith(".mdx")) {
|
|
5959
|
-
const content = measurePerfPhaseSync("read", () => fs.readFileSync(absolutePath, "utf-8"));
|
|
5960
|
-
await indexMarkdownFile(content, relativePath, repoPath, db);
|
|
5961
|
-
return;
|
|
5962
|
-
}
|
|
5963
|
-
if (relativePath.endsWith(".yml") || relativePath.endsWith(".yaml")) {
|
|
5964
|
-
const content = measurePerfPhaseSync("read", () => fs.readFileSync(absolutePath, "utf-8"));
|
|
5965
|
-
await indexYamlFile(content, relativePath, repoPath, db);
|
|
5966
|
-
return;
|
|
5967
|
-
}
|
|
5968
|
-
if (relativePath.endsWith(".csproj") ||
|
|
5969
|
-
relativePath.endsWith(".sln") ||
|
|
5970
|
-
relativePath.endsWith(".props") ||
|
|
5971
|
-
relativePath.endsWith(".targets")) {
|
|
5972
|
-
const content = measurePerfPhaseSync("read", () => fs.readFileSync(absolutePath, "utf-8"));
|
|
5973
|
-
await indexDotnetProjectFile(content, relativePath, repoPath, db);
|
|
5974
|
-
return;
|
|
5975
|
-
}
|
|
5976
|
-
if (relativePath.endsWith(".cshtml") || relativePath.endsWith(".razor")) {
|
|
5977
|
-
const content = measurePerfPhaseSync("read", () => fs.readFileSync(absolutePath, "utf-8"));
|
|
5978
|
-
await indexRazorFile(content, relativePath, repoPath, db);
|
|
5979
|
-
return;
|
|
5980
|
-
}
|
|
5981
|
-
const lang = await measurePerfPhase("parser_wasm_init", () => getLanguageForFile(absolutePath));
|
|
5982
|
-
if (!lang) {
|
|
5983
|
-
if (unparsed)
|
|
5984
|
-
tallyOne(unparsed, extensionBucket(relativePath));
|
|
5985
|
-
return;
|
|
6358
|
+
switch (route.kind) {
|
|
6359
|
+
case "lsif":
|
|
6360
|
+
await indexLsifDump(absolutePath, relativePath, repoPath, db);
|
|
6361
|
+
return;
|
|
6362
|
+
case "terraform":
|
|
6363
|
+
await indexTerraformFile(fs.readFileSync(absolutePath, "utf-8"), relativePath, repoPath, db);
|
|
6364
|
+
return;
|
|
6365
|
+
case "dockerfile":
|
|
6366
|
+
await indexDockerFile(fs.readFileSync(absolutePath, "utf-8"), relativePath, repoPath, db);
|
|
6367
|
+
return;
|
|
6368
|
+
case "salesforce-experience":
|
|
6369
|
+
await indexSalesforceExperienceFile(read(), relativePath, repoPath, db, route.file);
|
|
6370
|
+
return;
|
|
6371
|
+
case "lwc-bundle":
|
|
6372
|
+
await indexLwcBundleFile(read(), relativePath, repoPath, db, route.file);
|
|
6373
|
+
return;
|
|
6374
|
+
case "aura-bundle":
|
|
6375
|
+
await indexAuraBundleFile(read(), relativePath, repoPath, db, route.file);
|
|
6376
|
+
return;
|
|
6377
|
+
case "salesforce-metadata":
|
|
6378
|
+
await indexSalesforceMetadataFile(read(), relativePath, repoPath, db, route.file);
|
|
6379
|
+
return;
|
|
6380
|
+
case "markdown":
|
|
6381
|
+
await indexMarkdownFile(read(), relativePath, repoPath, db);
|
|
6382
|
+
return;
|
|
6383
|
+
case "yaml":
|
|
6384
|
+
await indexYamlFile(read(), relativePath, repoPath, db);
|
|
6385
|
+
return;
|
|
6386
|
+
case "dotnet-project":
|
|
6387
|
+
await indexDotnetProjectFile(read(), relativePath, repoPath, db);
|
|
6388
|
+
return;
|
|
6389
|
+
case "razor":
|
|
6390
|
+
await indexRazorFile(read(), relativePath, repoPath, db);
|
|
6391
|
+
return;
|
|
6392
|
+
default:
|
|
6393
|
+
break;
|
|
5986
6394
|
}
|
|
5987
|
-
const
|
|
5988
|
-
|
|
5989
|
-
// readable source and would inject thousands of junk "symbols" that pollute
|
|
5990
|
-
// the graph, communities, and hubs. Heuristic: a very high average line
|
|
5991
|
-
// length (minified code packs everything onto a few enormous lines).
|
|
5992
|
-
if (isMinifiedSource(content)) {
|
|
6395
|
+
const extraction = await extractGenericFileForIndex(absolutePath, relativePath, repoPath);
|
|
6396
|
+
if (extraction.kind === "unparsed") {
|
|
5993
6397
|
if (unparsed)
|
|
5994
6398
|
tallyOne(unparsed, extensionBucket(relativePath));
|
|
5995
6399
|
return;
|
|
5996
6400
|
}
|
|
5997
|
-
|
|
5998
|
-
|
|
5999
|
-
|
|
6000
|
-
|
|
6001
|
-
// The tree and every downstream indexer must see the SAME string, so pass
|
|
6002
|
-
// `parseContent` (not the raw file) forward.
|
|
6003
|
-
const parseContent = isSql ? sanitizeDollarQuotedSql(content) : content;
|
|
6004
|
-
const parser = new Parser();
|
|
6005
|
-
parser.setLanguage(lang);
|
|
6006
|
-
// A malformed/pathological input can make the wasm parser throw rather than
|
|
6007
|
-
// return an ERROR tree. Skip that one file instead of aborting the file's
|
|
6008
|
-
// index (and without dumping the wasm source into the log).
|
|
6009
|
-
let tree;
|
|
6010
|
-
try {
|
|
6011
|
-
tree = measurePerfPhaseSync("parse_extract", () => parser.parse(parseContent));
|
|
6012
|
-
}
|
|
6013
|
-
catch (e) {
|
|
6014
|
-
console.warn(`knodin: skipped ${relativePath} (parser error: ${e.message})`);
|
|
6015
|
-
return;
|
|
6016
|
-
}
|
|
6017
|
-
if (!tree)
|
|
6018
|
-
return;
|
|
6019
|
-
const mcpRegistrations = /\.[cm]?[jt]sx?$/.test(absolutePath) && /@modelcontextprotocol\/sdk/.test(content)
|
|
6020
|
-
? extractMcpToolRegistrations(content, relativePath, tree.rootNode)
|
|
6021
|
-
: [];
|
|
6022
|
-
const isPython = absolutePath.endsWith(".py");
|
|
6023
|
-
const isApex = absolutePath.endsWith(".cls") || absolutePath.endsWith(".trigger");
|
|
6024
|
-
const isVisualforce = absolutePath.endsWith(".page") || absolutePath.endsWith(".component");
|
|
6025
|
-
const isPrisma = absolutePath.endsWith(".prisma");
|
|
6026
|
-
const isWorkday = absolutePath.endsWith(".clp") ||
|
|
6027
|
-
absolutePath.endsWith(".ws") ||
|
|
6028
|
-
(absolutePath.endsWith(".xml") && isWorkdayStudioFile(content, absolutePath));
|
|
6029
|
-
if (isWorkday) {
|
|
6030
|
-
await indexWorkdayFile(content, tree, relativePath, repoPath, db);
|
|
6031
|
-
return;
|
|
6032
|
-
}
|
|
6033
|
-
if (isVisualforce) {
|
|
6034
|
-
await indexVisualforceFile(content, tree, relativePath, repoPath, db);
|
|
6035
|
-
return;
|
|
6036
|
-
}
|
|
6037
|
-
if (isSql) {
|
|
6038
|
-
await indexSqlFile(parseContent, tree, relativePath, repoPath, db);
|
|
6401
|
+
// A parse failure abandons this one file without tallying it as a
|
|
6402
|
+
// coverage gap: the file IS supported, it just did not parse, and
|
|
6403
|
+
// `extractGenericFileForIndex` has already warned about it.
|
|
6404
|
+
if (extraction.kind === "skipped")
|
|
6039
6405
|
return;
|
|
6040
|
-
|
|
6041
|
-
|
|
6042
|
-
|
|
6043
|
-
|
|
6044
|
-
|
|
6045
|
-
|
|
6046
|
-
|
|
6047
|
-
|
|
6048
|
-
|
|
6049
|
-
|
|
6050
|
-
|
|
6051
|
-
|
|
6052
|
-
|
|
6053
|
-
|
|
6054
|
-
|
|
6055
|
-
const resolveCalleeFile = (calleeName, explicitModule) => {
|
|
6056
|
-
// null means a member call on an ordinary value (`obj.run()`): do not
|
|
6057
|
-
// accidentally reuse an unrelated named import called `run`. undefined
|
|
6058
|
-
// means a bare call and is eligible for normal import-binding resolution.
|
|
6059
|
-
const mod = explicitModule === null ? undefined : (explicitModule ?? importBindings.get(calleeName));
|
|
6060
|
-
if (mod) {
|
|
6061
|
-
// Resolve the module (relative or workspace package alias) to a file,
|
|
6062
|
-
// then follow re-export barrels to the file that actually defines the
|
|
6063
|
-
// symbol (so `@pkg` -> index barrel -> definition file).
|
|
6064
|
-
const moduleFile = resolveModuleToFile(relativePath, mod, repoPath);
|
|
6065
|
-
if (moduleFile)
|
|
6066
|
-
return followReExports(repoPath, moduleFile, calleeName, 0);
|
|
6067
|
-
return null;
|
|
6068
|
-
}
|
|
6069
|
-
if (localDefNames.has(calleeName))
|
|
6070
|
-
return relativePath;
|
|
6071
|
-
return null;
|
|
6072
|
-
};
|
|
6073
|
-
// Execute all database operations inside a single transaction synchronously
|
|
6074
|
-
const finishWrite = beginPerfPhase("sqlite_write");
|
|
6075
|
-
db.run("BEGIN TRANSACTION;");
|
|
6076
|
-
try {
|
|
6077
|
-
db.run('DELETE FROM "references" WHERE callerFile = ?', [relativePath]);
|
|
6078
|
-
deleteSymbolsForFile(db, relativePath);
|
|
6079
|
-
db.run("DELETE FROM dependencies WHERE fromFile = ?", [relativePath]);
|
|
6080
|
-
db.run("DELETE FROM mcp_tools WHERE filePath = ?", [relativePath]);
|
|
6081
|
-
db.run("DELETE FROM api_contracts WHERE filePath = ?", [relativePath]);
|
|
6082
|
-
const insertMcp = db.prepare(`
|
|
6083
|
-
INSERT INTO mcp_tools(name, description, schemaSymbol, handlerSymbol, filePath, line, confidence, associationKey)
|
|
6084
|
-
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
|
|
6085
|
-
`);
|
|
6086
|
-
for (const registration of mcpRegistrations) {
|
|
6087
|
-
insertMcp.run(registration.name, registration.description ?? null, registration.schemaSymbol ?? null, registration.handlerSymbol ?? null, relativePath, registration.line, registration.confidence, registration.associationKey ?? null);
|
|
6088
|
-
}
|
|
6089
|
-
insertMcp.finalize();
|
|
6090
|
-
const insertSym = db.prepare(`
|
|
6091
|
-
INSERT INTO symbols (name, kind, filePath, startLine, endLine, startCol, endCol, summary)
|
|
6092
|
-
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
|
|
6093
|
-
`);
|
|
6094
|
-
for (const def of definitions) {
|
|
6095
|
-
insertSym.run(def.name, def.kind, relativePath, def.startLine, def.endLine, def.startCol, def.endCol, def.summary);
|
|
6096
|
-
}
|
|
6097
|
-
insertSym.finalize();
|
|
6098
|
-
const insertRef = db.prepare(`
|
|
6099
|
-
INSERT INTO "references" (callerSymbol, callerFile, calleeSymbol, calleeFile, line, column, kind)
|
|
6100
|
-
VALUES (?, ?, ?, ?, ?, ?, ?)
|
|
6101
|
-
`);
|
|
6102
|
-
for (const ref of references) {
|
|
6103
|
-
insertRef.run(ref.callerSymbol, relativePath, ref.calleeSymbol, resolveCalleeFile(ref.calleeSymbol, ref.calleeModule), ref.line, ref.column, ref.kind ?? "call");
|
|
6104
|
-
}
|
|
6105
|
-
insertRef.finalize();
|
|
6106
|
-
const insertApiContract = db.prepare(`
|
|
6107
|
-
INSERT INTO api_contracts (kind, framework, method, path, origin, handler, middleware, responseFields, responseKnown, declaredResponseFields, accessedFields, unresolved, filePath, line, confidence, evidence)
|
|
6108
|
-
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
6109
|
-
`);
|
|
6110
|
-
for (const fact of apiContracts) {
|
|
6111
|
-
insertApiContract.run(fact.kind, fact.framework, fact.method, fact.path, fact.origin, fact.handler, JSON.stringify(fact.middleware), JSON.stringify(fact.responseFields), fact.responseKnown ? 1 : 0, JSON.stringify(fact.declaredResponseFields), JSON.stringify(fact.accessedFields), fact.unresolved ? 1 : 0, relativePath, fact.line, fact.confidence, fact.evidence);
|
|
6112
|
-
}
|
|
6113
|
-
insertApiContract.finalize();
|
|
6114
|
-
const insertDep = db.prepare(`
|
|
6115
|
-
INSERT INTO dependencies (fromFile, toFile, kind)
|
|
6116
|
-
VALUES (?, ?, ?)
|
|
6117
|
-
`);
|
|
6118
|
-
for (const depSource of fileDependencies) {
|
|
6119
|
-
const resolved = resolveImportPath(relativePath, depSource, repoPath);
|
|
6120
|
-
if (resolved) {
|
|
6121
|
-
insertDep.run(relativePath, resolved, "import");
|
|
6122
|
-
}
|
|
6123
|
-
}
|
|
6124
|
-
if (tableDependencies) {
|
|
6125
|
-
for (const tableName of tableDependencies) {
|
|
6126
|
-
insertDep.run(relativePath, tableName, "table");
|
|
6127
|
-
}
|
|
6128
|
-
}
|
|
6129
|
-
if (ormDependencies) {
|
|
6130
|
-
for (const dep of ormDependencies) {
|
|
6131
|
-
insertDep.run(relativePath, dep.to, dep.kind);
|
|
6132
|
-
}
|
|
6133
|
-
}
|
|
6134
|
-
insertDep.finalize();
|
|
6135
|
-
const insertApexPlatformDependency = db.prepare("INSERT INTO dependencies (fromFile, toFile, kind, confidence, sourceEvidence) VALUES (?, ?, ?, ?, ?)");
|
|
6136
|
-
const seenApexPlatformDependencies = new Set();
|
|
6137
|
-
for (const dependency of apexPlatformDependencies) {
|
|
6138
|
-
const key = `${dependency.to}\0${dependency.kind}`;
|
|
6139
|
-
if (seenApexPlatformDependencies.has(key))
|
|
6140
|
-
continue;
|
|
6141
|
-
seenApexPlatformDependencies.add(key);
|
|
6142
|
-
insertApexPlatformDependency.run(relativePath, dependency.to, dependency.kind, 1, dependency.sourceEvidence);
|
|
6406
|
+
if (extraction.kind === "special") {
|
|
6407
|
+
const { content, parseContent, tree } = extraction;
|
|
6408
|
+
switch (extraction.route) {
|
|
6409
|
+
case "workday":
|
|
6410
|
+
await indexWorkdayFile(content, tree, relativePath, repoPath, db);
|
|
6411
|
+
return;
|
|
6412
|
+
case "visualforce":
|
|
6413
|
+
await indexVisualforceFile(content, tree, relativePath, repoPath, db);
|
|
6414
|
+
return;
|
|
6415
|
+
case "sql":
|
|
6416
|
+
await indexSqlFile(parseContent, tree, relativePath, repoPath, db);
|
|
6417
|
+
return;
|
|
6418
|
+
case "prisma":
|
|
6419
|
+
await indexPrismaFile(content, tree, relativePath, repoPath, db);
|
|
6420
|
+
return;
|
|
6143
6421
|
}
|
|
6144
|
-
insertApexPlatformDependency.finalize();
|
|
6145
|
-
db.run("COMMIT;");
|
|
6146
6422
|
}
|
|
6147
|
-
|
|
6148
|
-
|
|
6149
|
-
throw error;
|
|
6423
|
+
if (collectResult) {
|
|
6424
|
+
collectResult(extraction.result);
|
|
6150
6425
|
}
|
|
6151
|
-
|
|
6152
|
-
|
|
6426
|
+
else {
|
|
6427
|
+
writeFileIndexResult(db, extraction.result);
|
|
6153
6428
|
}
|
|
6154
6429
|
}
|
|
6155
6430
|
catch (error) {
|
|
@@ -7211,6 +7486,37 @@ function parseGitStatusPaths(output) {
|
|
|
7211
7486
|
* Only files actually re-indexed here are re-embedded (indexEmbeddings is
|
|
7212
7487
|
* incremental), so a clean reconcile does no work.
|
|
7213
7488
|
*/
|
|
7489
|
+
/**
|
|
7490
|
+
* Full mtime/size comparison against `index_state`, used both when there is
|
|
7491
|
+
* no git HEAD at all and as a correctness fallback when a git-diff baseline
|
|
7492
|
+
* turns out to be unresolvable (see `reconcileIndex`). Not a substitute for
|
|
7493
|
+
* the git fast path in general: it only detects drift relative to *this*
|
|
7494
|
+
* checkout's own recorded state, so a database whose `index_state` rows came
|
|
7495
|
+
* from a different physical checkout (fresh mtimes on every file) degrades
|
|
7496
|
+
* to "everything changed" here rather than a cheap targeted diff.
|
|
7497
|
+
*/
|
|
7498
|
+
function detectDriftByStat(repoPath, db, canonicalFiles) {
|
|
7499
|
+
const changed = new Set();
|
|
7500
|
+
const state = new Map();
|
|
7501
|
+
for (const row of db
|
|
7502
|
+
.query("SELECT filePath, mtimeMs, size FROM index_state")
|
|
7503
|
+
.all()) {
|
|
7504
|
+
state.set(row.filePath, row);
|
|
7505
|
+
}
|
|
7506
|
+
for (const file of canonicalFiles) {
|
|
7507
|
+
const prev = state.get(file);
|
|
7508
|
+
try {
|
|
7509
|
+
const st = fs.statSync(path.join(repoPath, file));
|
|
7510
|
+
if (!prev || Math.abs(prev.mtimeMs - st.mtimeMs) >= 1.0 || prev.size !== st.size) {
|
|
7511
|
+
changed.add(file);
|
|
7512
|
+
}
|
|
7513
|
+
}
|
|
7514
|
+
catch (_) {
|
|
7515
|
+
changed.add(file);
|
|
7516
|
+
}
|
|
7517
|
+
}
|
|
7518
|
+
return changed;
|
|
7519
|
+
}
|
|
7214
7520
|
async function reconcileIndex(repoPath, db, progress) {
|
|
7215
7521
|
try {
|
|
7216
7522
|
progress?.("collecting-files", 0, "Checking existing local graph for changes");
|
|
@@ -7230,6 +7536,7 @@ async function reconcileIndex(repoPath, db, progress) {
|
|
|
7230
7536
|
changed.add(file);
|
|
7231
7537
|
}
|
|
7232
7538
|
catch (_) { }
|
|
7539
|
+
let baselineUnresolved = false;
|
|
7233
7540
|
if (lastHead && lastHead !== head) {
|
|
7234
7541
|
try {
|
|
7235
7542
|
const diff = child_process.execFileSync("git", ["diff", "--name-only", "-z", lastHead, head, "--"], {
|
|
@@ -7240,7 +7547,26 @@ async function reconcileIndex(repoPath, db, progress) {
|
|
|
7240
7547
|
for (const file of splitNulPaths(diff))
|
|
7241
7548
|
changed.add(file);
|
|
7242
7549
|
}
|
|
7243
|
-
catch (_) {
|
|
7550
|
+
catch (_) {
|
|
7551
|
+
// `lastHead` is not resolvable in this checkout's object database
|
|
7552
|
+
// (a shallow clone missing that commit, or a database seeded from
|
|
7553
|
+
// an unrelated checkout). Silently treating this as "nothing
|
|
7554
|
+
// changed" would let the graph advance `lastIndexedHead` to the
|
|
7555
|
+
// new HEAD below while having missed every real change in
|
|
7556
|
+
// between — fall back to a full comparison instead.
|
|
7557
|
+
baselineUnresolved = true;
|
|
7558
|
+
}
|
|
7559
|
+
}
|
|
7560
|
+
if (baselineUnresolved) {
|
|
7561
|
+
console.warn(`knodin: cached baseline commit ${lastHead} is not resolvable in this checkout (shallow clone or foreign database) — falling back to a full file-drift comparison`);
|
|
7562
|
+
canonicalFiles = collectRepoFiles(repoPath);
|
|
7563
|
+
if (canonicalFiles.length > 50000) {
|
|
7564
|
+
console.warn(`knodin: skipping full-drift fallback reconcile (${canonicalFiles.length} files exceeds 50k bound)`);
|
|
7565
|
+
}
|
|
7566
|
+
else {
|
|
7567
|
+
for (const file of detectDriftByStat(repoPath, db, canonicalFiles))
|
|
7568
|
+
changed.add(file);
|
|
7569
|
+
}
|
|
7244
7570
|
}
|
|
7245
7571
|
}
|
|
7246
7572
|
else {
|
|
@@ -7250,24 +7576,8 @@ async function reconcileIndex(repoPath, db, progress) {
|
|
|
7250
7576
|
console.warn(`knodin: skipping freshness reconcile (${canonicalFiles.length} files exceeds 50k bound)`);
|
|
7251
7577
|
return 0;
|
|
7252
7578
|
}
|
|
7253
|
-
const
|
|
7254
|
-
|
|
7255
|
-
.query("SELECT filePath, mtimeMs, size FROM index_state")
|
|
7256
|
-
.all()) {
|
|
7257
|
-
state.set(row.filePath, row);
|
|
7258
|
-
}
|
|
7259
|
-
for (const file of canonicalFiles) {
|
|
7260
|
-
const prev = state.get(file);
|
|
7261
|
-
try {
|
|
7262
|
-
const st = fs.statSync(path.join(repoPath, file));
|
|
7263
|
-
if (!prev || Math.abs(prev.mtimeMs - st.mtimeMs) >= 1.0 || prev.size !== st.size) {
|
|
7264
|
-
changed.add(file);
|
|
7265
|
-
}
|
|
7266
|
-
}
|
|
7267
|
-
catch (_) {
|
|
7268
|
-
changed.add(file);
|
|
7269
|
-
}
|
|
7270
|
-
}
|
|
7579
|
+
for (const file of detectDriftByStat(repoPath, db, canonicalFiles))
|
|
7580
|
+
changed.add(file);
|
|
7271
7581
|
}
|
|
7272
7582
|
// A row in index_state whose file is gone or no longer satisfies the
|
|
7273
7583
|
// canonical source policy must be purged. Keep those legacy rows in the
|
|
@@ -7347,6 +7657,7 @@ async function reconcileIndex(repoPath, db, progress) {
|
|
|
7347
7657
|
// "dirty, and indexed that way" from "dirty, and moved since".
|
|
7348
7658
|
if (head)
|
|
7349
7659
|
setMeta(db, "lastIndexedHead", head);
|
|
7660
|
+
setMeta(db, "knodinVersion", KNODIN_VERSION);
|
|
7350
7661
|
setMeta(db, "lastSuccessfulReconciliation", new Date().toISOString());
|
|
7351
7662
|
recordFreshnessBaseline(repoPath, db);
|
|
7352
7663
|
return reindexed;
|
|
@@ -7726,20 +8037,99 @@ async function indexRepo(repoPath, db, progress, skipEmbeddings = false, publish
|
|
|
7726
8037
|
db.run("DELETE FROM dependencies;");
|
|
7727
8038
|
db.run("DELETE FROM mcp_tools;");
|
|
7728
8039
|
db.run("DELETE FROM index_state;");
|
|
7729
|
-
// Process files sequentially to ensure thread-safe SQLite transactions
|
|
8040
|
+
// Process files sequentially to ensure thread-safe SQLite transactions.
|
|
8041
|
+
// Extraction (buildFileIndexResult, pure) still happens one file at a
|
|
8042
|
+
// time here; only the write is batched, via `pendingWrites`, so the
|
|
8043
|
+
// generic tree-sitter path amortizes one BEGIN/COMMIT over many files
|
|
8044
|
+
// instead of paying it per file (per-file failure isolation is preserved
|
|
8045
|
+
// through per-file SAVEPOINTs inside `writeFileIndexResultBatch`).
|
|
7730
8046
|
let lastYieldAt = Date.now();
|
|
7731
|
-
|
|
7732
|
-
|
|
7733
|
-
|
|
8047
|
+
const pendingWrites = [];
|
|
8048
|
+
const flushPendingWrites = () => {
|
|
8049
|
+
if (pendingWrites.length === 0)
|
|
8050
|
+
return;
|
|
8051
|
+
const failures = writeFileIndexResultBatch(db, pendingWrites);
|
|
8052
|
+
for (const failure of failures) {
|
|
8053
|
+
console.error(`Error indexing file ${failure.relativePath}:`, failure.error);
|
|
8054
|
+
tallyOne(unparsedByExtension, extensionBucket(failure.relativePath));
|
|
8055
|
+
}
|
|
8056
|
+
pendingWrites.length = 0;
|
|
8057
|
+
};
|
|
8058
|
+
// Progress is reported against a single counter so the pool and the
|
|
8059
|
+
// sequential path can interleave without the count going backwards.
|
|
8060
|
+
let processed = 0;
|
|
8061
|
+
const afterFile = async (file) => {
|
|
8062
|
+
if (pendingWrites.length >= INDEX_WRITE_BATCH_SIZE)
|
|
8063
|
+
flushPendingWrites();
|
|
7734
8064
|
recordIndexState(db, repoPath, file);
|
|
7735
|
-
|
|
8065
|
+
processed++;
|
|
8066
|
+
progress?.("indexing-files", processed, `Indexing ${files.length.toLocaleString()} files`, {
|
|
7736
8067
|
phaseTotal: files.length,
|
|
7737
8068
|
});
|
|
7738
8069
|
if (Date.now() - lastYieldAt >= INDEX_EVENT_LOOP_YIELD_MS) {
|
|
7739
8070
|
await yieldToIndexEventLoop();
|
|
7740
8071
|
lastYieldAt = Date.now();
|
|
7741
8072
|
}
|
|
8073
|
+
};
|
|
8074
|
+
const indexOneInline = async (file) => {
|
|
8075
|
+
await indexFile(path.join(repoPath, file), file, repoPath, db, unparsedByExtension, (result) => {
|
|
8076
|
+
pendingWrites.push(result);
|
|
8077
|
+
});
|
|
8078
|
+
};
|
|
8079
|
+
// Opt-in for now. `createParsePool` returns null whenever the machine says
|
|
8080
|
+
// not to spawn, so an unsuitable host silently keeps today's behavior
|
|
8081
|
+
// rather than discovering the problem partway through an index.
|
|
8082
|
+
const pool = parseWorkersEnabled() ? createParsePool() : null;
|
|
8083
|
+
try {
|
|
8084
|
+
if (pool) {
|
|
8085
|
+
// Only the generic tree-sitter route may leave the main thread; every
|
|
8086
|
+
// other route calls an indexer that writes to the database, and the
|
|
8087
|
+
// database has exactly one writer. `isWorkerEligibleFile` is derived
|
|
8088
|
+
// from the same classifier `indexFile` dispatches on, so the two
|
|
8089
|
+
// cannot disagree about where a file belongs.
|
|
8090
|
+
const eligible = [];
|
|
8091
|
+
const inline = [];
|
|
8092
|
+
for (const file of files) {
|
|
8093
|
+
if (isWorkerEligibleFile(path.join(repoPath, file), file))
|
|
8094
|
+
eligible.push({ absolutePath: path.join(repoPath, file), relativePath: file, repoPath });
|
|
8095
|
+
else
|
|
8096
|
+
inline.push(file);
|
|
8097
|
+
}
|
|
8098
|
+
await pool.run(eligible, async (task, outcome) => {
|
|
8099
|
+
switch (outcome.kind) {
|
|
8100
|
+
case "result":
|
|
8101
|
+
pendingWrites.push(outcome.result);
|
|
8102
|
+
break;
|
|
8103
|
+
case "unparsed":
|
|
8104
|
+
tallyOne(unparsedByExtension, extensionBucket(task.relativePath));
|
|
8105
|
+
break;
|
|
8106
|
+
case "skipped":
|
|
8107
|
+
break;
|
|
8108
|
+
default:
|
|
8109
|
+
// The pool could not finish this file (worker crash, or it
|
|
8110
|
+
// turned out to need a main-thread indexer). Never dropped:
|
|
8111
|
+
// run it through the sequential path exactly as before.
|
|
8112
|
+
await indexOneInline(task.relativePath);
|
|
8113
|
+
break;
|
|
8114
|
+
}
|
|
8115
|
+
await afterFile(task.relativePath);
|
|
8116
|
+
});
|
|
8117
|
+
for (const file of inline) {
|
|
8118
|
+
await indexOneInline(file);
|
|
8119
|
+
await afterFile(file);
|
|
8120
|
+
}
|
|
8121
|
+
}
|
|
8122
|
+
else {
|
|
8123
|
+
for (const file of files) {
|
|
8124
|
+
await indexOneInline(file);
|
|
8125
|
+
await afterFile(file);
|
|
8126
|
+
}
|
|
8127
|
+
}
|
|
7742
8128
|
}
|
|
8129
|
+
finally {
|
|
8130
|
+
await pool?.close();
|
|
8131
|
+
}
|
|
8132
|
+
flushPendingWrites();
|
|
7743
8133
|
progress?.("finalizing", 0, "Persisting symbol identities and semantic index");
|
|
7744
8134
|
persistSymbolIdentities(db, repoPath);
|
|
7745
8135
|
reconcileTypeScriptDi(db, repoPath);
|
|
@@ -7751,6 +8141,7 @@ async function indexRepo(repoPath, db, progress, skipEmbeddings = false, publish
|
|
|
7751
8141
|
await indexEmbeddings(db, repoPath, progress);
|
|
7752
8142
|
// Remember the HEAD we indexed at so a later cold start can diff forward.
|
|
7753
8143
|
setMeta(db, "lastIndexedHead", gitHead(repoPath) ?? "");
|
|
8144
|
+
setMeta(db, "knodinVersion", KNODIN_VERSION);
|
|
7754
8145
|
setMeta(db, "lastSuccessfulReconciliation", new Date().toISOString());
|
|
7755
8146
|
setMeta(db, "mcpBackfillVersion", "17");
|
|
7756
8147
|
setMeta(db, COVERAGE_UNPARSED_META_KEY, JSON.stringify(Object.fromEntries(unparsedByExtension)));
|
|
@@ -7947,7 +8338,7 @@ async function suspendFileWatcher(repoPath) {
|
|
|
7947
8338
|
return true;
|
|
7948
8339
|
}
|
|
7949
8340
|
/** Open and initialize SQLite connection and full repo index. */
|
|
7950
|
-
async function getOrInitDb(repoPath, options = {}) {
|
|
8341
|
+
export async function getOrInitDb(repoPath, options = {}) {
|
|
7951
8342
|
const normalizedPath = path.resolve(repoPath);
|
|
7952
8343
|
const selectedDatabasePath = explicitDatabasePath(normalizedPath, options.databasePath);
|
|
7953
8344
|
if (!selectedDatabasePath)
|
|
@@ -8233,9 +8624,21 @@ async function getOrInitDb(repoPath, options = {}) {
|
|
|
8233
8624
|
DELETE FROM symbols_fts WHERE symbolId = old.id;
|
|
8234
8625
|
END;
|
|
8235
8626
|
`);
|
|
8627
|
+
// Scoped to the columns the FTS table actually mirrors. An unscoped
|
|
8628
|
+
// AFTER UPDATE fires for every column, so writing `identity` -- which
|
|
8629
|
+
// symbols_fts does not index -- rewrote an FTS row each time.
|
|
8630
|
+
// `persistSymbolIdentities` writes identity for every symbol on a full
|
|
8631
|
+
// index, and that alone measured 39.1s of a 40.0s pass (19,638 rows at
|
|
8632
|
+
// ~2ms each) on a 1,926-file corpus.
|
|
8633
|
+
//
|
|
8634
|
+
// Dropped unconditionally rather than guarded by `IF NOT EXISTS`,
|
|
8635
|
+
// because an existing database already carries the unscoped trigger and
|
|
8636
|
+
// would otherwise keep it forever. Recreating a trigger on open is
|
|
8637
|
+
// cheap and needs no schema-version bump or forced reindex.
|
|
8638
|
+
db.run("DROP TRIGGER IF EXISTS after_symbol_update;");
|
|
8236
8639
|
db.run(`
|
|
8237
|
-
CREATE TRIGGER
|
|
8238
|
-
AFTER UPDATE ON symbols
|
|
8640
|
+
CREATE TRIGGER after_symbol_update
|
|
8641
|
+
AFTER UPDATE OF name, kind, summary ON symbols
|
|
8239
8642
|
BEGIN
|
|
8240
8643
|
UPDATE symbols_fts
|
|
8241
8644
|
SET name = new.name,
|
|
@@ -8467,12 +8870,92 @@ function matchEndpoints(clientPath, serverPath) {
|
|
|
8467
8870
|
serverSegment.toLocaleLowerCase("en-US");
|
|
8468
8871
|
});
|
|
8469
8872
|
}
|
|
8470
|
-
/**
|
|
8471
|
-
|
|
8472
|
-
|
|
8473
|
-
|
|
8474
|
-
|
|
8475
|
-
|
|
8873
|
+
/**
|
|
8874
|
+
* Repositories being served from a sealed artifact rather than a working tree.
|
|
8875
|
+
*
|
|
8876
|
+
* Keyed by the repository path a caller passes in, so every existing reader
|
|
8877
|
+
* keeps its signature. This is the ONE place that decides where source bytes
|
|
8878
|
+
* come from: widening the chokepoint instead of threading a resolver through
|
|
8879
|
+
* ~20 call sites is what keeps a sealed answer from depending on whether a
|
|
8880
|
+
* particular reader remembered to ask.
|
|
8881
|
+
*/
|
|
8882
|
+
const sealedRepositories = new Map();
|
|
8883
|
+
export function registerSealedRepository(repoPath, db) {
|
|
8884
|
+
sealedRepositories.set(path.resolve(repoPath), db);
|
|
8885
|
+
}
|
|
8886
|
+
export function unregisterSealedRepository(repoPath) {
|
|
8887
|
+
sealedRepositories.delete(path.resolve(repoPath));
|
|
8888
|
+
}
|
|
8889
|
+
export function isSealedRepository(repoPath) {
|
|
8890
|
+
return sealedRepositories.has(path.resolve(repoPath));
|
|
8891
|
+
}
|
|
8892
|
+
/**
|
|
8893
|
+
* Whole-file source, from the sealed artifact when there is one.
|
|
8894
|
+
*
|
|
8895
|
+
* Returns an availability rather than a bare string, and NEVER `""` for
|
|
8896
|
+
* missing content: an empty string is indistinguishable from an empty file,
|
|
8897
|
+
* and downstream that reads as "this file has nothing in it" rather than "this
|
|
8898
|
+
* was not covered" — the silent false negative a consumer needs to detect in
|
|
8899
|
+
* order to fall back to live source.
|
|
8900
|
+
*/
|
|
8901
|
+
function readWholeSource(repoPath, relativePath) {
|
|
8902
|
+
const sealed = sealedRepositories.get(path.resolve(repoPath));
|
|
8903
|
+
if (sealed) {
|
|
8904
|
+
const text = readSealedSource(sealed, relativePath);
|
|
8905
|
+
return text === null
|
|
8906
|
+
? {
|
|
8907
|
+
text: "",
|
|
8908
|
+
availability: {
|
|
8909
|
+
state: "absent",
|
|
8910
|
+
reason: `not covered by the sealed artifact: ${relativePath}`,
|
|
8911
|
+
},
|
|
8912
|
+
}
|
|
8913
|
+
: { text, availability: { state: "available" } };
|
|
8914
|
+
}
|
|
8915
|
+
try {
|
|
8916
|
+
return {
|
|
8917
|
+
text: fs.readFileSync(path.resolve(repoPath, relativePath), "utf-8"),
|
|
8918
|
+
availability: { state: "available" },
|
|
8919
|
+
};
|
|
8920
|
+
}
|
|
8921
|
+
catch (error) {
|
|
8922
|
+
const reason = error instanceof Error ? error.message : String(error);
|
|
8923
|
+
return error?.code === "ENOENT"
|
|
8924
|
+
? { text: "", availability: { state: "absent", reason } }
|
|
8925
|
+
: { text: "", availability: { state: "read-failed", reason, retryable: true } };
|
|
8926
|
+
}
|
|
8927
|
+
}
|
|
8928
|
+
/**
|
|
8929
|
+
* Whether a file is present as source, without reading it.
|
|
8930
|
+
*
|
|
8931
|
+
* A sealed artifact answers from its embedded coverage, not the filesystem —
|
|
8932
|
+
* a sealed repository has no working tree, so `fs.existsSync` there would
|
|
8933
|
+
* report every file missing and quietly invert the meaning of any check built
|
|
8934
|
+
* on it (a test file "not present", a module path "unresolvable").
|
|
8935
|
+
*/
|
|
8936
|
+
function sourceFileExists(repoPath, relativePath) {
|
|
8937
|
+
const sealed = sealedRepositories.get(path.resolve(repoPath));
|
|
8938
|
+
if (sealed)
|
|
8939
|
+
return readSealedSource(sealed, relativePath) !== null;
|
|
8940
|
+
return fs.existsSync(path.resolve(repoPath, relativePath));
|
|
8941
|
+
}
|
|
8942
|
+
/**
|
|
8943
|
+
* Enumerate source files, from the sealed artifact's coverage when there is
|
|
8944
|
+
* one. A sealed repository has no directory tree, so anything that discovers
|
|
8945
|
+
* files by traversal must ask coverage instead of the filesystem.
|
|
8946
|
+
*/
|
|
8947
|
+
function listSourceFiles(repoPath, accept) {
|
|
8948
|
+
const sealed = sealedRepositories.get(path.resolve(repoPath));
|
|
8949
|
+
if (sealed)
|
|
8950
|
+
return listSealedFiles(sealed).filter(accept);
|
|
8951
|
+
return walkRepoFiles(repoPath, { accept });
|
|
8952
|
+
}
|
|
8953
|
+
/** Get verbatim source with line numbers, and say so when it is unavailable. */
|
|
8954
|
+
function readSourceRange(repoPath, relativePath, startLine, endLine, maxLines) {
|
|
8955
|
+
const whole = readWholeSource(repoPath, relativePath);
|
|
8956
|
+
if (whole.availability.state !== "available")
|
|
8957
|
+
return { text: "", availability: whole.availability };
|
|
8958
|
+
const content = whole.text;
|
|
8476
8959
|
const lines = content.split("\n");
|
|
8477
8960
|
const totalLines = endLine - startLine + 1;
|
|
8478
8961
|
let slice = lines.slice(startLine - 1, endLine);
|
|
@@ -8485,7 +8968,30 @@ function getSourceRange(repoPath, relativePath, startLine, endLine, maxLines) {
|
|
|
8485
8968
|
if (truncated) {
|
|
8486
8969
|
result += `\n... (truncated: showing lines ${startLine}-${startLine + slice.length - 1} of ${totalLines} total — use \`query large_functions\` to confirm size, or read the file directly for the remainder)`;
|
|
8487
8970
|
}
|
|
8488
|
-
return result;
|
|
8971
|
+
return { text: result, availability: { state: "available" } };
|
|
8972
|
+
}
|
|
8973
|
+
/**
|
|
8974
|
+
* One trimmed source line, used as edge evidence, or undefined if unreadable.
|
|
8975
|
+
*
|
|
8976
|
+
* The map builders previously read this inline and unguarded, so a single
|
|
8977
|
+
* missing file threw and aborted the ENTIRE map rather than costing one edge
|
|
8978
|
+
* its evidence string. Evidence is decoration on an edge; the edge itself is
|
|
8979
|
+
* derived from the graph and stands without it.
|
|
8980
|
+
*/
|
|
8981
|
+
function evidenceLine(root, relativePath, line) {
|
|
8982
|
+
const whole = readWholeSource(root, relativePath);
|
|
8983
|
+
if (whole.availability.state !== "available")
|
|
8984
|
+
return undefined;
|
|
8985
|
+
return whole.text.split(/\r?\n/)[line - 1]?.trim().slice(0, 240);
|
|
8986
|
+
}
|
|
8987
|
+
/**
|
|
8988
|
+
* Text-only view of `readSourceRange`, for callers that genuinely do not
|
|
8989
|
+
* surface availability. Kept deliberately thin and separate so that dropping
|
|
8990
|
+
* the availability is a visible choice at the call site rather than the
|
|
8991
|
+
* default behaviour of the only available helper.
|
|
8992
|
+
*/
|
|
8993
|
+
function getSourceRange(repoPath, relativePath, startLine, endLine, maxLines) {
|
|
8994
|
+
return readSourceRange(repoPath, relativePath, startLine, endLine, maxLines).text;
|
|
8489
8995
|
}
|
|
8490
8996
|
/** Recursively traverse backwards from callee to caller to build dependent trees. */
|
|
8491
8997
|
function _findCallersRecursive(db, targetSymbol, depth, visited = new Set()) {
|
|
@@ -10704,11 +11210,18 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
10704
11210
|
const candidateAudits = new Map();
|
|
10705
11211
|
const candidates = new Map();
|
|
10706
11212
|
let closed = false;
|
|
11213
|
+
const sealed = openPolicy.mode === "sealed";
|
|
10707
11214
|
const openDb = (repoPath, options = {}) => getOrInitDb(repoPath, {
|
|
10708
11215
|
...options,
|
|
10709
|
-
watcher: options.watcher ?? openPolicy.watcher,
|
|
11216
|
+
watcher: sealed ? "disabled" : (options.watcher ?? openPolicy.watcher),
|
|
10710
11217
|
watcherFileLimit: options.watcherFileLimit ?? openPolicy.watcherFileLimit,
|
|
10711
11218
|
databasePath: options.databasePath ?? openPolicy.databasePath,
|
|
11219
|
+
// A sealed artifact has no working tree. Reconciling against the empty
|
|
11220
|
+
// mount would find none of the indexed files on disk and prune every
|
|
11221
|
+
// symbol, so the artifact would answer "not found" for content it
|
|
11222
|
+
// actually carries — a negative that looks entirely correct.
|
|
11223
|
+
skipContentReconciliation: sealed || options.skipContentReconciliation,
|
|
11224
|
+
skipFreshnessGuard: sealed || options.skipFreshnessGuard,
|
|
10712
11225
|
});
|
|
10713
11226
|
const claimRepository = (repoPath) => {
|
|
10714
11227
|
if (closed)
|
|
@@ -10913,8 +11426,19 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
10913
11426
|
const header = fs.readFileSync(source).subarray(0, 16).toString("binary");
|
|
10914
11427
|
if (header !== "SQLite format 3\u0000")
|
|
10915
11428
|
throw new Error("candidate source is not a SQLite database");
|
|
10916
|
-
|
|
11429
|
+
const { reflinked } = reflinkCopyFile(source, candidate.databasePath);
|
|
11430
|
+
candidate.reflinked = reflinked;
|
|
10917
11431
|
fs.chmodSync(candidate.databasePath, 0o600);
|
|
11432
|
+
if (options.sourceLabel) {
|
|
11433
|
+
const stamp = new Database(candidate.databasePath);
|
|
11434
|
+
try {
|
|
11435
|
+
setMeta(stamp, "seededFromWorktree", options.sourceLabel);
|
|
11436
|
+
setMeta(stamp, "reflinkBacked", String(reflinked));
|
|
11437
|
+
}
|
|
11438
|
+
finally {
|
|
11439
|
+
stamp.close();
|
|
11440
|
+
}
|
|
11441
|
+
}
|
|
10918
11442
|
}
|
|
10919
11443
|
else {
|
|
10920
11444
|
await getOrInitDb(resolved, {
|
|
@@ -11001,11 +11525,27 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
11001
11525
|
}
|
|
11002
11526
|
}
|
|
11003
11527
|
}
|
|
11528
|
+
// A candidate populated from a foreign source (a downloaded shared-index
|
|
11529
|
+
// artifact, or another worktree's database) carries index_state mtimes
|
|
11530
|
+
// stamped wherever that source was built -- not this `resolved` checkout.
|
|
11531
|
+
// Every file outside `reconciled` is, by construction, content-identical
|
|
11532
|
+
// to what the caller already verified via its own diff, so refresh only
|
|
11533
|
+
// the mtime/size bookkeeping against this checkout's actual disk instead
|
|
11534
|
+
// of re-parsing/re-embedding: a plain stat, not a re-index.
|
|
11535
|
+
const reconciledSet = new Set(reconciled);
|
|
11536
|
+
const knownPaths = db
|
|
11537
|
+
.query("SELECT filePath FROM index_state")
|
|
11538
|
+
.all()
|
|
11539
|
+
.map((row) => row.filePath)
|
|
11540
|
+
.filter((file) => !reconciledSet.has(file));
|
|
11541
|
+
for (const file of knownPaths)
|
|
11542
|
+
recordIndexState(db, resolved, file);
|
|
11004
11543
|
await refreshFlowsForApexFiles(db, resolved, reconciled);
|
|
11005
11544
|
persistSymbolIdentities(db, resolved, reconciled);
|
|
11006
11545
|
reconcileTypeScriptDi(db, resolved);
|
|
11007
11546
|
await indexEmbeddings(db, resolved);
|
|
11008
11547
|
setMeta(db, "lastIndexedHead", gitHead(resolved) ?? "");
|
|
11548
|
+
setMeta(db, "knodinVersion", KNODIN_VERSION);
|
|
11009
11549
|
setMeta(db, "lastSuccessfulReconciliation", new Date().toISOString());
|
|
11010
11550
|
recordFreshnessBaseline(resolved, db);
|
|
11011
11551
|
return { reconciled };
|
|
@@ -11182,7 +11722,12 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
11182
11722
|
}
|
|
11183
11723
|
const summary = primaryDef.summary ||
|
|
11184
11724
|
`Symbol "${symbol}" is a ${primaryDef.kind} defined in ${primaryDef.filePath}.`;
|
|
11185
|
-
|
|
11725
|
+
// Availability is carried, not discarded: an empty `source` here would
|
|
11726
|
+
// otherwise be indistinguishable from the not-found and ambiguous
|
|
11727
|
+
// branches above, which also return "".
|
|
11728
|
+
const sourceRead = readSourceRange(targetRepoPath, primaryDef.filePath, primaryDef.startLine, primaryDef.endLine, 500);
|
|
11729
|
+
const source = sourceRead.text;
|
|
11730
|
+
const sourceAvailability = sourceRead.availability.state === "available" ? undefined : sourceRead.availability;
|
|
11186
11731
|
// DIRECT callers/callees (depth 1) — the immediate, navigable neighbours.
|
|
11187
11732
|
// A deep transitive closure is deliberately NOT returned: on a large repo
|
|
11188
11733
|
// it explodes to thousands of (often name-conflated) nodes and megabytes
|
|
@@ -11213,10 +11758,11 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
11213
11758
|
confidence: c.confidence,
|
|
11214
11759
|
...(c.kind?.startsWith("di_")
|
|
11215
11760
|
? {
|
|
11216
|
-
|
|
11217
|
-
|
|
11218
|
-
|
|
11219
|
-
|
|
11761
|
+
// Was an unguarded read: a single missing file threw and aborted
|
|
11762
|
+
// the whole `explain`, rather than costing one caller its
|
|
11763
|
+
// evidence string. `evidenceLine` also serves a sealed
|
|
11764
|
+
// repository, which has no working tree to read from.
|
|
11765
|
+
sourceEvidence: evidenceLine(c.repoPath || targetRepoPath, c.evidenceFile ?? c.filePath, c.evidenceLine ?? c.lineNumber) ?? "",
|
|
11220
11766
|
}
|
|
11221
11767
|
: {}),
|
|
11222
11768
|
...(definition
|
|
@@ -11237,15 +11783,7 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
11237
11783
|
.all(primaryDef.name, primaryDef.filePath)
|
|
11238
11784
|
.map((row) => {
|
|
11239
11785
|
const separator = row.calleeSymbol.indexOf(":");
|
|
11240
|
-
|
|
11241
|
-
try {
|
|
11242
|
-
evidence =
|
|
11243
|
-
fs
|
|
11244
|
-
.readFileSync(path.join(targetRepoPath, row.callerFile), "utf8")
|
|
11245
|
-
.split(/\r?\n/)[row.line - 1]?.trim()
|
|
11246
|
-
.slice(0, 240) ?? "";
|
|
11247
|
-
}
|
|
11248
|
-
catch { }
|
|
11786
|
+
const evidence = evidenceLine(targetRepoPath, row.callerFile, row.line) ?? "";
|
|
11249
11787
|
return {
|
|
11250
11788
|
kind: separator < 0 ? "di_unresolved" : row.calleeSymbol.slice(0, separator),
|
|
11251
11789
|
reason: separator < 0 ? row.calleeSymbol : row.calleeSymbol.slice(separator + 1),
|
|
@@ -11268,6 +11806,7 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
11268
11806
|
symbol,
|
|
11269
11807
|
summary,
|
|
11270
11808
|
source,
|
|
11809
|
+
...(sourceAvailability ? { sourceAvailability } : {}),
|
|
11271
11810
|
callers: allCallers.slice(0, cap),
|
|
11272
11811
|
callerCount: allCallers.length,
|
|
11273
11812
|
callees: allCallees.slice(0, cap),
|
|
@@ -11314,7 +11853,10 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
11314
11853
|
let hasTest = false;
|
|
11315
11854
|
const testPaths = getTestFilePaths(filePath);
|
|
11316
11855
|
for (const tp of testPaths) {
|
|
11317
|
-
|
|
11856
|
+
// Sealed repositories have no working tree, so an `fs.existsSync`
|
|
11857
|
+
// here would report every test file missing and invert the
|
|
11858
|
+
// meaning of this flag rather than failing visibly.
|
|
11859
|
+
if (sourceFileExists(repoPath, tp)) {
|
|
11318
11860
|
hasTest = true;
|
|
11319
11861
|
break;
|
|
11320
11862
|
}
|
|
@@ -11323,13 +11865,16 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
11323
11865
|
let tree = null;
|
|
11324
11866
|
try {
|
|
11325
11867
|
const absPath = path.resolve(repoPath, filePath);
|
|
11326
|
-
|
|
11327
|
-
|
|
11868
|
+
// Content comes from the resolver so a sealed repository can be
|
|
11869
|
+
// reviewed; the language is still chosen from the path, which
|
|
11870
|
+
// needs no working tree.
|
|
11871
|
+
const source = readWholeSource(repoPath, filePath);
|
|
11872
|
+
if (source.availability.state === "available") {
|
|
11328
11873
|
const lang = await getLanguageForFile(absPath);
|
|
11329
11874
|
if (lang) {
|
|
11330
11875
|
const parser = new Parser();
|
|
11331
11876
|
parser.setLanguage(lang);
|
|
11332
|
-
tree = parser.parse(
|
|
11877
|
+
tree = parser.parse(source.text);
|
|
11333
11878
|
}
|
|
11334
11879
|
}
|
|
11335
11880
|
}
|
|
@@ -11706,10 +12251,7 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
11706
12251
|
if (!calleeFile || calleeFile === r.callerFile)
|
|
11707
12252
|
continue;
|
|
11708
12253
|
addExtractedEdge(r.callerFile, calleeFile, r.kind.startsWith("di_") ? r.kind : "calls", r.confidence, r.kind.startsWith("di_")
|
|
11709
|
-
?
|
|
11710
|
-
.readFileSync(path.join(resolvedRepoPath, r.callerFile), "utf8")
|
|
11711
|
-
.split(/\r?\n/)[r.line - 1]?.trim()
|
|
11712
|
-
.slice(0, 240)
|
|
12254
|
+
? evidenceLine(resolvedRepoPath, r.callerFile, r.line)
|
|
11713
12255
|
: undefined);
|
|
11714
12256
|
}
|
|
11715
12257
|
return {
|
|
@@ -11823,10 +12365,7 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
11823
12365
|
}
|
|
11824
12366
|
if (calleePrefixed && calleePrefixed !== callerPrefixed) {
|
|
11825
12367
|
addExtractedEdge(callerPrefixed, calleePrefixed, r.kind.startsWith("di_") ? r.kind : "calls", r.confidence, r.kind.startsWith("di_")
|
|
11826
|
-
?
|
|
11827
|
-
.readFileSync(path.join(repo.path, r.callerFile), "utf8")
|
|
11828
|
-
.split(/\r?\n/)[r.line - 1]?.trim()
|
|
11829
|
-
.slice(0, 240)
|
|
12368
|
+
? evidenceLine(repo.path, r.callerFile, r.line)
|
|
11830
12369
|
: undefined);
|
|
11831
12370
|
}
|
|
11832
12371
|
// 2. Cross-repo Routing: If it does not exist locally but matches endpoints in other repos
|
|
@@ -12288,6 +12827,11 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
12288
12827
|
missing: { files: missingFiles, records },
|
|
12289
12828
|
lastSuccessfulReconciliation: getMeta(db, "lastSuccessfulReconciliation") ?? null,
|
|
12290
12829
|
lastIndexedHead: getMeta(db, "lastIndexedHead") ?? "",
|
|
12830
|
+
seededFromWorktree: getMeta(db, "seededFromWorktree") || undefined,
|
|
12831
|
+
reflinkBacked: (() => {
|
|
12832
|
+
const value = getMeta(db, "reflinkBacked");
|
|
12833
|
+
return value === "" || value === null ? undefined : value === "true";
|
|
12834
|
+
})(),
|
|
12291
12835
|
freshness,
|
|
12292
12836
|
indexGeneration,
|
|
12293
12837
|
verification: {
|
|
@@ -12675,6 +13219,7 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
12675
13219
|
if (after.freshness.state === "stale-head" || after.freshness.state === "unknown") {
|
|
12676
13220
|
setMeta(db, "lastIndexedHead", gitHead(resolved) ?? "");
|
|
12677
13221
|
}
|
|
13222
|
+
setMeta(db, "knodinVersion", KNODIN_VERSION);
|
|
12678
13223
|
setMeta(db, "lastSuccessfulReconciliation", new Date().toISOString());
|
|
12679
13224
|
recordFreshnessBaseline(resolved, db);
|
|
12680
13225
|
statusCache.delete(databaseCacheKey(resolved, openPolicy.databasePath));
|
|
@@ -13014,6 +13559,7 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
13014
13559
|
if (contentVerified &&
|
|
13015
13560
|
(health.freshness.state === "stale-head" || health.freshness.state === "unknown")) {
|
|
13016
13561
|
setMeta(db, "lastIndexedHead", gitHead(repoPath) ?? "");
|
|
13562
|
+
setMeta(db, "knodinVersion", KNODIN_VERSION);
|
|
13017
13563
|
setMeta(db, "lastSuccessfulReconciliation", new Date().toISOString());
|
|
13018
13564
|
recordFreshnessBaseline(repoPath, db);
|
|
13019
13565
|
statusCache.delete(databaseCacheKey(repoPath, openPolicy.databasePath));
|
|
@@ -13031,6 +13577,7 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
13031
13577
|
// the audit proves that no requested or unrequested drift remains.
|
|
13032
13578
|
if (files && files.length > 0) {
|
|
13033
13579
|
setMeta(db, "lastIndexedHead", gitHead(repoPath) ?? "");
|
|
13580
|
+
setMeta(db, "knodinVersion", KNODIN_VERSION);
|
|
13034
13581
|
setMeta(db, "lastSuccessfulReconciliation", new Date().toISOString());
|
|
13035
13582
|
recordFreshnessBaseline(repoPath, db);
|
|
13036
13583
|
}
|
|
@@ -14104,7 +14651,10 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
14104
14651
|
if (!impactOptions.includeDataFlow)
|
|
14105
14652
|
return undefined;
|
|
14106
14653
|
try {
|
|
14107
|
-
const
|
|
14654
|
+
const whole = readWholeSource(resolvedRepoPath, file);
|
|
14655
|
+
if (whole.availability.state !== "available")
|
|
14656
|
+
return undefined;
|
|
14657
|
+
const text = whole.text.split("\n")[line - 1];
|
|
14108
14658
|
if (!text)
|
|
14109
14659
|
return undefined;
|
|
14110
14660
|
const escaped = callee.replace(/[.*+?^${}()|[\]\\]/g, String.raw `\$&`);
|
|
@@ -14437,11 +14987,8 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
14437
14987
|
const readSource = (relFile) => {
|
|
14438
14988
|
if (sourceCache.has(relFile))
|
|
14439
14989
|
return sourceCache.get(relFile) ?? null;
|
|
14440
|
-
|
|
14441
|
-
|
|
14442
|
-
src = fs.readFileSync(path.join(repoPath, relFile), "utf-8");
|
|
14443
|
-
}
|
|
14444
|
-
catch { }
|
|
14990
|
+
const read = readWholeSource(repoPath, relFile);
|
|
14991
|
+
const src = read.availability.state === "available" ? read.text : null;
|
|
14445
14992
|
sourceCache.set(relFile, src);
|
|
14446
14993
|
return src;
|
|
14447
14994
|
};
|
|
@@ -14505,6 +15052,13 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
14505
15052
|
return false;
|
|
14506
15053
|
if (isTestFilePath(r.filePath))
|
|
14507
15054
|
return false;
|
|
15055
|
+
// Unreadable file: there is no basis to judge this symbol either
|
|
15056
|
+
// way. Everything below infers "unused" from the ABSENCE of
|
|
15057
|
+
// evidence, so a failed read silently produced a confident "dead"
|
|
15058
|
+
// verdict — and acting on that means deleting live code. Unknown
|
|
15059
|
+
// must not collapse into dead.
|
|
15060
|
+
if (readSource(r.filePath) === null)
|
|
15061
|
+
return false;
|
|
14508
15062
|
if (isExportedSymbol(r.filePath, r.name)) {
|
|
14509
15063
|
const importers = importersByFile.get(r.filePath) ?? [];
|
|
14510
15064
|
// An exported symbol in an otherwise unimported file may be a
|
|
@@ -14575,7 +15129,7 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
14575
15129
|
});
|
|
14576
15130
|
if (!destination.selected)
|
|
14577
15131
|
return empty("The destination endpoint was not resolved uniquely.");
|
|
14578
|
-
if (selectedTarget.kind !== "
|
|
15132
|
+
if (selectedTarget.kind !== "salesforce_flow" ||
|
|
14579
15133
|
!selectedTarget.filePath.endsWith(".flow-meta.xml") ||
|
|
14580
15134
|
!destination.selected.filePath.endsWith(".cls"))
|
|
14581
15135
|
return empty("The endpoints are not a supported Salesforce Flow-to-Apex boundary.");
|
|
@@ -14583,7 +15137,10 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
14583
15137
|
const actionName = `${className}.${destination.selected.name}`;
|
|
14584
15138
|
let flowEvidence = "";
|
|
14585
15139
|
try {
|
|
14586
|
-
const
|
|
15140
|
+
const flowRead = readWholeSource(primary.path, selectedTarget.filePath);
|
|
15141
|
+
if (flowRead.availability.state !== "available")
|
|
15142
|
+
throw new Error(flowRead.availability.reason);
|
|
15143
|
+
const flowSource = flowRead.text;
|
|
14587
15144
|
const matchingActions = [
|
|
14588
15145
|
...flowSource.matchAll(/<actionCalls>([\s\S]*?)<\/actionCalls>/g),
|
|
14589
15146
|
]
|
|
@@ -14917,7 +15474,10 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
14917
15474
|
return undefined;
|
|
14918
15475
|
try {
|
|
14919
15476
|
const absolutePath = path.join(resolvedRepoPath, ref.callerFile);
|
|
14920
|
-
const
|
|
15477
|
+
const read = readWholeSource(resolvedRepoPath, ref.callerFile);
|
|
15478
|
+
if (read.availability.state !== "available")
|
|
15479
|
+
return undefined;
|
|
15480
|
+
const content = read.text;
|
|
14921
15481
|
const language = await getLanguageForFile(absolutePath);
|
|
14922
15482
|
if (!language)
|
|
14923
15483
|
return undefined;
|
|
@@ -15109,17 +15669,19 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
15109
15669
|
}
|
|
15110
15670
|
else {
|
|
15111
15671
|
inputs = indexed.flatMap(({ filePath }) => {
|
|
15112
|
-
const
|
|
15113
|
-
|
|
15114
|
-
const content = fs.readFileSync(absolute, "utf8");
|
|
15115
|
-
if (fileDriftedFromIndexState(db, resolvedRepoPath, filePath))
|
|
15116
|
-
diskDrifted = true;
|
|
15117
|
-
return [{ path: filePath, content }];
|
|
15118
|
-
}
|
|
15119
|
-
catch {
|
|
15672
|
+
const read = readWholeSource(resolvedRepoPath, filePath);
|
|
15673
|
+
if (read.availability.state !== "available") {
|
|
15120
15674
|
diskDrifted = true;
|
|
15121
15675
|
return [];
|
|
15122
15676
|
}
|
|
15677
|
+
// Drift is a working-tree concept: a sealed artifact's bytes
|
|
15678
|
+
// came from the attested commit and cannot have moved since,
|
|
15679
|
+
// and the mtime/size comparison would report every file
|
|
15680
|
+
// drifted because there is no file to stat.
|
|
15681
|
+
if (!isSealedRepository(resolvedRepoPath) &&
|
|
15682
|
+
fileDriftedFromIndexState(db, resolvedRepoPath, filePath))
|
|
15683
|
+
diskDrifted = true;
|
|
15684
|
+
return [{ path: filePath, content: read.text }];
|
|
15123
15685
|
});
|
|
15124
15686
|
diskFingerprint = resourceFingerprint(inputs);
|
|
15125
15687
|
if (!diskDrifted)
|
|
@@ -15168,8 +15730,10 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
15168
15730
|
: undefined;
|
|
15169
15731
|
if (!language)
|
|
15170
15732
|
return finish([]);
|
|
15171
|
-
const
|
|
15172
|
-
|
|
15733
|
+
const flowRead = readWholeSource(resolvedRepoPath, selectedTarget.filePath);
|
|
15734
|
+
if (flowRead.availability.state !== "available")
|
|
15735
|
+
return finish([]);
|
|
15736
|
+
const sourceLines = flowRead.text.split(/\r?\n/);
|
|
15173
15737
|
const functionLines = sourceLines.slice(selectedTarget.startLine - 1, selectedTarget.endLine);
|
|
15174
15738
|
const facts = [];
|
|
15175
15739
|
const variableFilter = options.flowVariable;
|