@ngockhoale/ukit 2.6.8 → 2.6.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +52 -0
- package/package.json +1 -1
- package/scripts/index/build-index.mjs +2 -1
- package/scripts/index/query-index.mjs +2 -0
- package/src/cli/commands/indexTools.js +5 -0
- package/src/cli/commands/install.js +2 -1
- package/src/core/codeintel/analogy.js +197 -0
- package/src/core/codeintel/cochange.js +205 -0
- package/src/core/codeintel/compiler.js +82 -9
- package/src/core/codeintel/graph.js +291 -0
- package/src/core/codeintel/impact.js +23 -0
- package/src/core/codeintel/packet.js +1 -0
- package/src/core/codeintel/retriever.js +49 -3
- package/src/core/codeintel/summaries.js +194 -0
- package/src/core/codeintel/vectorProvider.js +213 -0
- package/src/core/runtimeConfig.js +50 -2
- package/src/index/buildIndex.js +29 -0
- package/src/index/paths.js +2 -0
- package/templates/ukit/storage/config.json +7 -2
|
@@ -0,0 +1,291 @@
|
|
|
1
|
+
import fs from 'node:fs/promises';
|
|
2
|
+
import path from 'node:path';
|
|
3
|
+
|
|
4
|
+
import { getArtifactPath, INDEX_ARTIFACTS, INDEX_SCHEMA_VERSION } from '../../index/paths.js';
|
|
5
|
+
import { readJsonIfExists, writeJson } from '../fileOps.js';
|
|
6
|
+
import { createEdge } from './providers.js';
|
|
7
|
+
import { COCHANGE_SCHEMA_VERSION } from './cochange.js';
|
|
8
|
+
|
|
9
|
+
// Graph contract + JSON store (SPEC §9 — CI-305). Builds `.cache/index/codegraph.json`
|
|
10
|
+
// from existing index artifacts (imports.json + calls.json, + cochange.json when
|
|
11
|
+
// present) and exposes a GraphProvider seam so SCIP/PDG/LSP backends can slot in
|
|
12
|
+
// later without touching consumers. No SQLite, no new deps — providers other than
|
|
13
|
+
// 'json' are contract-only NullGraphProvider stubs this cycle.
|
|
14
|
+
|
|
15
|
+
export const GRAPH_SCHEMA_VERSION = 1;
|
|
16
|
+
|
|
17
|
+
const VALID_GRAPH_STORES = new Set(['json', 'scip', 'pdg', 'lsp', 'null']);
|
|
18
|
+
|
|
19
|
+
async function readArtifact(rootDir, name, expectedSchemaVersion) {
|
|
20
|
+
try {
|
|
21
|
+
const raw = await fs.readFile(getArtifactPath(rootDir, name), 'utf8');
|
|
22
|
+
const parsed = JSON.parse(raw);
|
|
23
|
+
if (!parsed || typeof parsed !== 'object') return null;
|
|
24
|
+
if (expectedSchemaVersion !== undefined && parsed.schemaVersion !== expectedSchemaVersion) {
|
|
25
|
+
return null;
|
|
26
|
+
}
|
|
27
|
+
return parsed;
|
|
28
|
+
} catch {
|
|
29
|
+
return null;
|
|
30
|
+
}
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
// Same specifier-resolution discipline as impact.js — local copy so this lane
|
|
34
|
+
// does not import same-wave sibling internals.
|
|
35
|
+
function resolveRelativeSpecifier(fromFile, specifier, fileSet) {
|
|
36
|
+
const candidates = [];
|
|
37
|
+
if (specifier.startsWith('.')) {
|
|
38
|
+
const base = path.posix.normalize(path.posix.join(path.posix.dirname(fromFile), specifier));
|
|
39
|
+
candidates.push(base, `${base}.js`, `${base}.ts`, `${base}.mjs`, `${base}.jsx`, `${base}.tsx`, `${base}/index.js`, `${base}/index.ts`);
|
|
40
|
+
} else {
|
|
41
|
+
candidates.push(specifier);
|
|
42
|
+
}
|
|
43
|
+
for (const candidate of candidates) {
|
|
44
|
+
if (fileSet.has(candidate)) return candidate;
|
|
45
|
+
}
|
|
46
|
+
return null;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* buildGraphArtifact(projectRoot) → artifact | null
|
|
51
|
+
*
|
|
52
|
+
* Merges imports.json + calls.json (+ cochange.json when present) into a
|
|
53
|
+
* codegraph.json artifact: { schemaVersion, generatedAt, nodes, edges }.
|
|
54
|
+
* Node ids: files → their relative path, symbols → `symbol:<name>`.
|
|
55
|
+
* Edges use the canonical createEdge shape. Writes the artifact atomically via
|
|
56
|
+
* fileOps.writeJson; returns null when no source artifacts exist. Never throws
|
|
57
|
+
* on missing/partial sources.
|
|
58
|
+
*/
|
|
59
|
+
export async function buildGraphArtifact(projectRoot) {
|
|
60
|
+
const rootDir = path.resolve(projectRoot ?? process.cwd());
|
|
61
|
+
|
|
62
|
+
const [importsArtifact, callsArtifact, cochangeArtifact, filesArtifact] = await Promise.all([
|
|
63
|
+
readArtifact(rootDir, INDEX_ARTIFACTS.imports, INDEX_SCHEMA_VERSION),
|
|
64
|
+
readArtifact(rootDir, INDEX_ARTIFACTS.calls, INDEX_SCHEMA_VERSION),
|
|
65
|
+
readArtifact(rootDir, INDEX_ARTIFACTS.cochange, COCHANGE_SCHEMA_VERSION),
|
|
66
|
+
readArtifact(rootDir, INDEX_ARTIFACTS.files, INDEX_SCHEMA_VERSION),
|
|
67
|
+
]);
|
|
68
|
+
|
|
69
|
+
if (!importsArtifact && !callsArtifact && !cochangeArtifact) {
|
|
70
|
+
return null;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
const fileSet = new Set();
|
|
74
|
+
for (const file of filesArtifact?.items ?? []) {
|
|
75
|
+
if (file?.filePath) fileSet.add(file.filePath);
|
|
76
|
+
}
|
|
77
|
+
for (const imp of importsArtifact?.items ?? []) {
|
|
78
|
+
if (imp?.from) fileSet.add(imp.from);
|
|
79
|
+
}
|
|
80
|
+
for (const call of callsArtifact?.items ?? []) {
|
|
81
|
+
if (call?.filePath) fileSet.add(call.filePath);
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
const nodes = new Map(); // id → node
|
|
85
|
+
const edges = new Map(); // dedupe key → edge
|
|
86
|
+
const addFileNode = (filePath) => {
|
|
87
|
+
if (filePath && !nodes.has(filePath)) {
|
|
88
|
+
nodes.set(filePath, { id: filePath, kind: 'file', path: filePath });
|
|
89
|
+
}
|
|
90
|
+
};
|
|
91
|
+
const addSymbolNode = (name, filePath) => {
|
|
92
|
+
if (!name) return null;
|
|
93
|
+
const id = `symbol:${name}`;
|
|
94
|
+
if (!nodes.has(id)) {
|
|
95
|
+
nodes.set(id, { id, kind: 'symbol', path: filePath ?? null });
|
|
96
|
+
}
|
|
97
|
+
return id;
|
|
98
|
+
};
|
|
99
|
+
const addEdge = (edge, dedupeExtra = '') => {
|
|
100
|
+
const key = `${edge.from}»${edge.to}»${edge.kind}»${dedupeExtra}`;
|
|
101
|
+
if (!edges.has(key)) edges.set(key, edge);
|
|
102
|
+
};
|
|
103
|
+
|
|
104
|
+
// import edges: from-file → resolved target file
|
|
105
|
+
for (const imp of importsArtifact?.items ?? []) {
|
|
106
|
+
if (!imp?.from || !imp?.to) continue;
|
|
107
|
+
addFileNode(imp.from);
|
|
108
|
+
const target = resolveRelativeSpecifier(imp.from, imp.to, fileSet) ?? imp.to;
|
|
109
|
+
if (target === imp.from) continue;
|
|
110
|
+
addFileNode(target);
|
|
111
|
+
addEdge(createEdge({
|
|
112
|
+
from: imp.from,
|
|
113
|
+
to: target,
|
|
114
|
+
kind: 'import',
|
|
115
|
+
provider: 'json-graph',
|
|
116
|
+
evidence: INDEX_ARTIFACTS.imports,
|
|
117
|
+
}));
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
// call edges: caller file → callee file (resolved via symbol→file map);
|
|
121
|
+
// symbol nodes recorded regardless so queries can address them.
|
|
122
|
+
const symbolToFile = new Map();
|
|
123
|
+
for (const call of callsArtifact?.items ?? []) {
|
|
124
|
+
if (call?.filePath && call?.symbol && !symbolToFile.has(call.symbol)) {
|
|
125
|
+
symbolToFile.set(call.symbol, call.filePath);
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
for (const call of callsArtifact?.items ?? []) {
|
|
129
|
+
if (!call?.filePath) continue;
|
|
130
|
+
addFileNode(call.filePath);
|
|
131
|
+
if (call?.symbol) addSymbolNode(call.symbol, call.filePath);
|
|
132
|
+
for (const name of call.calls ?? []) {
|
|
133
|
+
const target = symbolToFile.get(name);
|
|
134
|
+
if (!target || target === call.filePath) continue;
|
|
135
|
+
addEdge(createEdge({
|
|
136
|
+
from: call.filePath,
|
|
137
|
+
to: target,
|
|
138
|
+
kind: 'call',
|
|
139
|
+
provider: 'json-graph',
|
|
140
|
+
evidence: INDEX_ARTIFACTS.calls,
|
|
141
|
+
}), name);
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
// co-change edges: pair endpoints become file nodes; count feeds confidence.
|
|
146
|
+
for (const pair of cochangeArtifact?.pairs ?? []) {
|
|
147
|
+
if (!pair?.a || !pair?.b) continue;
|
|
148
|
+
addFileNode(pair.a);
|
|
149
|
+
addFileNode(pair.b);
|
|
150
|
+
const confidence = Math.min(1, (pair.count ?? 1) / 10);
|
|
151
|
+
addEdge(createEdge({
|
|
152
|
+
from: pair.a,
|
|
153
|
+
to: pair.b,
|
|
154
|
+
kind: 'cochange',
|
|
155
|
+
confidence,
|
|
156
|
+
provider: 'json-graph',
|
|
157
|
+
evidence: INDEX_ARTIFACTS.cochange,
|
|
158
|
+
}), String(pair.count ?? ''));
|
|
159
|
+
addEdge(createEdge({
|
|
160
|
+
from: pair.b,
|
|
161
|
+
to: pair.a,
|
|
162
|
+
kind: 'cochange',
|
|
163
|
+
confidence,
|
|
164
|
+
provider: 'json-graph',
|
|
165
|
+
evidence: INDEX_ARTIFACTS.cochange,
|
|
166
|
+
}), String(pair.count ?? ''));
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
const artifact = {
|
|
170
|
+
schemaVersion: GRAPH_SCHEMA_VERSION,
|
|
171
|
+
generatedAt: new Date().toISOString(),
|
|
172
|
+
nodes: [...nodes.values()],
|
|
173
|
+
edges: [...edges.values()],
|
|
174
|
+
};
|
|
175
|
+
|
|
176
|
+
try {
|
|
177
|
+
// Content-gated write: unchanged graph (ignoring generatedAt) must not
|
|
178
|
+
// bump the artifact mtime — refresh's "no rewrite when nothing changed"
|
|
179
|
+
// guarantee covers derived artifacts too.
|
|
180
|
+
const graphPath = getArtifactPath(rootDir, INDEX_ARTIFACTS.codegraph);
|
|
181
|
+
const existing = await readJsonIfExists(graphPath);
|
|
182
|
+
const strip = ({ generatedAt, ...rest }) => rest;
|
|
183
|
+
if (!existing || JSON.stringify(strip(existing)) !== JSON.stringify(strip(artifact))) {
|
|
184
|
+
await writeJson(graphPath, artifact);
|
|
185
|
+
}
|
|
186
|
+
} catch {
|
|
187
|
+
// Artifact-write failure (read-only .cache, EACCES) is non-fatal — the
|
|
188
|
+
// in-memory artifact is still returned for the caller/summary.
|
|
189
|
+
}
|
|
190
|
+
return artifact;
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
/**
|
|
194
|
+
* JsonGraphStore — GraphProvider over `.cache/index/codegraph.json`.
|
|
195
|
+
* Schema-guarded (GRAPH_SCHEMA_VERSION); a missing/mismatched/corrupt artifact
|
|
196
|
+
* yields empty results per op, never throws.
|
|
197
|
+
*/
|
|
198
|
+
export class JsonGraphStore {
|
|
199
|
+
constructor(projectRoot) {
|
|
200
|
+
this.name = 'json';
|
|
201
|
+
this.rootDir = path.resolve(projectRoot ?? process.cwd());
|
|
202
|
+
this._artifactPromise = null;
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
capabilities() {
|
|
206
|
+
return { nodes: true, edges: true, queries: true };
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
async _load() {
|
|
210
|
+
if (!this._artifactPromise) {
|
|
211
|
+
this._artifactPromise = (async () => {
|
|
212
|
+
const artifact = await readJsonIfExists(getArtifactPath(this.rootDir, INDEX_ARTIFACTS.codegraph));
|
|
213
|
+
if (!artifact || typeof artifact !== 'object') return null;
|
|
214
|
+
if (artifact.schemaVersion !== GRAPH_SCHEMA_VERSION) return null;
|
|
215
|
+
if (!Array.isArray(artifact.nodes) || !Array.isArray(artifact.edges)) return null;
|
|
216
|
+
return artifact;
|
|
217
|
+
})();
|
|
218
|
+
}
|
|
219
|
+
return this._artifactPromise;
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
async nodes() {
|
|
223
|
+
const artifact = await this._load();
|
|
224
|
+
return artifact?.nodes ?? [];
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
async edgesFrom(id) {
|
|
228
|
+
const artifact = await this._load();
|
|
229
|
+
if (!artifact || typeof id !== 'string') return [];
|
|
230
|
+
return artifact.edges.filter((edge) => edge?.from === id);
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
/**
|
|
234
|
+
* query(q) → edge[] | null. Resolves a node id (file path or `symbol:` id) to
|
|
235
|
+
* its outgoing edges; null when the node is absent or no artifact exists.
|
|
236
|
+
*/
|
|
237
|
+
async query(q) {
|
|
238
|
+
const artifact = await this._load();
|
|
239
|
+
if (!artifact) return null;
|
|
240
|
+
const nodeId = typeof q === 'object' && q !== null ? q.path ?? q.id ?? null : q;
|
|
241
|
+
if (typeof nodeId !== 'string' || nodeId === '') return null;
|
|
242
|
+
const exists = artifact.nodes.some((node) => node?.id === nodeId);
|
|
243
|
+
if (!exists) return null;
|
|
244
|
+
return artifact.edges.filter((edge) => edge?.from === nodeId);
|
|
245
|
+
}
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
/**
|
|
249
|
+
* NullGraphProvider — contract-only stub for SCIP/PDG/LSP/'null' backends.
|
|
250
|
+
* All capabilities false; every op returns empty/null. Never throws.
|
|
251
|
+
*/
|
|
252
|
+
export class NullGraphProvider {
|
|
253
|
+
constructor(backend = 'null') {
|
|
254
|
+
this.name = backend;
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
capabilities() {
|
|
258
|
+
return { nodes: false, edges: false, queries: false };
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
async nodes() {
|
|
262
|
+
return [];
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
async edgesFrom() {
|
|
266
|
+
return [];
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
async query() {
|
|
270
|
+
return null;
|
|
271
|
+
}
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
/**
|
|
275
|
+
* createGraphProvider({ projectRoot?, config? }) → GraphProvider
|
|
276
|
+
* codeIntel.graph.store: 'json' (default/unset) → JsonGraphStore;
|
|
277
|
+
* 'scip' | 'pdg' | 'lsp' | 'null' → NullGraphProvider (contract-only seam).
|
|
278
|
+
*/
|
|
279
|
+
export function createGraphProvider({ projectRoot, config } = {}) {
|
|
280
|
+
if (config?.codeIntel?.graph?.enabled === false) {
|
|
281
|
+
return new NullGraphProvider('disabled');
|
|
282
|
+
}
|
|
283
|
+
const store = config?.codeIntel?.graph?.store;
|
|
284
|
+
if (store === undefined || store === 'json') {
|
|
285
|
+
return new JsonGraphStore(projectRoot);
|
|
286
|
+
}
|
|
287
|
+
if (VALID_GRAPH_STORES.has(store)) {
|
|
288
|
+
return new NullGraphProvider(store);
|
|
289
|
+
}
|
|
290
|
+
return new NullGraphProvider('null');
|
|
291
|
+
}
|
|
@@ -3,6 +3,7 @@ import path from 'node:path';
|
|
|
3
3
|
|
|
4
4
|
import { getArtifactPath, INDEX_ARTIFACTS, INDEX_SCHEMA_VERSION } from '../../index/paths.js';
|
|
5
5
|
import { loadRuntimeConfig } from '../runtimeConfig.js';
|
|
6
|
+
import { readCoChanges, coChangeEdges } from './cochange.js';
|
|
6
7
|
|
|
7
8
|
// Multi-hop impact/trace engine (SPEC §4) — cycle-safe BFS over imports.json +
|
|
8
9
|
// calls.json edges. `dependents` walks reverse edges (who imports the seed),
|
|
@@ -247,5 +248,27 @@ export async function impactSet(projectRoot, seeds, {
|
|
|
247
248
|
};
|
|
248
249
|
if (frontierCheck('rev') || (wantDeps && frontierCheck('fwd'))) truncated = true;
|
|
249
250
|
|
|
251
|
+
// Co-change lane (SPEC §6): when enabled (in-code default true) and the
|
|
252
|
+
// direction touches dependents, union kind:'cochange' edges for seed peers
|
|
253
|
+
// (count ≥ minCount) as context — peers join `nodes` at hop:0 with via:[],
|
|
254
|
+
// not as graph hops. Missing artifact → omitted reason; never fatal.
|
|
255
|
+
const cochangeConfig = config?.codeIntel?.cochange;
|
|
256
|
+
const cochangeEnabled = cochangeConfig?.enabled !== false;
|
|
257
|
+
if (cochangeEnabled && wantRev) {
|
|
258
|
+
const cochangeArtifact = await readCoChanges(rootDir);
|
|
259
|
+
if (!cochangeArtifact) {
|
|
260
|
+
omitted.push({ what: 'cochange-lane', why: 'index-missing' });
|
|
261
|
+
} else {
|
|
262
|
+
const minCount = typeof cochangeConfig?.minCount === 'number' ? cochangeConfig.minCount : 2;
|
|
263
|
+
const coEdges = await coChangeEdges(rootDir, seedFiles, { minCount });
|
|
264
|
+
for (const edge of coEdges) {
|
|
265
|
+
emitEdge(edge);
|
|
266
|
+
if (!nodes.has(edge.to)) {
|
|
267
|
+
nodes.set(edge.to, { file: edge.to, hop: 0, via: [] });
|
|
268
|
+
}
|
|
269
|
+
}
|
|
270
|
+
}
|
|
271
|
+
}
|
|
272
|
+
|
|
250
273
|
return { nodes: [...nodes.values()], edges, truncated, omitted };
|
|
251
274
|
}
|
|
@@ -108,6 +108,7 @@ export function packetToText(packet) {
|
|
|
108
108
|
lines.push('## Evidence');
|
|
109
109
|
for (const e of packet.evidence ?? []) {
|
|
110
110
|
lines.push(`- [${e.level}/${e.source}] ${e.path ?? ''} — ${e.why ?? ''}`);
|
|
111
|
+
if (e.summary) lines.push(` summary: ${e.summary}`);
|
|
111
112
|
if (e.excerpt) lines.push(` ${e.excerpt}`);
|
|
112
113
|
}
|
|
113
114
|
if ((packet.evidence ?? []).length === 0) lines.push('(none)');
|
|
@@ -5,6 +5,7 @@ import { getArtifactPath, INDEX_ARTIFACTS, INDEX_SCHEMA_VERSION, normalizeRelati
|
|
|
5
5
|
import { loadRuntimeConfig } from '../runtimeConfig.js';
|
|
6
6
|
import { createEdge, getSemanticProvider, IndexFileSyntaxProvider } from './providers.js';
|
|
7
7
|
import { createSemanticProvider } from './semanticProvider.js';
|
|
8
|
+
import { createEmbeddingProvider } from './vectorProvider.js';
|
|
8
9
|
|
|
9
10
|
// Hybrid Retriever (SPEC §3): each lane produces a ranked list of file paths,
|
|
10
11
|
// then lanes merge via Reciprocal Rank Fusion (k=60, per-lane weights from
|
|
@@ -15,8 +16,8 @@ import { createSemanticProvider } from './semanticProvider.js';
|
|
|
15
16
|
const DEFAULT_LIMIT = 20;
|
|
16
17
|
const DEFAULT_BM25 = { k1: 1.2, b: 0.75 };
|
|
17
18
|
const DEFAULT_RRF_K = 60;
|
|
18
|
-
const DEFAULT_WEIGHTS = { exact: 1.0, symbol: 1.2, bm25: 0.8, semantic: 1.0 };
|
|
19
|
-
const LANE_ORDER = ['exact', 'symbol', 'bm25', 'semantic'];
|
|
19
|
+
const DEFAULT_WEIGHTS = { exact: 1.0, symbol: 1.2, bm25: 0.8, vector: 0.6, semantic: 1.0 };
|
|
20
|
+
const LANE_ORDER = ['exact', 'symbol', 'bm25', 'vector', 'semantic'];
|
|
20
21
|
|
|
21
22
|
async function readArtifact(rootDir, name) {
|
|
22
23
|
try {
|
|
@@ -56,6 +57,7 @@ function retrieverDefaults(config) {
|
|
|
56
57
|
exact: weight('exact'),
|
|
57
58
|
symbol: weight('symbol'),
|
|
58
59
|
bm25: weight('bm25'),
|
|
60
|
+
vector: weight('vector'),
|
|
59
61
|
semantic: weight('semantic'),
|
|
60
62
|
},
|
|
61
63
|
};
|
|
@@ -192,6 +194,39 @@ async function semanticLane(rootDir, normalizedQuery, config, fileSet, snapshotI
|
|
|
192
194
|
return { lane, why: lane.length === 0 ? 'no-match' : null };
|
|
193
195
|
}
|
|
194
196
|
|
|
197
|
+
// Vector lane (SPEC §2/§3): dep-free hashed embedding provider; mirrors the
|
|
198
|
+
// semantic lane. Edges' `to` carries "file:0" — reduce to a ranked file list.
|
|
199
|
+
async function vectorLane(rootDir, normalizedQuery, config, fileSet, snapshotId) {
|
|
200
|
+
const provider = createEmbeddingProvider({ projectRoot: rootDir, config });
|
|
201
|
+
if (!provider || provider.name === 'null' || typeof provider.resolve !== 'function') {
|
|
202
|
+
return { lane: null, why: 'unavailable' };
|
|
203
|
+
}
|
|
204
|
+
let edges;
|
|
205
|
+
try {
|
|
206
|
+
edges = await provider.resolve(normalizedQuery, { limit: DEFAULT_LIMIT, snapshot: snapshotId });
|
|
207
|
+
} catch {
|
|
208
|
+
return { lane: null, why: 'provider-error' };
|
|
209
|
+
}
|
|
210
|
+
if (edges === null || edges === undefined) {
|
|
211
|
+
return { lane: null, why: 'unavailable' };
|
|
212
|
+
}
|
|
213
|
+
if (!Array.isArray(edges) || edges.length === 0) {
|
|
214
|
+
return { lane: [], why: 'no-match' };
|
|
215
|
+
}
|
|
216
|
+
const lane = [];
|
|
217
|
+
const seen = new Set();
|
|
218
|
+
for (const edge of edges) {
|
|
219
|
+
const target = typeof edge?.to === 'string' ? edge.to.split(':')[0] : null;
|
|
220
|
+
if (!target) continue;
|
|
221
|
+
const rel = path.isAbsolute(target) ? normalizeRelative(rootDir, target) : target;
|
|
222
|
+
if (seen.has(rel)) continue;
|
|
223
|
+
if (fileSet.size > 0 && !fileSet.has(rel)) continue;
|
|
224
|
+
seen.add(rel);
|
|
225
|
+
lane.push(laneEntry(rel, `vector ${edge.kind ?? 'match'} for "${normalizedQuery}"`));
|
|
226
|
+
}
|
|
227
|
+
return { lane, why: lane.length === 0 ? 'no-match' : null };
|
|
228
|
+
}
|
|
229
|
+
|
|
195
230
|
function rrfMerge(lanes, weights, rrfK) {
|
|
196
231
|
const scores = new Map();
|
|
197
232
|
const meta = new Map();
|
|
@@ -199,6 +234,7 @@ function rrfMerge(lanes, weights, rrfK) {
|
|
|
199
234
|
const lane = lanes.get(laneName);
|
|
200
235
|
if (!lane) continue;
|
|
201
236
|
const w = weights[laneName] ?? 0;
|
|
237
|
+
if (w <= 0) continue;
|
|
202
238
|
lane.forEach((entry, index) => {
|
|
203
239
|
scores.set(entry.path, (scores.get(entry.path) ?? 0) + w / (rrfK + index + 1));
|
|
204
240
|
if (!meta.has(entry.path)) meta.set(entry.path, entry);
|
|
@@ -312,7 +348,17 @@ export async function retrieve(projectRoot, query, { mode = 'search', limit, sna
|
|
|
312
348
|
lanes.set('bm25', bm25.slice(0, Math.max(effectiveLimit, 100)));
|
|
313
349
|
}
|
|
314
350
|
|
|
315
|
-
// Lane 4 —
|
|
351
|
+
// Lane 4 — dep-free hashed vector provider (index-bound; absent → omitted).
|
|
352
|
+
const vector = await vectorLane(rootDir, normalizedQuery, config, fileSet, snapshotId);
|
|
353
|
+
if (vector.lane === null) {
|
|
354
|
+
omitted.push({ what: 'vector-lane', why: vector.why });
|
|
355
|
+
} else if (vector.lane.length > 0) {
|
|
356
|
+
lanes.set('vector', vector.lane);
|
|
357
|
+
} else {
|
|
358
|
+
omitted.push({ what: 'vector-lane', why: vector.why });
|
|
359
|
+
}
|
|
360
|
+
|
|
361
|
+
// Lane 5 — semantic provider (defensive; absent → skipped, never fatal).
|
|
316
362
|
const semantic = await semanticLane(rootDir, normalizedQuery, config, fileSet, snapshotId);
|
|
317
363
|
if (semantic.lane === null) {
|
|
318
364
|
omitted.push({ what: 'semantic-lane', why: semantic.why });
|
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
import fs from 'node:fs/promises';
|
|
2
|
+
import path from 'node:path';
|
|
3
|
+
|
|
4
|
+
import { getSyntaxProvider } from './providers.js';
|
|
5
|
+
import { resolveProjectRelativePath } from '../fileOps.js';
|
|
6
|
+
|
|
7
|
+
// Deterministic extractive summaries (SPEC §7/§8, CI-304) — the L3 detail tier.
|
|
8
|
+
// No model calls, no clock/random, no new deps. File reads are bounded to the
|
|
9
|
+
// first MAX_READ_LINES lines; extraction is pure text scanning over docblocks,
|
|
10
|
+
// declaration regexes, and the index-file symbol provider.
|
|
11
|
+
|
|
12
|
+
const MAX_READ_LINES = 200;
|
|
13
|
+
const DEFAULT_MAX_CHARS = 240;
|
|
14
|
+
const DEFAULT_SYMBOL_MAX_CHARS = 160;
|
|
15
|
+
const DEFAULT_MAX_SYMBOLS = 8;
|
|
16
|
+
|
|
17
|
+
const DECL_RE = /^\s*(?:export\s+default\s+|export\s+)?(?:async\s+)?(?:function\*?|class|const|let|var)\s+([A-Za-z_$][\w$]*)/;
|
|
18
|
+
|
|
19
|
+
async function readHead(rootDir, relPath) {
|
|
20
|
+
try {
|
|
21
|
+
const abs = resolveProjectRelativePath(rootDir, relPath);
|
|
22
|
+
if (!abs) return null;
|
|
23
|
+
const content = await fs.readFile(abs, 'utf8');
|
|
24
|
+
return { content, lines: content.split('\n').slice(0, MAX_READ_LINES), bytes: Buffer.byteLength(content, 'utf8') };
|
|
25
|
+
} catch {
|
|
26
|
+
return null;
|
|
27
|
+
}
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
// Extract the first sentence of a docblock: with `endIndex` omitted, reads the
|
|
31
|
+
// leading file-level docblock at the top of `lines`; with `endIndex`, finds the
|
|
32
|
+
// docblock ending right before a declaration line (lines above decl passed in).
|
|
33
|
+
function docblockFirstSentence(lines, endIndex = lines.length) {
|
|
34
|
+
if (endIndex === lines.length) {
|
|
35
|
+
// File-level lane: the docblock must start at (or near) the top of file.
|
|
36
|
+
let start = 0;
|
|
37
|
+
while (start < lines.length && (lines[start].trim() === '' || lines[start].startsWith('#!'))) start += 1;
|
|
38
|
+
const first = (lines[start] ?? '').trim();
|
|
39
|
+
const isBlock = first.startsWith('/*');
|
|
40
|
+
const isLine = first.startsWith('//');
|
|
41
|
+
if (!isBlock && !isLine) return null;
|
|
42
|
+
const block = [];
|
|
43
|
+
if (isLine) {
|
|
44
|
+
for (let i = start; i < lines.length; i += 1) {
|
|
45
|
+
const t = lines[i].trim();
|
|
46
|
+
if (!t.startsWith('//')) break;
|
|
47
|
+
block.push(t.replace(/^\/\/+\s?/, ''));
|
|
48
|
+
}
|
|
49
|
+
} else {
|
|
50
|
+
for (let i = start; i < lines.length; i += 1) {
|
|
51
|
+
const t = lines[i].trim();
|
|
52
|
+
block.push(t.replace(/^\/\*\*?\s?/, '').replace(/\*\/\s?$/, '').replace(/^\*\s?/, ''));
|
|
53
|
+
if (t.endsWith('*/')) break;
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
const text = block.join(' ').replace(/\s+/g, ' ').trim();
|
|
57
|
+
if (!text) return null;
|
|
58
|
+
const m = text.match(/^.+?[.!?](?:\s|$)/);
|
|
59
|
+
return (m ? m[0] : text).trim();
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
// Find the last docblock ending at or before endIndex, scanning backwards
|
|
63
|
+
// for a '*/' terminator so decl-adjacent blocks are preferred.
|
|
64
|
+
let end = -1;
|
|
65
|
+
for (let i = Math.min(endIndex, lines.length) - 1; i >= 0; i -= 1) {
|
|
66
|
+
const t = lines[i].trim();
|
|
67
|
+
if (t === '' || t.startsWith('//') || t.startsWith('/*') || t.startsWith('*') || t.endsWith('*/')) {
|
|
68
|
+
if (t.endsWith('*/')) { end = i; break; }
|
|
69
|
+
if (t.startsWith('//')) { end = i; break; }
|
|
70
|
+
continue;
|
|
71
|
+
}
|
|
72
|
+
break;
|
|
73
|
+
}
|
|
74
|
+
if (end < 0) return null;
|
|
75
|
+
|
|
76
|
+
const block = [];
|
|
77
|
+
const tail = lines[end].trim();
|
|
78
|
+
if (tail.startsWith('//')) {
|
|
79
|
+
for (let i = end; i >= 0; i -= 1) {
|
|
80
|
+
const t = lines[i].trim();
|
|
81
|
+
if (!t.startsWith('//')) break;
|
|
82
|
+
block.unshift(t.replace(/^\/\/+\s?/, ''));
|
|
83
|
+
}
|
|
84
|
+
} else {
|
|
85
|
+
for (let i = end; i >= 0; i -= 1) {
|
|
86
|
+
const t = lines[i].trim();
|
|
87
|
+
const isBoundary = t.startsWith('/**') || t.startsWith('/*');
|
|
88
|
+
block.unshift(t.replace(/^\/\*\*?\s?/, '').replace(/\*\/\s?$/, '').replace(/^\*\s?/, ''));
|
|
89
|
+
if (isBoundary) break;
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
const text = block.join(' ').replace(/\s+/g, ' ').trim();
|
|
94
|
+
if (!text) return null;
|
|
95
|
+
const m = text.match(/^.+?[.!?](?:\s|$)/);
|
|
96
|
+
return (m ? m[0] : text).trim();
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
// Truncate on a word boundary — never mid-word.
|
|
100
|
+
function truncateWords(text, maxChars) {
|
|
101
|
+
if (text.length <= maxChars) return text;
|
|
102
|
+
const slice = text.slice(0, maxChars);
|
|
103
|
+
const lastSpace = slice.lastIndexOf(' ');
|
|
104
|
+
const cut = lastSpace > 0 ? slice.slice(0, lastSpace) : slice;
|
|
105
|
+
return cut.trimEnd();
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
// Regex fallback when the index symbols artifact is absent — bounded, dep-free.
|
|
109
|
+
function scanDeclarations(lines, maxSymbols) {
|
|
110
|
+
const names = [];
|
|
111
|
+
for (const line of lines) {
|
|
112
|
+
const m = line.match(DECL_RE);
|
|
113
|
+
if (m && !names.includes(m[1])) names.push(m[1]);
|
|
114
|
+
if (names.length >= maxSymbols) break;
|
|
115
|
+
}
|
|
116
|
+
return names;
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
async function symbolNames(rootDir, relPath, lines, maxSymbols) {
|
|
120
|
+
try {
|
|
121
|
+
const provider = getSyntaxProvider(relPath);
|
|
122
|
+
if (provider) {
|
|
123
|
+
const syms = await provider.symbols(relPath);
|
|
124
|
+
if (Array.isArray(syms) && syms.length > 0) {
|
|
125
|
+
return syms.slice(0, maxSymbols).map((s) => s.name).filter(Boolean);
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
} catch {
|
|
129
|
+
// provider failure falls through to the regex lane — never throws
|
|
130
|
+
}
|
|
131
|
+
return scanDeclarations(lines, maxSymbols);
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/**
|
|
135
|
+
* summarizeFile(rootDir, relPath, { maxSymbols=8, maxChars=240 })
|
|
136
|
+
* → { path, summary, symbolCount, bytes } | null
|
|
137
|
+
*
|
|
138
|
+
* Extractive summary: leading docblock first sentence + top symbol names.
|
|
139
|
+
* Falls back to a path/size description for docblock-free files. Returns null
|
|
140
|
+
* only when the file cannot be read. Deterministic across calls.
|
|
141
|
+
*/
|
|
142
|
+
export async function summarizeFile(rootDir, relPath, opts = {}) {
|
|
143
|
+
if (!relPath || typeof relPath !== 'string') return null;
|
|
144
|
+
const maxChars = typeof opts.maxChars === 'number' ? opts.maxChars : DEFAULT_MAX_CHARS;
|
|
145
|
+
const maxSymbols = typeof opts.maxSymbols === 'number' ? opts.maxSymbols : DEFAULT_MAX_SYMBOLS;
|
|
146
|
+
|
|
147
|
+
const head = await readHead(rootDir, relPath);
|
|
148
|
+
if (!head) return null;
|
|
149
|
+
|
|
150
|
+
const parts = [];
|
|
151
|
+
const doc = docblockFirstSentence(head.lines);
|
|
152
|
+
if (doc) parts.push(doc);
|
|
153
|
+
|
|
154
|
+
const names = await symbolNames(rootDir, relPath, head.lines, maxSymbols);
|
|
155
|
+
if (names.length > 0) parts.push(`exports: ${names.join(', ')}`);
|
|
156
|
+
|
|
157
|
+
const summary = parts.length > 0
|
|
158
|
+
? parts.join(' — ')
|
|
159
|
+
: `${relPath} (${head.bytes} bytes)`;
|
|
160
|
+
|
|
161
|
+
return {
|
|
162
|
+
path: relPath,
|
|
163
|
+
summary: truncateWords(summary, maxChars),
|
|
164
|
+
symbolCount: names.length,
|
|
165
|
+
bytes: head.bytes,
|
|
166
|
+
};
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
/**
|
|
170
|
+
* summarizeSymbol(rootDir, relPath, symbolName, { maxChars=160 })
|
|
171
|
+
* → { path, symbol, summary } | null
|
|
172
|
+
*
|
|
173
|
+
* Docblock-above-decl + the declaration signature line. Null when the file is
|
|
174
|
+
* unreadable or the symbol is absent. Deterministic.
|
|
175
|
+
*/
|
|
176
|
+
export async function summarizeSymbol(rootDir, relPath, symbolName, opts = {}) {
|
|
177
|
+
if (!relPath || !symbolName) return null;
|
|
178
|
+
const maxChars = typeof opts.maxChars === 'number' ? opts.maxChars : DEFAULT_SYMBOL_MAX_CHARS;
|
|
179
|
+
|
|
180
|
+
const head = await readHead(rootDir, relPath);
|
|
181
|
+
if (!head) return null;
|
|
182
|
+
|
|
183
|
+
const declIndex = head.lines.findIndex((line) => {
|
|
184
|
+
const m = line.match(DECL_RE);
|
|
185
|
+
return m && m[1] === symbolName;
|
|
186
|
+
});
|
|
187
|
+
if (declIndex < 0) return null;
|
|
188
|
+
|
|
189
|
+
const signature = head.lines[declIndex].trim();
|
|
190
|
+
const doc = docblockFirstSentence(head.lines, declIndex);
|
|
191
|
+
const summary = doc ? `${doc} ${signature}` : signature;
|
|
192
|
+
|
|
193
|
+
return { path: relPath, symbol: symbolName, summary: truncateWords(summary, maxChars) };
|
|
194
|
+
}
|