@ngockhoale/ukit 2.6.8 → 2.6.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +52 -0
- package/package.json +1 -1
- package/scripts/index/build-index.mjs +2 -1
- package/scripts/index/query-index.mjs +2 -0
- package/src/cli/commands/indexTools.js +5 -0
- package/src/cli/commands/install.js +2 -1
- package/src/cli/commands/metrics.js +127 -0
- package/src/cli/index.js +7 -0
- package/src/core/codeintel/analogy.js +197 -0
- package/src/core/codeintel/cochange.js +205 -0
- package/src/core/codeintel/compiler.js +82 -9
- package/src/core/codeintel/graph.js +291 -0
- package/src/core/codeintel/impact.js +23 -0
- package/src/core/codeintel/packet.js +1 -0
- package/src/core/codeintel/retriever.js +49 -3
- package/src/core/codeintel/summaries.js +194 -0
- package/src/core/codeintel/vectorProvider.js +213 -0
- package/src/core/runtimeConfig.js +50 -2
- package/src/diagnostics/failurePatterns.js +187 -0
- package/src/diagnostics/routeOutcomes.js +146 -0
- package/src/index/buildIndex.js +29 -0
- package/src/index/paths.js +2 -0
- package/templates/.claude/hooks/block-dangerous.sh +29 -5
- package/templates/.claude/hooks/context-hardcap-gate.sh +4 -1
- package/templates/.claude/hooks/protect-files.sh +26 -3
- package/templates/.claude/ukit/index/route-task.mjs +57 -6
- package/templates/.claude/ukit/runtime/hook-chain-runner.mjs +9 -1
- package/templates/.claude/ukit/runtime/hook-input.sh +26 -3
- package/templates/ukit/storage/config.json +7 -2
|
@@ -5,6 +5,7 @@ import { getArtifactPath, INDEX_ARTIFACTS, INDEX_SCHEMA_VERSION, normalizeRelati
|
|
|
5
5
|
import { loadRuntimeConfig } from '../runtimeConfig.js';
|
|
6
6
|
import { createEdge, getSemanticProvider, IndexFileSyntaxProvider } from './providers.js';
|
|
7
7
|
import { createSemanticProvider } from './semanticProvider.js';
|
|
8
|
+
import { createEmbeddingProvider } from './vectorProvider.js';
|
|
8
9
|
|
|
9
10
|
// Hybrid Retriever (SPEC §3): each lane produces a ranked list of file paths,
|
|
10
11
|
// then lanes merge via Reciprocal Rank Fusion (k=60, per-lane weights from
|
|
@@ -15,8 +16,8 @@ import { createSemanticProvider } from './semanticProvider.js';
|
|
|
15
16
|
const DEFAULT_LIMIT = 20;
|
|
16
17
|
const DEFAULT_BM25 = { k1: 1.2, b: 0.75 };
|
|
17
18
|
const DEFAULT_RRF_K = 60;
|
|
18
|
-
const DEFAULT_WEIGHTS = { exact: 1.0, symbol: 1.2, bm25: 0.8, semantic: 1.0 };
|
|
19
|
-
const LANE_ORDER = ['exact', 'symbol', 'bm25', 'semantic'];
|
|
19
|
+
const DEFAULT_WEIGHTS = { exact: 1.0, symbol: 1.2, bm25: 0.8, vector: 0.6, semantic: 1.0 };
|
|
20
|
+
const LANE_ORDER = ['exact', 'symbol', 'bm25', 'vector', 'semantic'];
|
|
20
21
|
|
|
21
22
|
async function readArtifact(rootDir, name) {
|
|
22
23
|
try {
|
|
@@ -56,6 +57,7 @@ function retrieverDefaults(config) {
|
|
|
56
57
|
exact: weight('exact'),
|
|
57
58
|
symbol: weight('symbol'),
|
|
58
59
|
bm25: weight('bm25'),
|
|
60
|
+
vector: weight('vector'),
|
|
59
61
|
semantic: weight('semantic'),
|
|
60
62
|
},
|
|
61
63
|
};
|
|
@@ -192,6 +194,39 @@ async function semanticLane(rootDir, normalizedQuery, config, fileSet, snapshotI
|
|
|
192
194
|
return { lane, why: lane.length === 0 ? 'no-match' : null };
|
|
193
195
|
}
|
|
194
196
|
|
|
197
|
+
// Vector lane (SPEC §2/§3): dep-free hashed embedding provider; mirrors the
|
|
198
|
+
// semantic lane. Edges' `to` carries "file:0" — reduce to a ranked file list.
|
|
199
|
+
async function vectorLane(rootDir, normalizedQuery, config, fileSet, snapshotId) {
|
|
200
|
+
const provider = createEmbeddingProvider({ projectRoot: rootDir, config });
|
|
201
|
+
if (!provider || provider.name === 'null' || typeof provider.resolve !== 'function') {
|
|
202
|
+
return { lane: null, why: 'unavailable' };
|
|
203
|
+
}
|
|
204
|
+
let edges;
|
|
205
|
+
try {
|
|
206
|
+
edges = await provider.resolve(normalizedQuery, { limit: DEFAULT_LIMIT, snapshot: snapshotId });
|
|
207
|
+
} catch {
|
|
208
|
+
return { lane: null, why: 'provider-error' };
|
|
209
|
+
}
|
|
210
|
+
if (edges === null || edges === undefined) {
|
|
211
|
+
return { lane: null, why: 'unavailable' };
|
|
212
|
+
}
|
|
213
|
+
if (!Array.isArray(edges) || edges.length === 0) {
|
|
214
|
+
return { lane: [], why: 'no-match' };
|
|
215
|
+
}
|
|
216
|
+
const lane = [];
|
|
217
|
+
const seen = new Set();
|
|
218
|
+
for (const edge of edges) {
|
|
219
|
+
const target = typeof edge?.to === 'string' ? edge.to.split(':')[0] : null;
|
|
220
|
+
if (!target) continue;
|
|
221
|
+
const rel = path.isAbsolute(target) ? normalizeRelative(rootDir, target) : target;
|
|
222
|
+
if (seen.has(rel)) continue;
|
|
223
|
+
if (fileSet.size > 0 && !fileSet.has(rel)) continue;
|
|
224
|
+
seen.add(rel);
|
|
225
|
+
lane.push(laneEntry(rel, `vector ${edge.kind ?? 'match'} for "${normalizedQuery}"`));
|
|
226
|
+
}
|
|
227
|
+
return { lane, why: lane.length === 0 ? 'no-match' : null };
|
|
228
|
+
}
|
|
229
|
+
|
|
195
230
|
function rrfMerge(lanes, weights, rrfK) {
|
|
196
231
|
const scores = new Map();
|
|
197
232
|
const meta = new Map();
|
|
@@ -199,6 +234,7 @@ function rrfMerge(lanes, weights, rrfK) {
|
|
|
199
234
|
const lane = lanes.get(laneName);
|
|
200
235
|
if (!lane) continue;
|
|
201
236
|
const w = weights[laneName] ?? 0;
|
|
237
|
+
if (w <= 0) continue;
|
|
202
238
|
lane.forEach((entry, index) => {
|
|
203
239
|
scores.set(entry.path, (scores.get(entry.path) ?? 0) + w / (rrfK + index + 1));
|
|
204
240
|
if (!meta.has(entry.path)) meta.set(entry.path, entry);
|
|
@@ -312,7 +348,17 @@ export async function retrieve(projectRoot, query, { mode = 'search', limit, sna
|
|
|
312
348
|
lanes.set('bm25', bm25.slice(0, Math.max(effectiveLimit, 100)));
|
|
313
349
|
}
|
|
314
350
|
|
|
315
|
-
// Lane 4 —
|
|
351
|
+
// Lane 4 — dep-free hashed vector provider (index-bound; absent → omitted).
|
|
352
|
+
const vector = await vectorLane(rootDir, normalizedQuery, config, fileSet, snapshotId);
|
|
353
|
+
if (vector.lane === null) {
|
|
354
|
+
omitted.push({ what: 'vector-lane', why: vector.why });
|
|
355
|
+
} else if (vector.lane.length > 0) {
|
|
356
|
+
lanes.set('vector', vector.lane);
|
|
357
|
+
} else {
|
|
358
|
+
omitted.push({ what: 'vector-lane', why: vector.why });
|
|
359
|
+
}
|
|
360
|
+
|
|
361
|
+
// Lane 5 — semantic provider (defensive; absent → skipped, never fatal).
|
|
316
362
|
const semantic = await semanticLane(rootDir, normalizedQuery, config, fileSet, snapshotId);
|
|
317
363
|
if (semantic.lane === null) {
|
|
318
364
|
omitted.push({ what: 'semantic-lane', why: semantic.why });
|
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
import fs from 'node:fs/promises';
|
|
2
|
+
import path from 'node:path';
|
|
3
|
+
|
|
4
|
+
import { getSyntaxProvider } from './providers.js';
|
|
5
|
+
import { resolveProjectRelativePath } from '../fileOps.js';
|
|
6
|
+
|
|
7
|
+
// Deterministic extractive summaries (SPEC §7/§8, CI-304) — the L3 detail tier.
|
|
8
|
+
// No model calls, no clock/random, no new deps. File reads are bounded to the
|
|
9
|
+
// first MAX_READ_LINES lines; extraction is pure text scanning over docblocks,
|
|
10
|
+
// declaration regexes, and the index-file symbol provider.
|
|
11
|
+
|
|
12
|
+
const MAX_READ_LINES = 200;
|
|
13
|
+
const DEFAULT_MAX_CHARS = 240;
|
|
14
|
+
const DEFAULT_SYMBOL_MAX_CHARS = 160;
|
|
15
|
+
const DEFAULT_MAX_SYMBOLS = 8;
|
|
16
|
+
|
|
17
|
+
const DECL_RE = /^\s*(?:export\s+default\s+|export\s+)?(?:async\s+)?(?:function\*?|class|const|let|var)\s+([A-Za-z_$][\w$]*)/;
|
|
18
|
+
|
|
19
|
+
async function readHead(rootDir, relPath) {
|
|
20
|
+
try {
|
|
21
|
+
const abs = resolveProjectRelativePath(rootDir, relPath);
|
|
22
|
+
if (!abs) return null;
|
|
23
|
+
const content = await fs.readFile(abs, 'utf8');
|
|
24
|
+
return { content, lines: content.split('\n').slice(0, MAX_READ_LINES), bytes: Buffer.byteLength(content, 'utf8') };
|
|
25
|
+
} catch {
|
|
26
|
+
return null;
|
|
27
|
+
}
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
// Extract the first sentence of a docblock: with `endIndex` omitted, reads the
|
|
31
|
+
// leading file-level docblock at the top of `lines`; with `endIndex`, finds the
|
|
32
|
+
// docblock ending right before a declaration line (lines above decl passed in).
|
|
33
|
+
function docblockFirstSentence(lines, endIndex = lines.length) {
|
|
34
|
+
if (endIndex === lines.length) {
|
|
35
|
+
// File-level lane: the docblock must start at (or near) the top of file.
|
|
36
|
+
let start = 0;
|
|
37
|
+
while (start < lines.length && (lines[start].trim() === '' || lines[start].startsWith('#!'))) start += 1;
|
|
38
|
+
const first = (lines[start] ?? '').trim();
|
|
39
|
+
const isBlock = first.startsWith('/*');
|
|
40
|
+
const isLine = first.startsWith('//');
|
|
41
|
+
if (!isBlock && !isLine) return null;
|
|
42
|
+
const block = [];
|
|
43
|
+
if (isLine) {
|
|
44
|
+
for (let i = start; i < lines.length; i += 1) {
|
|
45
|
+
const t = lines[i].trim();
|
|
46
|
+
if (!t.startsWith('//')) break;
|
|
47
|
+
block.push(t.replace(/^\/\/+\s?/, ''));
|
|
48
|
+
}
|
|
49
|
+
} else {
|
|
50
|
+
for (let i = start; i < lines.length; i += 1) {
|
|
51
|
+
const t = lines[i].trim();
|
|
52
|
+
block.push(t.replace(/^\/\*\*?\s?/, '').replace(/\*\/\s?$/, '').replace(/^\*\s?/, ''));
|
|
53
|
+
if (t.endsWith('*/')) break;
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
const text = block.join(' ').replace(/\s+/g, ' ').trim();
|
|
57
|
+
if (!text) return null;
|
|
58
|
+
const m = text.match(/^.+?[.!?](?:\s|$)/);
|
|
59
|
+
return (m ? m[0] : text).trim();
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
// Find the last docblock ending at or before endIndex, scanning backwards
|
|
63
|
+
// for a '*/' terminator so decl-adjacent blocks are preferred.
|
|
64
|
+
let end = -1;
|
|
65
|
+
for (let i = Math.min(endIndex, lines.length) - 1; i >= 0; i -= 1) {
|
|
66
|
+
const t = lines[i].trim();
|
|
67
|
+
if (t === '' || t.startsWith('//') || t.startsWith('/*') || t.startsWith('*') || t.endsWith('*/')) {
|
|
68
|
+
if (t.endsWith('*/')) { end = i; break; }
|
|
69
|
+
if (t.startsWith('//')) { end = i; break; }
|
|
70
|
+
continue;
|
|
71
|
+
}
|
|
72
|
+
break;
|
|
73
|
+
}
|
|
74
|
+
if (end < 0) return null;
|
|
75
|
+
|
|
76
|
+
const block = [];
|
|
77
|
+
const tail = lines[end].trim();
|
|
78
|
+
if (tail.startsWith('//')) {
|
|
79
|
+
for (let i = end; i >= 0; i -= 1) {
|
|
80
|
+
const t = lines[i].trim();
|
|
81
|
+
if (!t.startsWith('//')) break;
|
|
82
|
+
block.unshift(t.replace(/^\/\/+\s?/, ''));
|
|
83
|
+
}
|
|
84
|
+
} else {
|
|
85
|
+
for (let i = end; i >= 0; i -= 1) {
|
|
86
|
+
const t = lines[i].trim();
|
|
87
|
+
const isBoundary = t.startsWith('/**') || t.startsWith('/*');
|
|
88
|
+
block.unshift(t.replace(/^\/\*\*?\s?/, '').replace(/\*\/\s?$/, '').replace(/^\*\s?/, ''));
|
|
89
|
+
if (isBoundary) break;
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
const text = block.join(' ').replace(/\s+/g, ' ').trim();
|
|
94
|
+
if (!text) return null;
|
|
95
|
+
const m = text.match(/^.+?[.!?](?:\s|$)/);
|
|
96
|
+
return (m ? m[0] : text).trim();
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
// Truncate on a word boundary — never mid-word.
|
|
100
|
+
function truncateWords(text, maxChars) {
|
|
101
|
+
if (text.length <= maxChars) return text;
|
|
102
|
+
const slice = text.slice(0, maxChars);
|
|
103
|
+
const lastSpace = slice.lastIndexOf(' ');
|
|
104
|
+
const cut = lastSpace > 0 ? slice.slice(0, lastSpace) : slice;
|
|
105
|
+
return cut.trimEnd();
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
// Regex fallback when the index symbols artifact is absent — bounded, dep-free.
|
|
109
|
+
function scanDeclarations(lines, maxSymbols) {
|
|
110
|
+
const names = [];
|
|
111
|
+
for (const line of lines) {
|
|
112
|
+
const m = line.match(DECL_RE);
|
|
113
|
+
if (m && !names.includes(m[1])) names.push(m[1]);
|
|
114
|
+
if (names.length >= maxSymbols) break;
|
|
115
|
+
}
|
|
116
|
+
return names;
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
async function symbolNames(rootDir, relPath, lines, maxSymbols) {
|
|
120
|
+
try {
|
|
121
|
+
const provider = getSyntaxProvider(relPath);
|
|
122
|
+
if (provider) {
|
|
123
|
+
const syms = await provider.symbols(relPath);
|
|
124
|
+
if (Array.isArray(syms) && syms.length > 0) {
|
|
125
|
+
return syms.slice(0, maxSymbols).map((s) => s.name).filter(Boolean);
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
} catch {
|
|
129
|
+
// provider failure falls through to the regex lane — never throws
|
|
130
|
+
}
|
|
131
|
+
return scanDeclarations(lines, maxSymbols);
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/**
|
|
135
|
+
* summarizeFile(rootDir, relPath, { maxSymbols=8, maxChars=240 })
|
|
136
|
+
* → { path, summary, symbolCount, bytes } | null
|
|
137
|
+
*
|
|
138
|
+
* Extractive summary: leading docblock first sentence + top symbol names.
|
|
139
|
+
* Falls back to a path/size description for docblock-free files. Returns null
|
|
140
|
+
* only when the file cannot be read. Deterministic across calls.
|
|
141
|
+
*/
|
|
142
|
+
export async function summarizeFile(rootDir, relPath, opts = {}) {
|
|
143
|
+
if (!relPath || typeof relPath !== 'string') return null;
|
|
144
|
+
const maxChars = typeof opts.maxChars === 'number' ? opts.maxChars : DEFAULT_MAX_CHARS;
|
|
145
|
+
const maxSymbols = typeof opts.maxSymbols === 'number' ? opts.maxSymbols : DEFAULT_MAX_SYMBOLS;
|
|
146
|
+
|
|
147
|
+
const head = await readHead(rootDir, relPath);
|
|
148
|
+
if (!head) return null;
|
|
149
|
+
|
|
150
|
+
const parts = [];
|
|
151
|
+
const doc = docblockFirstSentence(head.lines);
|
|
152
|
+
if (doc) parts.push(doc);
|
|
153
|
+
|
|
154
|
+
const names = await symbolNames(rootDir, relPath, head.lines, maxSymbols);
|
|
155
|
+
if (names.length > 0) parts.push(`exports: ${names.join(', ')}`);
|
|
156
|
+
|
|
157
|
+
const summary = parts.length > 0
|
|
158
|
+
? parts.join(' — ')
|
|
159
|
+
: `${relPath} (${head.bytes} bytes)`;
|
|
160
|
+
|
|
161
|
+
return {
|
|
162
|
+
path: relPath,
|
|
163
|
+
summary: truncateWords(summary, maxChars),
|
|
164
|
+
symbolCount: names.length,
|
|
165
|
+
bytes: head.bytes,
|
|
166
|
+
};
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
/**
|
|
170
|
+
* summarizeSymbol(rootDir, relPath, symbolName, { maxChars=160 })
|
|
171
|
+
* → { path, symbol, summary } | null
|
|
172
|
+
*
|
|
173
|
+
* Docblock-above-decl + the declaration signature line. Null when the file is
|
|
174
|
+
* unreadable or the symbol is absent. Deterministic.
|
|
175
|
+
*/
|
|
176
|
+
export async function summarizeSymbol(rootDir, relPath, symbolName, opts = {}) {
|
|
177
|
+
if (!relPath || !symbolName) return null;
|
|
178
|
+
const maxChars = typeof opts.maxChars === 'number' ? opts.maxChars : DEFAULT_SYMBOL_MAX_CHARS;
|
|
179
|
+
|
|
180
|
+
const head = await readHead(rootDir, relPath);
|
|
181
|
+
if (!head) return null;
|
|
182
|
+
|
|
183
|
+
const declIndex = head.lines.findIndex((line) => {
|
|
184
|
+
const m = line.match(DECL_RE);
|
|
185
|
+
return m && m[1] === symbolName;
|
|
186
|
+
});
|
|
187
|
+
if (declIndex < 0) return null;
|
|
188
|
+
|
|
189
|
+
const signature = head.lines[declIndex].trim();
|
|
190
|
+
const doc = docblockFirstSentence(head.lines, declIndex);
|
|
191
|
+
const summary = doc ? `${doc} ${signature}` : signature;
|
|
192
|
+
|
|
193
|
+
return { path: relPath, symbol: symbolName, summary: truncateWords(summary, maxChars) };
|
|
194
|
+
}
|
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
import fs from 'node:fs/promises';
|
|
2
|
+
|
|
3
|
+
import { getArtifactPath, INDEX_ARTIFACTS, INDEX_SCHEMA_VERSION } from '../../index/paths.js';
|
|
4
|
+
import { createEdge } from './providers.js';
|
|
5
|
+
|
|
6
|
+
// Dep-free embedding lane (SPEC §2): feature hashing (FNV-1a 32-bit) over word
|
|
7
|
+
// tokens + word-boundary trigrams, L2-normalized, cosine-ranked against the
|
|
8
|
+
// same doc text the BM25 lane uses (file paths + symbol names). No file reads,
|
|
9
|
+
// no deps — real embeddings plug in later through the injectable `loader`
|
|
10
|
+
// seam (config.codeIntel.embedding.provider === 'auto'), degrading to the
|
|
11
|
+
// hashed provider on ANY failure. Never throws.
|
|
12
|
+
|
|
13
|
+
const DEFAULT_DIMENSIONS = 256;
|
|
14
|
+
const DEFAULT_LIMIT = 20;
|
|
15
|
+
|
|
16
|
+
function fnv1a(str) {
|
|
17
|
+
let h = 0x811c9dc5;
|
|
18
|
+
for (let i = 0; i < str.length; i += 1) {
|
|
19
|
+
h ^= str.charCodeAt(i);
|
|
20
|
+
h = Math.imul(h, 0x01000193) >>> 0;
|
|
21
|
+
}
|
|
22
|
+
return h >>> 0;
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
function wordTokens(text) {
|
|
26
|
+
return String(text ?? '')
|
|
27
|
+
.split(/[^A-Za-z0-9_$]+/)
|
|
28
|
+
.map((t) => t.toLowerCase())
|
|
29
|
+
.filter(Boolean);
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
function trigrams(token) {
|
|
33
|
+
const padded = `#${token}#`;
|
|
34
|
+
const grams = [];
|
|
35
|
+
for (let i = 0; i + 3 <= padded.length; i += 1) {
|
|
36
|
+
grams.push(padded.slice(i, i + 3));
|
|
37
|
+
}
|
|
38
|
+
return grams;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* hashVectorize(text, { dimensions }) → number[] of length `dimensions`,
|
|
43
|
+
* L2-normalized (zero vector for empty/null input). Deterministic.
|
|
44
|
+
*/
|
|
45
|
+
export function hashVectorize(text, { dimensions = DEFAULT_DIMENSIONS } = {}) {
|
|
46
|
+
const dims = Number.isInteger(dimensions) && dimensions > 0 ? dimensions : DEFAULT_DIMENSIONS;
|
|
47
|
+
const vec = new Array(dims).fill(0);
|
|
48
|
+
const tokens = wordTokens(text);
|
|
49
|
+
for (const token of tokens) {
|
|
50
|
+
vec[fnv1a(`w:${token}`) % dims] += 1;
|
|
51
|
+
for (const gram of trigrams(token)) {
|
|
52
|
+
vec[fnv1a(`t:${gram}`) % dims] += 0.1;
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
const norm = Math.sqrt(vec.reduce((sum, v) => sum + v * v, 0));
|
|
56
|
+
if (norm > 0) {
|
|
57
|
+
for (let i = 0; i < dims; i += 1) vec[i] /= norm;
|
|
58
|
+
}
|
|
59
|
+
return vec;
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/** cosineSimilarity(a, b) → number in [-1, 1]; 0 on dim mismatch/empty. */
|
|
63
|
+
export function cosineSimilarity(a, b) {
|
|
64
|
+
if (!Array.isArray(a) || !Array.isArray(b) || a.length === 0 || a.length !== b.length) {
|
|
65
|
+
return 0;
|
|
66
|
+
}
|
|
67
|
+
let dot = 0;
|
|
68
|
+
let na = 0;
|
|
69
|
+
let nb = 0;
|
|
70
|
+
for (let i = 0; i < a.length; i += 1) {
|
|
71
|
+
dot += a[i] * b[i];
|
|
72
|
+
na += a[i] * a[i];
|
|
73
|
+
nb += b[i] * b[i];
|
|
74
|
+
}
|
|
75
|
+
if (na === 0 || nb === 0) return 0;
|
|
76
|
+
return dot / (Math.sqrt(na) * Math.sqrt(nb));
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
async function readArtifact(rootDir, name) {
|
|
80
|
+
try {
|
|
81
|
+
const raw = await fs.readFile(getArtifactPath(rootDir, name), 'utf8');
|
|
82
|
+
const parsed = JSON.parse(raw);
|
|
83
|
+
if (parsed?.schemaVersion !== undefined && parsed.schemaVersion !== INDEX_SCHEMA_VERSION) {
|
|
84
|
+
return null;
|
|
85
|
+
}
|
|
86
|
+
return parsed;
|
|
87
|
+
} catch {
|
|
88
|
+
return null;
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
export class HashedEmbeddingProvider {
|
|
93
|
+
constructor({ projectRoot = process.cwd(), dimensions = DEFAULT_DIMENSIONS, embedFn = null } = {}) {
|
|
94
|
+
this.name = 'hashed-vector';
|
|
95
|
+
this.projectRoot = projectRoot;
|
|
96
|
+
this.dimensions = Number.isInteger(dimensions) && dimensions > 0 ? dimensions : DEFAULT_DIMENSIONS;
|
|
97
|
+
this._embedFn = typeof embedFn === 'function' ? embedFn : null;
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
async available() {
|
|
101
|
+
return true;
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
capabilities() {
|
|
105
|
+
return { embeddings: true, similarity: true };
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
async embed(text) {
|
|
109
|
+
if (this._embedFn) {
|
|
110
|
+
const vec = await this._embedFn(String(text ?? ''));
|
|
111
|
+
if (Array.isArray(vec)) return vec;
|
|
112
|
+
}
|
|
113
|
+
return hashVectorize(text, { dimensions: this.dimensions });
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
/**
|
|
117
|
+
* resolve(query, ctx { limit?, snapshot? }) → edge[] | null.
|
|
118
|
+
* Corpus: files.json paths + symbols.json names — index-bound, no file reads.
|
|
119
|
+
* Missing index → null (caller records 'unavailable'); empty index → [].
|
|
120
|
+
*/
|
|
121
|
+
async resolve(query, ctx = {}) {
|
|
122
|
+
try {
|
|
123
|
+
const [filesArtifact, symbolsArtifact] = await Promise.all([
|
|
124
|
+
readArtifact(this.projectRoot, INDEX_ARTIFACTS.files),
|
|
125
|
+
readArtifact(this.projectRoot, INDEX_ARTIFACTS.symbols),
|
|
126
|
+
]);
|
|
127
|
+
if (!filesArtifact && !symbolsArtifact) {
|
|
128
|
+
return null;
|
|
129
|
+
}
|
|
130
|
+
const limit = Number.isInteger(ctx?.limit) && ctx.limit > 0 ? ctx.limit : DEFAULT_LIMIT;
|
|
131
|
+
const queryVec = await this.embed(query);
|
|
132
|
+
const docs = new Map();
|
|
133
|
+
for (const item of filesArtifact?.items ?? []) {
|
|
134
|
+
if (!item?.filePath) continue;
|
|
135
|
+
docs.set(item.filePath, item.filePath);
|
|
136
|
+
}
|
|
137
|
+
for (const sym of symbolsArtifact?.items ?? []) {
|
|
138
|
+
if (!sym?.filePath || !sym?.name) continue;
|
|
139
|
+
docs.set(sym.filePath, `${docs.get(sym.filePath) ?? sym.filePath} ${sym.name}`);
|
|
140
|
+
}
|
|
141
|
+
const snapshot = ctx?.snapshot ?? null;
|
|
142
|
+
const scored = await Promise.all([...docs.entries()].map(async ([filePath, docText]) => ({
|
|
143
|
+
filePath,
|
|
144
|
+
score: cosineSimilarity(queryVec, await this.embed(docText)),
|
|
145
|
+
})));
|
|
146
|
+
return scored
|
|
147
|
+
.filter((entry) => entry.score > 0)
|
|
148
|
+
.sort((a, b) => b.score - a.score || a.filePath.localeCompare(b.filePath))
|
|
149
|
+
.slice(0, limit)
|
|
150
|
+
.map((entry) => createEdge({
|
|
151
|
+
from: String(query ?? ''),
|
|
152
|
+
to: `${entry.filePath}:0`,
|
|
153
|
+
kind: 'vector',
|
|
154
|
+
confidence: entry.score,
|
|
155
|
+
provider: this.name,
|
|
156
|
+
evidence: 'hash-vectorizer',
|
|
157
|
+
snapshot,
|
|
158
|
+
}));
|
|
159
|
+
} catch {
|
|
160
|
+
return null;
|
|
161
|
+
}
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
export class NullEmbeddingProvider {
|
|
166
|
+
constructor() {
|
|
167
|
+
this.name = 'null';
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
async available() {
|
|
171
|
+
return false;
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
capabilities() {
|
|
175
|
+
return { embeddings: false, similarity: false };
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
async embed() {
|
|
179
|
+
return [];
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
async resolve() {
|
|
183
|
+
return null;
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
/**
|
|
188
|
+
* createEmbeddingProvider({ projectRoot, config, loader }) → provider.
|
|
189
|
+
* config.codeIntel.embedding.provider:
|
|
190
|
+
* 'hashed' (default) → HashedEmbeddingProvider
|
|
191
|
+
* 'auto' → loader() module exposing embed(); ANY failure → Hashed
|
|
192
|
+
* 'null' → NullEmbeddingProvider
|
|
193
|
+
*/
|
|
194
|
+
export function createEmbeddingProvider({ projectRoot = process.cwd(), config, loader } = {}) {
|
|
195
|
+
const embedding = config?.codeIntel?.embedding;
|
|
196
|
+
const setting = embedding?.provider ?? 'hashed';
|
|
197
|
+
const dimensions = embedding?.dimensions;
|
|
198
|
+
if (setting === 'null') {
|
|
199
|
+
return new NullEmbeddingProvider();
|
|
200
|
+
}
|
|
201
|
+
if (setting === 'auto' && typeof loader === 'function') {
|
|
202
|
+
try {
|
|
203
|
+
const mod = loader();
|
|
204
|
+
if (mod && typeof mod.embed === 'function') {
|
|
205
|
+
return new HashedEmbeddingProvider({ projectRoot, dimensions, embedFn: mod.embed });
|
|
206
|
+
}
|
|
207
|
+
} catch {
|
|
208
|
+
// fall through — degrade to hashed
|
|
209
|
+
}
|
|
210
|
+
return new HashedEmbeddingProvider({ projectRoot, dimensions });
|
|
211
|
+
}
|
|
212
|
+
return new HashedEmbeddingProvider({ projectRoot, dimensions });
|
|
213
|
+
}
|
|
@@ -174,11 +174,17 @@ export function buildDefaultRuntimeConfig(overrides = {}) {
|
|
|
174
174
|
bm25: { k1: 1.2, b: 0.75 },
|
|
175
175
|
merge: 'rrf',
|
|
176
176
|
rrfK: 60,
|
|
177
|
-
weights: { exact: 1.0, symbol: 1.2, bm25: 0.8, semantic: 1.0 },
|
|
177
|
+
weights: { exact: 1.0, symbol: 1.2, bm25: 0.8, semantic: 1.0, vector: 0.6 },
|
|
178
178
|
},
|
|
179
179
|
impact: { defaultDepth: 2, maxDepth: 4, maxNodes: 200 },
|
|
180
180
|
diagnostics: { enabled: true, timeoutMs: 8000 },
|
|
181
181
|
providers: { semantic: 'null' },
|
|
182
|
+
// C30 additive blocks (SPEC §10 — TASK-222 single owner).
|
|
183
|
+
embedding: { provider: 'hashed', dimensions: 256 },
|
|
184
|
+
analogy: { enabled: true, limit: 5 },
|
|
185
|
+
cochange: { enabled: true, maxCommits: 500, minCount: 2, timeoutMs: 5000 },
|
|
186
|
+
summaries: { enabled: true, maxItems: 5 },
|
|
187
|
+
graph: { enabled: true, store: 'json' },
|
|
182
188
|
},
|
|
183
189
|
memoryV2: {
|
|
184
190
|
enabled: true,
|
|
@@ -433,7 +439,7 @@ export function validateRuntimeConfig(config) {
|
|
|
433
439
|
if (!isPlainObject(codeIntel.retriever.weights)) {
|
|
434
440
|
errors.push('codeIntel.retriever.weights must be an object.');
|
|
435
441
|
} else {
|
|
436
|
-
for (const lane of ['exact', 'symbol', 'bm25', 'semantic']) {
|
|
442
|
+
for (const lane of ['exact', 'symbol', 'bm25', 'semantic', 'vector']) {
|
|
437
443
|
if (typeof codeIntel.retriever.weights[lane] !== 'number' || Number.isNaN(codeIntel.retriever.weights[lane])) {
|
|
438
444
|
errors.push(`codeIntel.retriever.weights.${lane} must be a number.`);
|
|
439
445
|
}
|
|
@@ -458,6 +464,48 @@ export function validateRuntimeConfig(config) {
|
|
|
458
464
|
} else {
|
|
459
465
|
pushNonEmptyStringError(errors, codeIntel.providers.semantic, 'codeIntel.providers.semantic');
|
|
460
466
|
}
|
|
467
|
+
// C30 additive blocks (SPEC §10 — TASK-222).
|
|
468
|
+
if (!isPlainObject(codeIntel.embedding)) {
|
|
469
|
+
errors.push('codeIntel.embedding must be an object.');
|
|
470
|
+
} else {
|
|
471
|
+
const VALID_EMBEDDING_PROVIDERS = new Set(['hashed', 'auto', 'null']);
|
|
472
|
+
if (!VALID_EMBEDDING_PROVIDERS.has(codeIntel.embedding.provider)) {
|
|
473
|
+
errors.push(`codeIntel.embedding.provider must be one of: ${[...VALID_EMBEDDING_PROVIDERS].join(', ')}.`);
|
|
474
|
+
}
|
|
475
|
+
const dimensions = codeIntel.embedding.dimensions;
|
|
476
|
+
if (typeof dimensions !== 'number' || !Number.isInteger(dimensions) || dimensions <= 0 || dimensions > 4096) {
|
|
477
|
+
errors.push('codeIntel.embedding.dimensions must be a positive integer <= 4096.');
|
|
478
|
+
}
|
|
479
|
+
}
|
|
480
|
+
if (!isPlainObject(codeIntel.analogy)) {
|
|
481
|
+
errors.push('codeIntel.analogy must be an object.');
|
|
482
|
+
} else {
|
|
483
|
+
pushBooleanError(errors, codeIntel.analogy.enabled, 'codeIntel.analogy.enabled');
|
|
484
|
+
pushPositiveNumberError(errors, codeIntel.analogy.limit, 'codeIntel.analogy.limit');
|
|
485
|
+
}
|
|
486
|
+
if (!isPlainObject(codeIntel.cochange)) {
|
|
487
|
+
errors.push('codeIntel.cochange must be an object.');
|
|
488
|
+
} else {
|
|
489
|
+
pushBooleanError(errors, codeIntel.cochange.enabled, 'codeIntel.cochange.enabled');
|
|
490
|
+
pushPositiveNumberError(errors, codeIntel.cochange.maxCommits, 'codeIntel.cochange.maxCommits');
|
|
491
|
+
pushPositiveNumberError(errors, codeIntel.cochange.minCount, 'codeIntel.cochange.minCount');
|
|
492
|
+
pushPositiveNumberError(errors, codeIntel.cochange.timeoutMs, 'codeIntel.cochange.timeoutMs');
|
|
493
|
+
}
|
|
494
|
+
if (!isPlainObject(codeIntel.summaries)) {
|
|
495
|
+
errors.push('codeIntel.summaries must be an object.');
|
|
496
|
+
} else {
|
|
497
|
+
pushBooleanError(errors, codeIntel.summaries.enabled, 'codeIntel.summaries.enabled');
|
|
498
|
+
pushPositiveNumberError(errors, codeIntel.summaries.maxItems, 'codeIntel.summaries.maxItems');
|
|
499
|
+
}
|
|
500
|
+
if (!isPlainObject(codeIntel.graph)) {
|
|
501
|
+
errors.push('codeIntel.graph must be an object.');
|
|
502
|
+
} else {
|
|
503
|
+
pushBooleanError(errors, codeIntel.graph.enabled, 'codeIntel.graph.enabled');
|
|
504
|
+
const VALID_GRAPH_STORES = new Set(['json', 'scip', 'pdg', 'lsp', 'null']);
|
|
505
|
+
if (!VALID_GRAPH_STORES.has(codeIntel.graph.store)) {
|
|
506
|
+
errors.push(`codeIntel.graph.store must be one of: ${[...VALID_GRAPH_STORES].join(', ')}.`);
|
|
507
|
+
}
|
|
508
|
+
}
|
|
461
509
|
}
|
|
462
510
|
|
|
463
511
|
if (!isPlainObject(config.memoryV2)) {
|