@ngockhoale/ukit 2.6.8 → 2.6.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,291 @@
1
+ import fs from 'node:fs/promises';
2
+ import path from 'node:path';
3
+
4
+ import { getArtifactPath, INDEX_ARTIFACTS, INDEX_SCHEMA_VERSION } from '../../index/paths.js';
5
+ import { readJsonIfExists, writeJson } from '../fileOps.js';
6
+ import { createEdge } from './providers.js';
7
+ import { COCHANGE_SCHEMA_VERSION } from './cochange.js';
8
+
9
+ // Graph contract + JSON store (SPEC §9 — CI-305). Builds `.cache/index/codegraph.json`
10
+ // from existing index artifacts (imports.json + calls.json, + cochange.json when
11
+ // present) and exposes a GraphProvider seam so SCIP/PDG/LSP backends can slot in
12
+ // later without touching consumers. No SQLite, no new deps — providers other than
13
+ // 'json' are contract-only NullGraphProvider stubs this cycle.
14
+
15
+ export const GRAPH_SCHEMA_VERSION = 1;
16
+
17
+ const VALID_GRAPH_STORES = new Set(['json', 'scip', 'pdg', 'lsp', 'null']);
18
+
19
+ async function readArtifact(rootDir, name, expectedSchemaVersion) {
20
+ try {
21
+ const raw = await fs.readFile(getArtifactPath(rootDir, name), 'utf8');
22
+ const parsed = JSON.parse(raw);
23
+ if (!parsed || typeof parsed !== 'object') return null;
24
+ if (expectedSchemaVersion !== undefined && parsed.schemaVersion !== expectedSchemaVersion) {
25
+ return null;
26
+ }
27
+ return parsed;
28
+ } catch {
29
+ return null;
30
+ }
31
+ }
32
+
33
+ // Same specifier-resolution discipline as impact.js — local copy so this lane
34
+ // does not import same-wave sibling internals.
35
+ function resolveRelativeSpecifier(fromFile, specifier, fileSet) {
36
+ const candidates = [];
37
+ if (specifier.startsWith('.')) {
38
+ const base = path.posix.normalize(path.posix.join(path.posix.dirname(fromFile), specifier));
39
+ candidates.push(base, `${base}.js`, `${base}.ts`, `${base}.mjs`, `${base}.jsx`, `${base}.tsx`, `${base}/index.js`, `${base}/index.ts`);
40
+ } else {
41
+ candidates.push(specifier);
42
+ }
43
+ for (const candidate of candidates) {
44
+ if (fileSet.has(candidate)) return candidate;
45
+ }
46
+ return null;
47
+ }
48
+
49
+ /**
50
+ * buildGraphArtifact(projectRoot) → artifact | null
51
+ *
52
+ * Merges imports.json + calls.json (+ cochange.json when present) into a
53
+ * codegraph.json artifact: { schemaVersion, generatedAt, nodes, edges }.
54
+ * Node ids: files → their relative path, symbols → `symbol:<name>`.
55
+ * Edges use the canonical createEdge shape. Writes the artifact atomically via
56
+ * fileOps.writeJson; returns null when no source artifacts exist. Never throws
57
+ * on missing/partial sources.
58
+ */
59
+ export async function buildGraphArtifact(projectRoot) {
60
+ const rootDir = path.resolve(projectRoot ?? process.cwd());
61
+
62
+ const [importsArtifact, callsArtifact, cochangeArtifact, filesArtifact] = await Promise.all([
63
+ readArtifact(rootDir, INDEX_ARTIFACTS.imports, INDEX_SCHEMA_VERSION),
64
+ readArtifact(rootDir, INDEX_ARTIFACTS.calls, INDEX_SCHEMA_VERSION),
65
+ readArtifact(rootDir, INDEX_ARTIFACTS.cochange, COCHANGE_SCHEMA_VERSION),
66
+ readArtifact(rootDir, INDEX_ARTIFACTS.files, INDEX_SCHEMA_VERSION),
67
+ ]);
68
+
69
+ if (!importsArtifact && !callsArtifact && !cochangeArtifact) {
70
+ return null;
71
+ }
72
+
73
+ const fileSet = new Set();
74
+ for (const file of filesArtifact?.items ?? []) {
75
+ if (file?.filePath) fileSet.add(file.filePath);
76
+ }
77
+ for (const imp of importsArtifact?.items ?? []) {
78
+ if (imp?.from) fileSet.add(imp.from);
79
+ }
80
+ for (const call of callsArtifact?.items ?? []) {
81
+ if (call?.filePath) fileSet.add(call.filePath);
82
+ }
83
+
84
+ const nodes = new Map(); // id → node
85
+ const edges = new Map(); // dedupe key → edge
86
+ const addFileNode = (filePath) => {
87
+ if (filePath && !nodes.has(filePath)) {
88
+ nodes.set(filePath, { id: filePath, kind: 'file', path: filePath });
89
+ }
90
+ };
91
+ const addSymbolNode = (name, filePath) => {
92
+ if (!name) return null;
93
+ const id = `symbol:${name}`;
94
+ if (!nodes.has(id)) {
95
+ nodes.set(id, { id, kind: 'symbol', path: filePath ?? null });
96
+ }
97
+ return id;
98
+ };
99
+ const addEdge = (edge, dedupeExtra = '') => {
100
+ const key = `${edge.from}»${edge.to}»${edge.kind}»${dedupeExtra}`;
101
+ if (!edges.has(key)) edges.set(key, edge);
102
+ };
103
+
104
+ // import edges: from-file → resolved target file
105
+ for (const imp of importsArtifact?.items ?? []) {
106
+ if (!imp?.from || !imp?.to) continue;
107
+ addFileNode(imp.from);
108
+ const target = resolveRelativeSpecifier(imp.from, imp.to, fileSet) ?? imp.to;
109
+ if (target === imp.from) continue;
110
+ addFileNode(target);
111
+ addEdge(createEdge({
112
+ from: imp.from,
113
+ to: target,
114
+ kind: 'import',
115
+ provider: 'json-graph',
116
+ evidence: INDEX_ARTIFACTS.imports,
117
+ }));
118
+ }
119
+
120
+ // call edges: caller file → callee file (resolved via symbol→file map);
121
+ // symbol nodes recorded regardless so queries can address them.
122
+ const symbolToFile = new Map();
123
+ for (const call of callsArtifact?.items ?? []) {
124
+ if (call?.filePath && call?.symbol && !symbolToFile.has(call.symbol)) {
125
+ symbolToFile.set(call.symbol, call.filePath);
126
+ }
127
+ }
128
+ for (const call of callsArtifact?.items ?? []) {
129
+ if (!call?.filePath) continue;
130
+ addFileNode(call.filePath);
131
+ if (call?.symbol) addSymbolNode(call.symbol, call.filePath);
132
+ for (const name of call.calls ?? []) {
133
+ const target = symbolToFile.get(name);
134
+ if (!target || target === call.filePath) continue;
135
+ addEdge(createEdge({
136
+ from: call.filePath,
137
+ to: target,
138
+ kind: 'call',
139
+ provider: 'json-graph',
140
+ evidence: INDEX_ARTIFACTS.calls,
141
+ }), name);
142
+ }
143
+ }
144
+
145
+ // co-change edges: pair endpoints become file nodes; count feeds confidence.
146
+ for (const pair of cochangeArtifact?.pairs ?? []) {
147
+ if (!pair?.a || !pair?.b) continue;
148
+ addFileNode(pair.a);
149
+ addFileNode(pair.b);
150
+ const confidence = Math.min(1, (pair.count ?? 1) / 10);
151
+ addEdge(createEdge({
152
+ from: pair.a,
153
+ to: pair.b,
154
+ kind: 'cochange',
155
+ confidence,
156
+ provider: 'json-graph',
157
+ evidence: INDEX_ARTIFACTS.cochange,
158
+ }), String(pair.count ?? ''));
159
+ addEdge(createEdge({
160
+ from: pair.b,
161
+ to: pair.a,
162
+ kind: 'cochange',
163
+ confidence,
164
+ provider: 'json-graph',
165
+ evidence: INDEX_ARTIFACTS.cochange,
166
+ }), String(pair.count ?? ''));
167
+ }
168
+
169
+ const artifact = {
170
+ schemaVersion: GRAPH_SCHEMA_VERSION,
171
+ generatedAt: new Date().toISOString(),
172
+ nodes: [...nodes.values()],
173
+ edges: [...edges.values()],
174
+ };
175
+
176
+ try {
177
+ // Content-gated write: unchanged graph (ignoring generatedAt) must not
178
+ // bump the artifact mtime — refresh's "no rewrite when nothing changed"
179
+ // guarantee covers derived artifacts too.
180
+ const graphPath = getArtifactPath(rootDir, INDEX_ARTIFACTS.codegraph);
181
+ const existing = await readJsonIfExists(graphPath);
182
+ const strip = ({ generatedAt, ...rest }) => rest;
183
+ if (!existing || JSON.stringify(strip(existing)) !== JSON.stringify(strip(artifact))) {
184
+ await writeJson(graphPath, artifact);
185
+ }
186
+ } catch {
187
+ // Artifact-write failure (read-only .cache, EACCES) is non-fatal — the
188
+ // in-memory artifact is still returned for the caller/summary.
189
+ }
190
+ return artifact;
191
+ }
192
+
193
+ /**
194
+ * JsonGraphStore — GraphProvider over `.cache/index/codegraph.json`.
195
+ * Schema-guarded (GRAPH_SCHEMA_VERSION); a missing/mismatched/corrupt artifact
196
+ * yields empty results per op, never throws.
197
+ */
198
+ export class JsonGraphStore {
199
+ constructor(projectRoot) {
200
+ this.name = 'json';
201
+ this.rootDir = path.resolve(projectRoot ?? process.cwd());
202
+ this._artifactPromise = null;
203
+ }
204
+
205
+ capabilities() {
206
+ return { nodes: true, edges: true, queries: true };
207
+ }
208
+
209
+ async _load() {
210
+ if (!this._artifactPromise) {
211
+ this._artifactPromise = (async () => {
212
+ const artifact = await readJsonIfExists(getArtifactPath(this.rootDir, INDEX_ARTIFACTS.codegraph));
213
+ if (!artifact || typeof artifact !== 'object') return null;
214
+ if (artifact.schemaVersion !== GRAPH_SCHEMA_VERSION) return null;
215
+ if (!Array.isArray(artifact.nodes) || !Array.isArray(artifact.edges)) return null;
216
+ return artifact;
217
+ })();
218
+ }
219
+ return this._artifactPromise;
220
+ }
221
+
222
+ async nodes() {
223
+ const artifact = await this._load();
224
+ return artifact?.nodes ?? [];
225
+ }
226
+
227
+ async edgesFrom(id) {
228
+ const artifact = await this._load();
229
+ if (!artifact || typeof id !== 'string') return [];
230
+ return artifact.edges.filter((edge) => edge?.from === id);
231
+ }
232
+
233
+ /**
234
+ * query(q) → edge[] | null. Resolves a node id (file path or `symbol:` id) to
235
+ * its outgoing edges; null when the node is absent or no artifact exists.
236
+ */
237
+ async query(q) {
238
+ const artifact = await this._load();
239
+ if (!artifact) return null;
240
+ const nodeId = typeof q === 'object' && q !== null ? q.path ?? q.id ?? null : q;
241
+ if (typeof nodeId !== 'string' || nodeId === '') return null;
242
+ const exists = artifact.nodes.some((node) => node?.id === nodeId);
243
+ if (!exists) return null;
244
+ return artifact.edges.filter((edge) => edge?.from === nodeId);
245
+ }
246
+ }
247
+
248
+ /**
249
+ * NullGraphProvider — contract-only stub for SCIP/PDG/LSP/'null' backends.
250
+ * All capabilities false; every op returns empty/null. Never throws.
251
+ */
252
+ export class NullGraphProvider {
253
+ constructor(backend = 'null') {
254
+ this.name = backend;
255
+ }
256
+
257
+ capabilities() {
258
+ return { nodes: false, edges: false, queries: false };
259
+ }
260
+
261
+ async nodes() {
262
+ return [];
263
+ }
264
+
265
+ async edgesFrom() {
266
+ return [];
267
+ }
268
+
269
+ async query() {
270
+ return null;
271
+ }
272
+ }
273
+
274
+ /**
275
+ * createGraphProvider({ projectRoot?, config? }) → GraphProvider
276
+ * codeIntel.graph.store: 'json' (default/unset) → JsonGraphStore;
277
+ * 'scip' | 'pdg' | 'lsp' | 'null' → NullGraphProvider (contract-only seam).
278
+ */
279
+ export function createGraphProvider({ projectRoot, config } = {}) {
280
+ if (config?.codeIntel?.graph?.enabled === false) {
281
+ return new NullGraphProvider('disabled');
282
+ }
283
+ const store = config?.codeIntel?.graph?.store;
284
+ if (store === undefined || store === 'json') {
285
+ return new JsonGraphStore(projectRoot);
286
+ }
287
+ if (VALID_GRAPH_STORES.has(store)) {
288
+ return new NullGraphProvider(store);
289
+ }
290
+ return new NullGraphProvider('null');
291
+ }
@@ -3,6 +3,7 @@ import path from 'node:path';
3
3
 
4
4
  import { getArtifactPath, INDEX_ARTIFACTS, INDEX_SCHEMA_VERSION } from '../../index/paths.js';
5
5
  import { loadRuntimeConfig } from '../runtimeConfig.js';
6
+ import { readCoChanges, coChangeEdges } from './cochange.js';
6
7
 
7
8
  // Multi-hop impact/trace engine (SPEC §4) — cycle-safe BFS over imports.json +
8
9
  // calls.json edges. `dependents` walks reverse edges (who imports the seed),
@@ -247,5 +248,27 @@ export async function impactSet(projectRoot, seeds, {
247
248
  };
248
249
  if (frontierCheck('rev') || (wantDeps && frontierCheck('fwd'))) truncated = true;
249
250
 
251
+ // Co-change lane (SPEC §6): when enabled (in-code default true) and the
252
+ // direction touches dependents, union kind:'cochange' edges for seed peers
253
+ // (count ≥ minCount) as context — peers join `nodes` at hop:0 with via:[],
254
+ // not as graph hops. Missing artifact → omitted reason; never fatal.
255
+ const cochangeConfig = config?.codeIntel?.cochange;
256
+ const cochangeEnabled = cochangeConfig?.enabled !== false;
257
+ if (cochangeEnabled && wantRev) {
258
+ const cochangeArtifact = await readCoChanges(rootDir);
259
+ if (!cochangeArtifact) {
260
+ omitted.push({ what: 'cochange-lane', why: 'index-missing' });
261
+ } else {
262
+ const minCount = typeof cochangeConfig?.minCount === 'number' ? cochangeConfig.minCount : 2;
263
+ const coEdges = await coChangeEdges(rootDir, seedFiles, { minCount });
264
+ for (const edge of coEdges) {
265
+ emitEdge(edge);
266
+ if (!nodes.has(edge.to)) {
267
+ nodes.set(edge.to, { file: edge.to, hop: 0, via: [] });
268
+ }
269
+ }
270
+ }
271
+ }
272
+
250
273
  return { nodes: [...nodes.values()], edges, truncated, omitted };
251
274
  }
@@ -108,6 +108,7 @@ export function packetToText(packet) {
108
108
  lines.push('## Evidence');
109
109
  for (const e of packet.evidence ?? []) {
110
110
  lines.push(`- [${e.level}/${e.source}] ${e.path ?? ''} — ${e.why ?? ''}`);
111
+ if (e.summary) lines.push(` summary: ${e.summary}`);
111
112
  if (e.excerpt) lines.push(` ${e.excerpt}`);
112
113
  }
113
114
  if ((packet.evidence ?? []).length === 0) lines.push('(none)');
@@ -5,6 +5,7 @@ import { getArtifactPath, INDEX_ARTIFACTS, INDEX_SCHEMA_VERSION, normalizeRelati
5
5
  import { loadRuntimeConfig } from '../runtimeConfig.js';
6
6
  import { createEdge, getSemanticProvider, IndexFileSyntaxProvider } from './providers.js';
7
7
  import { createSemanticProvider } from './semanticProvider.js';
8
+ import { createEmbeddingProvider } from './vectorProvider.js';
8
9
 
9
10
  // Hybrid Retriever (SPEC §3): each lane produces a ranked list of file paths,
10
11
  // then lanes merge via Reciprocal Rank Fusion (k=60, per-lane weights from
@@ -15,8 +16,8 @@ import { createSemanticProvider } from './semanticProvider.js';
15
16
  const DEFAULT_LIMIT = 20;
16
17
  const DEFAULT_BM25 = { k1: 1.2, b: 0.75 };
17
18
  const DEFAULT_RRF_K = 60;
18
- const DEFAULT_WEIGHTS = { exact: 1.0, symbol: 1.2, bm25: 0.8, semantic: 1.0 };
19
- const LANE_ORDER = ['exact', 'symbol', 'bm25', 'semantic'];
19
+ const DEFAULT_WEIGHTS = { exact: 1.0, symbol: 1.2, bm25: 0.8, vector: 0.6, semantic: 1.0 };
20
+ const LANE_ORDER = ['exact', 'symbol', 'bm25', 'vector', 'semantic'];
20
21
 
21
22
  async function readArtifact(rootDir, name) {
22
23
  try {
@@ -56,6 +57,7 @@ function retrieverDefaults(config) {
56
57
  exact: weight('exact'),
57
58
  symbol: weight('symbol'),
58
59
  bm25: weight('bm25'),
60
+ vector: weight('vector'),
59
61
  semantic: weight('semantic'),
60
62
  },
61
63
  };
@@ -192,6 +194,39 @@ async function semanticLane(rootDir, normalizedQuery, config, fileSet, snapshotI
192
194
  return { lane, why: lane.length === 0 ? 'no-match' : null };
193
195
  }
194
196
 
197
+ // Vector lane (SPEC §2/§3): dep-free hashed embedding provider; mirrors the
198
+ // semantic lane. Edges' `to` carries "file:0" — reduce to a ranked file list.
199
+ async function vectorLane(rootDir, normalizedQuery, config, fileSet, snapshotId) {
200
+ const provider = createEmbeddingProvider({ projectRoot: rootDir, config });
201
+ if (!provider || provider.name === 'null' || typeof provider.resolve !== 'function') {
202
+ return { lane: null, why: 'unavailable' };
203
+ }
204
+ let edges;
205
+ try {
206
+ edges = await provider.resolve(normalizedQuery, { limit: DEFAULT_LIMIT, snapshot: snapshotId });
207
+ } catch {
208
+ return { lane: null, why: 'provider-error' };
209
+ }
210
+ if (edges === null || edges === undefined) {
211
+ return { lane: null, why: 'unavailable' };
212
+ }
213
+ if (!Array.isArray(edges) || edges.length === 0) {
214
+ return { lane: [], why: 'no-match' };
215
+ }
216
+ const lane = [];
217
+ const seen = new Set();
218
+ for (const edge of edges) {
219
+ const target = typeof edge?.to === 'string' ? edge.to.split(':')[0] : null;
220
+ if (!target) continue;
221
+ const rel = path.isAbsolute(target) ? normalizeRelative(rootDir, target) : target;
222
+ if (seen.has(rel)) continue;
223
+ if (fileSet.size > 0 && !fileSet.has(rel)) continue;
224
+ seen.add(rel);
225
+ lane.push(laneEntry(rel, `vector ${edge.kind ?? 'match'} for "${normalizedQuery}"`));
226
+ }
227
+ return { lane, why: lane.length === 0 ? 'no-match' : null };
228
+ }
229
+
195
230
  function rrfMerge(lanes, weights, rrfK) {
196
231
  const scores = new Map();
197
232
  const meta = new Map();
@@ -199,6 +234,7 @@ function rrfMerge(lanes, weights, rrfK) {
199
234
  const lane = lanes.get(laneName);
200
235
  if (!lane) continue;
201
236
  const w = weights[laneName] ?? 0;
237
+ if (w <= 0) continue;
202
238
  lane.forEach((entry, index) => {
203
239
  scores.set(entry.path, (scores.get(entry.path) ?? 0) + w / (rrfK + index + 1));
204
240
  if (!meta.has(entry.path)) meta.set(entry.path, entry);
@@ -312,7 +348,17 @@ export async function retrieve(projectRoot, query, { mode = 'search', limit, sna
312
348
  lanes.set('bm25', bm25.slice(0, Math.max(effectiveLimit, 100)));
313
349
  }
314
350
 
315
- // Lane 4 — semantic provider (defensive; absent → skipped, never fatal).
351
+ // Lane 4 — dep-free hashed vector provider (index-bound; absent → omitted).
352
+ const vector = await vectorLane(rootDir, normalizedQuery, config, fileSet, snapshotId);
353
+ if (vector.lane === null) {
354
+ omitted.push({ what: 'vector-lane', why: vector.why });
355
+ } else if (vector.lane.length > 0) {
356
+ lanes.set('vector', vector.lane);
357
+ } else {
358
+ omitted.push({ what: 'vector-lane', why: vector.why });
359
+ }
360
+
361
+ // Lane 5 — semantic provider (defensive; absent → skipped, never fatal).
316
362
  const semantic = await semanticLane(rootDir, normalizedQuery, config, fileSet, snapshotId);
317
363
  if (semantic.lane === null) {
318
364
  omitted.push({ what: 'semantic-lane', why: semantic.why });
@@ -0,0 +1,194 @@
1
+ import fs from 'node:fs/promises';
2
+ import path from 'node:path';
3
+
4
+ import { getSyntaxProvider } from './providers.js';
5
+ import { resolveProjectRelativePath } from '../fileOps.js';
6
+
7
+ // Deterministic extractive summaries (SPEC §7/§8, CI-304) — the L3 detail tier.
8
+ // No model calls, no clock/random, no new deps. File reads are bounded to the
9
+ // first MAX_READ_LINES lines; extraction is pure text scanning over docblocks,
10
+ // declaration regexes, and the index-file symbol provider.
11
+
12
+ const MAX_READ_LINES = 200;
13
+ const DEFAULT_MAX_CHARS = 240;
14
+ const DEFAULT_SYMBOL_MAX_CHARS = 160;
15
+ const DEFAULT_MAX_SYMBOLS = 8;
16
+
17
+ const DECL_RE = /^\s*(?:export\s+default\s+|export\s+)?(?:async\s+)?(?:function\*?|class|const|let|var)\s+([A-Za-z_$][\w$]*)/;
18
+
19
+ async function readHead(rootDir, relPath) {
20
+ try {
21
+ const abs = resolveProjectRelativePath(rootDir, relPath);
22
+ if (!abs) return null;
23
+ const content = await fs.readFile(abs, 'utf8');
24
+ return { content, lines: content.split('\n').slice(0, MAX_READ_LINES), bytes: Buffer.byteLength(content, 'utf8') };
25
+ } catch {
26
+ return null;
27
+ }
28
+ }
29
+
30
+ // Extract the first sentence of a docblock: with `endIndex` omitted, reads the
31
+ // leading file-level docblock at the top of `lines`; with `endIndex`, finds the
32
+ // docblock ending right before a declaration line (lines above decl passed in).
33
+ function docblockFirstSentence(lines, endIndex = lines.length) {
34
+ if (endIndex === lines.length) {
35
+ // File-level lane: the docblock must start at (or near) the top of file.
36
+ let start = 0;
37
+ while (start < lines.length && (lines[start].trim() === '' || lines[start].startsWith('#!'))) start += 1;
38
+ const first = (lines[start] ?? '').trim();
39
+ const isBlock = first.startsWith('/*');
40
+ const isLine = first.startsWith('//');
41
+ if (!isBlock && !isLine) return null;
42
+ const block = [];
43
+ if (isLine) {
44
+ for (let i = start; i < lines.length; i += 1) {
45
+ const t = lines[i].trim();
46
+ if (!t.startsWith('//')) break;
47
+ block.push(t.replace(/^\/\/+\s?/, ''));
48
+ }
49
+ } else {
50
+ for (let i = start; i < lines.length; i += 1) {
51
+ const t = lines[i].trim();
52
+ block.push(t.replace(/^\/\*\*?\s?/, '').replace(/\*\/\s?$/, '').replace(/^\*\s?/, ''));
53
+ if (t.endsWith('*/')) break;
54
+ }
55
+ }
56
+ const text = block.join(' ').replace(/\s+/g, ' ').trim();
57
+ if (!text) return null;
58
+ const m = text.match(/^.+?[.!?](?:\s|$)/);
59
+ return (m ? m[0] : text).trim();
60
+ }
61
+
62
+ // Find the last docblock ending at or before endIndex, scanning backwards
63
+ // for a '*/' terminator so decl-adjacent blocks are preferred.
64
+ let end = -1;
65
+ for (let i = Math.min(endIndex, lines.length) - 1; i >= 0; i -= 1) {
66
+ const t = lines[i].trim();
67
+ if (t === '' || t.startsWith('//') || t.startsWith('/*') || t.startsWith('*') || t.endsWith('*/')) {
68
+ if (t.endsWith('*/')) { end = i; break; }
69
+ if (t.startsWith('//')) { end = i; break; }
70
+ continue;
71
+ }
72
+ break;
73
+ }
74
+ if (end < 0) return null;
75
+
76
+ const block = [];
77
+ const tail = lines[end].trim();
78
+ if (tail.startsWith('//')) {
79
+ for (let i = end; i >= 0; i -= 1) {
80
+ const t = lines[i].trim();
81
+ if (!t.startsWith('//')) break;
82
+ block.unshift(t.replace(/^\/\/+\s?/, ''));
83
+ }
84
+ } else {
85
+ for (let i = end; i >= 0; i -= 1) {
86
+ const t = lines[i].trim();
87
+ const isBoundary = t.startsWith('/**') || t.startsWith('/*');
88
+ block.unshift(t.replace(/^\/\*\*?\s?/, '').replace(/\*\/\s?$/, '').replace(/^\*\s?/, ''));
89
+ if (isBoundary) break;
90
+ }
91
+ }
92
+
93
+ const text = block.join(' ').replace(/\s+/g, ' ').trim();
94
+ if (!text) return null;
95
+ const m = text.match(/^.+?[.!?](?:\s|$)/);
96
+ return (m ? m[0] : text).trim();
97
+ }
98
+
99
+ // Truncate on a word boundary — never mid-word.
100
+ function truncateWords(text, maxChars) {
101
+ if (text.length <= maxChars) return text;
102
+ const slice = text.slice(0, maxChars);
103
+ const lastSpace = slice.lastIndexOf(' ');
104
+ const cut = lastSpace > 0 ? slice.slice(0, lastSpace) : slice;
105
+ return cut.trimEnd();
106
+ }
107
+
108
+ // Regex fallback when the index symbols artifact is absent — bounded, dep-free.
109
+ function scanDeclarations(lines, maxSymbols) {
110
+ const names = [];
111
+ for (const line of lines) {
112
+ const m = line.match(DECL_RE);
113
+ if (m && !names.includes(m[1])) names.push(m[1]);
114
+ if (names.length >= maxSymbols) break;
115
+ }
116
+ return names;
117
+ }
118
+
119
+ async function symbolNames(rootDir, relPath, lines, maxSymbols) {
120
+ try {
121
+ const provider = getSyntaxProvider(relPath);
122
+ if (provider) {
123
+ const syms = await provider.symbols(relPath);
124
+ if (Array.isArray(syms) && syms.length > 0) {
125
+ return syms.slice(0, maxSymbols).map((s) => s.name).filter(Boolean);
126
+ }
127
+ }
128
+ } catch {
129
+ // provider failure falls through to the regex lane — never throws
130
+ }
131
+ return scanDeclarations(lines, maxSymbols);
132
+ }
133
+
134
+ /**
135
+ * summarizeFile(rootDir, relPath, { maxSymbols=8, maxChars=240 })
136
+ * → { path, summary, symbolCount, bytes } | null
137
+ *
138
+ * Extractive summary: leading docblock first sentence + top symbol names.
139
+ * Falls back to a path/size description for docblock-free files. Returns null
140
+ * only when the file cannot be read. Deterministic across calls.
141
+ */
142
+ export async function summarizeFile(rootDir, relPath, opts = {}) {
143
+ if (!relPath || typeof relPath !== 'string') return null;
144
+ const maxChars = typeof opts.maxChars === 'number' ? opts.maxChars : DEFAULT_MAX_CHARS;
145
+ const maxSymbols = typeof opts.maxSymbols === 'number' ? opts.maxSymbols : DEFAULT_MAX_SYMBOLS;
146
+
147
+ const head = await readHead(rootDir, relPath);
148
+ if (!head) return null;
149
+
150
+ const parts = [];
151
+ const doc = docblockFirstSentence(head.lines);
152
+ if (doc) parts.push(doc);
153
+
154
+ const names = await symbolNames(rootDir, relPath, head.lines, maxSymbols);
155
+ if (names.length > 0) parts.push(`exports: ${names.join(', ')}`);
156
+
157
+ const summary = parts.length > 0
158
+ ? parts.join(' — ')
159
+ : `${relPath} (${head.bytes} bytes)`;
160
+
161
+ return {
162
+ path: relPath,
163
+ summary: truncateWords(summary, maxChars),
164
+ symbolCount: names.length,
165
+ bytes: head.bytes,
166
+ };
167
+ }
168
+
169
+ /**
170
+ * summarizeSymbol(rootDir, relPath, symbolName, { maxChars=160 })
171
+ * → { path, symbol, summary } | null
172
+ *
173
+ * Docblock-above-decl + the declaration signature line. Null when the file is
174
+ * unreadable or the symbol is absent. Deterministic.
175
+ */
176
+ export async function summarizeSymbol(rootDir, relPath, symbolName, opts = {}) {
177
+ if (!relPath || !symbolName) return null;
178
+ const maxChars = typeof opts.maxChars === 'number' ? opts.maxChars : DEFAULT_SYMBOL_MAX_CHARS;
179
+
180
+ const head = await readHead(rootDir, relPath);
181
+ if (!head) return null;
182
+
183
+ const declIndex = head.lines.findIndex((line) => {
184
+ const m = line.match(DECL_RE);
185
+ return m && m[1] === symbolName;
186
+ });
187
+ if (declIndex < 0) return null;
188
+
189
+ const signature = head.lines[declIndex].trim();
190
+ const doc = docblockFirstSentence(head.lines, declIndex);
191
+ const summary = doc ? `${doc} ${signature}` : signature;
192
+
193
+ return { path: relPath, symbol: symbolName, summary: truncateWords(summary, maxChars) };
194
+ }