@compr/opscontext-mcp 2.5.8 → 2.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -5,12 +5,14 @@ import { z } from "zod";
5
5
  import { loadSources, loadProjectDirs, loadConfig, resolveProjectDir } from "./config.js";
6
6
  import { ingestSources } from "./ingest.js";
7
7
  import { searchChunks } from "./search.js";
8
- import { initEmbeddings, embedChunks, vectorSearch, isEmbeddingsReady, } from "./embeddings.js";
8
+ import { initEmbeddings, embedChunks, embedKeyOf, vectorSearch, isEmbeddingsReady, } from "./embeddings.js";
9
9
  import { collectProjectOps, collectSystemOps } from "./collectors.js";
10
- import { loadCache, saveCache } from "./cache.js";
10
+ import { loadEmbeddingStore, compactEmbeddingStore } from "./embedding-store.js";
11
+ import { sharedIndexEnabled, corpusId, electIndexer, writeSharedIndex, readSharedIndex, sharedIndexMtime, } from "./shared-index.js";
11
12
  import { listProjects, checkPorts, runComplianceAudit, formatProjectList, formatPortMap, formatPlan, scoreProject, formatScoreReport, runScoreCanary, } from "./agents.js";
12
13
  import { saveSession, loadSession, listSessions, deleteSession, formatSession, formatSessionList, } from "./sessions.js";
13
- import { verifyChain, readAuditLog, filterByRange, autoRotateAuditLog } from "./audit.js";
14
+ import { verifyChain, readAuditLog, filterByRange, autoRotateAuditLog, safeAppend } from "./audit.js";
15
+ import { registerServer, listServers, formatServers } from "./server-registry.js";
14
16
  import { startEventIngestServer } from "./http-server.js";
15
17
  import { detect } from "./detector.js";
16
18
  import { buildCostReport } from "./cost-report.js";
@@ -43,6 +45,18 @@ let chunks = [];
43
45
  let embeddedChunks = [];
44
46
  let activeProjectNames = [];
45
47
  const firewall = new ProtocolFirewall();
48
+ // One indexer, many readers. [LOCK] [ONE-INDEXER-MANY-READERS]
49
+ // With the shared index off (default until the trial), every server is its own indexer, exactly
50
+ // as before, and only the content-addressed vector store below is new.
51
+ let role = "indexer";
52
+ let corpus;
53
+ let indexerPid = null;
54
+ let setRegistryRole = null;
55
+ /** key -> vector, loaded from ~/.contextengine/embeddings.bin and grown by what we embed. */
56
+ let vectorStore = new Map();
57
+ let indexSeq = 0;
58
+ let lastIndexMtime = null;
59
+ let modelInit = null;
46
60
  // Wire up learning search for auto-injection (avoids circular import)
47
61
  firewall.setLearningSearchFn((query, projects) => {
48
62
  return searchLearnings(query)
@@ -58,9 +72,11 @@ firewall.setLearningSearchFn((query, projects) => {
58
72
  .map((l) => ({ rule: l.rule, project: l.project, category: l.category }));
59
73
  });
60
74
  /**
61
- * (Re-)ingest all sources. Called at startup and on file changes.
75
+ * Parse every source, collect ops and code, import learnings (indexer only), inject learnings,
76
+ * community rules and adapters. Sets `sources`, `chunks`, `activeProjectNames`. No embedding
77
+ * here. One body for startup and for every reindex; the two used to be separate copies.
62
78
  */
63
- async function reindex() {
79
+ async function buildIndex(opts) {
64
80
  sources = loadSources();
65
81
  chunks = ingestSources(sources);
66
82
  // Collect operational data from project directories
@@ -104,11 +120,16 @@ async function reindex() {
104
120
  console.error(`[ContextEngine] 💻 Parsed ${codeChunks} code chunks from source files`);
105
121
  }
106
122
  }
107
- // Auto-import learnings from discovered doc sources
108
- // Dedup is built-in — safe to call on every reindex, no duplicates created
109
- const autoImport = autoImportFromSources(sources.map((s) => ({ path: s.path, name: s.name })));
110
- if (autoImport.imported > 0) {
111
- console.error(`[ContextEngine] 📥 Auto-imported ${autoImport.imported} new learnings from ${autoImport.total} doc sources (${autoImport.updated} updated)`);
123
+ // Auto-import learnings from discovered doc sources. Dedup is built-in. Only the indexer of a
124
+ // corpus writes the store from a sweep; a reader leaves that to it (one writer, not N).
125
+ if (opts.importLearnings) {
126
+ const autoImport = autoImportFromSources(sources.map((s) => ({ path: s.path, name: s.name })));
127
+ if (autoImport.imported > 0) {
128
+ console.error(`[ContextEngine] 📥 Auto-imported ${autoImport.imported} new learnings from ${autoImport.total} doc sources (${autoImport.updated} updated)`);
129
+ }
130
+ if (autoImport.refused) {
131
+ console.error(`[ContextEngine] ⛔ Auto-import write refused: ${autoImport.refused}`);
132
+ }
112
133
  }
113
134
  // Inject learnings as searchable chunks (project-scoped to prevent IP leakage)
114
135
  const learningChunks = learningsToChunks(activeProjectNames);
@@ -131,17 +152,109 @@ async function reindex() {
131
152
  }
132
153
  // Collect from plugin adapters
133
154
  if (config.adapters && config.adapters.length > 0) {
155
+ if (opts.loadAdapters) {
156
+ const adapterCount = await loadAdapters(config.adapters);
157
+ if (adapterCount === 0)
158
+ return;
159
+ }
134
160
  const adapterChunks = await collectFromAdapters(config.adapters);
135
161
  if (adapterChunks.length > 0) {
136
162
  chunks.push(...adapterChunks);
137
163
  console.error(`[ContextEngine] 🔌 Adapters contributed ${adapterChunks.length} chunks`);
138
164
  }
139
165
  }
140
- if (isEmbeddingsReady()) {
141
- console.error(`[ContextEngine] 🧠 Re-embedding ${chunks.length} chunks...`);
142
- embeddedChunks = await embedChunks(chunks);
143
- saveCache(chunks, embeddedChunks);
166
+ }
167
+ /** Load the model once; every caller shares the same promise. */
168
+ function ensureModel() {
169
+ if (!modelInit)
170
+ modelInit = initEmbeddings();
171
+ return modelInit;
172
+ }
173
+ /**
174
+ * Vectors for the current chunks: from the store for every text it holds, embedded now for
175
+ * the rest. [LOCK] [EMBEDDINGS-ARE-CONTENT-ADDRESSED]
176
+ */
177
+ async function embedAll() {
178
+ if (!isEmbeddingsReady())
179
+ return;
180
+ const snapshot = chunks;
181
+ const r = await embedChunks(snapshot, vectorStore);
182
+ if (snapshot !== chunks)
183
+ return; // a newer build replaced these chunks meanwhile; its own embed follows
184
+ embeddedChunks = r.embedded;
185
+ console.error(`[ContextEngine] ✅ Semantic search ready: ${r.reused} vectors from the store, ${r.fresh} embedded now`);
186
+ if (role === "indexer") {
187
+ try {
188
+ const c = compactEmbeddingStore(new Set(snapshot.map(embedKeyOf)));
189
+ if (c.compacted)
190
+ console.error(`[ContextEngine] 🧹 Embedding store compacted: ${c.before} -> ${c.after} records`);
191
+ }
192
+ catch (err) {
193
+ console.error(`[ContextEngine] ⚠ embedding store compaction failed: ${err.message}`);
194
+ }
195
+ }
196
+ }
197
+ /** Indexer only: write the shared index for readers of this corpus. No-op otherwise. */
198
+ function publishIndex() {
199
+ if (!corpus || role !== "indexer")
200
+ return;
201
+ try {
202
+ indexSeq++;
203
+ const r = writeSharedIndex({
204
+ corpus,
205
+ seq: indexSeq,
206
+ writer: process.pid,
207
+ sources,
208
+ activeProjectNames,
209
+ chunks,
210
+ keys: chunks.map(embedKeyOf),
211
+ });
212
+ lastIndexMtime = sharedIndexMtime(corpus);
213
+ safeAppend("index.write", { corpus, seq: indexSeq, chunks: chunks.length, vectors: embeddedChunks.length, bytes: r.bytes, ms: r.ms });
214
+ console.error(`[ContextEngine] 📤 Shared index written: seq ${indexSeq}, ${chunks.length} chunks, ${Math.round(r.bytes / 1024)} KB, ${r.ms} ms`);
215
+ }
216
+ catch (err) {
217
+ console.error(`[ContextEngine] ⚠ shared index write failed: ${err.message}`);
218
+ }
219
+ }
220
+ /** Reader: take the indexer's chunks and resolve their vectors from the store. */
221
+ function adoptSharedIndex() {
222
+ if (!corpus)
223
+ return false;
224
+ const f = readSharedIndex(corpus);
225
+ if (!f)
226
+ return false;
227
+ sources = f.sources;
228
+ chunks = f.chunks;
229
+ activeProjectNames = f.activeProjectNames;
230
+ try {
231
+ firewall.setProjectDirs(loadProjectDirs());
144
232
  }
233
+ catch { /* scoping keeps its last value */ }
234
+ vectorStore = loadEmbeddingStore().vectors;
235
+ const vecs = [];
236
+ let missing = 0;
237
+ f.chunks.forEach((c, i) => {
238
+ const v = vectorStore.get(f.keys[i]);
239
+ if (v)
240
+ vecs.push({ chunk: c, vector: v });
241
+ else
242
+ missing++;
243
+ });
244
+ embeddedChunks = vecs;
245
+ indexSeq = f.seq;
246
+ lastIndexMtime = sharedIndexMtime(corpus);
247
+ console.error(`[ContextEngine] 📥 Shared index loaded: seq ${f.seq} from pid ${f.writer}, ${chunks.length} chunks, ${vecs.length} vectors${missing ? `, ${missing} not embedded yet` : ""}`);
248
+ return true;
249
+ }
250
+ /**
251
+ * (Re-)ingest all sources. Called at startup and on file changes by the indexer; a reader
252
+ * never calls it on its own except as the fallback when no shared index exists yet.
253
+ */
254
+ async function reindex() {
255
+ await buildIndex({ importLearnings: role === "indexer" });
256
+ await embedAll();
257
+ publishIndex();
145
258
  }
146
259
  // ---------------------------------------------------------------------------
147
260
  // Hybrid Search: combine keyword + vector scores with temporal decay
@@ -217,8 +330,11 @@ function hybridSearch(query, keywordResults, vectorResults, topK) {
217
330
  // File Watching
218
331
  // ---------------------------------------------------------------------------
219
332
  const watchers = [];
220
- function startWatching() {
221
- // Clean up old watchers
333
+ let indexPoll = null;
334
+ let rolePoll = null;
335
+ const INDEX_POLL_MS = 3_000;
336
+ const ROLE_POLL_MS = 15_000;
337
+ function stopWatching() {
222
338
  for (const w of watchers) {
223
339
  try {
224
340
  w.close();
@@ -228,6 +344,10 @@ function startWatching() {
228
344
  }
229
345
  }
230
346
  watchers.length = 0;
347
+ }
348
+ /** Indexer only: one fs.watch per source; a change rebuilds, embeds the new chunks, publishes. */
349
+ function startWatching() {
350
+ stopWatching();
231
351
  let debounceTimer = null;
232
352
  for (const source of sources) {
233
353
  if (!existsSync(source.path))
@@ -238,6 +358,8 @@ function startWatching() {
238
358
  if (debounceTimer)
239
359
  clearTimeout(debounceTimer);
240
360
  debounceTimer = setTimeout(async () => {
361
+ if (role !== "indexer")
362
+ return; // demoted while the timer ran
241
363
  console.error(`[ContextEngine] 📝 File changed: ${basename(source.path)} — re-indexing...`);
242
364
  await reindex();
243
365
  console.error(`[ContextEngine] ✅ Re-indexed: ${chunks.length} chunks from ${sources.length} sources`);
@@ -251,6 +373,61 @@ function startWatching() {
251
373
  }
252
374
  console.error(`[ContextEngine] 👁 Watching ${watchers.length} source files for changes`);
253
375
  }
376
+ /** Reader only: reload the shared index when its stamp moves. A stat every 3 s, nothing else. */
377
+ function startIndexPolling() {
378
+ if (indexPoll || !corpus)
379
+ return;
380
+ indexPoll = setInterval(() => {
381
+ if (role !== "reader" || !corpus)
382
+ return;
383
+ const m = sharedIndexMtime(corpus);
384
+ if (m !== null && m !== lastIndexMtime)
385
+ adoptSharedIndex();
386
+ }, INDEX_POLL_MS);
387
+ indexPoll.unref();
388
+ }
389
+ function stopIndexPolling() {
390
+ if (indexPoll)
391
+ clearInterval(indexPoll);
392
+ indexPoll = null;
393
+ }
394
+ /** Re-run the election; on a change of role, switch what this server does. */
395
+ function evaluateRole(reason) {
396
+ if (!corpus)
397
+ return;
398
+ let e;
399
+ try {
400
+ e = electIndexer(corpus, listServers().servers, process.pid);
401
+ }
402
+ catch (err) {
403
+ console.error(`[ContextEngine] ⚠ election failed, staying ${role}: ${err.message}`);
404
+ return;
405
+ }
406
+ indexerPid = e.indexer;
407
+ if (e.role === role)
408
+ return;
409
+ const was = role;
410
+ role = e.role;
411
+ setRegistryRole?.(role);
412
+ safeAppend("server.role", { pid: process.pid, corpus, role, indexer: indexerPid, reason });
413
+ console.error(`[ContextEngine] 🧭 Role ${was} -> ${role} (${reason}; indexer pid ${indexerPid ?? process.pid})`);
414
+ if (role === "indexer") {
415
+ stopIndexPolling();
416
+ ensureModel().then(() => reindex()).then(() => startWatching()).catch((err) => {
417
+ console.error(`[ContextEngine] ⚠ taking over as indexer failed: ${err.message}`);
418
+ });
419
+ }
420
+ else {
421
+ stopWatching();
422
+ startIndexPolling();
423
+ }
424
+ }
425
+ function startRolePolling() {
426
+ if (rolePoll || !corpus)
427
+ return;
428
+ rolePoll = setInterval(() => evaluateRole("periodic"), ROLE_POLL_MS);
429
+ rolePoll.unref();
430
+ }
254
431
  // ---------------------------------------------------------------------------
255
432
  // MCP Server
256
433
  // ---------------------------------------------------------------------------
@@ -289,6 +466,10 @@ server.tool("search_context", "Search across all indexed project knowledge (copi
289
466
  .describe("Search mode: hybrid (default), keyword-only, or semantic-only"),
290
467
  }, async ({ query, top_k, mode }) => {
291
468
  let results = [];
469
+ // A reader keeps its 300 MB model unloaded until someone asks for semantics; the first such
470
+ // query is answered by keyword while the model loads. [LOCK] [EMBEDDINGS-ARE-CONTENT-ADDRESSED]
471
+ if (mode !== "keyword" && !isEmbeddingsReady())
472
+ void ensureModel();
292
473
  if (mode === "keyword" || mode === "hybrid") {
293
474
  const kwResults = searchChunks(chunks, query, top_k * 2);
294
475
  if (mode === "keyword" || !isEmbeddingsReady()) {
@@ -424,6 +605,10 @@ server.tool("read_source", "Read the full content of a specific knowledge source
424
605
  // Tool: reindex
425
606
  // ---------------------------------------------------------------------------
426
607
  server.tool("reindex", "Force a full re-index of all knowledge sources. Use after adding new files or changing contextengine.json.", {}, async () => {
608
+ if (role === "reader" && corpus) {
609
+ adoptSharedIndex();
610
+ return respond("reindex", `This server reads the shared index of corpus ${corpus}, written by pid ${indexerPid ?? "?"}: reloaded seq ${indexSeq}, ${chunks.length} chunks, ${embeddedChunks.length} vectors. Saving a doc makes the indexer rebuild; every reader picks it up within ${INDEX_POLL_MS / 1000} s.`);
611
+ }
427
612
  await reindex();
428
613
  return respond("reindex", `Re-indexed: ${chunks.length} chunks from ${sources.length} sources. Embeddings: ${embeddedChunks.length} vectors.`);
429
614
  });
@@ -973,7 +1158,13 @@ server.tool("import_learnings", "Bulk-import learnings from a Markdown or JSON f
973
1158
  .optional()
974
1159
  .describe("Import every heading, bold bullet and table row as a rule (the pre-2.5.7 behaviour). Default false: only marked learnings."),
975
1160
  }, async ({ file_path, default_category, project, permissive }) => {
976
- const result = importLearningsFromFile(file_path, default_category || "other", project, { permissive: permissive === true });
1161
+ let result;
1162
+ try {
1163
+ result = importLearningsFromFile(file_path, default_category || "other", project, { permissive: permissive === true });
1164
+ }
1165
+ catch (e) {
1166
+ return respond("import_learnings", `⛔ Import refused: ${e?.message || e}`);
1167
+ }
977
1168
  // Re-inject learnings into search index (project-scoped)
978
1169
  const newChunks = learningsToChunks(activeProjectNames);
979
1170
  const nonLearningChunks = chunks.filter((c) => c.source !== "💡 Learnings Store");
@@ -1078,70 +1269,45 @@ function registerResources() {
1078
1269
  // Start
1079
1270
  // ---------------------------------------------------------------------------
1080
1271
  async function main() {
1081
- // 1. Ingest all sources (fast — keyword search available immediately)
1082
- sources = loadSources();
1083
- chunks = ingestSources(sources);
1084
- // 1b. Collect operational data (git, deps, env, docker, pm2, etc.)
1085
- const config = loadConfig();
1086
- const projectDirs = loadProjectDirs();
1087
- activeProjectNames = projectDirs.map((d) => d.name);
1088
- firewall.setProjectDirs(projectDirs);
1089
- if (config.collectOps !== false) {
1090
- let opsChunks = 0;
1091
- for (const dir of projectDirs) {
1092
- const ops = collectProjectOps(dir.path, dir.name);
1093
- chunks.push(...ops);
1094
- opsChunks += ops.length;
1095
- }
1096
- if (opsChunks > 0) {
1097
- console.error(`[ContextEngine] ⚙ Collected ${opsChunks} operational chunks from ${projectDirs.length} projects`);
1272
+ // 0. Inventory this server FIRST, before indexing takes minutes: a server exists the moment it
1273
+ // starts. [LOCK] [SERVERS-ARE-INVENTORIED]. With the shared index on, the registry is also
1274
+ // the electorate: the record carries the corpus and the role. [LOCK] [ONE-INDEXER-MANY-READERS]
1275
+ if (sharedIndexEnabled()) {
1276
+ try {
1277
+ corpus = corpusId();
1098
1278
  }
1099
- }
1100
- if (config.collectSystemOps !== false) {
1101
- const sysOps = collectSystemOps();
1102
- if (sysOps.length > 0) {
1103
- chunks.push(...sysOps);
1104
- console.error(`[ContextEngine] 🖥 Collected ${sysOps.length} system operational chunks`);
1279
+ catch (err) {
1280
+ console.error("[ContextEngine] ⚠ corpus id failed, shared index off for this server:", err);
1105
1281
  }
1106
1282
  }
1107
- // 1c. Scan code files (TS/JS/Python) if configured
1108
- if (config.codeDirs && config.codeDirs.length > 0) {
1109
- let codeChunks = 0;
1110
- for (const dir of projectDirs) {
1111
- for (const codeDir of config.codeDirs) {
1112
- const codePath = join(dir.path, codeDir);
1113
- if (existsSync(codePath)) {
1114
- const codeResults = scanCodeDir(codePath, dir.name);
1115
- chunks.push(...codeResults);
1116
- codeChunks += codeResults.length;
1117
- }
1118
- }
1119
- }
1120
- if (codeChunks > 0) {
1121
- console.error(`[ContextEngine] 💻 Parsed ${codeChunks} code chunks from source files`);
1283
+ try {
1284
+ const reg = registerServer({ version: PKG_VERSION, script: fileURLToPath(import.meta.url), corpus, role: corpus ? "reader" : undefined });
1285
+ setRegistryRole = reg.setRole;
1286
+ const fleet = listServers();
1287
+ if (corpus) {
1288
+ const e = electIndexer(corpus, fleet.servers, process.pid);
1289
+ role = e.role;
1290
+ indexerPid = e.indexer;
1291
+ reg.setRole(role);
1292
+ safeAppend("server.role", { pid: process.pid, corpus, role, indexer: indexerPid, reason: "start" });
1122
1293
  }
1294
+ console.error(`[ContextEngine] 🧭 ${formatServers(fleet)}`);
1295
+ if (corpus)
1296
+ console.error(`[ContextEngine] 🧭 This server: ${role} of corpus ${corpus}${role === "reader" ? ` (indexer pid ${indexerPid})` : ""}`);
1297
+ safeAppend("server.start", { pid: reg.record.pid, parent: reg.record.parent, version: reg.record.version, build: reg.record.build, cwd: reg.record.cwd, servers_running: fleet.servers.length, stale_builds: fleet.servers.filter((x) => x.staleBuild).length });
1123
1298
  }
1124
- // 1d. Auto-import learnings from discovered doc sources
1125
- const autoImport = autoImportFromSources(sources.map((s) => ({ path: s.path, name: s.name })));
1126
- if (autoImport.imported > 0) {
1127
- console.error(`[ContextEngine] 📥 Auto-imported ${autoImport.imported} new learnings from ${autoImport.total} doc sources (${autoImport.updated} updated)`);
1299
+ catch (err) {
1300
+ console.error("[ContextEngine] ⚠ Server registry failed:", err);
1128
1301
  }
1129
- // 1e. Inject learnings into search index (project-scoped)
1130
- const learningChunks = learningsToChunks(activeProjectNames);
1131
- if (learningChunks.length > 0) {
1132
- chunks.push(...learningChunks);
1133
- console.error(`[ContextEngine] 💡 Injected ${learningChunks.length} learning chunks into search index (scoped)`);
1134
- }
1135
- // 1f. Load and collect from plugin adapters
1136
- if (config.adapters && config.adapters.length > 0) {
1137
- const adapterCount = await loadAdapters(config.adapters);
1138
- if (adapterCount > 0) {
1139
- const adapterChunks = await collectFromAdapters(config.adapters);
1140
- if (adapterChunks.length > 0) {
1141
- chunks.push(...adapterChunks);
1142
- console.error(`[ContextEngine] 🔌 Adapters contributed ${adapterChunks.length} chunks from ${adapterCount} adapters`);
1143
- }
1144
- }
1302
+ // 1. The index: a reader takes the indexer's; everyone else, or a reader with nothing to
1303
+ // take yet, builds it (fast, keyword search available immediately).
1304
+ let adopted = false;
1305
+ if (role === "reader")
1306
+ adopted = adoptSharedIndex();
1307
+ if (!adopted) {
1308
+ if (role === "reader")
1309
+ console.error("[ContextEngine] 📥 No shared index yet; building locally once, without importing learnings");
1310
+ await buildIndex({ importLearnings: role === "indexer", loadAdapters: true });
1145
1311
  }
1146
1312
  // 2. Register MCP resources
1147
1313
  registerResources();
@@ -1208,24 +1374,28 @@ async function main() {
1208
1374
  catch (err) {
1209
1375
  console.error("[ContextEngine] ⚠ Failed to write server-meta.json:", err);
1210
1376
  }
1211
- // 4. Load embeddings — try cache first, then model (non-blocking)
1212
- const cached = loadCache(chunks);
1213
- if (cached) {
1214
- embeddedChunks = cached;
1215
- console.error(`[ContextEngine] ✅ Semantic search ready from cache (${embeddedChunks.length} vectors)`);
1216
- }
1217
- else {
1218
- initEmbeddings().then(async (ready) => {
1219
- if (ready) {
1220
- console.error(`[ContextEngine] 🧠 Embedding ${chunks.length} chunks...`);
1221
- embeddedChunks = await embedChunks(chunks);
1222
- saveCache(chunks, embeddedChunks);
1223
- console.error(`[ContextEngine] ✅ Semantic search ready (${embeddedChunks.length} vectors)`);
1224
- }
1377
+ // 4. Vectors. The store holds one vector per text ever embedded on this machine; the model
1378
+ // always loads for whoever embeds or answers queries. [LOCK] [EMBEDDINGS-ARE-CONTENT-ADDRESSED]
1379
+ // A reader that adopted the index already has its vectors and loads the model on its first
1380
+ // semantic query, not before: 300 MB per process is worth waiting for.
1381
+ if (!adopted) {
1382
+ vectorStore = loadEmbeddingStore().vectors;
1383
+ if (vectorStore.size > 0)
1384
+ console.error(`[ContextEngine] 💾 Embedding store: ${vectorStore.size} vectors`);
1385
+ publishIndex(); // keyword-searchable index for readers now; vectors follow
1386
+ ensureModel().then(async (ready) => {
1387
+ if (!ready)
1388
+ return;
1389
+ await embedAll();
1390
+ publishIndex();
1225
1391
  });
1226
1392
  }
1227
- // 5. Start file watchers
1228
- startWatching();
1393
+ // 5. Watch (indexer) or poll (reader), and keep the election running
1394
+ if (role === "indexer")
1395
+ startWatching();
1396
+ else
1397
+ startIndexPolling();
1398
+ startRolePolling();
1229
1399
  // 6. Boot the local HTTP event-ingest endpoint for the browser extension.
1230
1400
  // Local 127.0.0.1:7842 only; auth via shared secret at
1231
1401
  // ~/.contextengine/extension-secret (see init-extension-secret CLI).
@@ -26,6 +26,7 @@ export declare function withStoreLock<T>(fn: () => T): T;
26
26
  * loadStore() returns the same in-memory store and every saveStore() only marks it dirty.
27
27
  */
28
28
  export declare function withStoreBatch<T>(fn: () => T): T;
29
+ export declare const MAX_GROWTH_PER_WRITE = 200;
29
30
  /** Test seam for the writer's tripwire; not part of the API. */
30
31
  export declare function __writeStoreForTests(store: LearningsStore): void;
31
32
  /**
@@ -120,6 +121,7 @@ export declare function autoImportFromSources(sources: Array<{
120
121
  imported: number;
121
122
  updated: number;
122
123
  ignored: number;
124
+ refused?: string;
123
125
  };
124
126
  /**
125
127
  * Get the store stats.
package/dist/learnings.js CHANGED
@@ -266,6 +266,19 @@ function dailyBackup() {
266
266
  }
267
267
  catch { /* a missing backup must never block a save */ }
268
268
  }
269
+ // [LOCKED] [STORE-GROWTH-IS-A-TRIPWIRE-TOO] 2026-09-05
270
+ // [NEVER] let one write add more than MAX_GROWTH_PER_WRITE records to the store without the
271
+ // explicit override, and never raise the limit to make an import "just work".
272
+ // WHY: the shrink guard below caught the wipe of 2026-09-05; the same evening two stale servers
273
+ // wrote 1,766 records in one minute and nothing objected, because only shrinking was
274
+ // guarded. Every legitimate write is small: an agent saves one rule, the strict auto-import
275
+ // of a whole workspace produced 99 (replayed 2026-09-05), the largest real learnings file
276
+ // a few dozen. Thousands in one write is a bug or old code, never a lesson.
277
+ // FIX: a write that grows the store by more than MAX_GROWTH_PER_WRITE over the file on disk
278
+ // is refused with an audit event, unless CONTEXTENGINE_ALLOW_BULK=1 (set knowingly, for
279
+ // one deliberate bulk import). The auto-import catches the refusal and reports it; the
280
+ // server keeps running.
281
+ export const MAX_GROWTH_PER_WRITE = 200;
269
282
  function writeStoreToDisk(store) {
270
283
  ensureDir();
271
284
  store.count = store.learnings.length;
@@ -284,6 +297,21 @@ function writeStoreToDisk(store) {
284
297
  throw new Error(`refusing to write ${store.learnings.length} learnings over a store of ${onDisk}: that is the shape of a wipe, not an edit. Set CONTEXTENGINE_ALLOW_SHRINK=1 if this is deliberate.`);
285
298
  }
286
299
  }
300
+ // Growth tripwire. [LOCK] [STORE-GROWTH-IS-A-TRIPWIRE-TOO]
301
+ if (existsSync(LEARNINGS_PATH) && process.env.CONTEXTENGINE_ALLOW_BULK !== "1") {
302
+ let onDisk = -1;
303
+ try {
304
+ onDisk = (JSON.parse(readFileSync(LEARNINGS_PATH, "utf-8")).learnings || []).length;
305
+ }
306
+ catch {
307
+ onDisk = -1;
308
+ }
309
+ const growth = store.learnings.length - onDisk;
310
+ if (onDisk >= 0 && growth > MAX_GROWTH_PER_WRITE) {
311
+ safeAppend("learning.store_growth_refused", { on_disk: onDisk, attempted: store.learnings.length, growth });
312
+ throw new Error(`refusing to add ${growth} learnings in one write (store ${onDisk}, limit ${MAX_GROWTH_PER_WRITE}): that is the shape of a runaway import, not a lesson. Set CONTEXTENGINE_ALLOW_BULK=1 for one deliberate bulk import.`);
313
+ }
314
+ }
287
315
  dailyBackup();
288
316
  const tmp = `${LEARNINGS_PATH}.tmp-${process.pid}-${Date.now()}`;
289
317
  writeFileSync(tmp, JSON.stringify(store, null, 2));
@@ -975,27 +1003,38 @@ export function autoImportFromSources(sources) {
975
1003
  let totalUpdated = 0;
976
1004
  let totalIgnored = 0;
977
1005
  let processed = 0;
1006
+ let refused;
978
1007
  // One load and one save for the whole sweep (~880 files), instead of one full-file
979
1008
  // rewrite per rule per file. [LOCK] [STORE-NEVER-STARTS-FRESH-OVER-DATA]
980
- withStoreBatch(() => {
981
- for (const source of sources) {
982
- // Only process markdown files
983
- if (!source.path.endsWith(".md"))
984
- continue;
985
- if (!existsSync(source.path))
986
- continue;
987
- // Extract project name from source name (e.g., "ContextEngine — copilot-instructions.md")
988
- const project = source.name.split(" — ")[0]?.trim() || undefined;
989
- // Strict by construction: only marked learnings. [LOCK] [AUTO-IMPORT-ONLY-MARKED-LEARNINGS]
990
- const result = importLearningsFromFile(source.path, "other", project);
991
- totalImported += result.imported;
992
- totalUpdated += result.updated;
993
- totalIgnored += result.ignored;
994
- if (result.imported > 0 || result.updated > 0)
995
- processed++;
996
- }
997
- });
998
- return { total: processed, imported: totalImported, updated: totalUpdated, ignored: totalIgnored };
1009
+ try {
1010
+ withStoreBatch(() => {
1011
+ for (const source of sources) {
1012
+ // Only process markdown files
1013
+ if (!source.path.endsWith(".md"))
1014
+ continue;
1015
+ if (!existsSync(source.path))
1016
+ continue;
1017
+ // Extract project name from source name (e.g., "ContextEngine — copilot-instructions.md")
1018
+ const project = source.name.split(" — ")[0]?.trim() || undefined;
1019
+ // Strict by construction: only marked learnings. [LOCK] [AUTO-IMPORT-ONLY-MARKED-LEARNINGS]
1020
+ const result = importLearningsFromFile(source.path, "other", project);
1021
+ totalImported += result.imported;
1022
+ totalUpdated += result.updated;
1023
+ totalIgnored += result.ignored;
1024
+ if (result.imported > 0 || result.updated > 0)
1025
+ processed++;
1026
+ }
1027
+ });
1028
+ }
1029
+ catch (e) {
1030
+ // A refused write (growth or shrink tripwire, lock timeout) must not take the server down;
1031
+ // it is reported to the caller and the store is left as it was. [LOCK] [STORE-GROWTH-IS-A-TRIPWIRE-TOO]
1032
+ refused = String(e?.message || e);
1033
+ totalImported = 0;
1034
+ totalUpdated = 0;
1035
+ processed = 0;
1036
+ }
1037
+ return { total: processed, imported: totalImported, updated: totalUpdated, ignored: totalIgnored, refused };
999
1038
  }
1000
1039
  /**
1001
1040
  * Get the store stats.
@@ -0,0 +1,47 @@
1
+ export interface ServerRecord {
2
+ pid: number;
3
+ ppid: number;
4
+ parent: string;
5
+ started: string;
6
+ heartbeat: string;
7
+ version: string;
8
+ script: string;
9
+ build: string;
10
+ cwd: string;
11
+ node: string;
12
+ /** Since 2.6.0: what this server indexes (see shared-index.ts corpusId) and whether it is
13
+ * the one writing the shared index for it, or a reader of it. Absent on older builds. */
14
+ corpus?: string;
15
+ role?: "indexer" | "reader";
16
+ }
17
+ export interface ServerReport {
18
+ servers: Array<ServerRecord & {
19
+ alive: true;
20
+ currentBuild: string | null;
21
+ staleBuild: boolean;
22
+ }>;
23
+ removed: number;
24
+ warnings: string[];
25
+ }
26
+ /** More concurrent servers than this and every doc change costs that many re-embeds. */
27
+ export declare const SERVER_COUNT_WARN = 3;
28
+ /** Short content hash of the script a server loaded; the build identity. */
29
+ export declare function buildHashOf(scriptPath: string): string | null;
30
+ export declare function isAlive(pid: number): boolean;
31
+ /**
32
+ * Register the running server. Returns a stop() that removes the record; exit handlers call it too.
33
+ */
34
+ export declare function registerServer(opts: {
35
+ version: string;
36
+ script: string;
37
+ corpus?: string;
38
+ role?: "indexer" | "reader";
39
+ }): {
40
+ record: ServerRecord;
41
+ stop: () => void;
42
+ setRole: (role: "indexer" | "reader") => void;
43
+ };
44
+ /** Read every record, drop the dead ones, compare builds with the files on disk now. */
45
+ export declare function listServers(): ServerReport;
46
+ export declare function formatServers(report: ServerReport, home?: string): string;
47
+ //# sourceMappingURL=server-registry.d.ts.map