@compr/opscontext-mcp 2.5.8 → 2.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +52 -0
- package/dist/audit.d.ts +1 -1
- package/dist/cli-commands.js +1 -0
- package/dist/cli.js +26 -1
- package/dist/config.d.ts +5 -0
- package/dist/config.js +1 -1
- package/dist/embedding-store.d.ts +46 -0
- package/dist/embedding-store.js +141 -0
- package/dist/embeddings.d.ts +15 -3
- package/dist/embeddings.js +48 -18
- package/dist/index.js +261 -91
- package/dist/learnings.d.ts +2 -0
- package/dist/learnings.js +58 -19
- package/dist/server-registry.d.ts +47 -0
- package/dist/server-registry.js +162 -0
- package/dist/shared-index.d.ts +45 -0
- package/dist/shared-index.js +113 -0
- package/package.json +1 -1
package/dist/index.js
CHANGED
|
@@ -5,12 +5,14 @@ import { z } from "zod";
|
|
|
5
5
|
import { loadSources, loadProjectDirs, loadConfig, resolveProjectDir } from "./config.js";
|
|
6
6
|
import { ingestSources } from "./ingest.js";
|
|
7
7
|
import { searchChunks } from "./search.js";
|
|
8
|
-
import { initEmbeddings, embedChunks, vectorSearch, isEmbeddingsReady, } from "./embeddings.js";
|
|
8
|
+
import { initEmbeddings, embedChunks, embedKeyOf, vectorSearch, isEmbeddingsReady, } from "./embeddings.js";
|
|
9
9
|
import { collectProjectOps, collectSystemOps } from "./collectors.js";
|
|
10
|
-
import {
|
|
10
|
+
import { loadEmbeddingStore, compactEmbeddingStore } from "./embedding-store.js";
|
|
11
|
+
import { sharedIndexEnabled, corpusId, electIndexer, writeSharedIndex, readSharedIndex, sharedIndexMtime, } from "./shared-index.js";
|
|
11
12
|
import { listProjects, checkPorts, runComplianceAudit, formatProjectList, formatPortMap, formatPlan, scoreProject, formatScoreReport, runScoreCanary, } from "./agents.js";
|
|
12
13
|
import { saveSession, loadSession, listSessions, deleteSession, formatSession, formatSessionList, } from "./sessions.js";
|
|
13
|
-
import { verifyChain, readAuditLog, filterByRange, autoRotateAuditLog } from "./audit.js";
|
|
14
|
+
import { verifyChain, readAuditLog, filterByRange, autoRotateAuditLog, safeAppend } from "./audit.js";
|
|
15
|
+
import { registerServer, listServers, formatServers } from "./server-registry.js";
|
|
14
16
|
import { startEventIngestServer } from "./http-server.js";
|
|
15
17
|
import { detect } from "./detector.js";
|
|
16
18
|
import { buildCostReport } from "./cost-report.js";
|
|
@@ -43,6 +45,18 @@ let chunks = [];
|
|
|
43
45
|
let embeddedChunks = [];
|
|
44
46
|
let activeProjectNames = [];
|
|
45
47
|
const firewall = new ProtocolFirewall();
|
|
48
|
+
// One indexer, many readers. [LOCK] [ONE-INDEXER-MANY-READERS]
|
|
49
|
+
// With the shared index off (default until the trial), every server is its own indexer, exactly
|
|
50
|
+
// as before, and only the content-addressed vector store below is new.
|
|
51
|
+
let role = "indexer";
|
|
52
|
+
let corpus;
|
|
53
|
+
let indexerPid = null;
|
|
54
|
+
let setRegistryRole = null;
|
|
55
|
+
/** key -> vector, loaded from ~/.contextengine/embeddings.bin and grown by what we embed. */
|
|
56
|
+
let vectorStore = new Map();
|
|
57
|
+
let indexSeq = 0;
|
|
58
|
+
let lastIndexMtime = null;
|
|
59
|
+
let modelInit = null;
|
|
46
60
|
// Wire up learning search for auto-injection (avoids circular import)
|
|
47
61
|
firewall.setLearningSearchFn((query, projects) => {
|
|
48
62
|
return searchLearnings(query)
|
|
@@ -58,9 +72,11 @@ firewall.setLearningSearchFn((query, projects) => {
|
|
|
58
72
|
.map((l) => ({ rule: l.rule, project: l.project, category: l.category }));
|
|
59
73
|
});
|
|
60
74
|
/**
|
|
61
|
-
*
|
|
75
|
+
* Parse every source, collect ops and code, import learnings (indexer only), inject learnings,
|
|
76
|
+
* community rules and adapters. Sets `sources`, `chunks`, `activeProjectNames`. No embedding
|
|
77
|
+
* here. One body for startup and for every reindex; the two used to be separate copies.
|
|
62
78
|
*/
|
|
63
|
-
async function
|
|
79
|
+
async function buildIndex(opts) {
|
|
64
80
|
sources = loadSources();
|
|
65
81
|
chunks = ingestSources(sources);
|
|
66
82
|
// Collect operational data from project directories
|
|
@@ -104,11 +120,16 @@ async function reindex() {
|
|
|
104
120
|
console.error(`[ContextEngine] 💻 Parsed ${codeChunks} code chunks from source files`);
|
|
105
121
|
}
|
|
106
122
|
}
|
|
107
|
-
// Auto-import learnings from discovered doc sources
|
|
108
|
-
//
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
123
|
+
// Auto-import learnings from discovered doc sources. Dedup is built-in. Only the indexer of a
|
|
124
|
+
// corpus writes the store from a sweep; a reader leaves that to it (one writer, not N).
|
|
125
|
+
if (opts.importLearnings) {
|
|
126
|
+
const autoImport = autoImportFromSources(sources.map((s) => ({ path: s.path, name: s.name })));
|
|
127
|
+
if (autoImport.imported > 0) {
|
|
128
|
+
console.error(`[ContextEngine] 📥 Auto-imported ${autoImport.imported} new learnings from ${autoImport.total} doc sources (${autoImport.updated} updated)`);
|
|
129
|
+
}
|
|
130
|
+
if (autoImport.refused) {
|
|
131
|
+
console.error(`[ContextEngine] ⛔ Auto-import write refused: ${autoImport.refused}`);
|
|
132
|
+
}
|
|
112
133
|
}
|
|
113
134
|
// Inject learnings as searchable chunks (project-scoped to prevent IP leakage)
|
|
114
135
|
const learningChunks = learningsToChunks(activeProjectNames);
|
|
@@ -131,17 +152,109 @@ async function reindex() {
|
|
|
131
152
|
}
|
|
132
153
|
// Collect from plugin adapters
|
|
133
154
|
if (config.adapters && config.adapters.length > 0) {
|
|
155
|
+
if (opts.loadAdapters) {
|
|
156
|
+
const adapterCount = await loadAdapters(config.adapters);
|
|
157
|
+
if (adapterCount === 0)
|
|
158
|
+
return;
|
|
159
|
+
}
|
|
134
160
|
const adapterChunks = await collectFromAdapters(config.adapters);
|
|
135
161
|
if (adapterChunks.length > 0) {
|
|
136
162
|
chunks.push(...adapterChunks);
|
|
137
163
|
console.error(`[ContextEngine] 🔌 Adapters contributed ${adapterChunks.length} chunks`);
|
|
138
164
|
}
|
|
139
165
|
}
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
166
|
+
}
|
|
167
|
+
/** Load the model once; every caller shares the same promise. */
|
|
168
|
+
function ensureModel() {
|
|
169
|
+
if (!modelInit)
|
|
170
|
+
modelInit = initEmbeddings();
|
|
171
|
+
return modelInit;
|
|
172
|
+
}
|
|
173
|
+
/**
|
|
174
|
+
* Vectors for the current chunks: from the store for every text it holds, embedded now for
|
|
175
|
+
* the rest. [LOCK] [EMBEDDINGS-ARE-CONTENT-ADDRESSED]
|
|
176
|
+
*/
|
|
177
|
+
async function embedAll() {
|
|
178
|
+
if (!isEmbeddingsReady())
|
|
179
|
+
return;
|
|
180
|
+
const snapshot = chunks;
|
|
181
|
+
const r = await embedChunks(snapshot, vectorStore);
|
|
182
|
+
if (snapshot !== chunks)
|
|
183
|
+
return; // a newer build replaced these chunks meanwhile; its own embed follows
|
|
184
|
+
embeddedChunks = r.embedded;
|
|
185
|
+
console.error(`[ContextEngine] ✅ Semantic search ready: ${r.reused} vectors from the store, ${r.fresh} embedded now`);
|
|
186
|
+
if (role === "indexer") {
|
|
187
|
+
try {
|
|
188
|
+
const c = compactEmbeddingStore(new Set(snapshot.map(embedKeyOf)));
|
|
189
|
+
if (c.compacted)
|
|
190
|
+
console.error(`[ContextEngine] 🧹 Embedding store compacted: ${c.before} -> ${c.after} records`);
|
|
191
|
+
}
|
|
192
|
+
catch (err) {
|
|
193
|
+
console.error(`[ContextEngine] ⚠ embedding store compaction failed: ${err.message}`);
|
|
194
|
+
}
|
|
195
|
+
}
|
|
196
|
+
}
|
|
197
|
+
/** Indexer only: write the shared index for readers of this corpus. No-op otherwise. */
|
|
198
|
+
function publishIndex() {
|
|
199
|
+
if (!corpus || role !== "indexer")
|
|
200
|
+
return;
|
|
201
|
+
try {
|
|
202
|
+
indexSeq++;
|
|
203
|
+
const r = writeSharedIndex({
|
|
204
|
+
corpus,
|
|
205
|
+
seq: indexSeq,
|
|
206
|
+
writer: process.pid,
|
|
207
|
+
sources,
|
|
208
|
+
activeProjectNames,
|
|
209
|
+
chunks,
|
|
210
|
+
keys: chunks.map(embedKeyOf),
|
|
211
|
+
});
|
|
212
|
+
lastIndexMtime = sharedIndexMtime(corpus);
|
|
213
|
+
safeAppend("index.write", { corpus, seq: indexSeq, chunks: chunks.length, vectors: embeddedChunks.length, bytes: r.bytes, ms: r.ms });
|
|
214
|
+
console.error(`[ContextEngine] 📤 Shared index written: seq ${indexSeq}, ${chunks.length} chunks, ${Math.round(r.bytes / 1024)} KB, ${r.ms} ms`);
|
|
215
|
+
}
|
|
216
|
+
catch (err) {
|
|
217
|
+
console.error(`[ContextEngine] ⚠ shared index write failed: ${err.message}`);
|
|
218
|
+
}
|
|
219
|
+
}
|
|
220
|
+
/** Reader: take the indexer's chunks and resolve their vectors from the store. */
|
|
221
|
+
function adoptSharedIndex() {
|
|
222
|
+
if (!corpus)
|
|
223
|
+
return false;
|
|
224
|
+
const f = readSharedIndex(corpus);
|
|
225
|
+
if (!f)
|
|
226
|
+
return false;
|
|
227
|
+
sources = f.sources;
|
|
228
|
+
chunks = f.chunks;
|
|
229
|
+
activeProjectNames = f.activeProjectNames;
|
|
230
|
+
try {
|
|
231
|
+
firewall.setProjectDirs(loadProjectDirs());
|
|
144
232
|
}
|
|
233
|
+
catch { /* scoping keeps its last value */ }
|
|
234
|
+
vectorStore = loadEmbeddingStore().vectors;
|
|
235
|
+
const vecs = [];
|
|
236
|
+
let missing = 0;
|
|
237
|
+
f.chunks.forEach((c, i) => {
|
|
238
|
+
const v = vectorStore.get(f.keys[i]);
|
|
239
|
+
if (v)
|
|
240
|
+
vecs.push({ chunk: c, vector: v });
|
|
241
|
+
else
|
|
242
|
+
missing++;
|
|
243
|
+
});
|
|
244
|
+
embeddedChunks = vecs;
|
|
245
|
+
indexSeq = f.seq;
|
|
246
|
+
lastIndexMtime = sharedIndexMtime(corpus);
|
|
247
|
+
console.error(`[ContextEngine] 📥 Shared index loaded: seq ${f.seq} from pid ${f.writer}, ${chunks.length} chunks, ${vecs.length} vectors${missing ? `, ${missing} not embedded yet` : ""}`);
|
|
248
|
+
return true;
|
|
249
|
+
}
|
|
250
|
+
/**
|
|
251
|
+
* (Re-)ingest all sources. Called at startup and on file changes by the indexer; a reader
|
|
252
|
+
* never calls it on its own except as the fallback when no shared index exists yet.
|
|
253
|
+
*/
|
|
254
|
+
async function reindex() {
|
|
255
|
+
await buildIndex({ importLearnings: role === "indexer" });
|
|
256
|
+
await embedAll();
|
|
257
|
+
publishIndex();
|
|
145
258
|
}
|
|
146
259
|
// ---------------------------------------------------------------------------
|
|
147
260
|
// Hybrid Search: combine keyword + vector scores with temporal decay
|
|
@@ -217,8 +330,11 @@ function hybridSearch(query, keywordResults, vectorResults, topK) {
|
|
|
217
330
|
// File Watching
|
|
218
331
|
// ---------------------------------------------------------------------------
|
|
219
332
|
const watchers = [];
|
|
220
|
-
|
|
221
|
-
|
|
333
|
+
let indexPoll = null;
|
|
334
|
+
let rolePoll = null;
|
|
335
|
+
const INDEX_POLL_MS = 3_000;
|
|
336
|
+
const ROLE_POLL_MS = 15_000;
|
|
337
|
+
function stopWatching() {
|
|
222
338
|
for (const w of watchers) {
|
|
223
339
|
try {
|
|
224
340
|
w.close();
|
|
@@ -228,6 +344,10 @@ function startWatching() {
|
|
|
228
344
|
}
|
|
229
345
|
}
|
|
230
346
|
watchers.length = 0;
|
|
347
|
+
}
|
|
348
|
+
/** Indexer only: one fs.watch per source; a change rebuilds, embeds the new chunks, publishes. */
|
|
349
|
+
function startWatching() {
|
|
350
|
+
stopWatching();
|
|
231
351
|
let debounceTimer = null;
|
|
232
352
|
for (const source of sources) {
|
|
233
353
|
if (!existsSync(source.path))
|
|
@@ -238,6 +358,8 @@ function startWatching() {
|
|
|
238
358
|
if (debounceTimer)
|
|
239
359
|
clearTimeout(debounceTimer);
|
|
240
360
|
debounceTimer = setTimeout(async () => {
|
|
361
|
+
if (role !== "indexer")
|
|
362
|
+
return; // demoted while the timer ran
|
|
241
363
|
console.error(`[ContextEngine] 📝 File changed: ${basename(source.path)} — re-indexing...`);
|
|
242
364
|
await reindex();
|
|
243
365
|
console.error(`[ContextEngine] ✅ Re-indexed: ${chunks.length} chunks from ${sources.length} sources`);
|
|
@@ -251,6 +373,61 @@ function startWatching() {
|
|
|
251
373
|
}
|
|
252
374
|
console.error(`[ContextEngine] 👁 Watching ${watchers.length} source files for changes`);
|
|
253
375
|
}
|
|
376
|
+
/** Reader only: reload the shared index when its stamp moves. A stat every 3 s, nothing else. */
|
|
377
|
+
function startIndexPolling() {
|
|
378
|
+
if (indexPoll || !corpus)
|
|
379
|
+
return;
|
|
380
|
+
indexPoll = setInterval(() => {
|
|
381
|
+
if (role !== "reader" || !corpus)
|
|
382
|
+
return;
|
|
383
|
+
const m = sharedIndexMtime(corpus);
|
|
384
|
+
if (m !== null && m !== lastIndexMtime)
|
|
385
|
+
adoptSharedIndex();
|
|
386
|
+
}, INDEX_POLL_MS);
|
|
387
|
+
indexPoll.unref();
|
|
388
|
+
}
|
|
389
|
+
function stopIndexPolling() {
|
|
390
|
+
if (indexPoll)
|
|
391
|
+
clearInterval(indexPoll);
|
|
392
|
+
indexPoll = null;
|
|
393
|
+
}
|
|
394
|
+
/** Re-run the election; on a change of role, switch what this server does. */
|
|
395
|
+
function evaluateRole(reason) {
|
|
396
|
+
if (!corpus)
|
|
397
|
+
return;
|
|
398
|
+
let e;
|
|
399
|
+
try {
|
|
400
|
+
e = electIndexer(corpus, listServers().servers, process.pid);
|
|
401
|
+
}
|
|
402
|
+
catch (err) {
|
|
403
|
+
console.error(`[ContextEngine] ⚠ election failed, staying ${role}: ${err.message}`);
|
|
404
|
+
return;
|
|
405
|
+
}
|
|
406
|
+
indexerPid = e.indexer;
|
|
407
|
+
if (e.role === role)
|
|
408
|
+
return;
|
|
409
|
+
const was = role;
|
|
410
|
+
role = e.role;
|
|
411
|
+
setRegistryRole?.(role);
|
|
412
|
+
safeAppend("server.role", { pid: process.pid, corpus, role, indexer: indexerPid, reason });
|
|
413
|
+
console.error(`[ContextEngine] 🧭 Role ${was} -> ${role} (${reason}; indexer pid ${indexerPid ?? process.pid})`);
|
|
414
|
+
if (role === "indexer") {
|
|
415
|
+
stopIndexPolling();
|
|
416
|
+
ensureModel().then(() => reindex()).then(() => startWatching()).catch((err) => {
|
|
417
|
+
console.error(`[ContextEngine] ⚠ taking over as indexer failed: ${err.message}`);
|
|
418
|
+
});
|
|
419
|
+
}
|
|
420
|
+
else {
|
|
421
|
+
stopWatching();
|
|
422
|
+
startIndexPolling();
|
|
423
|
+
}
|
|
424
|
+
}
|
|
425
|
+
function startRolePolling() {
|
|
426
|
+
if (rolePoll || !corpus)
|
|
427
|
+
return;
|
|
428
|
+
rolePoll = setInterval(() => evaluateRole("periodic"), ROLE_POLL_MS);
|
|
429
|
+
rolePoll.unref();
|
|
430
|
+
}
|
|
254
431
|
// ---------------------------------------------------------------------------
|
|
255
432
|
// MCP Server
|
|
256
433
|
// ---------------------------------------------------------------------------
|
|
@@ -289,6 +466,10 @@ server.tool("search_context", "Search across all indexed project knowledge (copi
|
|
|
289
466
|
.describe("Search mode: hybrid (default), keyword-only, or semantic-only"),
|
|
290
467
|
}, async ({ query, top_k, mode }) => {
|
|
291
468
|
let results = [];
|
|
469
|
+
// A reader keeps its 300 MB model unloaded until someone asks for semantics; the first such
|
|
470
|
+
// query is answered by keyword while the model loads. [LOCK] [EMBEDDINGS-ARE-CONTENT-ADDRESSED]
|
|
471
|
+
if (mode !== "keyword" && !isEmbeddingsReady())
|
|
472
|
+
void ensureModel();
|
|
292
473
|
if (mode === "keyword" || mode === "hybrid") {
|
|
293
474
|
const kwResults = searchChunks(chunks, query, top_k * 2);
|
|
294
475
|
if (mode === "keyword" || !isEmbeddingsReady()) {
|
|
@@ -424,6 +605,10 @@ server.tool("read_source", "Read the full content of a specific knowledge source
|
|
|
424
605
|
// Tool: reindex
|
|
425
606
|
// ---------------------------------------------------------------------------
|
|
426
607
|
server.tool("reindex", "Force a full re-index of all knowledge sources. Use after adding new files or changing contextengine.json.", {}, async () => {
|
|
608
|
+
if (role === "reader" && corpus) {
|
|
609
|
+
adoptSharedIndex();
|
|
610
|
+
return respond("reindex", `This server reads the shared index of corpus ${corpus}, written by pid ${indexerPid ?? "?"}: reloaded seq ${indexSeq}, ${chunks.length} chunks, ${embeddedChunks.length} vectors. Saving a doc makes the indexer rebuild; every reader picks it up within ${INDEX_POLL_MS / 1000} s.`);
|
|
611
|
+
}
|
|
427
612
|
await reindex();
|
|
428
613
|
return respond("reindex", `Re-indexed: ${chunks.length} chunks from ${sources.length} sources. Embeddings: ${embeddedChunks.length} vectors.`);
|
|
429
614
|
});
|
|
@@ -973,7 +1158,13 @@ server.tool("import_learnings", "Bulk-import learnings from a Markdown or JSON f
|
|
|
973
1158
|
.optional()
|
|
974
1159
|
.describe("Import every heading, bold bullet and table row as a rule (the pre-2.5.7 behaviour). Default false: only marked learnings."),
|
|
975
1160
|
}, async ({ file_path, default_category, project, permissive }) => {
|
|
976
|
-
|
|
1161
|
+
let result;
|
|
1162
|
+
try {
|
|
1163
|
+
result = importLearningsFromFile(file_path, default_category || "other", project, { permissive: permissive === true });
|
|
1164
|
+
}
|
|
1165
|
+
catch (e) {
|
|
1166
|
+
return respond("import_learnings", `⛔ Import refused: ${e?.message || e}`);
|
|
1167
|
+
}
|
|
977
1168
|
// Re-inject learnings into search index (project-scoped)
|
|
978
1169
|
const newChunks = learningsToChunks(activeProjectNames);
|
|
979
1170
|
const nonLearningChunks = chunks.filter((c) => c.source !== "💡 Learnings Store");
|
|
@@ -1078,70 +1269,45 @@ function registerResources() {
|
|
|
1078
1269
|
// Start
|
|
1079
1270
|
// ---------------------------------------------------------------------------
|
|
1080
1271
|
async function main() {
|
|
1081
|
-
//
|
|
1082
|
-
|
|
1083
|
-
|
|
1084
|
-
|
|
1085
|
-
|
|
1086
|
-
|
|
1087
|
-
activeProjectNames = projectDirs.map((d) => d.name);
|
|
1088
|
-
firewall.setProjectDirs(projectDirs);
|
|
1089
|
-
if (config.collectOps !== false) {
|
|
1090
|
-
let opsChunks = 0;
|
|
1091
|
-
for (const dir of projectDirs) {
|
|
1092
|
-
const ops = collectProjectOps(dir.path, dir.name);
|
|
1093
|
-
chunks.push(...ops);
|
|
1094
|
-
opsChunks += ops.length;
|
|
1095
|
-
}
|
|
1096
|
-
if (opsChunks > 0) {
|
|
1097
|
-
console.error(`[ContextEngine] ⚙ Collected ${opsChunks} operational chunks from ${projectDirs.length} projects`);
|
|
1272
|
+
// 0. Inventory this server FIRST, before indexing takes minutes: a server exists the moment it
|
|
1273
|
+
// starts. [LOCK] [SERVERS-ARE-INVENTORIED]. With the shared index on, the registry is also
|
|
1274
|
+
// the electorate: the record carries the corpus and the role. [LOCK] [ONE-INDEXER-MANY-READERS]
|
|
1275
|
+
if (sharedIndexEnabled()) {
|
|
1276
|
+
try {
|
|
1277
|
+
corpus = corpusId();
|
|
1098
1278
|
}
|
|
1099
|
-
|
|
1100
|
-
|
|
1101
|
-
const sysOps = collectSystemOps();
|
|
1102
|
-
if (sysOps.length > 0) {
|
|
1103
|
-
chunks.push(...sysOps);
|
|
1104
|
-
console.error(`[ContextEngine] 🖥 Collected ${sysOps.length} system operational chunks`);
|
|
1279
|
+
catch (err) {
|
|
1280
|
+
console.error("[ContextEngine] ⚠ corpus id failed, shared index off for this server:", err);
|
|
1105
1281
|
}
|
|
1106
1282
|
}
|
|
1107
|
-
|
|
1108
|
-
|
|
1109
|
-
|
|
1110
|
-
|
|
1111
|
-
|
|
1112
|
-
|
|
1113
|
-
|
|
1114
|
-
|
|
1115
|
-
|
|
1116
|
-
|
|
1117
|
-
}
|
|
1118
|
-
}
|
|
1119
|
-
}
|
|
1120
|
-
if (codeChunks > 0) {
|
|
1121
|
-
console.error(`[ContextEngine] 💻 Parsed ${codeChunks} code chunks from source files`);
|
|
1283
|
+
try {
|
|
1284
|
+
const reg = registerServer({ version: PKG_VERSION, script: fileURLToPath(import.meta.url), corpus, role: corpus ? "reader" : undefined });
|
|
1285
|
+
setRegistryRole = reg.setRole;
|
|
1286
|
+
const fleet = listServers();
|
|
1287
|
+
if (corpus) {
|
|
1288
|
+
const e = electIndexer(corpus, fleet.servers, process.pid);
|
|
1289
|
+
role = e.role;
|
|
1290
|
+
indexerPid = e.indexer;
|
|
1291
|
+
reg.setRole(role);
|
|
1292
|
+
safeAppend("server.role", { pid: process.pid, corpus, role, indexer: indexerPid, reason: "start" });
|
|
1122
1293
|
}
|
|
1294
|
+
console.error(`[ContextEngine] 🧭 ${formatServers(fleet)}`);
|
|
1295
|
+
if (corpus)
|
|
1296
|
+
console.error(`[ContextEngine] 🧭 This server: ${role} of corpus ${corpus}${role === "reader" ? ` (indexer pid ${indexerPid})` : ""}`);
|
|
1297
|
+
safeAppend("server.start", { pid: reg.record.pid, parent: reg.record.parent, version: reg.record.version, build: reg.record.build, cwd: reg.record.cwd, servers_running: fleet.servers.length, stale_builds: fleet.servers.filter((x) => x.staleBuild).length });
|
|
1123
1298
|
}
|
|
1124
|
-
|
|
1125
|
-
|
|
1126
|
-
if (autoImport.imported > 0) {
|
|
1127
|
-
console.error(`[ContextEngine] 📥 Auto-imported ${autoImport.imported} new learnings from ${autoImport.total} doc sources (${autoImport.updated} updated)`);
|
|
1299
|
+
catch (err) {
|
|
1300
|
+
console.error("[ContextEngine] ⚠ Server registry failed:", err);
|
|
1128
1301
|
}
|
|
1129
|
-
//
|
|
1130
|
-
|
|
1131
|
-
|
|
1132
|
-
|
|
1133
|
-
|
|
1134
|
-
|
|
1135
|
-
|
|
1136
|
-
|
|
1137
|
-
|
|
1138
|
-
if (adapterCount > 0) {
|
|
1139
|
-
const adapterChunks = await collectFromAdapters(config.adapters);
|
|
1140
|
-
if (adapterChunks.length > 0) {
|
|
1141
|
-
chunks.push(...adapterChunks);
|
|
1142
|
-
console.error(`[ContextEngine] 🔌 Adapters contributed ${adapterChunks.length} chunks from ${adapterCount} adapters`);
|
|
1143
|
-
}
|
|
1144
|
-
}
|
|
1302
|
+
// 1. The index: a reader takes the indexer's; everyone else, or a reader with nothing to
|
|
1303
|
+
// take yet, builds it (fast, keyword search available immediately).
|
|
1304
|
+
let adopted = false;
|
|
1305
|
+
if (role === "reader")
|
|
1306
|
+
adopted = adoptSharedIndex();
|
|
1307
|
+
if (!adopted) {
|
|
1308
|
+
if (role === "reader")
|
|
1309
|
+
console.error("[ContextEngine] 📥 No shared index yet; building locally once, without importing learnings");
|
|
1310
|
+
await buildIndex({ importLearnings: role === "indexer", loadAdapters: true });
|
|
1145
1311
|
}
|
|
1146
1312
|
// 2. Register MCP resources
|
|
1147
1313
|
registerResources();
|
|
@@ -1208,24 +1374,28 @@ async function main() {
|
|
|
1208
1374
|
catch (err) {
|
|
1209
1375
|
console.error("[ContextEngine] ⚠ Failed to write server-meta.json:", err);
|
|
1210
1376
|
}
|
|
1211
|
-
// 4.
|
|
1212
|
-
|
|
1213
|
-
|
|
1214
|
-
|
|
1215
|
-
|
|
1216
|
-
|
|
1217
|
-
|
|
1218
|
-
|
|
1219
|
-
|
|
1220
|
-
|
|
1221
|
-
|
|
1222
|
-
|
|
1223
|
-
|
|
1224
|
-
|
|
1377
|
+
// 4. Vectors. The store holds one vector per text ever embedded on this machine; the model
|
|
1378
|
+
// always loads for whoever embeds or answers queries. [LOCK] [EMBEDDINGS-ARE-CONTENT-ADDRESSED]
|
|
1379
|
+
// A reader that adopted the index already has its vectors and loads the model on its first
|
|
1380
|
+
// semantic query, not before: 300 MB per process is worth waiting for.
|
|
1381
|
+
if (!adopted) {
|
|
1382
|
+
vectorStore = loadEmbeddingStore().vectors;
|
|
1383
|
+
if (vectorStore.size > 0)
|
|
1384
|
+
console.error(`[ContextEngine] 💾 Embedding store: ${vectorStore.size} vectors`);
|
|
1385
|
+
publishIndex(); // keyword-searchable index for readers now; vectors follow
|
|
1386
|
+
ensureModel().then(async (ready) => {
|
|
1387
|
+
if (!ready)
|
|
1388
|
+
return;
|
|
1389
|
+
await embedAll();
|
|
1390
|
+
publishIndex();
|
|
1225
1391
|
});
|
|
1226
1392
|
}
|
|
1227
|
-
// 5.
|
|
1228
|
-
|
|
1393
|
+
// 5. Watch (indexer) or poll (reader), and keep the election running
|
|
1394
|
+
if (role === "indexer")
|
|
1395
|
+
startWatching();
|
|
1396
|
+
else
|
|
1397
|
+
startIndexPolling();
|
|
1398
|
+
startRolePolling();
|
|
1229
1399
|
// 6. Boot the local HTTP event-ingest endpoint for the browser extension.
|
|
1230
1400
|
// Local 127.0.0.1:7842 only; auth via shared secret at
|
|
1231
1401
|
// ~/.contextengine/extension-secret (see init-extension-secret CLI).
|
package/dist/learnings.d.ts
CHANGED
|
@@ -26,6 +26,7 @@ export declare function withStoreLock<T>(fn: () => T): T;
|
|
|
26
26
|
* loadStore() returns the same in-memory store and every saveStore() only marks it dirty.
|
|
27
27
|
*/
|
|
28
28
|
export declare function withStoreBatch<T>(fn: () => T): T;
|
|
29
|
+
export declare const MAX_GROWTH_PER_WRITE = 200;
|
|
29
30
|
/** Test seam for the writer's tripwire; not part of the API. */
|
|
30
31
|
export declare function __writeStoreForTests(store: LearningsStore): void;
|
|
31
32
|
/**
|
|
@@ -120,6 +121,7 @@ export declare function autoImportFromSources(sources: Array<{
|
|
|
120
121
|
imported: number;
|
|
121
122
|
updated: number;
|
|
122
123
|
ignored: number;
|
|
124
|
+
refused?: string;
|
|
123
125
|
};
|
|
124
126
|
/**
|
|
125
127
|
* Get the store stats.
|
package/dist/learnings.js
CHANGED
|
@@ -266,6 +266,19 @@ function dailyBackup() {
|
|
|
266
266
|
}
|
|
267
267
|
catch { /* a missing backup must never block a save */ }
|
|
268
268
|
}
|
|
269
|
+
// [LOCKED] [STORE-GROWTH-IS-A-TRIPWIRE-TOO] 2026-09-05
|
|
270
|
+
// [NEVER] let one write add more than MAX_GROWTH_PER_WRITE records to the store without the
|
|
271
|
+
// explicit override, and never raise the limit to make an import "just work".
|
|
272
|
+
// WHY: the shrink guard below caught the wipe of 2026-09-05; the same evening two stale servers
|
|
273
|
+
// wrote 1,766 records in one minute and nothing objected, because only shrinking was
|
|
274
|
+
// guarded. Every legitimate write is small: an agent saves one rule, the strict auto-import
|
|
275
|
+
// of a whole workspace produced 99 (replayed 2026-09-05), the largest real learnings file
|
|
276
|
+
// a few dozen. Thousands in one write is a bug or old code, never a lesson.
|
|
277
|
+
// FIX: a write that grows the store by more than MAX_GROWTH_PER_WRITE over the file on disk
|
|
278
|
+
// is refused with an audit event, unless CONTEXTENGINE_ALLOW_BULK=1 (set knowingly, for
|
|
279
|
+
// one deliberate bulk import). The auto-import catches the refusal and reports it; the
|
|
280
|
+
// server keeps running.
|
|
281
|
+
export const MAX_GROWTH_PER_WRITE = 200;
|
|
269
282
|
function writeStoreToDisk(store) {
|
|
270
283
|
ensureDir();
|
|
271
284
|
store.count = store.learnings.length;
|
|
@@ -284,6 +297,21 @@ function writeStoreToDisk(store) {
|
|
|
284
297
|
throw new Error(`refusing to write ${store.learnings.length} learnings over a store of ${onDisk}: that is the shape of a wipe, not an edit. Set CONTEXTENGINE_ALLOW_SHRINK=1 if this is deliberate.`);
|
|
285
298
|
}
|
|
286
299
|
}
|
|
300
|
+
// Growth tripwire. [LOCK] [STORE-GROWTH-IS-A-TRIPWIRE-TOO]
|
|
301
|
+
if (existsSync(LEARNINGS_PATH) && process.env.CONTEXTENGINE_ALLOW_BULK !== "1") {
|
|
302
|
+
let onDisk = -1;
|
|
303
|
+
try {
|
|
304
|
+
onDisk = (JSON.parse(readFileSync(LEARNINGS_PATH, "utf-8")).learnings || []).length;
|
|
305
|
+
}
|
|
306
|
+
catch {
|
|
307
|
+
onDisk = -1;
|
|
308
|
+
}
|
|
309
|
+
const growth = store.learnings.length - onDisk;
|
|
310
|
+
if (onDisk >= 0 && growth > MAX_GROWTH_PER_WRITE) {
|
|
311
|
+
safeAppend("learning.store_growth_refused", { on_disk: onDisk, attempted: store.learnings.length, growth });
|
|
312
|
+
throw new Error(`refusing to add ${growth} learnings in one write (store ${onDisk}, limit ${MAX_GROWTH_PER_WRITE}): that is the shape of a runaway import, not a lesson. Set CONTEXTENGINE_ALLOW_BULK=1 for one deliberate bulk import.`);
|
|
313
|
+
}
|
|
314
|
+
}
|
|
287
315
|
dailyBackup();
|
|
288
316
|
const tmp = `${LEARNINGS_PATH}.tmp-${process.pid}-${Date.now()}`;
|
|
289
317
|
writeFileSync(tmp, JSON.stringify(store, null, 2));
|
|
@@ -975,27 +1003,38 @@ export function autoImportFromSources(sources) {
|
|
|
975
1003
|
let totalUpdated = 0;
|
|
976
1004
|
let totalIgnored = 0;
|
|
977
1005
|
let processed = 0;
|
|
1006
|
+
let refused;
|
|
978
1007
|
// One load and one save for the whole sweep (~880 files), instead of one full-file
|
|
979
1008
|
// rewrite per rule per file. [LOCK] [STORE-NEVER-STARTS-FRESH-OVER-DATA]
|
|
980
|
-
|
|
981
|
-
|
|
982
|
-
|
|
983
|
-
|
|
984
|
-
|
|
985
|
-
|
|
986
|
-
|
|
987
|
-
|
|
988
|
-
|
|
989
|
-
|
|
990
|
-
|
|
991
|
-
|
|
992
|
-
|
|
993
|
-
|
|
994
|
-
|
|
995
|
-
|
|
996
|
-
|
|
997
|
-
|
|
998
|
-
|
|
1009
|
+
try {
|
|
1010
|
+
withStoreBatch(() => {
|
|
1011
|
+
for (const source of sources) {
|
|
1012
|
+
// Only process markdown files
|
|
1013
|
+
if (!source.path.endsWith(".md"))
|
|
1014
|
+
continue;
|
|
1015
|
+
if (!existsSync(source.path))
|
|
1016
|
+
continue;
|
|
1017
|
+
// Extract project name from source name (e.g., "ContextEngine — copilot-instructions.md")
|
|
1018
|
+
const project = source.name.split(" — ")[0]?.trim() || undefined;
|
|
1019
|
+
// Strict by construction: only marked learnings. [LOCK] [AUTO-IMPORT-ONLY-MARKED-LEARNINGS]
|
|
1020
|
+
const result = importLearningsFromFile(source.path, "other", project);
|
|
1021
|
+
totalImported += result.imported;
|
|
1022
|
+
totalUpdated += result.updated;
|
|
1023
|
+
totalIgnored += result.ignored;
|
|
1024
|
+
if (result.imported > 0 || result.updated > 0)
|
|
1025
|
+
processed++;
|
|
1026
|
+
}
|
|
1027
|
+
});
|
|
1028
|
+
}
|
|
1029
|
+
catch (e) {
|
|
1030
|
+
// A refused write (growth or shrink tripwire, lock timeout) must not take the server down;
|
|
1031
|
+
// it is reported to the caller and the store is left as it was. [LOCK] [STORE-GROWTH-IS-A-TRIPWIRE-TOO]
|
|
1032
|
+
refused = String(e?.message || e);
|
|
1033
|
+
totalImported = 0;
|
|
1034
|
+
totalUpdated = 0;
|
|
1035
|
+
processed = 0;
|
|
1036
|
+
}
|
|
1037
|
+
return { total: processed, imported: totalImported, updated: totalUpdated, ignored: totalIgnored, refused };
|
|
999
1038
|
}
|
|
1000
1039
|
/**
|
|
1001
1040
|
* Get the store stats.
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
export interface ServerRecord {
|
|
2
|
+
pid: number;
|
|
3
|
+
ppid: number;
|
|
4
|
+
parent: string;
|
|
5
|
+
started: string;
|
|
6
|
+
heartbeat: string;
|
|
7
|
+
version: string;
|
|
8
|
+
script: string;
|
|
9
|
+
build: string;
|
|
10
|
+
cwd: string;
|
|
11
|
+
node: string;
|
|
12
|
+
/** Since 2.6.0: what this server indexes (see shared-index.ts corpusId) and whether it is
|
|
13
|
+
* the one writing the shared index for it, or a reader of it. Absent on older builds. */
|
|
14
|
+
corpus?: string;
|
|
15
|
+
role?: "indexer" | "reader";
|
|
16
|
+
}
|
|
17
|
+
export interface ServerReport {
|
|
18
|
+
servers: Array<ServerRecord & {
|
|
19
|
+
alive: true;
|
|
20
|
+
currentBuild: string | null;
|
|
21
|
+
staleBuild: boolean;
|
|
22
|
+
}>;
|
|
23
|
+
removed: number;
|
|
24
|
+
warnings: string[];
|
|
25
|
+
}
|
|
26
|
+
/** More concurrent servers than this and every doc change costs that many re-embeds. */
|
|
27
|
+
export declare const SERVER_COUNT_WARN = 3;
|
|
28
|
+
/** Short content hash of the script a server loaded; the build identity. */
|
|
29
|
+
export declare function buildHashOf(scriptPath: string): string | null;
|
|
30
|
+
export declare function isAlive(pid: number): boolean;
|
|
31
|
+
/**
|
|
32
|
+
* Register the running server. Returns a stop() that removes the record; exit handlers call it too.
|
|
33
|
+
*/
|
|
34
|
+
export declare function registerServer(opts: {
|
|
35
|
+
version: string;
|
|
36
|
+
script: string;
|
|
37
|
+
corpus?: string;
|
|
38
|
+
role?: "indexer" | "reader";
|
|
39
|
+
}): {
|
|
40
|
+
record: ServerRecord;
|
|
41
|
+
stop: () => void;
|
|
42
|
+
setRole: (role: "indexer" | "reader") => void;
|
|
43
|
+
};
|
|
44
|
+
/** Read every record, drop the dead ones, compare builds with the files on disk now. */
|
|
45
|
+
export declare function listServers(): ServerReport;
|
|
46
|
+
export declare function formatServers(report: ServerReport, home?: string): string;
|
|
47
|
+
//# sourceMappingURL=server-registry.d.ts.map
|