rag-memory-epf-mcp 3.5.1 → 3.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -24,10 +24,52 @@ import { MigrationManager } from './src/migrations/migration-manager.js';
24
24
  // Import chunk text algorithm (extracted for publish-time invariant testing)
25
25
  import { chunkText as splitTextIntoChunks } from './src/chunkText.js';
26
26
  import { migrations } from './src/migrations/migrations.js';
27
+ // v3.6 lite install: model lifecycle + version-independent cache (A′ boundary)
28
+ import { EmbeddingGate, GateNotReadyError, GateDisabledError, TerminalConfigError } from './src/embeddingGate.js';
29
+ import { resolveModelCacheDir, preflightCacheDir, artifactKey, ModelDownloadLock, handleLoaderFailure } from './src/modelCache.js';
30
+ import { BackfillCoordinator } from './src/backfillCoordinator.js';
31
+ import os from 'node:os';
27
32
  import { createHash } from 'crypto';
28
33
  import { createRequire } from 'module';
29
34
  const require = createRequire(import.meta.url);
30
35
  const PKG_VERSION = require('../package.json').version;
36
+ // v3.6: runtime Node floor (engines is advisory only under default npm config).
37
+ // Limitation: static native imports above may fail before this runs on very old
38
+ // Node — documented in docs/UPDATING.md.
39
+ const NODE_MAJOR = Number(process.versions.node.split('.')[0]);
40
+ function assertNodeVersion() {
41
+ if (NODE_MAJOR < 24) {
42
+ console.error(`❌ rag-memory-epf-mcp v${PKG_VERSION} requires Node >= 24 (current: ${process.versions.node}).`);
43
+ console.error(' See docs/UPDATING.md for the supported runtime matrix.');
44
+ process.exitCode = 1;
45
+ throw new Error('unsupported Node version');
46
+ }
47
+ }
48
+ // v3.6: strip tokens / auth material / long URLs from operator-facing error text.
49
+ function sanitizeErrorMessage(msg) {
50
+ return msg
51
+ .replace(/(hf_|api[_-]?key=|authorization:\s*)\S+/gi, '$1[redacted]')
52
+ .replace(/https?:\/\/\S{60,}/g, '[url]')
53
+ .slice(0, 500);
54
+ }
55
+ // v3.6: startup self-report banner (version reliability — spec §8).
56
+ function printBanner(opts) {
57
+ console.error(`🚀 rag-memory-epf-mcp v${PKG_VERSION} | node v${process.versions.node} | model ${opts.model}@${opts.revision} (${opts.dtype}) | cache ${opts.cachePath} | db ${opts.dbPath}`);
58
+ }
59
+ // v3.6 (spec §5): ONE FTS5 literal-query compiler shared by chunk and entity
60
+ // search — raw user input can never produce MATCH syntax errors or trigger
61
+ // operators (every term is double-quoted; special characters stripped exactly
62
+ // as the pre-3.6 hybridSearch sanitizer did). Returns null when nothing
63
+ // searchable remains (contract: caller returns empty results + warning).
64
+ export function compileFtsLiteralQuery(q) {
65
+ const sanitized = q.replace(/["\*\(\)\-]/g, ' ').trim();
66
+ if (!sanitized)
67
+ return null;
68
+ const terms = sanitized.split(/\s+/).filter(t => t.length > 0);
69
+ if (terms.length === 0)
70
+ return null;
71
+ return terms.map(t => `"${t}"`).join(' OR ');
72
+ }
31
73
  // Configure Hugging Face transformers for better compatibility
32
74
  if (env.backends?.onnx?.wasm) {
33
75
  env.backends.onnx.wasm.wasmPaths = './node_modules/@huggingface/transformers/dist/';
@@ -40,6 +82,21 @@ const DB_FILE_PATH = process.env.DB_FILE_PATH
40
82
  : path.join(path.dirname(fileURLToPath(import.meta.url)), process.env.DB_FILE_PATH)
41
83
  : defaultDbPath;
42
84
  const EMBEDDING_MODEL = process.env.EMBEDDING_MODEL || 'Xenova/bge-m3';
85
+ // v3.5 default model config — grandfathering legacy vectors is only automatic
86
+ // when the current config matches this (spec §6b custom-model guard). An
87
+ // EXPLICIT `EMBEDDING_MODEL=Xenova/bge-m3` counts as default: same weights,
88
+ // same pin, same grandfather policy (beta 1R consistency fix).
89
+ const IS_DEFAULT_MODEL_CONFIG = !process.env.EMBEDDING_MODEL || process.env.EMBEDDING_MODEL === 'Xenova/bge-m3';
90
+ // Default model pinned to an upstream commit (spec §6c): a shared version-
91
+ // independent cache must never silently swap weights under 'main'. Verified
92
+ // 2026-07-18 via `git ls-remote https://huggingface.co/Xenova/bge-m3` — the
93
+ // same revision the local v3.5 cache was downloaded from. Custom models stay
94
+ // on 'main' (their vectors are never auto-grandfathered anyway).
95
+ const MODEL_REVISION = IS_DEFAULT_MODEL_CONFIG ? '4de13258303883538bd53b696b452bf8099f0858' : 'main';
96
+ const MODEL_DTYPE = 'fp16';
97
+ // Entity embedding text builder version — mixed into entity input hashes so a
98
+ // builder change re-backfills entities without touching chunk vectors (spec §6c).
99
+ const TEXT_BUILDER_VERSION = 'tb1';
43
100
  // Safe rowid for vec0 virtual tables (require literal integer, not parameterized)
44
101
  // Trim incomplete UTF-8 multi-byte sequences at chunk boundaries.
45
102
  // Continuation bytes match 10xxxxxx (0x80-0xBF); lead bytes indicate how many
@@ -59,18 +116,23 @@ function safeRowid(value) {
59
116
  export class RAGKnowledgeGraphManager {
60
117
  db = null;
61
118
  encoding = null;
62
- embeddingModel = null;
63
- modelInitialized = false;
119
+ gate;
120
+ embeddingsMode = 'lazy';
121
+ currentProfileId = 0;
122
+ // Automatic grandfathering of legacy vectors is only allowed under the v3.5
123
+ // default model config, or with the explicit trust opt-in (spec §6b guard).
124
+ grandfatherAllowed = IS_DEFAULT_MODEL_CONFIG || process.env.RAG_MEMORY_TRUST_LEGACY_VECTORS === '1';
125
+ coordinator = null;
64
126
  embeddingCache = new Map();
65
127
  EMBEDDING_CACHE_MAX = 500;
66
128
  dictionaryCache = null;
129
+ // v3.6 (spec §3): initialize = DB + migrations + profile only. The embedding
130
+ // model is NEVER awaited here — main() connects the MCP server first and the
131
+ // gate loads in the background (lazy) or is awaited explicitly (eager).
67
132
  async initialize(opts = {}) {
68
133
  console.error('🚀 Initializing RAG Knowledge Graph MCP Server...');
69
- // Initialize database
70
134
  this.db = new Database(DB_FILE_PATH);
71
- // Load sqlite-vec extension
72
135
  sqliteVec.load(this.db);
73
- // SQLite performance & safety optimizations
74
136
  this.db.pragma('journal_mode = WAL');
75
137
  this.db.pragma('synchronous = NORMAL');
76
138
  this.db.pragma('busy_timeout = 5000');
@@ -78,37 +140,286 @@ export class RAGKnowledgeGraphManager {
78
140
  this.db.pragma('temp_store = MEMORY');
79
141
  this.db.pragma('mmap_size = 268435456');
80
142
  this.db.pragma('foreign_keys = ON');
81
- // Initialize tiktoken
82
143
  this.encoding = get_encoding("cl100k_base");
83
- // Initialize embedding model (skippable for tests / FTS-only environments)
84
- if (!opts.skipModel) {
85
- await this.initializeEmbeddingModel();
86
- }
87
- else {
88
- console.error('⏭️ Skipping embedding model load (skipModel=true)');
89
- }
90
- // Run database migrations
91
144
  await this.runMigrations();
92
- console.error('✅ RAG-enabled knowledge graph initialized');
93
- // Log system info
145
+ this.currentProfileId = this.ensureCurrentProfile();
146
+ this.embeddingsMode = opts.skipModel
147
+ ? 'off'
148
+ : (process.env.RAG_MEMORY_EMBEDDINGS || 'lazy');
149
+ if (!['lazy', 'eager', 'off'].includes(this.embeddingsMode))
150
+ this.embeddingsMode = 'lazy';
151
+ this.gate = opts.gate ?? new EmbeddingGate({
152
+ mode: this.embeddingsMode,
153
+ loadModel: () => this.buildRealLoader(),
154
+ onReady: () => this.coordinator?.kick(),
155
+ });
156
+ // Late-bound deps (closures): tests swap manager.gate / flip the guard.
157
+ this.coordinator = new BackfillCoordinator({
158
+ db: () => this.db,
159
+ gateIsReady: () => this.gate.isReady,
160
+ gateIsDisabled: () => this.gate.isDisabled,
161
+ mode: () => this.embeddingsMode,
162
+ grandfatherAllowed: () => this.grandfatherAllowed,
163
+ currentProfileId: () => this.currentProfileId,
164
+ buildEntityInputHash: (entityId) => this.entityInputHash(entityId),
165
+ hashEntityText: (text) => this.hashWithBuilderVersion(text),
166
+ chunkInputHash: (text) => createHash('sha256').update(text).digest('hex'),
167
+ reembedEntity: async (entityId) => this.embedEntity(entityId, 'backfill'),
168
+ reembedChunk: async (rowid) => this.reembedChunkByRowid(rowid),
169
+ });
170
+ console.error('✅ RAG-enabled knowledge graph initialized (embedding model deferred)');
94
171
  const systemInfo = getSystemInfo();
95
172
  console.error(`📊 System Info: ${systemInfo.toolCounts.total} tools available (${systemInfo.toolCounts.knowledgeGraph} knowledge graph, ${systemInfo.toolCounts.rag} RAG, ${systemInfo.toolCounts.graphQuery} query)`);
96
173
  }
97
- async initializeEmbeddingModel() {
174
+ // Upsert the stored-vector compatibility profile (spec §6c layer 2) and record
175
+ // retrieval config in server_meta (layer 3 — never a backfill trigger).
176
+ ensureCurrentProfile() {
177
+ if (!this.db)
178
+ throw new Error('Database not initialized');
179
+ const dims = 1024;
180
+ if (dims !== 1024)
181
+ throw new Error('unsupported embedding dims (vec0 tables are fixed at 1024)'); // fail-fast contract
182
+ this.db.prepare(`INSERT OR IGNORE INTO embedding_profiles
183
+ (model_id, revision, dtype, dims, pooling, normalize) VALUES (?,?,?,?,?,?)`)
184
+ .run(EMBEDDING_MODEL, MODEL_REVISION, MODEL_DTYPE, dims, 'cls', 1);
185
+ const row = this.db.prepare(`SELECT id FROM embedding_profiles
186
+ WHERE model_id=? AND revision=? AND dtype=? AND dims=? AND pooling=? AND normalize=?`)
187
+ .get(EMBEDDING_MODEL, MODEL_REVISION, MODEL_DTYPE, dims, 'cls', 1);
188
+ this.db.prepare(`INSERT INTO server_meta(key,value) VALUES('current_profile_id',?)
189
+ ON CONFLICT(key) DO UPDATE SET value=excluded.value`).run(String(row.id));
190
+ this.db.prepare(`INSERT INTO server_meta(key,value) VALUES('query_prefix_version','1')
191
+ ON CONFLICT(key) DO NOTHING`).run();
192
+ return row.id;
193
+ }
194
+ // Real model loader used by the gate: version-independent cache dir with a
195
+ // cross-process download lock. Preflight failure throws (gate -> failed);
196
+ // silently falling back to the package-internal cache is forbidden (spec §7).
197
+ async buildRealLoader() {
198
+ const cacheDir = resolveModelCacheDir(process.env, process.platform, os.homedir());
199
+ const pf = preflightCacheDir(cacheDir);
200
+ if (!pf.ok)
201
+ throw new Error(`model cache dir not writable (${cacheDir}): ${pf.error}`);
202
+ const key = artifactKey(EMBEDDING_MODEL, MODEL_REVISION, MODEL_DTYPE);
203
+ const lock = new ModelDownloadLock(cacheDir, key);
204
+ // Shutdown aborts the lock wait via the gate's AbortController (spec §3).
205
+ console.error('⏳ acquiring model download lock...'); // deterministic lock-wait marker (5R test residual)
206
+ const role = await lock.acquireOrWait({ timeoutMs: 10 * 60_000, signal: this.gate.abort.signal });
98
207
  try {
99
- console.error(`🤖 Loading embedding model: ${EMBEDDING_MODEL} (1024-dim, 100+ languages)...`);
100
- // Configure environment to allow remote model downloads
208
+ this.gate.markDownloading();
101
209
  env.allowRemoteModels = true;
102
210
  env.allowLocalModels = true;
103
- this.embeddingModel = await pipeline('feature-extraction', EMBEDDING_MODEL, { revision: 'main', dtype: 'fp16' });
104
- this.modelInitialized = true;
105
- console.error(`✅ ${EMBEDDING_MODEL} model loaded successfully (fp16)`);
211
+ console.error(`🤖 Loading embedding model: ${EMBEDDING_MODEL} (1024-dim, cache=${cacheDir})...`);
212
+ const model = await pipeline('feature-extraction', EMBEDDING_MODEL, { revision: MODEL_REVISION, dtype: MODEL_DTYPE, cache_dir: cacheDir });
213
+ // dims fail-fast (spec §2 / beta B8): probe the ACTUAL output length — a
214
+ // 384/768-dim custom model must fail here with a clear message, not at
215
+ // every subsequent vector write.
216
+ const probe = await model('dimension probe', { pooling: 'cls', normalize: true });
217
+ const actualDims = probe.data.length;
218
+ if (actualDims !== 1024) {
219
+ // Config incompatibility, NOT cache corruption (beta 2R B3): the
220
+ // download and load both succeeded — quarantining or retrying cannot
221
+ // change the model's dimensions.
222
+ if (role === 'owner')
223
+ lock.markComplete(); // cache itself is valid
224
+ throw new TerminalConfigError(`embedding model ${EMBEDDING_MODEL} outputs ${actualDims} dims — this engine's vec0 tables are fixed at 1024. Use a 1024-dim model.`);
225
+ }
226
+ if (role === 'owner')
227
+ lock.markComplete();
228
+ console.error(`✅ ${EMBEDDING_MODEL} model loaded (${MODEL_DTYPE})`);
229
+ return async (text, dims, isQuery) => {
230
+ const input = isQuery ? `Represent this sentence for searching relevant passages: ${text}` : text;
231
+ const r = await model(input, { pooling: 'cls', normalize: true });
232
+ return new Float32Array(r.data.slice(0, dims));
233
+ };
106
234
  }
107
- catch (error) {
108
- console.error('❌ Failed to load embedding model:', error instanceof Error ? error.message : error);
109
- console.error('⚠️ Semantic search will be unavailable. Other tools (CRUD, graph queries) still work.');
110
- console.error('💡 Fix: Ensure ONNX model cache is accessible. Try: rm -rf ~/.npm/_onnx_models && restart.');
111
- this.modelInitialized = false;
235
+ catch (e) {
236
+ // Cache policy by CAUSE and ROLE (beta 2R B3 -> 4R M1 -> 5R M1), unit-
237
+ // tested in modelCache: config errors touch nothing; integrity errors
238
+ // invalidate the marker, and only a lock-holding OWNER may quarantine
239
+ // (a ready-role process racing other readers never deletes shared
240
+ // files); OOM/network/unknown preserve everything.
241
+ const action = handleLoaderFailure({
242
+ role, error: e, lock, cacheDir, modelId: EMBEDDING_MODEL,
243
+ terminal: e instanceof TerminalConfigError,
244
+ });
245
+ if (action === 'quarantined')
246
+ console.error('🧹 cache-integrity failure (owner) — model cache quarantined');
247
+ else if (action === 'marker-invalidated')
248
+ console.error('… cache-integrity failure (reader) — marker dropped, next retry re-proves as locked owner');
249
+ else if (!(e instanceof TerminalConfigError))
250
+ console.error('… non-integrity load failure — model cache preserved');
251
+ throw e;
252
+ }
253
+ finally {
254
+ lock.release();
255
+ }
256
+ }
257
+ // Background provenance reconciliation (spec §6b). Runs in parallel with the
258
+ // model load; vector search and automatic backfill stay closed until it
259
+ // settles (eligibility barrier in the coordinator).
260
+ async startReconciliation() {
261
+ if (!this.coordinator)
262
+ return;
263
+ await this.coordinator.runReconciliation();
264
+ this.coordinator.sweepStart();
265
+ }
266
+ // sha256 with the entity text-builder version mixed in: a builder change
267
+ // re-backfills entities only, never chunks (spec §6c N2).
268
+ hashWithBuilderVersion(text) {
269
+ return createHash('sha256').update(`${TEXT_BUILDER_VERSION}\n${text}`).digest('hex');
270
+ }
271
+ // Rebuild the CURRENT embedding input hash for an entity. null = entity gone
272
+ // or malformed observations — reconciliation fail-closes to missing.
273
+ entityInputHash(entityId) {
274
+ try {
275
+ const entity = this.db.prepare(`SELECT name, entityType, observations FROM entities WHERE id = ?`)
276
+ .get(entityId);
277
+ if (!entity)
278
+ return null;
279
+ const built = this.buildEntityEmbeddingText({
280
+ name: entity.name,
281
+ entityType: entity.entityType,
282
+ observations: JSON.parse(entity.observations),
283
+ });
284
+ return this.hashWithBuilderVersion(built.text);
285
+ }
286
+ catch {
287
+ return null;
288
+ }
289
+ }
290
+ // Mutation-path embedding wrapper (spec §5): CRUD success never depends on
291
+ // model availability. On not-ready/disabled the stale vector is deleted in
292
+ // the same breath (dirty = missing, §6a-1) and the coordinator is kicked so
293
+ // the row is recovered without a restart (§5 kick column).
294
+ async tryEmbedEntity(entityId, priority = 'bulk') {
295
+ try {
296
+ const ok = await this.embedEntity(entityId, priority);
297
+ if (ok) {
298
+ // Success clears any stale backfill-failure record for this target.
299
+ this.db.prepare(`DELETE FROM embedding_backfill_failures WHERE kind = 'entity' AND target_id = ?`).run(entityId);
300
+ this.coordinator?.invalidateCoverage();
301
+ return 'embedded';
302
+ }
303
+ this.invalidateEntityVector(entityId);
304
+ this.coordinator?.kick();
305
+ return 'queued';
306
+ }
307
+ catch (e) {
308
+ // Any embedding-layer failure (not-ready, disabled, OR a ready-state
309
+ // inference error) must not fail the CRUD that already committed. The
310
+ // vector is invalidated (§6a-1) and recovery is owned by the backfill
311
+ // scanner with its attempts cap — never by rethrowing here (spec §5).
312
+ if (!(e instanceof GateNotReadyError) && !(e instanceof GateDisabledError)) {
313
+ console.error(`⚠️ embedding failed for ${entityId} (queued for backfill): ${e instanceof Error ? e.message : e}`);
314
+ }
315
+ this.invalidateEntityVector(entityId);
316
+ this.coordinator?.kick();
317
+ return e instanceof GateDisabledError ? 'disabled' : 'queued';
318
+ }
319
+ }
320
+ // §6a-1 (beta B2): the entity change and the stale-vector removal commit in
321
+ // ONE synchronous transaction, BEFORE any inference await. No window exists
322
+ // where another tool call can retrieve the pre-mutation vector, and a crash
323
+ // between mutation and re-embed leaves a clean missing state (backfill
324
+ // target), never a stale-searchable one.
325
+ mutateEntityAndInvalidate(entityId, mutate) {
326
+ const tx = this.db.transaction(() => {
327
+ mutate();
328
+ const meta = this.db.prepare(`SELECT rowid FROM entity_embedding_metadata WHERE entity_id = ?`)
329
+ .get(entityId);
330
+ if (meta) {
331
+ this.db.exec(`DELETE FROM entity_embeddings WHERE rowid = ${Number(meta.rowid)}`);
332
+ this.db.prepare(`DELETE FROM entity_embedding_metadata WHERE entity_id = ?`).run(entityId);
333
+ }
334
+ });
335
+ tx();
336
+ this.coordinator?.invalidateCoverage();
337
+ }
338
+ // §6a-1 invariant: when an entity's embedding input changed but re-embedding
339
+ // is unavailable, its old vector must not stay searchable.
340
+ invalidateEntityVector(entityId) {
341
+ if (!this.db)
342
+ return;
343
+ const meta = this.db.prepare(`SELECT rowid FROM entity_embedding_metadata WHERE entity_id = ?`)
344
+ .get(entityId);
345
+ if (!meta)
346
+ return;
347
+ const tx = this.db.transaction(() => {
348
+ this.db.exec(`DELETE FROM entity_embeddings WHERE rowid = ${Number(meta.rowid)}`);
349
+ this.db.prepare(`DELETE FROM entity_embedding_metadata WHERE entity_id = ?`).run(entityId);
350
+ });
351
+ tx();
352
+ this.coordinator?.invalidateCoverage();
353
+ }
354
+ // Backfill callback: re-embed one chunk and commit vector + provenance in a
355
+ // single transaction (§6a-2).
356
+ async reembedChunkByRowid(rowid) {
357
+ if (!this.db)
358
+ return false;
359
+ const row = this.db.prepare(`SELECT text FROM chunk_metadata WHERE rowid = ?`)
360
+ .get(rowid);
361
+ if (!row || row.text === null)
362
+ return false;
363
+ try {
364
+ const embedding = await this.generateEmbedding(row.text, 1024, false, 'backfill');
365
+ const hash = createHash('sha256').update(row.text).digest('hex');
366
+ const safe = Number(rowid);
367
+ // Write-back CAS (beta 2R B1): the rowid may have been deleted and reused
368
+ // by a re-sync while inference ran — re-read the CURRENT text in the
369
+ // transaction and only write when it still matches what was embedded.
370
+ const tx = this.db.transaction(() => {
371
+ const cur = this.db.prepare(`SELECT text FROM chunk_metadata WHERE rowid = ?`).get(rowid);
372
+ if (!cur || cur.text !== row.text)
373
+ return false; // superseded — discard
374
+ this.db.exec(`DELETE FROM chunks WHERE rowid = ${safe}`);
375
+ this.db.prepare(`INSERT INTO chunks (rowid, embedding) VALUES (${safe}, ?)`).run(Buffer.from(embedding.buffer));
376
+ this.db.prepare(`UPDATE chunk_metadata SET input_hash = ?, profile_id = ?, provenance_state = 'verified' WHERE rowid = ?`)
377
+ .run(hash, this.currentProfileId, rowid);
378
+ this.db.prepare(`DELETE FROM embedding_backfill_failures WHERE kind = 'chunk' AND target_id = ?`).run(String(rowid));
379
+ return true;
380
+ });
381
+ const written = tx();
382
+ this.coordinator?.invalidateCoverage();
383
+ return written;
384
+ }
385
+ catch (e) {
386
+ if (e instanceof GateNotReadyError || e instanceof GateDisabledError)
387
+ return false;
388
+ throw e;
389
+ }
390
+ }
391
+ // spec §3 shutdown order (beta B1): block new batches -> settle coordinator
392
+ // (INCLUDING an in-flight reconciliation pass) -> settle gate (INCLUDING an
393
+ // in-flight model load, bounded) -> close DB -> natural exit.
394
+ //
395
+ // Bounded-exit rationale, re-derived after beta 1R: the exit decision is made
396
+ // AFTER the settle wait, not before — if the load completed during settling
397
+ // (ONNX session now exists) we take the natural-exit path. Only when the load
398
+ // is STILL pending after the deadline (dominant case: the 1.2GB download,
399
+ // which is un-abortable through transformers.js and would hold the event
400
+ // loop indefinitely) do we exit(). At that point the DB is already closed
401
+ // cleanly, so even the residual worst case — the load being inside ONNX
402
+ // session construction at exit — risks an ugly abort message, never data
403
+ // loss. Hanging forever is the alternative and is worse.
404
+ async shutdownAll() {
405
+ console.error('\n🧹 Cleaning up...');
406
+ try {
407
+ await this.coordinator?.shutdown(5000);
408
+ }
409
+ catch { /* settle best-effort */ }
410
+ try {
411
+ await this.gate?.shutdown(5000);
412
+ }
413
+ catch { /* settle best-effort */ }
414
+ const loadStillPending = this.gate?.loadInFlight ?? false;
415
+ try {
416
+ this.cleanup();
417
+ }
418
+ catch { /* DB close */ }
419
+ process.exitCode = process.exitCode ?? 0;
420
+ if (loadStillPending) {
421
+ console.error('… model load/download still in flight after settle deadline — bounded exit (DB already closed)');
422
+ process.exit(process.exitCode);
112
423
  }
113
424
  }
114
425
  async runMigrations() {
@@ -140,11 +451,6 @@ export class RAGKnowledgeGraphManager {
140
451
  this.encoding.free();
141
452
  this.encoding = null;
142
453
  }
143
- if (this.embeddingModel) {
144
- // Clean up the embedding model if it has cleanup methods
145
- this.embeddingModel = null;
146
- this.modelInitialized = false;
147
- }
148
454
  this.embeddingCache.clear();
149
455
  if (this.db) {
150
456
  this.db.close();
@@ -173,10 +479,11 @@ export class RAGKnowledgeGraphManager {
173
479
  // Try insert first
174
480
  const insertResult = insertStmt.run(entityId, entity.name, entity.entityType, JSON.stringify(timestamped), '{}');
175
481
  if (insertResult.changes > 0) {
176
- // New entity created
177
- result.push({ ...entity, observations: timestamped });
482
+ // New entity created. CRUD success is independent of model readiness
483
+ // (spec §5): not-ready -> row stays vectorless (queued for backfill).
178
484
  console.error(`🔮 Generating embedding for new entity: ${entity.name}`);
179
- await this.embedEntity(entityId);
485
+ const embedding_status = await this.tryEmbedEntity(entityId, 'bulk');
486
+ result.push({ ...entity, observations: timestamped, embedding_status });
180
487
  }
181
488
  else {
182
489
  // Entity already exists — upsert: merge observations and update entityType
@@ -192,11 +499,13 @@ export class RAGKnowledgeGraphManager {
192
499
  if (newObs.length > 0 || needsTypeUpdate) {
193
500
  const mergedObs = [...currentObs, ...newObs];
194
501
  const updatedType = needsTypeUpdate ? entity.entityType : existing.entityType;
195
- this.db.prepare(`UPDATE entities SET observations = ?, entityType = ? WHERE id = ?`)
196
- .run(JSON.stringify(mergedObs), updatedType, entityId);
502
+ this.mutateEntityAndInvalidate(entityId, () => {
503
+ this.db.prepare(`UPDATE entities SET observations = ?, entityType = ? WHERE id = ?`)
504
+ .run(JSON.stringify(mergedObs), updatedType, entityId);
505
+ });
197
506
  console.error(`♻️ Upserted entity: ${entity.name} (+${newObs.length} obs${needsTypeUpdate ? ', type→' + updatedType : ''})`);
198
- await this.embedEntity(entityId);
199
- result.push({ ...entity, observations: mergedObs });
507
+ const embedding_status = await this.tryEmbedEntity(entityId, 'bulk');
508
+ result.push({ ...entity, observations: mergedObs, embedding_status });
200
509
  }
201
510
  }
202
511
  }
@@ -208,24 +517,32 @@ export class RAGKnowledgeGraphManager {
208
517
  throw new Error('Database not initialized');
209
518
  const newRelations = [];
210
519
  for (const relation of relations) {
211
- // Ensure entities exist
212
- await this.createEntities([
520
+ // Ensure entities exist. v3.6 (spec §5c): auto-created endpoints may be
521
+ // embedded/queued/disabled independently — report per endpoint; 'n/a'
522
+ // means the endpoint already existed (no embedding work happened here).
523
+ const ensured = await this.createEntities([
213
524
  { name: relation.from, entityType: 'CONCEPT', observations: [] },
214
525
  { name: relation.to, entityType: 'CONCEPT', observations: [] }
215
526
  ]);
527
+ const statusOf = (name) => {
528
+ const hit = ensured.find(e => e.name === name);
529
+ return hit?.embedding_status ?? 'n/a';
530
+ };
531
+ const endpoint_embedding_status = { from: statusOf(relation.from), to: statusOf(relation.to) };
216
532
  const sourceId = `entity_${relation.from.toLowerCase().replace(/[^\p{L}\p{N}]/gu, '_')}`;
217
533
  const targetId = `entity_${relation.to.toLowerCase().replace(/[^\p{L}\p{N}]/gu, '_')}`;
218
534
  const relationId = `rel_${sourceId}_${relation.relationType}_${targetId}`.toLowerCase();
219
535
  const stmt = this.db.prepare(`
220
- INSERT OR IGNORE INTO relationships
536
+ INSERT OR IGNORE INTO relationships
221
537
  (id, source_entity, target_entity, relationType, confidence, metadata)
222
538
  VALUES (?, ?, ?, ?, ?, ?)
223
539
  `);
224
540
  const result = stmt.run(relationId, sourceId, targetId, relation.relationType, 1.0, '{}');
225
541
  if (result.changes > 0) {
226
- newRelations.push(relation);
542
+ newRelations.push({ ...relation, endpoint_embedding_status });
227
543
  }
228
544
  }
545
+ this.coordinator?.kick();
229
546
  return newRelations;
230
547
  }
231
548
  async addObservations(observations) {
@@ -248,12 +565,16 @@ export class RAGKnowledgeGraphManager {
248
565
  const newObservations = timestamped.filter(c => !currentBare.has(stripDate(c)));
249
566
  if (newObservations.length > 0) {
250
567
  const updatedObservations = [...currentObservations, ...newObservations];
251
- this.db.prepare(`
252
- UPDATE entities SET observations = ? WHERE id = ?
253
- `).run(JSON.stringify(updatedObservations), entityId);
254
- // Regenerate embedding for the updated entity
568
+ this.mutateEntityAndInvalidate(entityId, () => {
569
+ this.db.prepare(`
570
+ UPDATE entities SET observations = ? WHERE id = ?
571
+ `).run(JSON.stringify(updatedObservations), entityId);
572
+ });
573
+ // Regenerate embedding for the updated entity (queued when not ready)
255
574
  console.error(`🔮 Regenerating embedding for updated entity: ${obs.entityName}`);
256
- await this.embedEntity(entityId);
575
+ const embedding_status = await this.tryEmbedEntity(entityId, 'bulk');
576
+ results.push({ entityName: obs.entityName, addedObservations: newObservations, embedding_status });
577
+ continue;
257
578
  }
258
579
  results.push({ entityName: obs.entityName, addedObservations: newObservations });
259
580
  }
@@ -322,22 +643,40 @@ export class RAGKnowledgeGraphManager {
322
643
  }
323
644
  console.error(`✅ Entity deletion process completed`);
324
645
  }
646
+ // v3.6 (spec §5c, breaking): structured per-entity results + re-embedding.
647
+ // Pre-3.6 this method silently left STALE entity vectors behind (the input
648
+ // text changed but the vector was never regenerated) — fixed via
649
+ // tryEmbedEntity, which also covers the not-ready dirty contract.
325
650
  async deleteObservations(deletions) {
326
651
  if (!this.db)
327
652
  throw new Error('Database not initialized');
653
+ const results = [];
654
+ let total = 0;
328
655
  for (const deletion of deletions) {
329
656
  const entityId = `entity_${deletion.entityName.toLowerCase().replace(/[^\p{L}\p{N}]/gu, '_')}`;
330
657
  const entity = this.db.prepare(`
331
658
  SELECT observations FROM entities WHERE id = ?
332
659
  `).get(entityId);
333
- if (entity) {
334
- const currentObservations = JSON.parse(entity.observations);
335
- const filteredObservations = currentObservations.filter((obs) => !deletion.observations.includes(obs));
336
- this.db.prepare(`
337
- UPDATE entities SET observations = ? WHERE id = ?
338
- `).run(JSON.stringify(filteredObservations), entityId);
660
+ if (!entity) {
661
+ results.push({ entityName: deletion.entityName, deleted: 0, embedding_status: 'n/a' });
662
+ continue;
663
+ }
664
+ const currentObservations = JSON.parse(entity.observations);
665
+ const filteredObservations = currentObservations.filter((obs) => !deletion.observations.includes(obs));
666
+ const deleted = currentObservations.length - filteredObservations.length;
667
+ if (deleted === 0) {
668
+ results.push({ entityName: deletion.entityName, deleted: 0, embedding_status: 'n/a' });
669
+ continue;
339
670
  }
671
+ this.mutateEntityAndInvalidate(entityId, () => {
672
+ this.db.prepare(`UPDATE entities SET observations = ? WHERE id = ?`)
673
+ .run(JSON.stringify(filteredObservations), entityId);
674
+ });
675
+ const embedding_status = await this.tryEmbedEntity(entityId, 'bulk');
676
+ total += deleted;
677
+ results.push({ entityName: deletion.entityName, deleted, embedding_status });
340
678
  }
679
+ return { results, total_deleted: total };
341
680
  }
342
681
  async deleteRelations(relations) {
343
682
  if (!this.db)
@@ -524,10 +863,24 @@ export class RAGKnowledgeGraphManager {
524
863
  console.error(`✅ getNeighbors: Found ${entities.length} entities, ${relations.length} relations, ${paths.length} paths (depth=${effectiveDepth})`);
525
864
  return { entities, relations, paths };
526
865
  }
866
+ // v3.6 (spec §5·§5c, additive): FTS lexical fallback when vector search is
867
+ // not eligible, hybrid-partial merge while backfill is catching up, and
868
+ // top-level state fields on every response.
527
869
  async searchNodes(query, limit = 10, since, until) {
528
870
  if (!this.db)
529
871
  throw new Error('Database not initialized');
530
872
  console.error(`🔍 Semantic entity search: "${query}"`);
873
+ const covS = this.coordinator?.coverage();
874
+ const entityPct = covS && covS.entity.total > 0 ? Math.round((covS.entity.embedded / covS.entity.total) * 100) : 100;
875
+ const stateFields = () => ({
876
+ model_state: this.gate.status.state,
877
+ coverage: { entity_pct: entityPct },
878
+ });
879
+ if (!(this.coordinator?.eligible ?? false)) {
880
+ // No waiting on the model (spec §5) — lexical entities_fts fallback.
881
+ return { ...this.searchNodesFts(query, limit, since, until), search_mode: 'fts-only',
882
+ ...stateFields(), degradation_reason: this.degradationReason() };
883
+ }
531
884
  const queryVariants = this.buildCrossLingualVariants(query);
532
885
  if (queryVariants.length > 1) {
533
886
  console.error(`🌐 searchNodes variants: ${queryVariants.slice(1).join(' | ')}`);
@@ -551,16 +904,26 @@ export class RAGKnowledgeGraphManager {
551
904
  `).all(Buffer.from(embedding.buffer), k);
552
905
  };
553
906
  const resultMap = new Map();
554
- for (const variant of queryVariants) {
555
- const embedding = await this.generateEmbedding(variant, 1024, true);
556
- const variantResults = searchEntities(embedding, limit * 2);
557
- for (const result of variantResults) {
558
- const existing = resultMap.get(result.entity_id);
559
- if (!existing || result.distance < existing.distance) {
560
- resultMap.set(result.entity_id, result);
907
+ try {
908
+ for (const variant of queryVariants) {
909
+ const embedding = await this.generateEmbedding(variant, 1024, true);
910
+ const variantResults = searchEntities(embedding, limit * 2);
911
+ for (const result of variantResults) {
912
+ const existing = resultMap.get(result.entity_id);
913
+ if (!existing || result.distance < existing.distance) {
914
+ resultMap.set(result.entity_id, result);
915
+ }
561
916
  }
562
917
  }
563
918
  }
919
+ catch (embErr) {
920
+ // Ready-state inference failure degrades to FTS instead of failing the
921
+ // tool (beta B6) — same contract as hybridSearch. The gate's own
922
+ // consecutive-failure counter handles the systemic transition.
923
+ console.error(`⚠️ searchNodes vector path failed — FTS fallback:`, embErr instanceof Error ? embErr.message : embErr);
924
+ return { ...this.searchNodesFts(query, limit, since, until), search_mode: 'fts-only',
925
+ ...stateFields(), degradation_reason: this.degradationReason() ?? 'inference_error' };
926
+ }
564
927
  const entityResults = Array.from(resultMap.values()).sort((a, b) => a.distance - b.distance).slice(0, limit);
565
928
  // Filter by temporal range if specified
566
929
  let filteredResults = entityResults;
@@ -576,35 +939,79 @@ export class RAGKnowledgeGraphManager {
576
939
  return true;
577
940
  });
578
941
  }
579
- if (filteredResults.length === 0) {
580
- console.error(`ℹ️ No semantic matches found for "${query}"`);
581
- return { entities: [], relations: [] };
582
- }
583
942
  const entities = filteredResults.map(result => ({
584
943
  name: result.name,
585
944
  entityType: result.entityType,
586
945
  observations: JSON.parse(result.observations),
587
946
  similarity: Math.max(0, 1 - result.distance / 2) // Convert cosine distance (0-2) to similarity (1-0)
588
947
  }));
589
- // Get relationships between the found entities
590
- const entityNames = entities.map(e => e.name);
591
- const relations = this.db.prepare(`
592
- SELECT
948
+ // hybrid-partial (spec §4): entities without vectors must not vanish from
949
+ // search while backfill catches up — merge lexical FTS hits for the gap.
950
+ let search_mode = 'hybrid';
951
+ if (entityPct < 100) {
952
+ search_mode = 'hybrid-partial';
953
+ const seen = new Set(entities.map(e => e.name));
954
+ const ftsExtra = this.searchNodesFts(query, limit, since, until);
955
+ for (const e of ftsExtra.entities) {
956
+ if (entities.length >= limit)
957
+ break;
958
+ if (!seen.has(e.name)) {
959
+ seen.add(e.name);
960
+ entities.push(e);
961
+ }
962
+ }
963
+ }
964
+ if (entities.length === 0) {
965
+ console.error(`ℹ️ No semantic matches found for "${query}"`);
966
+ return { entities: [], relations: [], search_mode, ...stateFields() };
967
+ }
968
+ const relations = this.relationsAmong(entities.map(e => e.name));
969
+ console.error(`✅ Found ${entities.length} semantically similar entities with ${relations.length} relationships`);
970
+ return { entities, relations, search_mode, ...stateFields() };
971
+ }
972
+ // Lexical entity search over entities_fts (spec §5 contract: name /
973
+ // observations / entityType lexical match — no semantic-equivalence claim).
974
+ // Temporal filters apply in SQL so LIMIT is not distorted.
975
+ searchNodesFts(query, limit, since, until) {
976
+ const expr = compileFtsLiteralQuery(query);
977
+ if (expr === null) {
978
+ return { entities: [], relations: [], warning: 'query has no searchable terms' };
979
+ }
980
+ const rows = this.db.prepare(`
981
+ SELECT e.name, e.entityType, e.observations
982
+ FROM entities_fts f
983
+ JOIN entities e ON f.rowid = e.rowid
984
+ WHERE entities_fts MATCH @expr
985
+ ${since ? 'AND e.created_at >= @since' : ''}
986
+ ${until ? 'AND e.created_at <= @until' : ''}
987
+ ORDER BY bm25(entities_fts)
988
+ LIMIT @limit
989
+ `).all({ expr, since, until, limit });
990
+ const entities = rows.map(r => ({
991
+ name: r.name,
992
+ entityType: r.entityType,
993
+ observations: JSON.parse(r.observations),
994
+ }));
995
+ return { entities, relations: this.relationsAmong(entities.map(e => e.name)) };
996
+ }
997
+ relationsAmong(entityNames) {
998
+ if (entityNames.length === 0)
999
+ return [];
1000
+ return this.db.prepare(`
1001
+ SELECT
593
1002
  e1.name as from_name,
594
1003
  e2.name as to_name,
595
1004
  r.relationType
596
1005
  FROM relationships r
597
1006
  JOIN entities e1 ON r.source_entity = e1.id
598
1007
  JOIN entities e2 ON r.target_entity = e2.id
599
- WHERE e1.name IN (${entityNames.map(() => '?').join(',')})
1008
+ WHERE e1.name IN (${entityNames.map(() => '?').join(',')})
600
1009
  AND e2.name IN (${entityNames.map(() => '?').join(',')})
601
1010
  `).all(...entityNames, ...entityNames).map((row) => ({
602
1011
  from: row.from_name,
603
1012
  to: row.to_name,
604
1013
  relationType: row.relationType
605
1014
  }));
606
- console.error(`✅ Found ${entities.length} semantically similar entities with ${relations.length} relationships`);
607
- return { entities, relations };
608
1015
  }
609
1016
  async openNodes(names) {
610
1017
  if (!this.db)
@@ -638,12 +1045,49 @@ export class RAGKnowledgeGraphManager {
638
1045
  return { entities, relations };
639
1046
  }
640
1047
  // === NEW RAG FUNCTIONALITY ===
641
- // Generate embedding text for an entity (combines name, type, and observations)
1048
+ // Generate embedding text for an entity (identity + newest observations within a char budget).
642
1049
  generateEntityEmbeddingText(entity) {
643
- const observationsText = entity.observations
644
- .filter(o => !o.startsWith('Source:') && !o.startsWith('Created:') && !o.startsWith('Type:') && !o.startsWith('Tags:') && !o.startsWith('Content length:'))
645
- .join('. ');
646
- return `${entity.entityType}: ${entity.name}. ${observationsText}`.trim();
1050
+ return this.buildEntityEmbeddingText(entity).text;
1051
+ }
1052
+ // Build entity embedding text plus stats for instrumentation.
1053
+ // The char budget keeps the entity vector representative of CURRENT state and stays under the
1054
+ // bge-m3 8192-token ceiling; older history lives in RAG document chunks / dated entities.
1055
+ // Returns selected/total observation counts and filtered (pre-cap) vs capped observation char
1056
+ // sizes so callers can log truncation accurately (identity prefix is excluded from these sizes).
1057
+ buildEntityEmbeddingText(entity) {
1058
+ const maxObservationChars = Math.max(1000, Number.parseInt(process.env.ENTITY_EMBED_OBS_CHAR_BUDGET || '12000', 10) || 12000);
1059
+ const observations = entity.observations.filter(o => !o.startsWith('Source:') && !o.startsWith('Created:') && !o.startsWith('Type:') &&
1060
+ !o.startsWith('Tags:') && !o.startsWith('Content length:'));
1061
+ const filteredObsChars = observations.join('. ').length;
1062
+ const selected = [];
1063
+ let remaining = maxObservationChars;
1064
+ for (let i = observations.length - 1; i >= 0 && remaining > 0; i--) {
1065
+ const obs = observations[i];
1066
+ const separatorCost = selected.length > 0 ? 2 : 0; // '. ' joiner
1067
+ const available = remaining - separatorCost;
1068
+ if (available <= 0)
1069
+ break;
1070
+ if (obs.length <= available) {
1071
+ selected.push(obs);
1072
+ remaining -= obs.length + separatorCost;
1073
+ }
1074
+ else if (selected.length === 0) {
1075
+ selected.push(obs.slice(0, available)); // single giant obs: keep a truncated head, never empty
1076
+ break;
1077
+ }
1078
+ else {
1079
+ break;
1080
+ }
1081
+ }
1082
+ const observationsText = selected.reverse().join('. ');
1083
+ const text = `${entity.entityType}: ${entity.name}. ${observationsText}`.trim();
1084
+ return {
1085
+ text,
1086
+ filteredObsChars,
1087
+ cappedObsChars: observationsText.length,
1088
+ selectedObsCount: selected.length,
1089
+ totalObsCount: observations.length,
1090
+ };
647
1091
  }
648
1092
  // NEW: Generic semantic summary generation methods
649
1093
  splitIntoSentences(text) {
@@ -763,7 +1207,7 @@ export class RAGKnowledgeGraphManager {
763
1207
  };
764
1208
  }
765
1209
  // Generate and store embedding for a single entity
766
- async embedEntity(entityId) {
1210
+ async embedEntity(entityId, priority = 'bulk') {
767
1211
  if (!this.db)
768
1212
  throw new Error('Database not initialized');
769
1213
  // Get entity data
@@ -775,32 +1219,53 @@ export class RAGKnowledgeGraphManager {
775
1219
  return false;
776
1220
  }
777
1221
  const parsedObservations = JSON.parse(entity.observations);
778
- const embeddingText = this.generateEntityEmbeddingText({
1222
+ const built = this.buildEntityEmbeddingText({
779
1223
  name: entity.name,
780
1224
  entityType: entity.entityType,
781
1225
  observations: parsedObservations
782
1226
  });
783
- // Generate embedding
784
- const embedding = await this.generateEmbedding(embeddingText);
1227
+ const embeddingText = built.text;
1228
+ // Instrumentation (stderr only): observations kept vs total, filtered pre-cap vs capped obs
1229
+ // char size (identity excluded), and embed duration. `capped` = some observation chars dropped.
1230
+ const capped = built.cappedObsChars < built.filteredObsChars;
1231
+ const embedStart = Date.now();
1232
+ const embedding = await this.generateEmbedding(embeddingText, 1024, false, priority);
1233
+ const embedMs = Date.now() - embedStart;
1234
+ console.error(`[embed] ${entity.name}: ${built.selectedObsCount}/${built.totalObsCount} obs, ${built.filteredObsChars}ch -> ${built.cappedObsChars}ch${capped ? ' (capped)' : ''}, ${embedMs}ms`);
785
1235
  try {
786
- // Delete existing embedding if any
787
- const existingMetadata = this.db.prepare(`
788
- SELECT rowid FROM entity_embedding_metadata WHERE entity_id = ?
789
- `).get(entityId);
790
- if (existingMetadata) {
791
- this.db.prepare(`DELETE FROM entity_embeddings WHERE rowid = ?`).run(existingMetadata.rowid);
792
- this.db.prepare(`DELETE FROM entity_embedding_metadata WHERE entity_id = ?`).run(entityId);
1236
+ // v3.6 (§6a-2): vector replace + provenance stamp commit atomically.
1237
+ // Write-back CAS (beta 2R B1): the entity may have been mutated again
1238
+ // while THIS inference was in flight — a late writer must never
1239
+ // re-insert a vector for superseded content as 'verified'. Inside the
1240
+ // write transaction the CURRENT entity text is rebuilt and hashed; on
1241
+ // mismatch the result is discarded and the row stays missing/queued for
1242
+ // the backfill pass that the newer mutation already kicked.
1243
+ const inputHash = this.hashWithBuilderVersion(embeddingText);
1244
+ const writeTx = this.db.transaction(() => {
1245
+ const currentHash = this.entityInputHash(entityId);
1246
+ if (currentHash !== inputHash)
1247
+ return false; // superseded — discard
1248
+ const existingMetadata = this.db.prepare(`
1249
+ SELECT rowid FROM entity_embedding_metadata WHERE entity_id = ?
1250
+ `).get(entityId);
1251
+ if (existingMetadata) {
1252
+ this.db.exec(`DELETE FROM entity_embeddings WHERE rowid = ${Number(existingMetadata.rowid)}`);
1253
+ this.db.prepare(`DELETE FROM entity_embedding_metadata WHERE entity_id = ?`).run(entityId);
1254
+ }
1255
+ const result = this.db.prepare(`
1256
+ INSERT INTO entity_embeddings (embedding) VALUES (?)
1257
+ `).run(Buffer.from(embedding.buffer));
1258
+ this.db.prepare(`
1259
+ INSERT INTO entity_embedding_metadata (rowid, entity_id, embedding_text, input_hash, profile_id, provenance_state)
1260
+ VALUES (?, ?, ?, ?, ?, 'verified')
1261
+ `).run(result.lastInsertRowid, entityId, embeddingText, inputHash, this.currentProfileId);
1262
+ return true;
1263
+ });
1264
+ const written = writeTx();
1265
+ if (!written) {
1266
+ console.error(`⏭️ discarded superseded embedding for ${entityId} (entity changed during inference)`);
793
1267
  }
794
- // Insert new embedding
795
- const result = this.db.prepare(`
796
- INSERT INTO entity_embeddings (embedding) VALUES (?)
797
- `).run(Buffer.from(embedding.buffer));
798
- // Store metadata
799
- this.db.prepare(`
800
- INSERT INTO entity_embedding_metadata (rowid, entity_id, embedding_text)
801
- VALUES (?, ?, ?)
802
- `).run(result.lastInsertRowid, entityId, embeddingText);
803
- return true;
1268
+ return written;
804
1269
  }
805
1270
  catch (error) {
806
1271
  console.error(`Failed to embed entity ${entityId}:`, error);
@@ -823,6 +1288,7 @@ export class RAGKnowledgeGraphManager {
823
1288
  embeddedCount += results.filter(Boolean).length;
824
1289
  }
825
1290
  console.error(`✅ Entity embeddings completed: ${embeddedCount}/${entities.length} entities embedded`);
1291
+ this.coordinator?.invalidateCoverage();
826
1292
  return {
827
1293
  totalEntities: entities.length,
828
1294
  embeddedEntities: embeddedCount
@@ -902,15 +1368,20 @@ export class RAGKnowledgeGraphManager {
902
1368
  const errors = [];
903
1369
  for (const chunk of chunks) {
904
1370
  // Generate embedding
905
- const embedding = await this.generateEmbedding(chunk.text);
1371
+ const embedding = await this.generateEmbedding(chunk.text, 1024, false, 'bulk');
906
1372
  const rowid = safeRowid(chunk.rowid);
907
1373
  try {
908
- // Delete existing embedding if any
909
- this.db.exec(`DELETE FROM chunks WHERE rowid = ${rowid}`);
910
- // Insert new embedding - rowid as literal integer for vec0 compatibility
911
- this.db.prepare(`
912
- INSERT INTO chunks (rowid, embedding) VALUES (${rowid}, ?)
913
- `).run(Buffer.from(embedding.buffer));
1374
+ // vector + verified provenance in one transaction (§6a-2) — KG chunks
1375
+ // must never become vector-bearing provenance-NULL rows post-recon.
1376
+ const tx = this.db.transaction(() => {
1377
+ this.db.exec(`DELETE FROM chunks WHERE rowid = ${rowid}`);
1378
+ this.db.prepare(`
1379
+ INSERT INTO chunks (rowid, embedding) VALUES (${rowid}, ?)
1380
+ `).run(Buffer.from(embedding.buffer));
1381
+ this.db.prepare(`UPDATE chunk_metadata SET input_hash = ?, profile_id = ?, provenance_state = 'verified' WHERE rowid = ?`)
1382
+ .run(createHash('sha256').update(chunk.text).digest('hex'), this.currentProfileId, chunk.rowid);
1383
+ });
1384
+ tx();
914
1385
  embeddedCount++;
915
1386
  }
916
1387
  catch (error) {
@@ -919,6 +1390,7 @@ export class RAGKnowledgeGraphManager {
919
1390
  errors.push(errMsg);
920
1391
  }
921
1392
  }
1393
+ this.coordinator?.invalidateCoverage();
922
1394
  console.error(`✅ Knowledge graph chunks embedded: ${embeddedCount}/${chunks.length}`);
923
1395
  return { embeddedChunks: embeddedCount, totalChunks: chunks.length, ...(errors.length > 0 && { errors: errors.slice(0, 5) }) };
924
1396
  }
@@ -1142,37 +1614,23 @@ export class RAGKnowledgeGraphManager {
1142
1614
  }
1143
1615
  // Generate embeddings using sentence transformers
1144
1616
  // isQuery: true for search queries (adds instruction prefix), false for documents/entities
1145
- async generateEmbedding(text, dimensions = 1024, isQuery = false) {
1617
+ async generateEmbedding(text, dimensions = 1024, isQuery = false, priority = 'interactive') {
1146
1618
  // Check cache first (hash-based key to avoid collisions on long texts)
1147
1619
  const cacheKey = createHash('md5').update(`${text}_${dimensions}_${isQuery}`).digest('hex');
1148
1620
  const cached = this.embeddingCache.get(cacheKey);
1149
1621
  if (cached)
1150
1622
  return cached;
1151
- if (this.modelInitialized && this.embeddingModel) {
1152
- try {
1153
- // BGE-M3: instruction prefix improves retrieval quality for queries
1154
- const inputText = isQuery
1155
- ? `Represent this sentence for searching relevant passages: ${text}`
1156
- : text;
1157
- const result = await this.embeddingModel(inputText, { pooling: 'cls', normalize: true });
1158
- // Extract the embedding array and convert to Float32Array
1159
- const embedding = result.data;
1160
- const modelResult = new Float32Array(embedding.slice(0, dimensions));
1161
- // Cache the result (LRU: evict oldest if full)
1162
- if (this.embeddingCache.size >= this.EMBEDDING_CACHE_MAX) {
1163
- const firstKey = this.embeddingCache.keys().next().value;
1164
- if (firstKey)
1165
- this.embeddingCache.delete(firstKey);
1166
- }
1167
- this.embeddingCache.set(cacheKey, modelResult);
1168
- return modelResult;
1169
- }
1170
- catch (error) {
1171
- console.error(`⚠️ Embedding model failed for text "${text.slice(0, 50)}...":`, error instanceof Error ? error.message : error);
1172
- throw new Error(`Embedding model not available. Ensure the model is loaded. Original error: ${error instanceof Error ? error.message : error}`);
1173
- }
1174
- }
1175
- throw new Error('Embedding model not initialized. The server may still be loading the model — retry in a few seconds.');
1623
+ // v3.6: all inference goes through the gate — state check + execution in one
1624
+ // atomic boundary (TOCTOU-safe). GateNotReadyError / GateDisabledError
1625
+ // propagate so each consumer honors its own not-ready contract (spec §5).
1626
+ const modelResult = await this.gate.embed(text, { dims: dimensions, isQuery, priority });
1627
+ if (this.embeddingCache.size >= this.EMBEDDING_CACHE_MAX) {
1628
+ const firstKey = this.embeddingCache.keys().next().value;
1629
+ if (firstKey)
1630
+ this.embeddingCache.delete(firstKey);
1631
+ }
1632
+ this.embeddingCache.set(cacheKey, modelResult);
1633
+ return modelResult;
1176
1634
  }
1177
1635
  // === NEW SEPARATE TOOLS ===
1178
1636
  async syncDocumentFromFile(filePath, documentId, options = {}) {
@@ -1200,31 +1658,56 @@ export class RAGKnowledgeGraphManager {
1200
1658
  catch { /* ignore */ }
1201
1659
  if (existingHash === contentHash) {
1202
1660
  const cmCount = this.db.prepare(`SELECT count(*) AS n FROM chunk_metadata WHERE document_id = ?`).get(documentId).n;
1661
+ // "Embedded" for dedup completeness = vector exists AND its profile is
1662
+ // current (or legacy-NULL awaiting grandfather). Raw vector counts
1663
+ // would misjudge old-profile rows as complete (beta 1R supplement).
1203
1664
  const embCount = this.db.prepare(`
1204
- SELECT count(*) AS n FROM chunks c JOIN chunk_metadata m ON c.rowid = m.rowid WHERE m.document_id = ?
1665
+ SELECT count(*) AS n FROM chunks c JOIN chunk_metadata m ON c.rowid = m.rowid
1666
+ WHERE m.document_id = ? AND (m.provenance_state IS NULL OR m.profile_id = ?)
1667
+ `).get(documentId, this.currentProfileId).n;
1668
+ const linked = this.db.prepare(`
1669
+ SELECT count(DISTINCT ce.entity_id) AS n FROM chunk_entities ce
1670
+ JOIN chunk_metadata m ON ce.chunk_rowid = m.rowid WHERE m.document_id = ?
1205
1671
  `).get(documentId).n;
1206
1672
  if (cmCount > 0 && cmCount === embCount) {
1207
- const linked = this.db.prepare(`
1208
- SELECT count(DISTINCT ce.entity_id) AS n FROM chunk_entities ce
1209
- JOIN chunk_metadata m ON ce.chunk_rowid = m.rowid WHERE m.document_id = ?
1210
- `).get(documentId).n;
1211
1673
  console.error(`⏭️ syncDocumentFromFile: ${documentId} unchanged (hash match, ${cmCount} chunks embedded) — skipped`);
1212
1674
  return { documentId, bytes, chunks: cmCount, embeddedChunks: embCount, linkedEntities: linked, skipped: true, reason: 'unchanged' };
1213
1675
  }
1676
+ if (cmCount > 0 && embCount < cmCount) {
1677
+ // v3.6 (spec §5b M12): identical content with incomplete/stale vectors
1678
+ // keeps the document, chunks, rowids and entity links — only the
1679
+ // missing vectors are re-queued via the coordinator. Full re-chunking
1680
+ // here would churn rowids and links for no content change.
1681
+ console.error(`♻️ syncDocumentFromFile: ${documentId} unchanged but ${cmCount - embCount} vectors missing — re-queued (chunks preserved)`);
1682
+ this.coordinator?.kick();
1683
+ return { documentId, bytes, chunks: cmCount, embeddedChunks: embCount, linkedEntities: linked, skipped: true, reason: 'unchanged-revectorizing', embedding_status: this.gate.isDisabled ? 'disabled' : 'queued' };
1684
+ }
1214
1685
  }
1215
1686
  }
1216
1687
  console.error(`🔄 syncDocumentFromFile: ${documentId} <- ${filePath} (${bytes} bytes)`);
1217
- // 3. Pre-compute chunks + embeddings BEFORE any DB mutation. If embedding
1218
- // throws (model down), the existing document is left completely intact.
1688
+ // 3. Two contracts (spec §5b):
1689
+ // ready — pre-compute ALL embeddings BEFORE any DB mutation; if
1690
+ // inference throws mid-way the old document stays intact
1691
+ // (v3.5.0 atomicity, unchanged).
1692
+ // not-ready — intentional lazy sync: store document + chunks + FTS in
1693
+ // one transaction with NO vectors (embedding_status:
1694
+ // queued); the backfill coordinator recovers them.
1219
1695
  const { maxTokens = 800, overlap = 160 } = options.chunkParams || {};
1220
1696
  const segments = this.chunkText(content, maxTokens, overlap);
1697
+ const lazySync = !this.gate.isReady;
1221
1698
  const embedded = [];
1222
- for (const seg of segments) {
1223
- const embedding = await this.generateEmbedding(seg.text);
1224
- embedded.push({ seg, embedding });
1699
+ if (lazySync) {
1700
+ for (const seg of segments)
1701
+ embedded.push({ seg, embedding: null });
1225
1702
  }
1226
- // 4. Atomic swap: delete old -> insert doc -> insert chunks + embeddings,
1227
- // all in a single synchronous better-sqlite3 transaction (all-or-nothing).
1703
+ else {
1704
+ for (const seg of segments) {
1705
+ const embedding = await this.generateEmbedding(seg.text, 1024, false, 'bulk');
1706
+ embedded.push({ seg, embedding });
1707
+ }
1708
+ }
1709
+ // 4. Atomic swap: delete old -> insert doc -> insert chunks (+ embeddings
1710
+ // with verified provenance when ready), one synchronous transaction.
1228
1711
  const applyTx = this.db.transaction(() => {
1229
1712
  const db = this.db;
1230
1713
  // 4a. cleanup old doc (inlined sync version of cleanupDocument).
@@ -1238,7 +1721,8 @@ export class RAGKnowledgeGraphManager {
1238
1721
  // 4b. insert document.
1239
1722
  db.prepare(`INSERT INTO documents (id, content, metadata) VALUES (?, ?, ?)`)
1240
1723
  .run(documentId, content, JSON.stringify(metadata));
1241
- // 4c. insert chunk_metadata (FTS5 chunks_fts auto-filled by trigger) + embeddings.
1724
+ // 4c. insert chunk_metadata (FTS5 chunks_fts auto-filled by trigger);
1725
+ // vectors + provenance only on the ready path (§6a-2).
1242
1726
  for (const { seg, embedding } of embedded) {
1243
1727
  const chunkId = `${documentId}_chunk_${seg.chunk_index}`;
1244
1728
  const info = db.prepare(`
@@ -1246,11 +1730,18 @@ export class RAGKnowledgeGraphManager {
1246
1730
  VALUES (?, ?, ?, ?, ?, ?, ?, ?)
1247
1731
  `).run(chunkId, documentId, seg.chunk_index, seg.text, seg.start_pos, seg.end_pos, seg.start_token, seg.end_token);
1248
1732
  const rowid = Number(info.lastInsertRowid);
1249
- db.prepare(`INSERT INTO chunks (rowid, embedding) VALUES (${rowid}, ?)`).run(Buffer.from(embedding.buffer));
1733
+ if (embedding) {
1734
+ db.prepare(`INSERT INTO chunks (rowid, embedding) VALUES (${rowid}, ?)`).run(Buffer.from(embedding.buffer));
1735
+ db.prepare(`UPDATE chunk_metadata SET input_hash = ?, profile_id = ?, provenance_state = 'verified' WHERE rowid = ?`)
1736
+ .run(createHash('sha256').update(seg.text).digest('hex'), this.currentProfileId, rowid);
1737
+ }
1250
1738
  }
1251
1739
  });
1252
1740
  applyTx();
1253
- const embeddedChunks = embedded.length;
1741
+ this.coordinator?.invalidateCoverage();
1742
+ if (lazySync)
1743
+ this.coordinator?.kick();
1744
+ const embeddedChunks = lazySync ? 0 : embedded.length;
1254
1745
  // 5. Entity linking AFTER commit. Non-destructive + idempotent (INSERT OR
1255
1746
  // IGNORE), so a linking failure cannot corrupt the doc/embeddings.
1256
1747
  const linkedEntities = await this.autoLinkEntities(documentId);
@@ -1266,6 +1757,7 @@ export class RAGKnowledgeGraphManager {
1266
1757
  chunks: segments.length,
1267
1758
  embeddedChunks,
1268
1759
  linkedEntities,
1760
+ embedding_status: lazySync ? (this.gate.isDisabled ? 'disabled' : 'queued') : 'embedded',
1269
1761
  ...(explicitlyLinked !== undefined ? { explicitlyLinked } : {}),
1270
1762
  };
1271
1763
  if (linkedEntities === 0 && explicitlyLinked === undefined) {
@@ -1323,6 +1815,10 @@ export class RAGKnowledgeGraphManager {
1323
1815
  });
1324
1816
  }
1325
1817
  console.error(`✅ Document chunked: ${chunks.length} chunks created`);
1818
+ // Indirect missing-row producer (spec §5): freshly chunked rows have no
1819
+ // vectors yet — let the coordinator recover them without a restart.
1820
+ this.coordinator?.invalidateCoverage();
1821
+ this.coordinator?.kick();
1326
1822
  return { documentId, chunks: resultChunks };
1327
1823
  }
1328
1824
  async embedChunks(documentId) {
@@ -1339,18 +1835,20 @@ export class RAGKnowledgeGraphManager {
1339
1835
  let embeddedCount = 0;
1340
1836
  const errors = [];
1341
1837
  for (const chunk of chunks) {
1342
- // Generate embedding
1343
- const embedding = await this.generateEmbedding(chunk.text);
1838
+ // Generate embedding (foreground-bulk priority)
1839
+ const embedding = await this.generateEmbedding(chunk.text, 1024, false, 'bulk');
1344
1840
  const rowid = Number(chunk.rowid);
1345
- // Store in vector table
1841
+ // Store in vector table (+ verified provenance, §6a-2 atomic)
1346
1842
  try {
1347
- // First, delete any existing embedding for this rowid
1348
- this.db.exec(`DELETE FROM chunks WHERE rowid = ${safeRowid(rowid)}`);
1349
- // Insert new embedding with explicit rowid to match chunk_metadata
1350
- // Use parameterized only for embedding blob, rowid as literal integer
1351
- this.db.prepare(`
1352
- INSERT INTO chunks (rowid, embedding) VALUES (${rowid}, ?)
1353
- `).run(Buffer.from(embedding.buffer));
1843
+ const tx = this.db.transaction(() => {
1844
+ this.db.exec(`DELETE FROM chunks WHERE rowid = ${safeRowid(rowid)}`);
1845
+ this.db.prepare(`
1846
+ INSERT INTO chunks (rowid, embedding) VALUES (${rowid}, ?)
1847
+ `).run(Buffer.from(embedding.buffer));
1848
+ this.db.prepare(`UPDATE chunk_metadata SET input_hash = ?, profile_id = ?, provenance_state = 'verified' WHERE rowid = ?`)
1849
+ .run(createHash('sha256').update(chunk.text).digest('hex'), this.currentProfileId, rowid);
1850
+ });
1851
+ tx();
1354
1852
  embeddedCount++;
1355
1853
  }
1356
1854
  catch (error) {
@@ -1360,6 +1858,7 @@ export class RAGKnowledgeGraphManager {
1360
1858
  }
1361
1859
  }
1362
1860
  console.error(`✅ Chunks embedded: ${embeddedCount}/${chunks.length}`);
1861
+ this.coordinator?.invalidateCoverage();
1363
1862
  // Auto-link entities to document after embedding
1364
1863
  const linkedCount = await this.autoLinkEntities(documentId);
1365
1864
  return { documentId, embeddedChunks: embeddedCount, totalChunks: chunks.length, linkedEntities: linkedCount, ...(errors.length > 0 && { errors: errors.slice(0, 5) }) };
@@ -1768,6 +2267,9 @@ export class RAGKnowledgeGraphManager {
1768
2267
  }
1769
2268
  }
1770
2269
  console.error(`✅ Import completed: ${imported.entities} entities, ${imported.relations} relations, ${imported.documents} documents imported`);
2270
+ // Indirect missing-row producer (spec §5): imported rows may lack vectors.
2271
+ this.coordinator?.invalidateCoverage();
2272
+ this.coordinator?.kick();
1771
2273
  return { imported, skipped };
1772
2274
  }
1773
2275
  async hybridSearch(query, limit = 5, useGraph = true) {
@@ -1776,6 +2278,9 @@ export class RAGKnowledgeGraphManager {
1776
2278
  if (!this.encoding)
1777
2279
  throw new Error('Tokenizer not initialized');
1778
2280
  console.error(`🔍 Enhanced hybrid search: "${query}"`);
2281
+ // Parity with searchNodes (beta 1R supplement): an unsearchable query gets
2282
+ // an explicit warning instead of a silent empty envelope.
2283
+ const ftsUnsearchable = compileFtsLiteralQuery(query) === null;
1779
2284
  const queryVariants = this.buildCrossLingualVariants(query);
1780
2285
  if (queryVariants.length > 1) {
1781
2286
  console.error(`🌐 Cross-lingual variants: ${queryVariants.slice(1).join(' | ')}`);
@@ -1810,37 +2315,47 @@ export class RAGKnowledgeGraphManager {
1810
2315
  `).all(Buffer.from(embedding.buffer), k);
1811
2316
  };
1812
2317
  // Search original query plus cross-lingual expansions and keep best match per chunk.
2318
+ // v3.6 eligibility gate (spec §3): vector usage requires model_ready AND
2319
+ // reconciliation settled — otherwise FTS5-only, no waiting.
1813
2320
  const resultMap = new Map();
1814
- try {
1815
- primaryQueryEmbedding = await this.generateEmbedding(queryVariants[0], 1024, true);
1816
- for (const variant of queryVariants) {
1817
- const embedding = await this.generateEmbedding(variant, 1024, true);
1818
- const variantResults = searchChunks(embedding, limit * 3);
1819
- for (const r of variantResults) {
1820
- const existing = resultMap.get(r.chunk_id);
1821
- if (!existing || r.distance < existing.distance) {
1822
- resultMap.set(r.chunk_id, r);
2321
+ let degradationReason;
2322
+ if (!(this.coordinator?.eligible ?? false)) {
2323
+ vectorDegraded = true;
2324
+ degradationReason = this.degradationReason();
2325
+ console.error(`ℹ️ vector search not eligible (${degradationReason ?? 'unknown'}) — FTS5-only`);
2326
+ }
2327
+ else {
2328
+ try {
2329
+ primaryQueryEmbedding = await this.generateEmbedding(queryVariants[0], 1024, true);
2330
+ for (const variant of queryVariants) {
2331
+ const embedding = await this.generateEmbedding(variant, 1024, true);
2332
+ const variantResults = searchChunks(embedding, limit * 3);
2333
+ for (const r of variantResults) {
2334
+ const existing = resultMap.get(r.chunk_id);
2335
+ if (!existing || r.distance < existing.distance) {
2336
+ resultMap.set(r.chunk_id, r);
2337
+ }
1823
2338
  }
1824
2339
  }
1825
2340
  }
1826
- }
1827
- catch (embErr) {
1828
- vectorDegraded = true;
1829
- console.error(`⚠️ Vector search unavailable (embedding model down) — degrading to FTS5-only:`, embErr instanceof Error ? embErr.message : embErr);
2341
+ catch (embErr) {
2342
+ vectorDegraded = true;
2343
+ // 'inference_error' (not 'model_not_ready'): model_state may still read
2344
+ // 'ready' here — a contradictory reason pair confused callers (beta B6).
2345
+ degradationReason = this.degradationReason() ?? 'inference_error';
2346
+ console.error(`⚠️ Vector search unavailable — degrading to FTS5-only:`, embErr instanceof Error ? embErr.message : embErr);
2347
+ }
1830
2348
  }
1831
2349
  const vectorResults = Array.from(resultMap.values()).sort((a, b) => a.distance - b.distance);
1832
2350
  // FTS5 full-text search as additional signal (Reciprocal Rank Fusion)
1833
2351
  const ftsBoostMap = new Map();
1834
2352
  try {
1835
2353
  const ftsSearchQuery = (q) => {
1836
- // Escape FTS5 special characters and build a query with OR between terms
1837
- const sanitized = q.replace(/["\*\(\)\-]/g, ' ').trim();
1838
- if (!sanitized)
1839
- return [];
1840
- const terms = sanitized.split(/\s+/).filter(t => t.length > 0);
1841
- if (terms.length === 0)
2354
+ // Shared compiler (spec §5) — same sanitize rules as pre-3.6, extracted
2355
+ // so entity FTS fallback uses identical MATCH-safety guarantees.
2356
+ const ftsExpr = compileFtsLiteralQuery(q);
2357
+ if (ftsExpr === null)
1842
2358
  return [];
1843
- const ftsExpr = terms.map(t => `"${t}"`).join(' OR ');
1844
2359
  return this.db.prepare(`
1845
2360
  SELECT cm.rowid, cm.chunk_id, bm25(chunks_fts) as fts_score
1846
2361
  FROM chunks_fts
@@ -1910,7 +2425,19 @@ export class RAGKnowledgeGraphManager {
1910
2425
  }
1911
2426
  if (vectorResults.length === 0) {
1912
2427
  console.error(`ℹ️ No vector or FTS5 matches found for "${query}"`);
1913
- return [];
2428
+ // Empty results still carry state (spec §5c: envelope exists so callers
2429
+ // can distinguish "nothing matched" from "vector search was degraded").
2430
+ const covE = this.coordinator?.coverage();
2431
+ const chunkPctE = covE && covE.chunk.total > 0 ? Math.round((covE.chunk.embedded / covE.chunk.total) * 100) : 100;
2432
+ const graphPctE = covE && covE.entity.total > 0 ? Math.round((covE.entity.embedded / covE.entity.total) * 100) : 100;
2433
+ return {
2434
+ results: [],
2435
+ search_mode: vectorDegraded ? 'fts-only' : (chunkPctE < 100 ? 'hybrid-partial' : 'hybrid'),
2436
+ model_state: this.gate.status.state,
2437
+ coverage: { chunk_pct: chunkPctE, graph_coverage_pct: graphPctE },
2438
+ ...(degradationReason ? { degradation_reason: degradationReason } : {}),
2439
+ ...(ftsUnsearchable ? { warning: 'query has no searchable terms for FTS' } : {}),
2440
+ };
1914
2441
  }
1915
2442
  // Get entity information for graph enhancement via vector similarity
1916
2443
  let connectedEntities = new Set();
@@ -2114,8 +2641,7 @@ export class RAGKnowledgeGraphManager {
2114
2641
  fts_boost: ftsBoost > 0 ? ftsBoost : undefined,
2115
2642
  full_context_available: true,
2116
2643
  chunk_type: result.chunk_type,
2117
- source_id: sourceId,
2118
- search_mode: vectorDegraded ? 'fts-only' : 'hybrid'
2644
+ source_id: sourceId
2119
2645
  });
2120
2646
  }
2121
2647
  // Sort by relevance and return top results
@@ -2127,7 +2653,34 @@ export class RAGKnowledgeGraphManager {
2127
2653
  const entityResults = finalResults.filter(r => r.chunk_type === 'entity').length;
2128
2654
  const relResults = finalResults.filter(r => r.chunk_type === 'relationship').length;
2129
2655
  console.error(`✅ Enhanced hybrid search completed: ${finalResults.length} results (${docResults} docs, ${entityResults} entities, ${relResults} relationships)`);
2130
- return finalResults;
2656
+ // v3.6 envelope (spec §5c, breaking): search_mode moved from per-item to
2657
+ // top-level so state is visible even on empty results; coverage tells the
2658
+ // caller how much of the corpus is actually vector-searchable.
2659
+ const cov = this.coordinator?.coverage();
2660
+ const chunkPct = cov && cov.chunk.total > 0 ? Math.round((cov.chunk.embedded / cov.chunk.total) * 100) : 100;
2661
+ const graphPct = cov && cov.entity.total > 0 ? Math.round((cov.entity.embedded / cov.entity.total) * 100) : 100;
2662
+ const search_mode = vectorDegraded ? 'fts-only' : (chunkPct < 100 ? 'hybrid-partial' : 'hybrid');
2663
+ return {
2664
+ results: finalResults,
2665
+ search_mode,
2666
+ model_state: this.gate.status.state,
2667
+ coverage: { chunk_pct: chunkPct, graph_coverage_pct: graphPct },
2668
+ ...(degradationReason ? { degradation_reason: degradationReason } : {}),
2669
+ ...(ftsUnsearchable ? { warning: 'query has no searchable terms for FTS' } : {}),
2670
+ };
2671
+ }
2672
+ // v3.6 (spec §5c / 6R note 2): why is vector search degraded right now?
2673
+ degradationReason() {
2674
+ if (this.gate.isDisabled)
2675
+ return 'disabled';
2676
+ if (!this.gate.isReady)
2677
+ return 'model_not_ready';
2678
+ const rs = this.coordinator?.reconState;
2679
+ if (rs === 'failed')
2680
+ return 'reconciliation_failed';
2681
+ if (rs && rs !== 'complete' && rs !== 'n/a')
2682
+ return 'reconciling';
2683
+ return undefined;
2131
2684
  }
2132
2685
  // NEW: Get detailed context for a specific chunk
2133
2686
  async getDetailedContext(chunkId, includeSurrounding = true) {
@@ -2218,6 +2771,10 @@ export class RAGKnowledgeGraphManager {
2218
2771
  const chunkCount = this.db.prepare(`
2219
2772
  SELECT COUNT(*) as count FROM chunk_metadata
2220
2773
  `).get();
2774
+ // v3.6 (spec §8-2, additive): server self-report — the framework's /start
2775
+ // reads version, model/reconciliation state, and provenance coverage here.
2776
+ const gs = this.gate.status;
2777
+ const cov = this.coordinator?.coverage();
2221
2778
  return {
2222
2779
  entities: {
2223
2780
  total: entityStats.reduce((sum, stat) => sum + stat.count, 0),
@@ -2228,7 +2785,23 @@ export class RAGKnowledgeGraphManager {
2228
2785
  by_type: Object.fromEntries(relationshipStats.map(s => [s.relationType, s.count]))
2229
2786
  },
2230
2787
  documents: documentCount.count,
2231
- chunks: chunkCount.count
2788
+ chunks: chunkCount.count,
2789
+ server: {
2790
+ version: PKG_VERSION,
2791
+ node: process.versions.node,
2792
+ embeddings_mode: this.embeddingsMode,
2793
+ model: `${EMBEDDING_MODEL}@${MODEL_REVISION}`,
2794
+ model_state: gs.state,
2795
+ ready_since: gs.readySince ?? null,
2796
+ last_error: gs.lastError ? sanitizeErrorMessage(gs.lastError) : null,
2797
+ retry_at: gs.retryAt ?? null,
2798
+ reconciliation_state: this.coordinator?.reconState ?? 'n/a',
2799
+ reconciliation_last_error: this.coordinator?.reconLastError ?? null,
2800
+ coverage: cov ? {
2801
+ chunk: { total: cov.chunk.total, embedded: cov.chunk.embedded, verified: cov.chunk.verified, legacy_assumed: cov.chunk.legacy_assumed, missing: cov.chunk.total - cov.chunk.embedded },
2802
+ entity: { total: cov.entity.total, embedded: cov.entity.embedded, verified: cov.entity.verified, legacy_assumed: cov.entity.legacy_assumed, missing: cov.entity.total - cov.entity.embedded },
2803
+ } : null,
2804
+ }
2232
2805
  };
2233
2806
  }
2234
2807
  // === GRAPH ANALYTICS TOOLS (graphology) ===
@@ -2557,8 +3130,9 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
2557
3130
  await ragKgManager.deleteEntities(validatedArgs.entityNames);
2558
3131
  return { content: [{ type: "text", text: "Entities deleted successfully" }] };
2559
3132
  case "deleteObservations":
2560
- await ragKgManager.deleteObservations(validatedArgs.deletions);
2561
- return { content: [{ type: "text", text: "Observations deleted successfully" }] };
3133
+ // v3.6 (spec §5c, breaking): structured per-entity results replace the
3134
+ // bare success string — mixed embedded/queued/no-op states are visible.
3135
+ return { content: [{ type: "text", text: JSON.stringify(await ragKgManager.deleteObservations(validatedArgs.deletions), null, 2) }] };
2562
3136
  case "deleteRelations":
2563
3137
  await ragKgManager.deleteRelations(validatedArgs.relations);
2564
3138
  return { content: [{ type: "text", text: "Relations deleted successfully" }] };
@@ -2629,35 +3203,95 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
2629
3203
  }
2630
3204
  }
2631
3205
  catch (error) {
3206
+ // v3.6 (spec §5c): machine-distinguishable failures. Embedding-gate errors
3207
+ // become structured retryable/terminal payloads; every error response now
3208
+ // sets isError so clients stop parsing "Error: ..." strings.
3209
+ if (error instanceof GateNotReadyError) {
3210
+ return { isError: true, content: [{ type: "text", text: JSON.stringify({
3211
+ code: error.code, state: error.state,
3212
+ ...(error.retryAfterMs !== undefined ? { retry_after_ms: error.retryAfterMs } : {}),
3213
+ message: error.message
3214
+ }) }] };
3215
+ }
3216
+ if (error instanceof GateDisabledError) {
3217
+ return { isError: true, content: [{ type: "text", text: JSON.stringify({
3218
+ code: error.code, state: error.state, message: error.message
3219
+ }) }] };
3220
+ }
2632
3221
  if (error instanceof Error) {
2633
3222
  console.error(`❌ Tool execution error for ${name}:`, error.message);
2634
- return { content: [{ type: "text", text: `Error: ${error.message}` }] };
3223
+ return { isError: true, content: [{ type: "text", text: `Error: ${error.message}` }] };
2635
3224
  }
2636
3225
  throw error;
2637
3226
  }
2638
3227
  });
2639
3228
  async function main() {
2640
3229
  try {
3230
+ assertNodeVersion();
2641
3231
  await ragKgManager.initialize();
3232
+ printBanner({
3233
+ model: EMBEDDING_MODEL, revision: MODEL_REVISION, dtype: MODEL_DTYPE,
3234
+ cachePath: resolveModelCacheDir(process.env, process.platform, os.homedir()),
3235
+ dbPath: DB_FILE_PATH,
3236
+ });
3237
+ if (ragKgManager.embeddingsMode === 'eager') {
3238
+ // eager = wait for BOTH the first model load attempt and reconciliation to
3239
+ // settle (success or failure) before connecting — v3.5-equivalent boot
3240
+ // extended to legacy DBs (spec §9). Failures fall back to background retry.
3241
+ await Promise.allSettled([ragKgManager.gate.start(), ragKgManager.startReconciliation()]);
3242
+ }
2642
3243
  const transport = new StdioServerTransport();
2643
3244
  await server.connect(transport);
2644
3245
  console.error("🚀 Enhanced RAG Knowledge Graph MCP Server running on stdio");
2645
- // Cleanup on exit — avoid process.exit() to prevent ONNX runtime mutex crash
3246
+ if (ragKgManager.embeddingsMode === 'lazy') {
3247
+ // Background: model load + provenance reconciliation run in parallel.
3248
+ // Failures surface via gate/coordinator state, never as rejections.
3249
+ void ragKgManager.gate.start().catch(() => { });
3250
+ void ragKgManager.startReconciliation().catch(() => { });
3251
+ }
3252
+ else if (ragKgManager.embeddingsMode === 'off') {
3253
+ // off mode still CLASSIFIES reconciliation state (deferred vs n/a) so
3254
+ // stats honor the mode matrix — no sanitation, no inference (beta B7).
3255
+ void ragKgManager.startReconciliation().catch(() => { });
3256
+ }
3257
+ // Graceful shutdown (spec §3 order, beta-2R-amended) — transport close
3258
+ // FIRST so stdin stops holding the event loop, then settle coordinator and
3259
+ // gate, then DB close. process.exit is forbidden EXCEPT the one spec'd
3260
+ // case: a model load/download still pending after the settle deadline
3261
+ // (un-abortable fetch would hold the loop forever) — see shutdownAll.
3262
+ let shuttingDown = false;
2646
3263
  const shutdown = () => {
2647
- console.error('\n🧹 Cleaning up...');
2648
- try {
2649
- ragKgManager.cleanup();
2650
- }
2651
- catch { }
3264
+ if (shuttingDown)
3265
+ return;
3266
+ shuttingDown = true;
3267
+ void (async () => {
3268
+ try {
3269
+ await server.close();
3270
+ }
3271
+ catch { /* transport already gone */ }
3272
+ try {
3273
+ process.stdin.pause();
3274
+ process.stdin.unref?.();
3275
+ }
3276
+ catch { /* best-effort */ }
3277
+ await ragKgManager.shutdownAll();
3278
+ })();
2652
3279
  };
2653
3280
  process.on('SIGINT', shutdown);
2654
3281
  process.on('SIGTERM', shutdown);
2655
- process.on('exit', shutdown);
3282
+ process.on('exit', () => { try {
3283
+ ragKgManager.cleanup();
3284
+ }
3285
+ catch { /* idempotent */ } });
3286
+ console.error('🛡️ shutdown handlers registered'); // deterministic handler-ready marker (5R test residual)
2656
3287
  }
2657
3288
  catch (error) {
2658
3289
  console.error("Failed to initialize server:", error);
2659
- ragKgManager.cleanup();
2660
- process.exit(1);
3290
+ try {
3291
+ ragKgManager.cleanup();
3292
+ }
3293
+ catch { /* already down */ }
3294
+ process.exitCode = 1;
2661
3295
  }
2662
3296
  }
2663
3297
  // Boot the server unless explicitly suppressed. Tests import this module with
@@ -2668,7 +3302,10 @@ async function main() {
2668
3302
  if (process.env.RAG_MEMORY_NO_AUTOSTART !== '1') {
2669
3303
  main().catch((error) => {
2670
3304
  console.error("Fatal error in main():", error);
2671
- ragKgManager.cleanup();
2672
- process.exit(1);
3305
+ try {
3306
+ ragKgManager.cleanup();
3307
+ }
3308
+ catch { /* already down */ }
3309
+ process.exitCode = 1;
2673
3310
  });
2674
3311
  }