rag-memory-epf-mcp 3.6.0 → 5.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -21,8 +21,12 @@ import modularity from 'graphology-metrics/graph/modularity.js';
21
21
  import { getAllMCPTools, validateToolArgs, getSystemInfo } from './src/tools/tool-registry.js';
22
22
  // Import migration system
23
23
  import { MigrationManager } from './src/migrations/migration-manager.js';
24
+ import { backupBeforeMigration } from './src/backup/preflight.js';
25
+ import { rebuildProjection, deleteStaleKgChunks } from './src/observations/projection.js';
26
+ import { addRevision, correctRevision, transitionStatus, linkSources, nextProjectionOrder } from './src/observations/lifecycle.js';
27
+ import { getObservationHistory } from './src/observations/history.js';
24
28
  // Import chunk text algorithm (extracted for publish-time invariant testing)
25
- import { chunkText as splitTextIntoChunks } from './src/chunkText.js';
29
+ import { chunkStructured as chunkStructuredText, effectiveSignature, isCurrentFormatSignature, LEGACY_SIGNATURE, DEFAULT_MAX_TOKENS } from './src/chunkerC.js';
26
30
  import { migrations } from './src/migrations/migrations.js';
27
31
  // v3.6 lite install: model lifecycle + version-independent cache (A′ boundary)
28
32
  import { EmbeddingGate, GateNotReadyError, GateDisabledError, TerminalConfigError } from './src/embeddingGate.js';
@@ -105,6 +109,36 @@ const TEXT_BUILDER_VERSION = 'tb1';
105
109
  // edge belongs to an adjacent chunk and must be removed so TextDecoder does
106
110
  // not emit U+FFFD. Pass trimHead/trimTail=false to preserve head/tail bytes.
107
111
  // (Implementation moved to src/chunkText.ts for testability.)
112
+ export class SyncCasConflictError extends Error {
113
+ constructor(documentId) { super(`sync CAS conflict on ${documentId}`); this.name = 'SyncCasConflictError'; }
114
+ }
115
+ // Test-only fault hook (v13 setMigrationFaultPoint 선례 — 환경변수 금지: 상시 스위치는
116
+ // 오설정 한 줄로 sync 를 깬다).
117
+ let __syncFaultHook = null;
118
+ export function setSyncFaultPoint(point, fn) {
119
+ __syncFaultHook = point && fn ? (p) => { if (p === point)
120
+ fn(); } : null;
121
+ }
122
+ // Pure vector-reuse decision (spec §5.2 조건 1~5). Returns an OWNED Buffer copy:
123
+ // a Buffer read back from SQLite has no byteOffset-0 guarantee, and inserting
124
+ // `.buffer` of a subarray would write the wrong 4,096 bytes (advisor r5-9).
125
+ export function selectReusableVector(candidates, text, currentProfileId, sha256hex) {
126
+ const h = sha256hex(text);
127
+ for (const r of candidates) {
128
+ if (!r.embedding)
129
+ continue; // 조건 1: 벡터 실존
130
+ if (r.input_hash !== h)
131
+ continue; // 조건 2: input_hash 일치
132
+ if (r.text !== text)
133
+ continue; // 조건 3: exact text 최종판정
134
+ if (r.profile_id !== currentProfileId)
135
+ continue; // 조건 4: 현행 프로필
136
+ if (r.provenance_state !== 'verified' && r.provenance_state !== 'legacy_assumed')
137
+ continue; // 조건 5 (NULL 제외)
138
+ return { vec: Buffer.from(r.embedding), provenance: r.provenance_state }; // owned copy
139
+ }
140
+ return null;
141
+ }
108
142
  function safeRowid(value) {
109
143
  const n = Number(value);
110
144
  if (!Number.isInteger(n) || n < 0) {
@@ -129,6 +163,9 @@ export class RAGKnowledgeGraphManager {
129
163
  // v3.6 (spec §3): initialize = DB + migrations + profile only. The embedding
130
164
  // model is NEVER awaited here — main() connects the MCP server first and the
131
165
  // gate loads in the background (lazy) or is awaited explicitly (eager).
166
+ // __testForceFkOff: 음성 대조군 전용. FK 게이트가 실제로 부팅을 막는지 시험한다.
167
+ // 이름에 __test 를 박아 둔 이유는 이것이 프로덕션 설정 표면이 아니라는 것을
168
+ // 호출부에서 읽히게 하기 위해서다.
132
169
  async initialize(opts = {}) {
133
170
  console.error('🚀 Initializing RAG Knowledge Graph MCP Server...');
134
171
  this.db = new Database(DB_FILE_PATH);
@@ -140,9 +177,34 @@ export class RAGKnowledgeGraphManager {
140
177
  this.db.pragma('temp_store = MEMORY');
141
178
  this.db.pragma('mmap_size = 268435456');
142
179
  this.db.pragma('foreign_keys = ON');
180
+ // spec §5.2: 관찰 lifecycle 의 무결성은 전부 FK CASCADE 를 전제한다 — root 를 지우면
181
+ // revision 이, revision 을 지우면 source 가 따라가야 history 가 고아로 남지 않는다.
182
+ // FK 가 꺼진 채 돌면 그 계약이 조용히 무효가 되고, 그게 최악이다. 스키마를 건드리기
183
+ // 전에(= runMigrations 앞에서) 멈춘다.
184
+ // 트랜잭션 내부에서는 이 pragma 가 no-op 이므로(실측 before=1·during=1·after=1)
185
+ // 부팅 시점 확인이 유일한 방어 지점이다.
186
+ //
187
+ // 음성 대조군 주입은 **인자로만** 받는다. 환경변수로 두면 프로덕션 경로에
188
+ // "부팅을 막는 스위치"가 상시 존재하게 되고, 오설정 한 줄로 서버가 안 뜬다
189
+ // (advisor beta 자기의심 2 = "더 나쁘다"). 테스트는 manager 를 직접 만들므로
190
+ // 인자 주입으로 충분하다.
191
+ if (opts.__testForceFkOff)
192
+ this.db.pragma('foreign_keys = OFF');
193
+ {
194
+ const fk = this.db.pragma('foreign_keys', { simple: true });
195
+ if (Number(fk) !== 1) {
196
+ throw new Error(`foreign_keys is ${fk}, expected 1. The observation lifecycle relies on FK CASCADE ` +
197
+ `for history integrity; refusing to run migrations without it.`);
198
+ }
199
+ }
143
200
  this.encoding = get_encoding("cl100k_base");
144
201
  await this.runMigrations();
145
202
  this.currentProfileId = this.ensureCurrentProfile();
203
+ // v14 (spec §7.2): 런타임이 기본 chunker 의 SSOT — 마이그레이션의 리터럴은 동결된
204
+ // 역사이고, 기본값이 진화하면(c2 등) 이 upsert 가 부팅마다 현재값을 기록한다.
205
+ this.db.prepare(`INSERT INTO server_meta (key, value) VALUES ('current_default_chunker', ?)
206
+ ON CONFLICT(key) DO UPDATE SET value = excluded.value`)
207
+ .run(effectiveSignature(DEFAULT_MAX_TOKENS));
146
208
  this.embeddingsMode = opts.skipModel
147
209
  ? 'off'
148
210
  : (process.env.RAG_MEMORY_EMBEDDINGS || 'lazy');
@@ -322,18 +384,44 @@ export class RAGKnowledgeGraphManager {
322
384
  // where another tool call can retrieve the pre-mutation vector, and a crash
323
385
  // between mutation and re-embed leaves a clean missing state (backfill
324
386
  // target), never a stale-searchable one.
387
+ // spec §4.5 단계 1: 관찰 변경 · projection 재합성 · entity vector 무효화 ·
388
+ // stale KG chunk 제거를 한 트랜잭션으로 묶는다. 하나만 되면 검색이 낡은
389
+ // 사실을 계속 반환한다.
390
+ // mutate 가 명시적으로 false 를 반환하면 "아무것도 바꾸지 않았다"는 뜻이고
391
+ // projection 재합성·벡터 무효화·KG 정리를 건너뛴다. 이 경로가 없으면
392
+ // 무변경 upsert 나 dedup-only add 가 **정상 벡터를 지우고 재임베딩도 안 해서**
393
+ // 검색 품질만 깎는다(advisor 구현리뷰 r1 발견 1, 실행 재현).
394
+ // 반환값 = 실제로 변경이 있었는가.
325
395
  mutateEntityAndInvalidate(entityId, mutate) {
396
+ let changed = false;
326
397
  const tx = this.db.transaction(() => {
327
- mutate();
328
- const meta = this.db.prepare(`SELECT rowid FROM entity_embedding_metadata WHERE entity_id = ?`)
329
- .get(entityId);
330
- if (meta) {
331
- this.db.exec(`DELETE FROM entity_embeddings WHERE rowid = ${Number(meta.rowid)}`);
332
- this.db.prepare(`DELETE FROM entity_embedding_metadata WHERE entity_id = ?`).run(entityId);
333
- }
398
+ changed = mutate() !== false;
399
+ if (!changed)
400
+ return;
401
+ rebuildProjection(this.db, entityId);
402
+ this.invalidateDerivedForEntity(entityId);
334
403
  });
335
404
  tx();
336
- this.coordinator?.invalidateCoverage();
405
+ if (changed)
406
+ this.coordinator?.invalidateCoverage();
407
+ return changed;
408
+ }
409
+ // 관찰이 바뀐 entity 의 파생 상태를 무효화한다: entity vector + stale KG chunk.
410
+ // **트랜잭션을 열지 않는다** — 호출자가 이미 하나의 단위 안에 있다고 가정한다.
411
+ //
412
+ // importGraph 가 이 단계를 건너뛰고 있었다: projection 만 재합성하고 파생 상태를
413
+ // 그대로 둬서, 이미 존재하는 entity 를 import 로 덮으면 옛 벡터·옛 KG chunk 가
414
+ // 계속 검색에 나왔다(advisor beta 발견 2, hybridSearch 로 실측 재현).
415
+ // 그래서 "모든 관찰 변경은 mutateEntityAndInvalidate 를 통한다"는 규칙에
416
+ // 예외가 하나 있었고, 그 예외가 정확히 그 규칙이 막으려던 결함을 만들었다.
417
+ invalidateDerivedForEntity(entityId) {
418
+ const meta = this.db.prepare(`SELECT rowid FROM entity_embedding_metadata WHERE entity_id = ?`)
419
+ .get(entityId);
420
+ if (meta) {
421
+ this.db.exec(`DELETE FROM entity_embeddings WHERE rowid = ${Number(meta.rowid)}`);
422
+ this.db.prepare(`DELETE FROM entity_embedding_metadata WHERE entity_id = ?`).run(entityId);
423
+ }
424
+ deleteStaleKgChunks(this.db, entityId);
337
425
  }
338
426
  // §6a-1 invariant: when an entity's embedding input changed but re-embedding
339
427
  // is unavailable, its old vector must not stay searchable.
@@ -434,6 +522,11 @@ export class RAGKnowledgeGraphManager {
434
522
  });
435
523
  // Get pending migrations before running them
436
524
  const pendingBefore = migrationManager.getPendingMigrations();
525
+ // spec §5.1: 대기 중 마이그레이션이 있으면 먼저 일관 스냅샷을 남긴다.
526
+ // 실패는 throw = fail-closed (백업 없이 스키마를 바꾸지 않는다).
527
+ // await: 백업은 Online Backup API 를 쓰므로 비동기다. 여기서 await 를 빠뜨리면
528
+ // 백업이 끝나기 전에 마이그레이션이 시작한다 = 백업 없이 스키마를 바꾸는 것이다.
529
+ await backupBeforeMigration(this.db, DB_FILE_PATH, pendingBefore.map(m => m.version), migrationManager.getCurrentVersion());
437
530
  // Run pending migrations
438
531
  const result = await migrationManager.runMigrations();
439
532
  console.error(`🔧 Database schema ready (version ${result.currentVersion}, ${result.applied} migrations applied)`);
@@ -469,45 +562,73 @@ export class RAGKnowledgeGraphManager {
469
562
  if (!this.db)
470
563
  throw new Error('Database not initialized');
471
564
  const result = [];
565
+ // v13: 관찰은 lifecycle 테이블이 정본이고 entities.observations 는 projection 이다.
566
+ // entity 행은 빈 배열로 만들고 rebuildProjection 이 채운다.
472
567
  const insertStmt = this.db.prepare(`
473
568
  INSERT OR IGNORE INTO entities (id, name, entityType, observations, metadata)
474
- VALUES (?, ?, ?, ?, ?)
569
+ VALUES (?, ?, ?, '[]', ?)
475
570
  `);
571
+ const stripDate = (s2) => s2.replace(/^\[\d{4}-\d{2}-\d{2}\]\s*/, '');
476
572
  for (const entity of entities) {
477
573
  const entityId = `entity_${entity.name.toLowerCase().replace(/[^\p{L}\p{N}]/gu, '_')}`;
478
- const timestamped = (entity.observations || []).map(o => this._timestampObservation(o));
479
- // Try insert first
480
- const insertResult = insertStmt.run(entityId, entity.name, entity.entityType, JSON.stringify(timestamped), '{}');
481
- if (insertResult.changes > 0) {
482
- // New entity created. CRUD success is independent of model readiness
483
- // (spec §5): not-ready -> row stays vectorless (queued for backfill).
484
- console.error(`🔮 Generating embedding for new entity: ${entity.name}`);
574
+ const ts = new Date().toISOString();
575
+ const ids = [];
576
+ let created = false;
577
+ let addedCount = 0;
578
+ let typeUpdated = false;
579
+ // entity INSERT 도 같은 트랜잭션 안이다. 밖에 두면 lifecycle INSERT 가
580
+ // 실패할 때 entity 행만 남는 split state 가 생긴다
581
+ // (advisor 구현리뷰 r1 발견 2, 실행 재현).
582
+ const changed = this.mutateEntityAndInvalidate(entityId, () => {
583
+ created = insertStmt.run(entityId, entity.name, entity.entityType, '{}').changes > 0;
584
+ if (!created && entity.entityType && entity.entityType !== 'CONCEPT') {
585
+ const cur = this.db.prepare(`SELECT entityType FROM entities WHERE id = ?`)
586
+ .get(entityId);
587
+ if (cur && cur.entityType !== entity.entityType) {
588
+ this.db.prepare(`UPDATE entities SET entityType = ? WHERE id = ?`)
589
+ .run(entity.entityType, entityId);
590
+ typeUpdated = true;
591
+ }
592
+ }
593
+ const activeRows = this.db.prepare(`SELECT observation_id, content FROM entity_observations
594
+ WHERE entity_id = ? AND status = 'active'`).all(entityId);
595
+ const activeByBare = new Map(activeRows.map(r => [stripDate(r.content), r.observation_id]));
596
+ let sourcesAdded = 0;
597
+ for (const raw of (entity.observations || [])) {
598
+ const content = this._timestampObservation(raw);
599
+ const bare = stripDate(content);
600
+ const dupId = activeByBare.get(bare);
601
+ if (dupId) {
602
+ // 같은 사실이 다른 출처에서 다시 왔다 = evidence 추가, 새 revision 아님.
603
+ if (entity.sources?.length)
604
+ sourcesAdded += linkSources(this.db, dupId, entity.sources, ts);
605
+ ids.push(null);
606
+ continue;
607
+ }
608
+ const id = addRevision(this.db, {
609
+ entityId, content, status: entity.status ?? 'active', sources: entity.sources, ts
610
+ });
611
+ activeByBare.set(bare, id);
612
+ ids.push(id);
613
+ addedCount++;
614
+ }
615
+ // 아무것도 안 바뀌었으면 projection·벡터·KG 를 건드리지 않는다.
616
+ return created || typeUpdated || addedCount > 0 || sourcesAdded > 0;
617
+ });
618
+ const projected = JSON.parse(this.db.prepare(`SELECT observations FROM entities WHERE id = ?`)
619
+ .get(entityId).observations);
620
+ // 재임베딩은 무효화가 실제로 일어났을 때만. 조건이 갈리면
621
+ // "벡터를 지우고 다시 만들지 않는" 창이 생긴다.
622
+ if (changed) {
623
+ console.error(created
624
+ ? `🔮 Generating embedding for new entity: ${entity.name}`
625
+ : `♻️ Upserted entity: ${entity.name} (+${addedCount} obs${typeUpdated ? ', type→' + entity.entityType : ''})`);
485
626
  const embedding_status = await this.tryEmbedEntity(entityId, 'bulk');
486
- result.push({ ...entity, observations: timestamped, embedding_status });
627
+ result.push({ ...entity, observations: projected, created,
628
+ observation_ids: ids, embedding_status });
487
629
  }
488
630
  else {
489
- // Entity already exists — upsert: merge observations and update entityType
490
- const existing = this.db.prepare(`SELECT observations, entityType FROM entities WHERE id = ?`)
491
- .get(entityId);
492
- if (existing) {
493
- const currentObs = JSON.parse(existing.observations);
494
- // Strip date prefix for dedup comparison
495
- const stripDate = (s) => s.replace(/^\[\d{4}-\d{2}-\d{2}\]\s*/, '');
496
- const currentBare = new Set(currentObs.map(stripDate));
497
- const newObs = timestamped.filter(o => !currentBare.has(stripDate(o)));
498
- const needsTypeUpdate = entity.entityType && entity.entityType !== 'CONCEPT' && entity.entityType !== existing.entityType;
499
- if (newObs.length > 0 || needsTypeUpdate) {
500
- const mergedObs = [...currentObs, ...newObs];
501
- const updatedType = needsTypeUpdate ? entity.entityType : existing.entityType;
502
- this.mutateEntityAndInvalidate(entityId, () => {
503
- this.db.prepare(`UPDATE entities SET observations = ?, entityType = ? WHERE id = ?`)
504
- .run(JSON.stringify(mergedObs), updatedType, entityId);
505
- });
506
- console.error(`♻️ Upserted entity: ${entity.name} (+${newObs.length} obs${needsTypeUpdate ? ', type→' + updatedType : ''})`);
507
- const embedding_status = await this.tryEmbedEntity(entityId, 'bulk');
508
- result.push({ ...entity, observations: mergedObs, embedding_status });
509
- }
510
- }
631
+ result.push({ ...entity, observations: projected, created, observation_ids: ids });
511
632
  }
512
633
  }
513
634
  return result;
@@ -551,35 +672,144 @@ export class RAGKnowledgeGraphManager {
551
672
  const results = [];
552
673
  for (const obs of observations) {
553
674
  const entityId = `entity_${obs.entityName.toLowerCase().replace(/[^\p{L}\p{N}]/gu, '_')}`;
554
- // Get current observations
555
- const entity = this.db.prepare(`
556
- SELECT observations FROM entities WHERE id = ?
557
- `).get(entityId);
558
- if (!entity) {
675
+ const entity = this.db.prepare(`SELECT id FROM entities WHERE id = ?`).get(entityId);
676
+ if (!entity)
559
677
  throw new Error(`Entity with name ${obs.entityName} not found`);
560
- }
561
- const currentObservations = JSON.parse(entity.observations);
562
- const stripDate = (s) => s.replace(/^\[\d{4}-\d{2}-\d{2}\]\s*/, '');
563
- const currentBare = new Set(currentObservations.map(stripDate));
564
- const timestamped = obs.contents.map(c => this._timestampObservation(c));
565
- const newObservations = timestamped.filter(c => !currentBare.has(stripDate(c)));
566
- if (newObservations.length > 0) {
567
- const updatedObservations = [...currentObservations, ...newObservations];
568
- this.mutateEntityAndInvalidate(entityId, () => {
569
- this.db.prepare(`
570
- UPDATE entities SET observations = ? WHERE id = ?
571
- `).run(JSON.stringify(updatedObservations), entityId);
572
- });
573
- // Regenerate embedding for the updated entity (queued when not ready)
678
+ // dedup 기준은 v3.6 과 같다: 날짜 prefix 를 뗀 본문이 active 에 이미 있으면
679
+ // 새 revision 을 만들지 않는다. 다만 v13 에서는 같은 사실이 다른 출처에서 다시
680
+ // 온 것이므로 그 revision 에 source link 를 더한다(spec §8.3 T13).
681
+ const stripDate = (s2) => s2.replace(/^\[\d{4}-\d{2}-\d{2}\]\s*/, '');
682
+ const activeRows = this.db.prepare(`SELECT observation_id, content FROM entity_observations
683
+ WHERE entity_id = ? AND status = 'active'`).all(entityId);
684
+ const activeByBare = new Map(activeRows.map(r => [stripDate(r.content), r.observation_id]));
685
+ const ts = new Date().toISOString();
686
+ const ids = [];
687
+ const added = [];
688
+ let sourcesAdded = 0;
689
+ const changed = this.mutateEntityAndInvalidate(entityId, () => {
690
+ for (const raw of obs.contents) {
691
+ const content = this._timestampObservation(raw);
692
+ const bare = stripDate(content);
693
+ const dupId = activeByBare.get(bare);
694
+ if (dupId) {
695
+ if (obs.sources?.length)
696
+ sourcesAdded += linkSources(this.db, dupId, obs.sources, ts);
697
+ ids.push(null);
698
+ continue;
699
+ }
700
+ const id = addRevision(this.db, {
701
+ entityId, content, status: obs.status ?? 'active', sources: obs.sources, ts
702
+ });
703
+ activeByBare.set(bare, id);
704
+ ids.push(id);
705
+ added.push(content);
706
+ }
707
+ // 아무것도 안 바뀌었으면 projection·벡터·KG 를 건드리지 않는다.
708
+ // 이 반환이 없으면 빈 contents 나 dedup-only add 가 정상 벡터를
709
+ // 지우고 재임베딩도 안 한다(advisor 구현리뷰 r1 발견 1).
710
+ return added.length > 0 || sourcesAdded > 0;
711
+ });
712
+ let embedding_status;
713
+ if (changed) {
574
714
  console.error(`🔮 Regenerating embedding for updated entity: ${obs.entityName}`);
575
- const embedding_status = await this.tryEmbedEntity(entityId, 'bulk');
576
- results.push({ entityName: obs.entityName, addedObservations: newObservations, embedding_status });
577
- continue;
715
+ embedding_status = await this.tryEmbedEntity(entityId, 'bulk');
578
716
  }
579
- results.push({ entityName: obs.entityName, addedObservations: newObservations });
717
+ results.push({ entityName: obs.entityName, observation_ids: ids,
718
+ addedObservations: added, embedding_status });
580
719
  }
581
720
  return results;
582
721
  }
722
+ async correctObservation(observationId, content, changeKind = 'correction', reason) {
723
+ if (!this.db)
724
+ throw new Error('Database not initialized');
725
+ const row = this.db.prepare(`SELECT entity_id FROM entity_observations WHERE observation_id = ?`)
726
+ .get(observationId);
727
+ if (!row)
728
+ throw new Error(`observation ${observationId} not found`);
729
+ let newId = '';
730
+ const ts = new Date().toISOString();
731
+ this.mutateEntityAndInvalidate(row.entity_id, () => {
732
+ newId = correctRevision(this.db, {
733
+ observationId, content: this._timestampObservation(content),
734
+ changeKind, reason: reason ?? null, ts
735
+ });
736
+ });
737
+ await this.tryEmbedEntity(row.entity_id, 'bulk');
738
+ return newId;
739
+ }
740
+ async _transition(observationId, event, reason) {
741
+ if (!this.db)
742
+ throw new Error('Database not initialized');
743
+ const row = this.db.prepare(`SELECT entity_id FROM entity_observations WHERE observation_id = ?`)
744
+ .get(observationId);
745
+ if (!row)
746
+ throw new Error(`observation ${observationId} not found`);
747
+ const ts = new Date().toISOString();
748
+ this.mutateEntityAndInvalidate(row.entity_id, () => {
749
+ transitionStatus(this.db, { observationId, event, reason: reason ?? null, ts });
750
+ });
751
+ await this.tryEmbedEntity(row.entity_id, 'bulk');
752
+ }
753
+ async retractObservation(observationId, reason) {
754
+ return this._transition(observationId, 'retract', reason);
755
+ }
756
+ async restoreObservation(observationId, reason) {
757
+ return this._transition(observationId, 'restore', reason);
758
+ }
759
+ async approveObservation(observationId, reason) {
760
+ return this._transition(observationId, 'approve', reason);
761
+ }
762
+ async declineObservation(observationId, reason) {
763
+ return this._transition(observationId, 'decline', reason);
764
+ }
765
+ // DESTRUCTIVE. Physically removes revisions (and their sources via CASCADE).
766
+ //
767
+ // Chain contract (advisor 구현리뷰 r1 발견 4): a revision chain is
768
+ // rev1 <- rev2 <- ... and purging a middle revision would either fail on the
769
+ // supersedes_id FK or leave a chain pointing at a deleted row, plus events
770
+ // whose from_id/to_id dangle. So purge is defined as **suffix purge from the
771
+ // target to the newest revision of that root**, newest-first:
772
+ // - purging the newest revision removes exactly it
773
+ // - purging rev2 of a 3-revision chain removes rev3 then rev2
774
+ // - purging rev1 removes the whole chain
775
+ // Events for purged revisions are removed too, so no event dangles.
776
+ // The root row is always kept: its projection_order stays reserved, because
777
+ // reusing an order would make a later restore/approve fail on the
778
+ // active-order index.
779
+ async purgeObservation(observationId, confirm) {
780
+ if (!this.db)
781
+ throw new Error('Database not initialized');
782
+ if (confirm !== 'PURGE') {
783
+ throw new Error(`purgeObservation refused: pass confirm='PURGE' to physically delete a revision. ` +
784
+ `This destroys history — retractObservation() is almost always what you want.`);
785
+ }
786
+ const row = this.db.prepare(`SELECT entity_id, root_id, revision_no FROM entity_observations WHERE observation_id = ?`)
787
+ .get(observationId);
788
+ if (!row)
789
+ return { purged: 0 };
790
+ let purged = 0;
791
+ this.mutateEntityAndInvalidate(row.entity_id, () => {
792
+ // newest-first so each DELETE has no successor referencing it
793
+ const victims = this.db.prepare(`SELECT observation_id FROM entity_observations
794
+ WHERE root_id = ? AND revision_no >= ?
795
+ ORDER BY revision_no DESC`).all(row.root_id, row.revision_no);
796
+ for (const v of victims) {
797
+ this.db.prepare(`DELETE FROM observation_events WHERE from_id = ? OR to_id = ?`)
798
+ .run(v.observation_id, v.observation_id);
799
+ purged += this.db.prepare(`DELETE FROM entity_observations WHERE observation_id = ?`)
800
+ .run(v.observation_id).changes;
801
+ }
802
+ return purged > 0;
803
+ });
804
+ await this.tryEmbedEntity(row.entity_id, 'bulk');
805
+ return { purged };
806
+ }
807
+ // spec §6.2: 과거 판본은 여기서만 나온다. 일반 검색은 active 만 반환한다.
808
+ async getObservationHistory(sel) {
809
+ if (!this.db)
810
+ throw new Error('Database not initialized');
811
+ return getObservationHistory(this.db, sel);
812
+ }
583
813
  async deleteEntities(entityNames) {
584
814
  if (!this.db)
585
815
  throw new Error('Database not initialized');
@@ -647,34 +877,85 @@ export class RAGKnowledgeGraphManager {
647
877
  // Pre-3.6 this method silently left STALE entity vectors behind (the input
648
878
  // text changed but the vector was never regenerated) — fixed via
649
879
  // tryEmbedEntity, which also covers the not-ready dirty contract.
880
+ // DEPRECATED shim (v13, one version only). Content-addressed deletion cannot
881
+ // express "which revision" — use retractObservation(observation_id) instead.
882
+ // Semantics: exact-string match against ACTIVE revisions -> soft retract.
883
+ //
884
+ // The whole call is one transaction and ambiguity is judged before any
885
+ // mutation: if any item matches 2+ active revisions the call aborts with 0
886
+ // mutations (spec §6.3). That is a deliberate change from v3.6, which deleted
887
+ // every duplicate and carried on — a machine cannot pick which revision was meant.
888
+ //
889
+ // Duplicate ids across items are collapsed. Without that, listing the same
890
+ // (entity, content) twice retracted it once and then failed on an illegal
891
+ // transition, returning an error *after* committing part of the batch
892
+ // (advisor 구현리뷰 r1 발견 3, 실행 재현). Embedding runs after the commit.
650
893
  async deleteObservations(deletions) {
651
894
  if (!this.db)
652
895
  throw new Error('Database not initialized');
896
+ const plan = [];
897
+ const ambiguous = [];
898
+ const claimed = new Set(); // 항목 간 중복 id 흡수
899
+ for (const d of deletions) {
900
+ const entityId = `entity_${d.entityName.toLowerCase().replace(/[^\p{L}\p{N}]/gu, '_')}`;
901
+ const ids = [];
902
+ for (const content of d.observations) {
903
+ const rows = this.db.prepare(`SELECT observation_id FROM entity_observations
904
+ WHERE entity_id = ? AND status = 'active' AND content = ?`).all(entityId, content);
905
+ if (rows.length > 1) {
906
+ ambiguous.push({ entityName: d.entityName, content, matches: rows.length });
907
+ continue;
908
+ }
909
+ if (rows.length === 1 && !claimed.has(rows[0].observation_id)) {
910
+ claimed.add(rows[0].observation_id);
911
+ ids.push(rows[0].observation_id);
912
+ }
913
+ // rows.length === 0 -> no-op (v3.6 behaviour, spec §6.3-3)
914
+ }
915
+ plan.push({ entityName: d.entityName, entityId, ids });
916
+ }
917
+ if (ambiguous.length > 0) {
918
+ throw new Error(`AMBIGUOUS_OBSERVATION_MATCH: ${ambiguous.length} item(s) matched multiple active ` +
919
+ `revisions; 0 mutations were applied. Use retractObservation(observation_id) instead. ` +
920
+ `Conflicts: ${JSON.stringify(ambiguous)}`);
921
+ }
922
+ // pass 2 — mutate everything in ONE transaction so a failure anywhere
923
+ // leaves zero mutations. Per-plan transactions plus an awaited embedding
924
+ // in between made a partial commit observable.
925
+ const touched = plan.filter(p => p.ids.length > 0);
926
+ const ts = new Date().toISOString();
927
+ if (touched.length > 0) {
928
+ const tx = this.db.transaction(() => {
929
+ for (const p of touched) {
930
+ for (const id of p.ids) {
931
+ transitionStatus(this.db, { observationId: id, event: 'retract',
932
+ reason: 'deleteObservations (deprecated shim)', ts });
933
+ }
934
+ rebuildProjection(this.db, p.entityId);
935
+ const meta = this.db.prepare(`SELECT rowid FROM entity_embedding_metadata WHERE entity_id = ?`)
936
+ .get(p.entityId);
937
+ if (meta) {
938
+ this.db.exec(`DELETE FROM entity_embeddings WHERE rowid = ${Number(meta.rowid)}`);
939
+ this.db.prepare(`DELETE FROM entity_embedding_metadata WHERE entity_id = ?`)
940
+ .run(p.entityId);
941
+ }
942
+ deleteStaleKgChunks(this.db, p.entityId);
943
+ }
944
+ });
945
+ tx();
946
+ this.coordinator?.invalidateCoverage();
947
+ }
948
+ // pass 3 — embedding after the commit
653
949
  const results = [];
654
950
  let total = 0;
655
- for (const deletion of deletions) {
656
- const entityId = `entity_${deletion.entityName.toLowerCase().replace(/[^\p{L}\p{N}]/gu, '_')}`;
657
- const entity = this.db.prepare(`
658
- SELECT observations FROM entities WHERE id = ?
659
- `).get(entityId);
660
- if (!entity) {
661
- results.push({ entityName: deletion.entityName, deleted: 0, embedding_status: 'n/a' });
662
- continue;
663
- }
664
- const currentObservations = JSON.parse(entity.observations);
665
- const filteredObservations = currentObservations.filter((obs) => !deletion.observations.includes(obs));
666
- const deleted = currentObservations.length - filteredObservations.length;
667
- if (deleted === 0) {
668
- results.push({ entityName: deletion.entityName, deleted: 0, embedding_status: 'n/a' });
951
+ for (const p of plan) {
952
+ if (p.ids.length === 0) {
953
+ results.push({ entityName: p.entityName, deleted: 0, embedding_status: 'n/a' });
669
954
  continue;
670
955
  }
671
- this.mutateEntityAndInvalidate(entityId, () => {
672
- this.db.prepare(`UPDATE entities SET observations = ? WHERE id = ?`)
673
- .run(JSON.stringify(filteredObservations), entityId);
674
- });
675
- const embedding_status = await this.tryEmbedEntity(entityId, 'bulk');
676
- total += deleted;
677
- results.push({ entityName: deletion.entityName, deleted, embedding_status });
956
+ const embedding_status = await this.tryEmbedEntity(p.entityId, 'bulk');
957
+ results.push({ entityName: p.entityName, deleted: p.ids.length, embedding_status });
958
+ total += p.ids.length;
678
959
  }
679
960
  return { results, total_deleted: total };
680
961
  }
@@ -1597,21 +1878,32 @@ export class RAGKnowledgeGraphManager {
1597
1878
  // CJK), so the function maintains parallel UTF-16 and codepoint cursors and
1598
1879
  // reports codepoint offsets. On a coincidental indexOf miss the char offsets
1599
1880
  // are NULL.
1600
- chunkText(text, maxTokens = 800, overlap = 160) {
1881
+ chunkStructured(text, maxTokens = DEFAULT_MAX_TOKENS) {
1601
1882
  if (!this.encoding)
1602
1883
  throw new Error('Tokenizer not initialized');
1603
- const segments = splitTextIntoChunks(text, this.encoding, maxTokens, overlap);
1604
- return segments.map((seg, idx) => ({
1884
+ return chunkStructuredText(text, this.encoding, maxTokens).map((seg, idx) => ({
1605
1885
  id: '',
1606
1886
  document_id: '',
1607
1887
  chunk_index: idx,
1608
1888
  text: seg.text,
1609
1889
  start_pos: seg.start_pos,
1610
1890
  end_pos: seg.end_pos,
1611
- start_token: seg.start_token,
1612
- end_token: seg.end_token
1891
+ // c1 has no token-space offsets (spec §4.3, r4 D4). Legacy rows keep theirs.
1892
+ start_token: null,
1893
+ end_token: null
1613
1894
  }));
1614
1895
  }
1896
+ // spec §7.1 (r4·r5-8): overlap is rejected on BOTH public paths, BEFORE any
1897
+ // content/dedup judgment — silently accepting it on unchanged content would
1898
+ // void the contract. maxTokens must be a positive integer.
1899
+ validateChunkParams(params) {
1900
+ const { maxTokens = DEFAULT_MAX_TOKENS, overlap = 0 } = params || {};
1901
+ if (!Number.isInteger(maxTokens) || maxTokens <= 0)
1902
+ throw new Error(`chunkParams.maxTokens must be a positive integer (got ${maxTokens})`);
1903
+ if (overlap !== 0)
1904
+ throw new Error(`chunkParams.overlap is no longer supported (chunker c1 has no overlap); omit it or pass 0 (got ${overlap})`);
1905
+ return { maxTokens };
1906
+ }
1615
1907
  // Generate embeddings using sentence transformers
1616
1908
  // isQuery: true for search queries (adds instruction prefix), false for documents/entities
1617
1909
  async generateEmbedding(text, dimensions = 1024, isQuery = false, priority = 'interactive') {
@@ -1636,31 +1928,32 @@ export class RAGKnowledgeGraphManager {
1636
1928
  async syncDocumentFromFile(filePath, documentId, options = {}) {
1637
1929
  if (!this.db)
1638
1930
  throw new Error('Database not initialized');
1639
- // 1. Resolve content: raw file verbatim (default) or explicit override.
1640
- // Content is read on the server and never routed through the model context.
1641
- const content = options.content !== undefined
1642
- ? options.content
1643
- : fsSync.readFileSync(filePath, 'utf-8');
1644
- const bytes = Buffer.byteLength(content, 'utf-8');
1645
- // 2. Metadata: default source=path, updated=today, content_hash; caller can override.
1646
- const today = new Date().toISOString().slice(0, 10);
1647
- const contentHash = createHash('sha256').update(content).digest('hex');
1648
- const metadata = { source: filePath, updated: today, content_hash: contentHash, ...(options.metadata || {}) };
1649
- // 2b. Dedup gate: skip the full delete/store/chunk/embed pipeline when the
1650
- // file is unchanged AND the existing document is fully embedded. The
1651
- // completeness check avoids wrongly skipping a partial/failed prior sync.
1652
- const existingDoc = this.db.prepare(`SELECT metadata FROM documents WHERE id = ?`).get(documentId);
1653
- if (existingDoc) {
1931
+ // spec §7.1 + r5-8: 검증은 content 해석·dedup 판정보다 앞 (첫 실행문).
1932
+ const { maxTokens } = this.validateChunkParams(options.chunkParams);
1933
+ const signature = effectiveSignature(maxTokens);
1934
+ const shaHex = (t) => createHash('sha256').update(t).digest('hex');
1935
+ const zero = { reusedChunks: 0, newlyEmbeddedChunks: 0, queuedChunks: 0, deletedChunks: 0, chunkerTransitioned: false };
1936
+ for (let attempt = 1; attempt <= 3; attempt++) {
1937
+ // r6-3: CAS 재시작 = 처음부터 — 파일 읽기·hash·metadata 도 attempt 안에서 재계산한다.
1938
+ const content = options.content !== undefined ? options.content : fsSync.readFileSync(filePath, 'utf-8');
1939
+ const bytes = Buffer.byteLength(content, 'utf-8');
1940
+ const today = new Date().toISOString().slice(0, 10);
1941
+ const contentHash = shaHex(content);
1942
+ // spec §5.1: content_hash 는 system-owned — user metadata 뒤에 쓴다 (r1: spread 가 덮어쓸 수 있었다).
1943
+ const metadata = { source: filePath, updated: today, ...(options.metadata || {}), content_hash: contentHash };
1944
+ const snap = this.db.prepare(`SELECT content, metadata, chunking_signature FROM documents WHERE id = ?`)
1945
+ .get(documentId);
1654
1946
  let existingHash;
1655
- try {
1656
- existingHash = JSON.parse(existingDoc.metadata)?.content_hash;
1947
+ if (snap) {
1948
+ try {
1949
+ existingHash = JSON.parse(snap.metadata)?.content_hash;
1950
+ }
1951
+ catch { /* hash 없으면 full 경로 */ }
1657
1952
  }
1658
- catch { /* ignore */ }
1659
- if (existingHash === contentHash) {
1953
+ // dedup gate — spec §5.1 그대로: "content_hash 동일" 만 (r6-8: content=== 확장 금지.
1954
+ // hash 가 없거나 낡은 문서는 full 경로로 가서 hash 가 복구된다). signature 무관.
1955
+ if (snap && existingHash === contentHash) {
1660
1956
  const cmCount = this.db.prepare(`SELECT count(*) AS n FROM chunk_metadata WHERE document_id = ?`).get(documentId).n;
1661
- // "Embedded" for dedup completeness = vector exists AND its profile is
1662
- // current (or legacy-NULL awaiting grandfather). Raw vector counts
1663
- // would misjudge old-profile rows as complete (beta 1R supplement).
1664
1957
  const embCount = this.db.prepare(`
1665
1958
  SELECT count(*) AS n FROM chunks c JOIN chunk_metadata m ON c.rowid = m.rowid
1666
1959
  WHERE m.document_id = ? AND (m.provenance_state IS NULL OR m.profile_id = ?)
@@ -1671,100 +1964,141 @@ export class RAGKnowledgeGraphManager {
1671
1964
  `).get(documentId).n;
1672
1965
  if (cmCount > 0 && cmCount === embCount) {
1673
1966
  console.error(`⏭️ syncDocumentFromFile: ${documentId} unchanged (hash match, ${cmCount} chunks embedded) — skipped`);
1674
- return { documentId, bytes, chunks: cmCount, embeddedChunks: embCount, linkedEntities: linked, skipped: true, reason: 'unchanged' };
1967
+ return { documentId, bytes, chunks: cmCount, embeddedChunks: embCount, linkedEntities: linked,
1968
+ skipped: true, reason: 'unchanged', ...zero };
1675
1969
  }
1676
1970
  if (cmCount > 0 && embCount < cmCount) {
1677
- // v3.6 (spec §5b M12): identical content with incomplete/stale vectors
1678
- // keeps the document, chunks, rowids and entity links — only the
1679
- // missing vectors are re-queued via the coordinator. Full re-chunking
1680
- // here would churn rowids and links for no content change.
1971
+ // v3.6 (spec §5b M12): identical content with incomplete/stale vectors keeps the
1972
+ // document, chunks, rowids and entity links — only missing vectors are re-queued.
1681
1973
  console.error(`♻️ syncDocumentFromFile: ${documentId} unchanged but ${cmCount - embCount} vectors missing — re-queued (chunks preserved)`);
1682
1974
  this.coordinator?.kick();
1683
- return { documentId, bytes, chunks: cmCount, embeddedChunks: embCount, linkedEntities: linked, skipped: true, reason: 'unchanged-revectorizing', embedding_status: this.gate.isDisabled ? 'disabled' : 'queued' };
1975
+ return { documentId, bytes, chunks: cmCount, embeddedChunks: embCount, linkedEntities: linked,
1976
+ skipped: true, reason: 'unchanged-revectorizing',
1977
+ embedding_status: this.gate.isDisabled ? 'disabled' : 'queued',
1978
+ ...zero, queuedChunks: cmCount - embCount };
1684
1979
  }
1980
+ // cmCount === 0 이면 아래 full 경로로 계속 (최초 생성).
1981
+ }
1982
+ console.error(`🔄 syncDocumentFromFile: ${documentId} <- ${filePath} (${bytes} bytes)`);
1983
+ const segments = this.chunkStructured(content, maxTokens);
1984
+ // spec §5.2-2: 옛 행을 트랜잭션 밖에서 읽는다 (벡터 재사용 후보).
1985
+ const oldRows = this.db.prepare(`
1986
+ SELECT m.rowid, m.text, m.input_hash, m.profile_id, m.provenance_state, c.embedding
1987
+ FROM chunk_metadata m LEFT JOIN chunks c ON c.rowid = m.rowid
1988
+ WHERE m.document_id = ?`).all(documentId);
1989
+ const oldRowids = oldRows.map(r => r.rowid);
1990
+ const byHash = new Map();
1991
+ for (const r of oldRows) {
1992
+ if (!r.input_hash)
1993
+ continue;
1994
+ const arr = byHash.get(r.input_hash);
1995
+ if (arr)
1996
+ arr.push(r);
1997
+ else
1998
+ byHash.set(r.input_hash, [r]);
1999
+ }
2000
+ // 임베딩/재사용 — 트랜잭션 밖, ready 경로 한정 (N2: not-ready 계약 불변).
2001
+ const lazySync = !this.gate.isReady;
2002
+ const slots = [];
2003
+ let reusedChunks = 0, newlyEmbeddedChunks = 0;
2004
+ if (lazySync) {
2005
+ for (const seg of segments)
2006
+ slots.push({ seg, vec: null, provenance: null });
1685
2007
  }
1686
- }
1687
- console.error(`🔄 syncDocumentFromFile: ${documentId} <- ${filePath} (${bytes} bytes)`);
1688
- // 3. Two contracts (spec §5b):
1689
- // ready — pre-compute ALL embeddings BEFORE any DB mutation; if
1690
- // inference throws mid-way the old document stays intact
1691
- // (v3.5.0 atomicity, unchanged).
1692
- // not-ready — intentional lazy sync: store document + chunks + FTS in
1693
- // one transaction with NO vectors (embedding_status:
1694
- // queued); the backfill coordinator recovers them.
1695
- const { maxTokens = 800, overlap = 160 } = options.chunkParams || {};
1696
- const segments = this.chunkText(content, maxTokens, overlap);
1697
- const lazySync = !this.gate.isReady;
1698
- const embedded = [];
1699
- if (lazySync) {
1700
- for (const seg of segments)
1701
- embedded.push({ seg, embedding: null });
1702
- }
1703
- else {
1704
- for (const seg of segments) {
1705
- const embedding = await this.generateEmbedding(seg.text, 1024, false, 'bulk');
1706
- embedded.push({ seg, embedding });
1707
- }
1708
- }
1709
- // 4. Atomic swap: delete old -> insert doc -> insert chunks (+ embeddings
1710
- // with verified provenance when ready), one synchronous transaction.
1711
- const applyTx = this.db.transaction(() => {
1712
- const db = this.db;
1713
- // 4a. cleanup old doc (inlined sync version of cleanupDocument).
1714
- const existing = db.prepare(`SELECT rowid FROM chunk_metadata WHERE document_id = ?`).all(documentId);
1715
- for (const ch of existing) {
1716
- db.prepare(`DELETE FROM chunk_entities WHERE chunk_rowid = ?`).run(ch.rowid);
1717
- db.exec(`DELETE FROM chunks WHERE rowid = ${safeRowid(ch.rowid)}`);
1718
- }
1719
- db.prepare(`DELETE FROM chunk_metadata WHERE document_id = ?`).run(documentId);
1720
- db.prepare(`DELETE FROM documents WHERE id = ?`).run(documentId);
1721
- // 4b. insert document.
1722
- db.prepare(`INSERT INTO documents (id, content, metadata) VALUES (?, ?, ?)`)
1723
- .run(documentId, content, JSON.stringify(metadata));
1724
- // 4c. insert chunk_metadata (FTS5 chunks_fts auto-filled by trigger);
1725
- // vectors + provenance only on the ready path (§6a-2).
1726
- for (const { seg, embedding } of embedded) {
1727
- const chunkId = `${documentId}_chunk_${seg.chunk_index}`;
1728
- const info = db.prepare(`
1729
- INSERT INTO chunk_metadata (chunk_id, document_id, chunk_index, text, start_pos, end_pos, start_token, end_token)
1730
- VALUES (?, ?, ?, ?, ?, ?, ?, ?)
1731
- `).run(chunkId, documentId, seg.chunk_index, seg.text, seg.start_pos, seg.end_pos, seg.start_token, seg.end_token);
1732
- const rowid = Number(info.lastInsertRowid);
1733
- if (embedding) {
1734
- db.prepare(`INSERT INTO chunks (rowid, embedding) VALUES (${rowid}, ?)`).run(Buffer.from(embedding.buffer));
1735
- db.prepare(`UPDATE chunk_metadata SET input_hash = ?, profile_id = ?, provenance_state = 'verified' WHERE rowid = ?`)
1736
- .run(createHash('sha256').update(seg.text).digest('hex'), this.currentProfileId, rowid);
2008
+ else {
2009
+ for (const seg of segments) {
2010
+ const hit = selectReusableVector(byHash.get(shaHex(seg.text)) ?? [], seg.text, this.currentProfileId, shaHex);
2011
+ if (hit) {
2012
+ slots.push({ seg, vec: hit.vec, provenance: hit.provenance });
2013
+ reusedChunks++;
2014
+ }
2015
+ else {
2016
+ const embedding = await this.generateEmbedding(seg.text, 1024, false, 'bulk');
2017
+ slots.push({ seg, vec: Buffer.from(embedding.buffer), provenance: 'verified' });
2018
+ newlyEmbeddedChunks++;
2019
+ }
1737
2020
  }
1738
2021
  }
1739
- });
1740
- applyTx();
1741
- this.coordinator?.invalidateCoverage();
1742
- if (lazySync)
1743
- this.coordinator?.kick();
1744
- const embeddedChunks = lazySync ? 0 : embedded.length;
1745
- // 5. Entity linking AFTER commit. Non-destructive + idempotent (INSERT OR
1746
- // IGNORE), so a linking failure cannot corrupt the doc/embeddings.
1747
- const linkedEntities = await this.autoLinkEntities(documentId);
1748
- let explicitlyLinked;
1749
- if (options.entityNames && options.entityNames.length > 0) {
1750
- const linkResult = await this.linkEntitiesToDocument(documentId, options.entityNames);
1751
- explicitlyLinked = linkResult.linkedEntities;
1752
- }
1753
- // 6. Terse summary only (no chunk text / content echo) to keep caller context flat.
1754
- const result = {
1755
- documentId,
1756
- bytes,
1757
- chunks: segments.length,
1758
- embeddedChunks,
1759
- linkedEntities,
1760
- embedding_status: lazySync ? (this.gate.isDisabled ? 'disabled' : 'queued') : 'embedded',
1761
- ...(explicitlyLinked !== undefined ? { explicitlyLinked } : {}),
1762
- };
1763
- if (linkedEntities === 0 && explicitlyLinked === undefined) {
1764
- result.warning = 'linkedEntities=0: ensure the file content contains entity-name literals (e.g. a wiki anchor line "RAG entity: ...") so term-matching can link entities.';
2022
+ __syncFaultHook?.('pre-transaction');
2023
+ // 한 트랜잭션: CAS 첫 문장 -> full delete/insert -> failure 정리 (spec §5.2-4·5).
2024
+ const applyTx = this.db.transaction(() => {
2025
+ const db = this.db;
2026
+ const now = db.prepare(`SELECT content, metadata, chunking_signature FROM documents WHERE id = ?`)
2027
+ .get(documentId);
2028
+ const same = (snap === undefined && now === undefined) ||
2029
+ (snap !== undefined && now !== undefined && now.content === snap.content &&
2030
+ now.metadata === snap.metadata && now.chunking_signature === snap.chunking_signature);
2031
+ if (!same)
2032
+ throw new SyncCasConflictError(documentId);
2033
+ const existing = db.prepare(`SELECT rowid FROM chunk_metadata WHERE document_id = ?`).all(documentId);
2034
+ for (const ch of existing) {
2035
+ db.prepare(`DELETE FROM chunk_entities WHERE chunk_rowid = ?`).run(ch.rowid);
2036
+ db.exec(`DELETE FROM chunks WHERE rowid = ${safeRowid(ch.rowid)}`);
2037
+ }
2038
+ db.prepare(`DELETE FROM chunk_metadata WHERE document_id = ?`).run(documentId);
2039
+ db.prepare(`DELETE FROM documents WHERE id = ?`).run(documentId);
2040
+ db.prepare(`INSERT INTO documents (id, content, metadata, chunking_signature) VALUES (?, ?, ?, ?)`)
2041
+ .run(documentId, content, JSON.stringify(metadata), signature);
2042
+ for (const { seg, vec, provenance } of slots) {
2043
+ const chunkId = `${documentId}_chunk_${seg.chunk_index}`;
2044
+ const info = db.prepare(`
2045
+ INSERT INTO chunk_metadata (chunk_id, document_id, chunk_index, text, start_pos, end_pos, start_token, end_token)
2046
+ VALUES (?, ?, ?, ?, ?, ?, ?, ?)
2047
+ `).run(chunkId, documentId, seg.chunk_index, seg.text, seg.start_pos, seg.end_pos, seg.start_token, seg.end_token);
2048
+ const rowid = Number(info.lastInsertRowid);
2049
+ if (vec) {
2050
+ db.prepare(`INSERT INTO chunks (rowid, embedding) VALUES (${rowid}, ?)`).run(vec);
2051
+ db.prepare(`UPDATE chunk_metadata SET input_hash = ?, profile_id = ?, provenance_state = ? WHERE rowid = ?`)
2052
+ .run(shaHex(seg.text), this.currentProfileId, provenance, rowid);
2053
+ }
2054
+ }
2055
+ if (oldRowids.length > 0) {
2056
+ // r5-9: 키는 (kind, target_id) — kind 조건 없이 지우면 같은 숫자 ID 의 entity failure 까지 지운다.
2057
+ const ph = oldRowids.map(() => '?').join(',');
2058
+ db.prepare(`DELETE FROM embedding_backfill_failures WHERE kind = 'chunk' AND target_id IN (${ph})`)
2059
+ .run(...oldRowids.map(String));
2060
+ }
2061
+ });
2062
+ try {
2063
+ applyTx();
2064
+ }
2065
+ catch (e) {
2066
+ if (e instanceof SyncCasConflictError) {
2067
+ console.error(`↻ sync CAS conflict on ${documentId} (attempt ${attempt}/3) — restarting from file read`);
2068
+ if (attempt === 3)
2069
+ throw e;
2070
+ continue;
2071
+ }
2072
+ throw e;
2073
+ }
2074
+ this.coordinator?.invalidateCoverage();
2075
+ if (lazySync)
2076
+ this.coordinator?.kick();
2077
+ // Entity linking AFTER commit. Non-destructive + idempotent (INSERT OR IGNORE).
2078
+ const linkedEntities = await this.autoLinkEntities(documentId);
2079
+ let explicitlyLinked;
2080
+ if (options.entityNames && options.entityNames.length > 0) {
2081
+ const linkResult = await this.linkEntitiesToDocument(documentId, options.entityNames);
2082
+ explicitlyLinked = linkResult.linkedEntities;
2083
+ }
2084
+ const result = {
2085
+ documentId, bytes, chunks: segments.length,
2086
+ embeddedChunks: reusedChunks + newlyEmbeddedChunks, // spec §5.3
2087
+ linkedEntities,
2088
+ embedding_status: lazySync ? (this.gate.isDisabled ? 'disabled' : 'queued') : 'embedded',
2089
+ reusedChunks, newlyEmbeddedChunks,
2090
+ queuedChunks: lazySync ? segments.length : 0,
2091
+ deletedChunks: oldRowids.length,
2092
+ chunkerTransitioned: snap !== undefined && snap.chunking_signature !== signature,
2093
+ ...(explicitlyLinked !== undefined ? { explicitlyLinked } : {}),
2094
+ };
2095
+ if (linkedEntities === 0 && explicitlyLinked === undefined) {
2096
+ result.warning = 'linkedEntities=0: ensure the file content contains entity-name literals (e.g. a wiki anchor line "RAG entity: ...") so term-matching can link entities.';
2097
+ }
2098
+ console.error(`✅ syncDocumentFromFile done: ${documentId} (${result.chunks} chunks, reused ${reusedChunks}, embedded ${newlyEmbeddedChunks})`);
2099
+ return result;
1765
2100
  }
1766
- console.error(`✅ syncDocumentFromFile done: ${documentId} (${result.chunks} chunks, ${result.embeddedChunks} embedded, ${linkedEntities} linked)`);
1767
- return result;
2101
+ throw new Error('unreachable');
1768
2102
  }
1769
2103
  async storeDocument(id, content, metadata = {}) {
1770
2104
  if (!this.db)
@@ -1790,12 +2124,12 @@ export class RAGKnowledgeGraphManager {
1790
2124
  if (!document) {
1791
2125
  throw new Error(`Document with ID ${documentId} not found`);
1792
2126
  }
1793
- const { maxTokens = 800, overlap = 160 } = options;
1794
- console.error(`🔪 Chunking document: ${documentId} (maxTokens: ${maxTokens}, overlap: ${overlap})`);
2127
+ const { maxTokens } = this.validateChunkParams(options);
2128
+ console.error(`🔪 Chunking document: ${documentId} (maxTokens: ${maxTokens}, chunker: c1)`);
1795
2129
  // Clean up existing chunks
1796
2130
  await this.cleanupDocument(documentId);
1797
2131
  // Create chunks
1798
- const chunks = this.chunkText(document.content, maxTokens, overlap);
2132
+ const chunks = this.chunkStructured(document.content, maxTokens);
1799
2133
  const resultChunks = [];
1800
2134
  for (const chunk of chunks) {
1801
2135
  const chunkId = `${documentId}_chunk_${chunk.chunk_index}`;
@@ -1815,6 +2149,9 @@ export class RAGKnowledgeGraphManager {
1815
2149
  });
1816
2150
  }
1817
2151
  console.error(`✅ Document chunked: ${chunks.length} chunks created`);
2152
+ // spec §7.1: 두 번째 chunk 생성 경로 — 스탬프를 안 박으면 §5.1 관측이 조용히 샌다.
2153
+ this.db.prepare(`UPDATE documents SET chunking_signature = ? WHERE id = ?`)
2154
+ .run(effectiveSignature(maxTokens), documentId);
1818
2155
  // Indirect missing-row producer (spec §5): freshly chunked rows have no
1819
2156
  // vectors yet — let the coordinator recover them without a restart.
1820
2157
  this.coordinator?.invalidateCoverage();
@@ -1867,6 +2204,73 @@ export class RAGKnowledgeGraphManager {
1867
2204
  hasCJK(text) {
1868
2205
  return /[\u3000-\u9fff\uac00-\ud7af\uff00-\uffef]/.test(text);
1869
2206
  }
2207
+ // spec §5.4 (r7-2·r8-1·r9): primary name 의 본문 occurrence range [sCp, eCp).
2208
+ // 의미 = buildEntityMatcher 와 동일 (CJK substring / Latin word-boundary) — 여기서
2209
+ // 어긋나면 'Data' 가 'Database' 에 새로 링크되는 식으로 의미가 확장된다.
2210
+ buildEntityRangeFinder(content) {
2211
+ // 원문 UTF-16 -> codepoint 표. Latin 경로는 folded 가 아니라 **원문**에 regex 를 건다
2212
+ // (r9-1: folded 에 걸면 fooİ -> fooi̇ 로 접힌 뒤 매치돼 현행 matcher 의미가 확장된다).
2213
+ const origU16ToCp = [];
2214
+ let origTotalCp = 0;
2215
+ for (let u = 0; u < content.length;) {
2216
+ const c = content.codePointAt(u);
2217
+ origU16ToCp.push(origTotalCp);
2218
+ if (c > 0xffff) {
2219
+ origU16ToCp.push(origTotalCp);
2220
+ u += 2;
2221
+ }
2222
+ else
2223
+ u += 1;
2224
+ origTotalCp++;
2225
+ }
2226
+ const origCpAt = (u16) => (u16 < origU16ToCp.length ? origU16ToCp[u16] : origTotalCp);
2227
+ // folded 표 (CJK substring / fallback 경로 전용). unit -> 유래한 원문 cp.
2228
+ let folded = '';
2229
+ const u16ToCp = [];
2230
+ let cp = 0;
2231
+ for (const ch of content) { // for..of = codepoint 순회
2232
+ const f = ch.toLowerCase(); // 다단위 fold 가능 (İ -> 'i̇')
2233
+ for (let i = 0; i < f.length; i++)
2234
+ u16ToCp.push(cp);
2235
+ folded += f;
2236
+ cp++;
2237
+ }
2238
+ // r9-1: exclusive end = "마지막으로 소비한 unit 의 원문 cp + 1".
2239
+ // 경계 unit 을 읽으면 매치가 fold 전개 중간에서 끝날 때 1 모자란다 (漢İ/漢i 실측 [0,1)).
2240
+ const endCp = (u16) => (u16 === 0 ? 0 : u16ToCp[Math.min(u16, u16ToCp.length) - 1] + 1);
2241
+ return (name, isCjk) => {
2242
+ const out = [];
2243
+ const lower = name.toLowerCase();
2244
+ const pushAllSubstr = () => {
2245
+ let from = 0;
2246
+ while (true) {
2247
+ const u = folded.indexOf(lower, from);
2248
+ if (u < 0)
2249
+ break;
2250
+ out.push({ s: u16ToCp[u], e: endCp(u + lower.length) });
2251
+ from = u + 1; // r9-2: 중첩 occurrence 보존
2252
+ }
2253
+ };
2254
+ if (isCjk) {
2255
+ pushAllSubstr();
2256
+ return out;
2257
+ }
2258
+ try {
2259
+ const escaped = lower.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
2260
+ const re = new RegExp(`\\b${escaped}\\b`, 'gi'); // buildEntityMatcher 와 동일 규칙,
2261
+ let m; // 단 원문에 실행 (의미 확장 방지)
2262
+ while ((m = re.exec(content)) !== null) {
2263
+ out.push({ s: origCpAt(m.index), e: origCpAt(m.index + m[0].length) });
2264
+ if (re.lastIndex === m.index)
2265
+ re.lastIndex++;
2266
+ }
2267
+ }
2268
+ catch {
2269
+ pushAllSubstr();
2270
+ } // matcher 의 fallback 과 동일
2271
+ return out;
2272
+ };
2273
+ }
1870
2274
  // Build a match pattern for an entity name — word-boundary for Latin, substring for CJK
1871
2275
  buildEntityMatcher(name) {
1872
2276
  const lower = name.toLowerCase();
@@ -1890,9 +2294,12 @@ export class RAGKnowledgeGraphManager {
1890
2294
  return 0;
1891
2295
  try {
1892
2296
  // Get all chunk text for this document
1893
- const chunks = this.db.prepare(`SELECT rowid, text FROM chunk_metadata WHERE document_id = ?`).all(documentId);
2297
+ const chunks = this.db.prepare(`SELECT rowid, text, start_pos, end_pos FROM chunk_metadata WHERE document_id = ?`).all(documentId);
1894
2298
  if (chunks.length === 0)
1895
2299
  return 0;
2300
+ // spec §5.4: range 링킹용 — 문서 본문과 finder 를 1회 준비
2301
+ const docRow = this.db.prepare(`SELECT content FROM documents WHERE id = ?`).get(documentId);
2302
+ const findRanges = docRow ? this.buildEntityRangeFinder(docRow.content) : null;
1896
2303
  // Get all entities with observations for richer matching
1897
2304
  const entities = this.db.prepare(`SELECT id, name, entityType, observations FROM entities`).all();
1898
2305
  // Minimum name length: 2 for CJK (e.g. "할랄"), 4 for Latin (avoid "API", "Bug")
@@ -1938,6 +2345,19 @@ export class RAGKnowledgeGraphManager {
1938
2345
  entityLinked = true;
1939
2346
  }
1940
2347
  }
2348
+ // spec §5.4 (r7-2): chunk 단위 매칭은 경계에 잘린 이름을 영원히 놓친다 — c1 은
2349
+ // overlap 이 없어 흡수도 안 된다. primary name 의 본문 occurrence range 와
2350
+ // 교차하는 chunk 에 링크한다 (aliases 는 predicate 라 chunk 단위 유지).
2351
+ if (findRanges) {
2352
+ for (const { s, e } of findRanges(entity.name, this.hasCJK(entity.name))) {
2353
+ for (const chunk of chunks) {
2354
+ if (chunk.start_pos !== null && chunk.end_pos !== null && chunk.start_pos < e && chunk.end_pos > s) {
2355
+ insertStmt.run(chunk.rowid, entity.id); // INSERT OR IGNORE — 중복 무해
2356
+ entityLinked = true;
2357
+ }
2358
+ }
2359
+ }
2360
+ }
1941
2361
  if (entityLinked)
1942
2362
  linkedCount++;
1943
2363
  }
@@ -2192,10 +2612,20 @@ export class RAGKnowledgeGraphManager {
2192
2612
  created_at: row.created_at
2193
2613
  }));
2194
2614
  console.error(`✅ Export completed: ${entities.length} entities, ${relations.length} relations, ${documents.length} documents`);
2615
+ // spec §6.4: lifecycle 정본을 함께 내보낸다. 이게 없으면 export->import 뒤
2616
+ // 관찰의 신원·출처·이력이 사라지고 projection 만 남는다.
2617
+ const observation_roots = this.db.prepare(`SELECT * FROM observation_roots ORDER BY entity_id, projection_order`).all();
2618
+ const entity_observations = this.db.prepare(`SELECT * FROM entity_observations ORDER BY root_id, revision_no`).all();
2619
+ const observation_sources = this.db.prepare(`SELECT * FROM observation_sources ORDER BY observation_id, source_kind, source_ref`).all();
2620
+ const observation_events = this.db.prepare(`SELECT * FROM observation_events ORDER BY root_id, recorded_at, event_id`).all();
2195
2621
  return {
2196
2622
  entities,
2197
2623
  relations,
2198
2624
  documents,
2625
+ observation_roots,
2626
+ entity_observations,
2627
+ observation_sources,
2628
+ observation_events,
2199
2629
  metadata: {
2200
2630
  exportedAt: new Date().toISOString(),
2201
2631
  version: PKG_VERSION,
@@ -2211,66 +2641,245 @@ export class RAGKnowledgeGraphManager {
2211
2641
  console.error(`📥 Importing knowledge graph (merge: ${options.merge !== false})...`);
2212
2642
  const imported = { entities: 0, relations: 0, documents: 0 };
2213
2643
  const skipped = { entities: 0, relations: 0, documents: 0 };
2214
- // If merge=false, clear existing data first
2215
- if (options.merge === false) {
2216
- this.db.exec(`DELETE FROM relationships`);
2217
- this.db.exec(`DELETE FROM entities`);
2218
- this.db.exec(`DELETE FROM documents`);
2219
- console.error('🗑️ Cleared existing data for full import');
2220
- }
2221
- // Import entities using INSERT OR IGNORE
2222
- if (data.entities && Array.isArray(data.entities)) {
2223
- const stmt = this.db.prepare(`
2644
+ // merge 로 배열 위치가 재배정된 관찰. 조용히 순서를 바꾸면 호출자가 알 수 없으므로
2645
+ // 응답으로 내보낸다(advisor beta r3 발견 3).
2646
+ const remapReport = [];
2647
+ // spec §6.4: abort 는 0 mutation 이다. lifecycle 만 트랜잭션으로 감싸면
2648
+ // 충돌로 throw 할 때 그 앞에서 넣은 entity·relation·document 가 살아남는다
2649
+ // (T17b 가 ghost entity 로 실증). import 전체가 한 단위여야 한다.
2650
+ // 내부 transaction() 호출은 better-sqlite3 에서 savepoint 로 중첩된다.
2651
+ const importAll = this.db.transaction(() => {
2652
+ // If merge=false, clear existing data first
2653
+ if (options.merge === false) {
2654
+ this.db.exec(`DELETE FROM relationships`);
2655
+ // entities 삭제가 FK CASCADE 로 lifecycle 4테이블을 지우지만, 순서를 계약으로
2656
+ // 두어 FK 가 꺼진 환경에서도 잔존 행이 남지 않게 한다.
2657
+ this.db.exec(`DELETE FROM observation_events`);
2658
+ this.db.exec(`DELETE FROM observation_sources`);
2659
+ this.db.exec(`DELETE FROM entity_observations`);
2660
+ this.db.exec(`DELETE FROM observation_roots`);
2661
+ this.db.exec(`DELETE FROM entities`);
2662
+ this.db.exec(`DELETE FROM documents`);
2663
+ // entities 를 지워도 파생 데이터는 따라오지 않는다: chunk_metadata 에는
2664
+ // entities 로 가는 FK 가 없고 entity_embedding_metadata.entity_id 는 UNIQUE 일
2665
+ // 뿐이다. 그래서 replace-import 뒤에 **사라진 entity 의 벡터와 KG chunk 가
2666
+ // 검색에 남았다**(advisor beta 발견 2). document chunk 는 documents 의
2667
+ // CASCADE 로 이미 정리되므로 여기서는 entity·relationship chunk 만 지운다.
2668
+ const orphanChunks = this.db.prepare(`SELECT rowid FROM chunk_metadata WHERE chunk_type IN ('entity','relationship')`)
2669
+ .all();
2670
+ for (const c of orphanChunks) {
2671
+ this.db.exec(`DELETE FROM chunks WHERE rowid = ${Number(c.rowid)}`);
2672
+ this.db.prepare(`DELETE FROM chunk_metadata WHERE rowid = ?`).run(c.rowid);
2673
+ }
2674
+ this.db.exec(`DELETE FROM entity_embeddings WHERE rowid IN (SELECT rowid FROM entity_embedding_metadata)`);
2675
+ this.db.exec(`DELETE FROM entity_embedding_metadata`);
2676
+ console.error('🗑️ Cleared existing data for full import');
2677
+ }
2678
+ // Import entities using INSERT OR IGNORE
2679
+ if (data.entities && Array.isArray(data.entities)) {
2680
+ const stmt = this.db.prepare(`
2224
2681
  INSERT OR IGNORE INTO entities (id, name, entityType, observations, metadata, created_at)
2225
2682
  VALUES (?, ?, ?, ?, ?, ?)
2226
2683
  `);
2227
- for (const entity of data.entities) {
2228
- const result = stmt.run(entity.id, entity.name, entity.entityType || 'CONCEPT', JSON.stringify(entity.observations || []), JSON.stringify(entity.metadata || {}), entity.created_at || new Date().toISOString());
2229
- if (result.changes > 0) {
2230
- imported.entities++;
2231
- }
2232
- else {
2233
- skipped.entities++;
2684
+ for (const entity of data.entities) {
2685
+ const result = stmt.run(entity.id, entity.name, entity.entityType || 'CONCEPT',
2686
+ // v13: observations 는 projection 이다. lifecycle 행을 넣은 뒤
2687
+ // rebuildProjection 이 채운다 — 여기서 배열을 심으면 정본과 갈라진다.
2688
+ '[]', JSON.stringify(entity.metadata || {}), entity.created_at || new Date().toISOString());
2689
+ if (result.changes > 0) {
2690
+ imported.entities++;
2691
+ }
2692
+ else {
2693
+ skipped.entities++;
2694
+ }
2234
2695
  }
2235
2696
  }
2236
- }
2237
- // Import relations using INSERT OR IGNORE
2238
- if (data.relations && Array.isArray(data.relations)) {
2239
- const stmt = this.db.prepare(`
2697
+ // Import relations using INSERT OR IGNORE
2698
+ if (data.relations && Array.isArray(data.relations)) {
2699
+ const stmt = this.db.prepare(`
2240
2700
  INSERT OR IGNORE INTO relationships (id, source_entity, target_entity, relationType, confidence, metadata, created_at)
2241
2701
  VALUES (?, ?, ?, ?, ?, ?, ?)
2242
2702
  `);
2243
- for (const relation of data.relations) {
2244
- const result = stmt.run(relation.id, relation.source_entity, relation.target_entity, relation.relationType, relation.confidence ?? 1.0, JSON.stringify(relation.metadata || {}), relation.created_at || new Date().toISOString());
2245
- if (result.changes > 0) {
2246
- imported.relations++;
2247
- }
2248
- else {
2249
- skipped.relations++;
2703
+ for (const relation of data.relations) {
2704
+ const result = stmt.run(relation.id, relation.source_entity, relation.target_entity, relation.relationType, relation.confidence ?? 1.0, JSON.stringify(relation.metadata || {}), relation.created_at || new Date().toISOString());
2705
+ if (result.changes > 0) {
2706
+ imported.relations++;
2707
+ }
2708
+ else {
2709
+ skipped.relations++;
2710
+ }
2250
2711
  }
2251
2712
  }
2252
- }
2253
- // Import documents using INSERT OR REPLACE
2254
- if (data.documents && Array.isArray(data.documents)) {
2255
- const stmt = this.db.prepare(`
2713
+ // Import documents using INSERT OR REPLACE
2714
+ if (data.documents && Array.isArray(data.documents)) {
2715
+ const stmt = this.db.prepare(`
2256
2716
  INSERT OR REPLACE INTO documents (id, content, metadata, created_at)
2257
2717
  VALUES (?, ?, ?, ?)
2258
2718
  `);
2259
- for (const doc of data.documents) {
2260
- const result = stmt.run(doc.id, doc.content, JSON.stringify(doc.metadata || {}), doc.created_at || new Date().toISOString());
2261
- if (result.changes > 0) {
2262
- imported.documents++;
2263
- }
2264
- else {
2265
- skipped.documents++;
2719
+ for (const doc of data.documents) {
2720
+ const result = stmt.run(doc.id, doc.content, JSON.stringify(doc.metadata || {}), doc.created_at || new Date().toISOString());
2721
+ if (result.changes > 0) {
2722
+ imported.documents++;
2723
+ }
2724
+ else {
2725
+ skipped.documents++;
2726
+ }
2266
2727
  }
2267
2728
  }
2268
- }
2729
+ // ---- spec §6.4: lifecycle import ----
2730
+ // 순서가 계약이다: entities -> roots -> revisions(root별 revision_no ↑)
2731
+ // -> sources/events. §4.1 트리거가 root 선행과 체인 연속성을 요구하므로
2732
+ // importer 는 입력 순서와 무관하게 재정렬한다 (역순 export 를 그대로
2733
+ // 스트리밍하면 'immediately preceding revision' 으로 죽는다).
2734
+ const sameRow = (a, b, cols) => cols.every(c => (a[c] ?? null) === (b[c] ?? null));
2735
+ const hasLifecycle = Array.isArray(data.observation_roots);
2736
+ if (hasLifecycle) {
2737
+ const tx = this.db.transaction(() => {
2738
+ // 새 root 가 이미 점유된 (entity_id, projection_order) 슬롯을 요구할 수 있다:
2739
+ // 두 DB 가 같은 entity 이름을 갖고 서로 다른 관찰을 배열 0번에 두면 그렇다.
2740
+ // 이건 §6.4 의 "같은 키 다른 값" 충돌이 아니라 **슬롯 충돌**이고, 규칙이 없어서
2741
+ // raw UNIQUE 오류로 터졌다(내 MCP 왕복 테스트가 잡았다). merge 의 뜻은
2742
+ // "더한다"이므로 들어오는 root 에 다음 빈 순번을 준다 — 남의 관찰을 덮지 않고,
2743
+ // 배열 끝에 붙는다. remap 은 그 root 의 revision 들에도 그대로 적용해야 한다
2744
+ // (trg_obs_matches_root 가 둘의 일치를 요구한다).
2745
+ // 입력 순서에 결과가 의존하면 같은 dump 를 두 번 넣었을 때 배열 순서가 달라진다.
2746
+ // (entity_id, projection_order, root_id) 로 정렬해 결정론을 만든다.
2747
+ const incomingRoots = [...(data.observation_roots ?? [])].sort((a, b) => String(a.entity_id).localeCompare(String(b.entity_id)) ||
2748
+ (a.projection_order - b.projection_order) ||
2749
+ String(a.root_id).localeCompare(String(b.root_id)));
2750
+ const remappedOrder = new Map();
2751
+ for (const r of incomingRoots) {
2752
+ const cur = this.db.prepare(`SELECT * FROM observation_roots WHERE root_id = ?`)
2753
+ .get(r.root_id);
2754
+ if (cur) {
2755
+ // projection_order 는 **target-local** 속성이다: merge 는 배열 위치를
2756
+ // 이 DB 기준으로 재배정하므로, 이미 remap 된 root 를 같은 dump 로 다시
2757
+ // 넣으면 dump 의 옛 순번과 다를 수밖에 없다. 그걸 충돌로 보면 동일
2758
+ // 재수입이 실패한다(advisor beta r3 발견 3, 실행 재현).
2759
+ if (!sameRow(cur, r, ['entity_id', 'created_at']))
2760
+ throw new Error(`import conflict: observation_roots ${r.root_id} differs from the existing row`);
2761
+ remappedOrder.set(r.root_id, cur.projection_order);
2762
+ continue;
2763
+ }
2764
+ let order = r.projection_order;
2765
+ const taken = this.db.prepare(`SELECT root_id FROM observation_roots WHERE entity_id = ? AND projection_order = ?`)
2766
+ .get(r.entity_id, order);
2767
+ if (taken) {
2768
+ order = nextProjectionOrder(this.db, r.entity_id);
2769
+ remappedOrder.set(r.root_id, order);
2770
+ remapReport.push({ root_id: r.root_id, entity_id: r.entity_id,
2771
+ from: r.projection_order, to: order });
2772
+ console.error(` ├─ ↪️ import: ${r.entity_id} position ${r.projection_order} is held by ` +
2773
+ `${taken.root_id}; appending imported observation at ${order}`);
2774
+ }
2775
+ this.db.prepare(`INSERT INTO observation_roots
2776
+ (root_id, entity_id, projection_order, created_at) VALUES (?, ?, ?, ?)`)
2777
+ .run(r.root_id, r.entity_id, order, r.created_at);
2778
+ }
2779
+ // projection_order 는 root 와 같은 이유로 비교 대상이 아니다(target-local).
2780
+ const revCols = ['root_id', 'entity_id', 'revision_no', 'content',
2781
+ 'status', 'supersedes_id', 'recorded_at', 'superseded_at'];
2782
+ const revs = [...(data.entity_observations ?? [])]
2783
+ .sort((a, b) => a.root_id === b.root_id
2784
+ ? a.revision_no - b.revision_no
2785
+ : String(a.root_id).localeCompare(String(b.root_id)));
2786
+ for (const v of revs) {
2787
+ const cur = this.db.prepare(`SELECT * FROM entity_observations WHERE observation_id = ?`)
2788
+ .get(v.observation_id);
2789
+ if (cur) {
2790
+ if (!sameRow(cur, v, revCols))
2791
+ throw new Error(`import conflict: entity_observations ${v.observation_id} differs from the existing row`);
2792
+ continue;
2793
+ }
2794
+ const order = remappedOrder.has(v.root_id)
2795
+ ? remappedOrder.get(v.root_id) : v.projection_order;
2796
+ this.db.prepare(`INSERT INTO entity_observations
2797
+ (observation_id, root_id, entity_id, revision_no, projection_order,
2798
+ content, status, supersedes_id, recorded_at, superseded_at)
2799
+ VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`)
2800
+ .run(v.observation_id, v.root_id, v.entity_id, v.revision_no, order, v.content, v.status, v.supersedes_id ?? null, v.recorded_at, v.superseded_at ?? null);
2801
+ }
2802
+ for (const so of (data.observation_sources ?? [])) {
2803
+ const cur = this.db.prepare(`SELECT * FROM observation_sources
2804
+ WHERE observation_id=? AND source_kind=? AND source_ref=?`)
2805
+ .get(so.observation_id, so.source_kind, so.source_ref);
2806
+ if (cur) {
2807
+ if (!sameRow(cur, so, ['source_hash', 'recorded_at']))
2808
+ throw new Error(`import conflict: observation_sources ` +
2809
+ `${so.observation_id}/${so.source_kind}/${so.source_ref} differs from the existing row`);
2810
+ continue;
2811
+ }
2812
+ this.db.prepare(`INSERT INTO observation_sources
2813
+ (observation_id, source_kind, source_ref, source_hash, recorded_at) VALUES (?, ?, ?, ?, ?)`)
2814
+ .run(so.observation_id, so.source_kind, so.source_ref, so.source_hash ?? null, so.recorded_at);
2815
+ }
2816
+ const evCols = ['root_id', 'from_id', 'to_id', 'event', 'change_kind', 'reason', 'actor', 'batch_id', 'recorded_at'];
2817
+ for (const e of (data.observation_events ?? [])) {
2818
+ const cur = this.db.prepare(`SELECT * FROM observation_events WHERE event_id = ?`)
2819
+ .get(e.event_id);
2820
+ if (cur) {
2821
+ if (!sameRow(cur, e, evCols))
2822
+ throw new Error(`import conflict: observation_events ${e.event_id} differs from the existing row`);
2823
+ continue;
2824
+ }
2825
+ this.db.prepare(`INSERT INTO observation_events
2826
+ (event_id, root_id, from_id, to_id, event, change_kind, reason, actor, batch_id, recorded_at)
2827
+ VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`)
2828
+ .run(e.event_id, e.root_id, e.from_id ?? null, e.to_id ?? null, e.event, e.change_kind ?? null, e.reason ?? null, e.actor ?? null, e.batch_id ?? null, e.recorded_at);
2829
+ }
2830
+ });
2831
+ tx();
2832
+ }
2833
+ else {
2834
+ // 구(舊) 형식 export: lifecycle 필드가 없으므로 entities.observations 를
2835
+ // 신규 root 로 승격한다. legacy import 필수 필드값 = spec §6.4.
2836
+ const ts = new Date().toISOString();
2837
+ const tx = this.db.transaction(() => {
2838
+ for (const ent of (data.entities ?? [])) {
2839
+ const entityId = ent.id ??
2840
+ `entity_${String(ent.name).toLowerCase().replace(/[^\p{L}\p{N}]/gu, '_')}`;
2841
+ for (const content of (ent.observations ?? [])) {
2842
+ addRevision(this.db, {
2843
+ entityId, content, status: 'active',
2844
+ sources: [{ source_kind: 'import', source_ref: 'legacy-export', source_hash: null }],
2845
+ actor: 'import', ts, event: 'import'
2846
+ });
2847
+ }
2848
+ }
2849
+ });
2850
+ tx();
2851
+ }
2852
+ // projection 재합성 + 파생 상태 무효화.
2853
+ // 무효화가 없으면 이미 있던 entity 를 덮어쓴 뒤에도 옛 벡터·옛 KG chunk 가
2854
+ // 검색에 남는다. import 는 관찰을 바꾸는 writer 이므로 다른 writer 와 같은
2855
+ // 계약을 져야 한다(advisor beta 발견 2).
2856
+ //
2857
+ // 대상은 `data.entities` 가 아니라 **영향받은 entity 전부**다. lifecycle import 는
2858
+ // observation_roots 만 있어도 활성화되므로, entities 없이 lifecycle 배열만 보내면
2859
+ // revision 은 들어가는데 projection 이 갱신되지 않아 새 사실이 reader 에 안 보이고
2860
+ // 옛 벡터가 남는다(advisor beta r3 발견 2, 실행 재현).
2861
+ const affected = new Set();
2862
+ for (const ent of (data.entities ?? [])) {
2863
+ affected.add(ent.id ??
2864
+ `entity_${String(ent.name).toLowerCase().replace(/[^\p{L}\p{N}]/gu, '_')}`);
2865
+ }
2866
+ for (const r of (data.observation_roots ?? []))
2867
+ if (r.entity_id)
2868
+ affected.add(r.entity_id);
2869
+ for (const v of (data.entity_observations ?? []))
2870
+ if (v.entity_id)
2871
+ affected.add(v.entity_id);
2872
+ for (const entityId of affected) {
2873
+ rebuildProjection(this.db, entityId);
2874
+ this.invalidateDerivedForEntity(entityId);
2875
+ }
2876
+ });
2877
+ importAll();
2269
2878
  console.error(`✅ Import completed: ${imported.entities} entities, ${imported.relations} relations, ${imported.documents} documents imported`);
2270
2879
  // Indirect missing-row producer (spec §5): imported rows may lack vectors.
2271
2880
  this.coordinator?.invalidateCoverage();
2272
2881
  this.coordinator?.kick();
2273
- return { imported, skipped };
2882
+ return { imported, skipped, observation_order_remap: remapReport };
2274
2883
  }
2275
2884
  async hybridSearch(query, limit = 5, useGraph = true) {
2276
2885
  if (!this.db)
@@ -2596,8 +3205,12 @@ export class RAGKnowledgeGraphManager {
2596
3205
  graphBoost += Math.min(entityBoost, 0.4);
2597
3206
  }
2598
3207
  // Generate semantic summary (skip when degraded — no embeddings available).
3208
+ // RAG_MEMORY_SEARCH_SUMMARIES=off: diagnostic escape hatch (v5) — the summary
3209
+ // path embeds EVERY sentence of EVERY candidate (~100+ inferences per search,
3210
+ // measured 90-120s cold). Off = preview slices + relevanceScore 0; ranking
3211
+ // then rests on vectorSimilarity + boosts. Default unchanged.
2599
3212
  let summary, keyHighlight, relevanceScore;
2600
- if (vectorDegraded || !primaryQueryEmbedding) {
3213
+ if (vectorDegraded || !primaryQueryEmbedding || process.env.RAG_MEMORY_SEARCH_SUMMARIES === 'off') {
2601
3214
  keyHighlight = result.text.slice(0, 150);
2602
3215
  summary = result.text.slice(0, 300);
2603
3216
  relevanceScore = 0;
@@ -2775,6 +3388,20 @@ export class RAGKnowledgeGraphManager {
2775
3388
  // reads version, model/reconciliation state, and provenance coverage here.
2776
3389
  const gs = this.gate.status;
2777
3390
  const cov = this.coordinator?.coverage();
3391
+ // v14 (spec §7.2): document 기준 chunking 전환 상태 — 상호배타, 합 = documents.
3392
+ // regex 분류는 SQL 밖(JS)에서: current = 런타임이 인식하는 c1 형식(강한 파서),
3393
+ // legacy = 'legacy-unknown', unknown = 그 외 전부.
3394
+ const sigRows = this.db.prepare(`SELECT chunking_signature AS s, count(*) AS n FROM documents GROUP BY chunking_signature`)
3395
+ .all();
3396
+ let sigCur = 0, sigLeg = 0, sigUnk = 0;
3397
+ for (const r of sigRows) {
3398
+ if (r.s === LEGACY_SIGNATURE)
3399
+ sigLeg += r.n;
3400
+ else if (isCurrentFormatSignature(r.s))
3401
+ sigCur += r.n;
3402
+ else
3403
+ sigUnk += r.n;
3404
+ }
2778
3405
  return {
2779
3406
  entities: {
2780
3407
  total: entityStats.reduce((sum, stat) => sum + stat.count, 0),
@@ -2786,6 +3413,8 @@ export class RAGKnowledgeGraphManager {
2786
3413
  },
2787
3414
  documents: documentCount.count,
2788
3415
  chunks: chunkCount.count,
3416
+ chunking: { current: sigCur, legacy: sigLeg, unknown: sigUnk,
3417
+ default_signature: effectiveSignature(DEFAULT_MAX_TOKENS) },
2789
3418
  server: {
2790
3419
  version: PKG_VERSION,
2791
3420
  node: process.versions.node,
@@ -3082,7 +3711,7 @@ export class RAGKnowledgeGraphManager {
3082
3711
  .filter(m => m.version > targetVersion && m.version <= currentVersion)
3083
3712
  .sort((a, b) => b.version - a.version);
3084
3713
  migrationManager.rollback(targetVersion);
3085
- return {
3714
+ const result = {
3086
3715
  rolledBack: migrationsToRollback.length,
3087
3716
  currentVersion: migrationManager.getCurrentVersion(),
3088
3717
  rolledBackMigrations: migrationsToRollback.map(m => ({
@@ -3090,6 +3719,18 @@ export class RAGKnowledgeGraphManager {
3090
3719
  description: m.description
3091
3720
  }))
3092
3721
  };
3722
+ // v14 rollback is a compatibility rollback ONLY (spec §6.3): dropping the
3723
+ // chunking_signature column does not restore old chunk boundaries — c1 rows
3724
+ // read fine on v13 code. Say so in the RESPONSE, not just the tool
3725
+ // description, so a caller who rolled back sees the limit (advisor r5-10).
3726
+ if (result.rolledBackMigrations.some(m => m.version === 14)) {
3727
+ return {
3728
+ ...result,
3729
+ semanticRollback: false,
3730
+ warning: 'v14 rollback removes the chunking_signature column only; chunk boundaries produced by chunker c1 are NOT restored (compatibility rollback). Data restore path = pre-migration backup snapshot.'
3731
+ };
3732
+ }
3733
+ return result;
3093
3734
  }
3094
3735
  }
3095
3736
  // Initialize the manager
@@ -3125,7 +3766,32 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
3125
3766
  case "createRelations":
3126
3767
  return { content: [{ type: "text", text: JSON.stringify(await ragKgManager.createRelations(validatedArgs.relations), null, 2) }] };
3127
3768
  case "addObservations":
3769
+ // v13: status·sources 를 그대로 넘긴다. 여기서 떨어뜨리면 스키마가 받아도
3770
+ // 엔진에 도달하지 않아 provenance 가 조용히 사라진다.
3128
3771
  return { content: [{ type: "text", text: JSON.stringify(await ragKgManager.addObservations(validatedArgs.observations), null, 2) }] };
3772
+ // v13 observation lifecycle (spec §6.1 / §6.2)
3773
+ case "correctObservation":
3774
+ return { content: [{ type: "text", text: JSON.stringify({ observation_id: await ragKgManager.correctObservation(validatedArgs.observation_id, validatedArgs.content, validatedArgs.change_kind ?? 'correction', validatedArgs.reason) }, null, 2) }] };
3775
+ case "retractObservation":
3776
+ await ragKgManager.retractObservation(validatedArgs.observation_id, validatedArgs.reason);
3777
+ return { content: [{ type: "text", text: JSON.stringify({ observation_id: validatedArgs.observation_id, status: 'retracted' }, null, 2) }] };
3778
+ case "restoreObservation":
3779
+ await ragKgManager.restoreObservation(validatedArgs.observation_id, validatedArgs.reason);
3780
+ return { content: [{ type: "text", text: JSON.stringify({ observation_id: validatedArgs.observation_id, status: 'active' }, null, 2) }] };
3781
+ case "approveObservation":
3782
+ await ragKgManager.approveObservation(validatedArgs.observation_id, validatedArgs.reason);
3783
+ return { content: [{ type: "text", text: JSON.stringify({ observation_id: validatedArgs.observation_id, status: 'active' }, null, 2) }] };
3784
+ case "declineObservation":
3785
+ await ragKgManager.declineObservation(validatedArgs.observation_id, validatedArgs.reason);
3786
+ return { content: [{ type: "text", text: JSON.stringify({ observation_id: validatedArgs.observation_id, status: 'retracted' }, null, 2) }] };
3787
+ case "purgeObservation":
3788
+ return { content: [{ type: "text", text: JSON.stringify(await ragKgManager.purgeObservation(validatedArgs.observation_id, validatedArgs.confirm), null, 2) }] };
3789
+ case "getObservationHistory":
3790
+ return { content: [{ type: "text", text: JSON.stringify(await ragKgManager.getObservationHistory({
3791
+ entity_name: validatedArgs.entity_name,
3792
+ observation_id: validatedArgs.observation_id,
3793
+ root_id: validatedArgs.root_id,
3794
+ }), null, 2) }] };
3129
3795
  case "deleteEntities":
3130
3796
  await ragKgManager.deleteEntities(validatedArgs.entityNames);
3131
3797
  return { content: [{ type: "text", text: "Entities deleted successfully" }] };