rag-memory-epf-mcp 3.6.0 → 5.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,5 @@
1
+ import Database from 'better-sqlite3';
2
+ export declare function backupBeforeMigration(db: Database.Database, dbPath: string, pendingVersions: number[], currentVersion: number): Promise<{
3
+ path: string;
4
+ sha256: string;
5
+ } | null>;
@@ -0,0 +1,185 @@
1
+ import { existsSync, openSync, readSync, closeSync, linkSync, unlinkSync } from 'node:fs';
2
+ import { createHash } from 'node:crypto';
3
+ import Database from 'better-sqlite3';
4
+ // 이 파일이 지켜야 하는 것은 두 줄이다:
5
+ // ① 복구점을 절대 덮어쓰지 않는다
6
+ // ② 재시작을 막지 않는다
7
+ //
8
+ // 이걸 "남아 있는 .bak 이 live 와 같은 상태인가"를 증명해서 풀려고 세 판본을 썼고
9
+ // 세 번 다 틀렸다(스키마 버전을 신원으로 착각 → DB 내부 UUID+집계 지문 → 전체 논리
10
+ // 다이제스트). 등가 증명은 애초에 필요하지 않았다: **백업은 스키마를 바꾸기 전에
11
+ // 만들어지므로, 시도마다 다음 빈 슬롯에 하나 더 만들면 그 모든 파일이 정의상 유효한
12
+ // pre-migration 스냅샷이다.** 덮어쓰지 않으니 ①, 막지 않으니 ②. 비교가 없으니
13
+ // 비교 정확성 문제가 전부 사라진다. (advisor beta r2~r7 결론)
14
+ //
15
+ // 대가 = 디스크. 슬롯을 유한하게 두고 다 차면 fail-closed 한다.
16
+ const MAX_RECOVERY_POINTS = 3;
17
+ // spec §5.1: 마이그레이션 전 일관 스냅샷.
18
+ //
19
+ // **`VACUUM INTO` 를 쓰지 않는다.** SQLite 문서는 VACUUM 이 명시적
20
+ // `INTEGER PRIMARY KEY` 가 없는 테이블의 ROWID 를 바꿀 수 있다고 명시한다. 이 스키마의
21
+ // `entities` 는 `TEXT PRIMARY KEY` 라 hidden ROWID 테이블이고, `entities_fts` 는
22
+ // `content_rowid='rowid'` 로 그 hidden ROWID 를 참조한다 — 재번호가 일어나면 백업
23
+ // **자체가** external-content 불일치를 안고 태어나며, `quick_check` 는 그것을 통과시킨다.
24
+ // 현재 빌드(SQLite 3.51.3, 이 스키마)에서는 재번호가 관측되지 않았다. 하지만
25
+ // **관측되지 않은 것과 계약으로 금지된 것은 다르고**, `better-sqlite3` 는 `^12.8.0`
26
+ // 부동이며 이 파일은 유일한 복구선이다. 그래서 bitwise-identical 스냅샷을 보장하는
27
+ // Online Backup API(`db.backup()`)를 쓴다. (advisor beta r7 P0-1)
28
+ //
29
+ // 실패는 전부 throw = fail-closed. 백업 없이 스키마를 바꾸지 않는다:
30
+ // 이 프로젝트군은 non-git 환경(Google Drive 폴더)에 배포되어 git 롤백이 없다.
31
+ //
32
+ // **단일 writer 전제**: 백업과 마이그레이션 사이에 다른 프로세스가 커밋하면 그 커밋은
33
+ // 마이그레이션에는 들어가고 복구점에는 없다. 이 엔진은 프로젝트당 서버 하나로 배포되며
34
+ // (프로젝트별 `.mcp.json`), 마이그레이션은 `server.connect()` 전에 끝난다. 같은 DB 를
35
+ // 두 프로세스가 동시에 여는 것은 지원 대상이 아니다 — 상세 = docs/UPDATING.md.
36
+ export async function backupBeforeMigration(db, dbPath, pendingVersions, currentVersion) {
37
+ if (pendingVersions.length === 0)
38
+ return null; // 대기 없으면 no-op
39
+ if (!dbPath || dbPath === ':memory:')
40
+ return null; // 메모리 DB 는 대상 아님
41
+ // 백업 대상 판정 = "잃을 데이터가 있는가". 테이블 목록을 하드코딩하면 그 목록 밖의
42
+ // 데이터가 안 보인다 — `embedding_profiles` 에만 행이 있는 version-0 DB 가 그렇게
43
+ // 무백업으로 통과했다(advisor beta r6 P0-3). 그래서 사용자 테이블 전체를 본다.
44
+ if (currentVersion <= 0 && !hasAnyUserRows(db))
45
+ return null;
46
+ const base = `${dbPath}.v${currentVersion}.bak`;
47
+ // 임시 파일에 먼저 만든다 = 이 프로세스가 독점 소유한다. 검증까지 통과한 뒤에야
48
+ // 슬롯에 게시하므로, 슬롯에는 절대 반쯤 만들어진 파일이 놓이지 않는다.
49
+ const tmp = `${base}.partial-${process.pid}`;
50
+ if (existsSync(tmp))
51
+ unlinkSync(tmp);
52
+ await db.backup(tmp);
53
+ try {
54
+ verifyRecoveryPoint(tmp);
55
+ const target = publishNoClobber(tmp, base);
56
+ const sha256 = streamSha256(target);
57
+ console.error(` ├─ 🛟 backup ${target} (sha256 ${sha256.slice(0, 12)}…)`);
58
+ return { path: target, sha256 };
59
+ }
60
+ catch (e) {
61
+ try {
62
+ if (existsSync(tmp))
63
+ unlinkSync(tmp);
64
+ }
65
+ catch { /* 정리 실패는 원인을 가리지 않는다 */ }
66
+ throw e;
67
+ }
68
+ }
69
+ // 이 파일이 실제로 복구점인가. `quick_check` 는 페이지 구조만 본다 — external-content
70
+ // FTS5 의 rowid 불일치는 통과시킨다. FTS5 자신의 `integrity-check` 가 그 대조를 한다.
71
+ function verifyRecoveryPoint(path) {
72
+ // 우리가 독점 소유한 임시 파일이므로 쓰기 모드로 연다(integrity-check 는 명령 삽입이다).
73
+ const v = new Database(path);
74
+ try {
75
+ const ok = v.pragma('quick_check', { simple: true });
76
+ if (ok !== 'ok')
77
+ throw new Error(`migration backup failed quick_check: ${ok}`);
78
+ const n = v.prepare(`SELECT COUNT(*) c FROM sqlite_master`).get();
79
+ if (!n || n.c === 0)
80
+ throw new Error('migration backup has empty schema');
81
+ const fts = v.prepare(`SELECT name FROM sqlite_schema WHERE type='table' AND sql LIKE '%USING fts5%'`)
82
+ .all();
83
+ for (const { name } of fts) {
84
+ const q = `"${name.replace(/"/g, '""')}"`;
85
+ try {
86
+ // **`rank = 1` 이 필수다.** 인자 없는 integrity-check 는 인덱스가 자기 자신과
87
+ // 정합한지만 본다 — external-content 테이블과의 대조는 하지 않는다. 실측:
88
+ // 인덱스에 content 없는 rowid 를 심어 놓으면 `quick_check` 도, 인자 없는
89
+ // integrity-check 도 **통과**하고 `rank=1` 만 malformed 를 던진다.
90
+ // 즉 rank 없이 부르면 이 검사를 넣은 이유였던 실패 모드를 못 잡는다
91
+ // (advisor beta r8 P0, 같은 런타임에서 재현).
92
+ v.exec(`INSERT INTO ${q}(${q}, rank) VALUES('integrity-check', 1)`);
93
+ }
94
+ catch (e) {
95
+ throw new Error(`migration backup failed FTS5 integrity-check on ${name}: ${e.message}. ` +
96
+ `The snapshot would not be a usable recovery point.`);
97
+ }
98
+ }
99
+ }
100
+ finally {
101
+ v.close();
102
+ }
103
+ }
104
+ // 슬롯 게시. `link()` 는 목적지가 있으면 EEXIST 로 실패하므로 **원자적 no-clobber** 다
105
+ // (rename 은 조용히 덮어쓴다). 그래서 경쟁하는 프로세스가 있어도 복구점을 잃지 않는다.
106
+ function publishNoClobber(tmp, base) {
107
+ for (let attempt = 0; attempt < MAX_RECOVERY_POINTS; attempt++) {
108
+ const slot = pickRecoverySlot(base);
109
+ try {
110
+ linkSync(tmp, slot);
111
+ unlinkSync(tmp);
112
+ return slot;
113
+ }
114
+ catch (e) {
115
+ if (e.code !== 'EEXIST')
116
+ throw e;
117
+ // 그 슬롯을 누가 먼저 가져갔다 — 다음 빈 슬롯으로.
118
+ }
119
+ }
120
+ throw slotsFullError(base);
121
+ }
122
+ // 다음 빈 복구점 슬롯. 기존 파일은 읽지도, 검증하지도, 건드리지도 않는다 —
123
+ // 그것들이 무엇인지 판정하려는 시도가 앞선 세 판본의 결함 전부였다.
124
+ function pickRecoverySlot(base) {
125
+ if (!existsSync(base))
126
+ return base;
127
+ for (let i = 1; i < MAX_RECOVERY_POINTS; i++) {
128
+ const candidate = `${base}.${i}`;
129
+ if (!existsSync(candidate))
130
+ return candidate;
131
+ }
132
+ throw slotsFullError(base);
133
+ }
134
+ // 슬롯이 찬 상태는 **이 스키마 버전에 대한 circuit breaker** 다. 전역 용량 상한이 아니고
135
+ // (버전마다 자기 세트를 갖는다), "3회 실패"의 증거도 아니다 — 손상 파일이나 남의 파일이
136
+ // 슬롯을 차지할 수도 있다. 그리고 이 오류는 `server.connect()` **전에** 나가므로
137
+ // 클라이언트에는 MCP unavailable 로 보인다. 복구 절차를 아는 유일한 경로가 stderr 다.
138
+ function slotsFullError(base) {
139
+ return new Error(`migration refused: all ${MAX_RECOVERY_POINTS} recovery-point slots for this schema version ` +
140
+ `are taken (${base} plus ${MAX_RECOVERY_POINTS - 1} numbered siblings). Nothing was written ` +
141
+ `and no existing file was touched. This is a circuit breaker for this version, not a disk ` +
142
+ `quota, and the files are not necessarily failed attempts — a stale or unrelated file occupies ` +
143
+ `a slot just the same. Inspect them, move the ones you do not need aside, then start the ` +
144
+ `server again. Runbook: docs/UPDATING.md "Recovery-point slots are full".`);
145
+ }
146
+ // 사용자 테이블 중 하나라도 행이 있으면 잃을 데이터가 있다.
147
+ // 목록은 스키마에서 얻는다(하드코딩하면 새 테이블이 조용히 검사 밖에 남는다).
148
+ // `schema_migrations` 는 순수 메타데이터라 제외한다 — 그것만 있는 DB 는 빈 DB 다.
149
+ // 가상 테이블은 자기 shadow 테이블을 통해 이미 세어진다.
150
+ function hasAnyUserRows(db) {
151
+ const tables = db.prepare(`SELECT name, sql FROM sqlite_schema WHERE type='table'
152
+ AND name <> 'schema_migrations'
153
+ ORDER BY name`).all();
154
+ for (const t of tables) {
155
+ if (/CREATE VIRTUAL TABLE/i.test(t.sql ?? ''))
156
+ continue;
157
+ const qt = `"${t.name.replace(/"/g, '""')}"`;
158
+ try {
159
+ const n = db.prepare(`SELECT EXISTS(SELECT 1 FROM ${qt}) e`).get();
160
+ if (n.e)
161
+ return true;
162
+ }
163
+ catch {
164
+ // 읽을 수 없는 테이블이 있으면 "빈 DB"라고 단정하지 않는다 = 백업한다.
165
+ return true;
166
+ }
167
+ }
168
+ return false;
169
+ }
170
+ // 스트리밍 해시. readFileSync 로 전체를 메모리에 올리면 fleet 의 큰 DB 에서
171
+ // OOM 위험이 있다(advisor 구현리뷰 r1 발견 6).
172
+ function streamSha256(path) {
173
+ const h = createHash('sha256');
174
+ const fd = openSync(path, 'r');
175
+ try {
176
+ const buf = Buffer.allocUnsafe(1 << 20);
177
+ let n;
178
+ while ((n = readSync(fd, buf, 0, buf.length, null)) > 0)
179
+ h.update(buf.subarray(0, n));
180
+ }
181
+ finally {
182
+ closeSync(fd);
183
+ }
184
+ return h.digest('hex');
185
+ }
@@ -0,0 +1,18 @@
1
+ import type { Tiktoken } from 'tiktoken';
2
+ export interface CSegment {
3
+ text: string;
4
+ start_pos: number;
5
+ end_pos: number;
6
+ }
7
+ export declare const DEFAULT_MAX_TOKENS = 800;
8
+ export declare const LEGACY_SIGNATURE = "legacy-unknown";
9
+ export declare function mergeMinTokens(maxTokens: number): number;
10
+ export declare function effectiveSignature(maxTokens: number): string;
11
+ export declare function isCurrentFormatSignature(sig: string): boolean;
12
+ type Block = {
13
+ lines: string[];
14
+ heading: boolean;
15
+ };
16
+ export declare function buildBlocks(lines: string[]): Block[];
17
+ export declare function chunkStructured(text: string, enc: Tiktoken, maxTokens?: number): CSegment[];
18
+ export {};
@@ -0,0 +1,210 @@
1
+ export const DEFAULT_MAX_TOKENS = 800;
2
+ export const LEGACY_SIGNATURE = 'legacy-unknown';
3
+ // r11: heading is a preferred cut, not a mandatory one. A cut happens only when
4
+ // the accumulated section group AND the next section are both >= this floor —
5
+ // short adjacent sections merge, so cross-section queries still land in one
6
+ // chunk (gate regression k05) and short log sections stop forming over-sharp
7
+ // competitor chunks (k09). Derived, not independent: one knob (maxTokens).
8
+ export function mergeMinTokens(maxTokens) {
9
+ return Math.floor(maxTokens / 2); // NOT >>1 — bitshift wraps at 2^31 (r12)
10
+ }
11
+ export function effectiveSignature(maxTokens) {
12
+ return `c1:enc=cl100k_base:max=${maxTokens}:overlap=0:fence=on:merge=${mergeMinTokens(maxTokens)}:fallback=cp-exact-${maxTokens}`;
13
+ }
14
+ // Strict parse (r5-15): the regex alone classified impossible signatures
15
+ // (max=0, max/fallback mismatch) as current. Cross-check the components.
16
+ export function isCurrentFormatSignature(sig) {
17
+ const m = /^c1:enc=cl100k_base:max=(\d+):overlap=0:fence=on:merge=(\d+):fallback=cp-exact-(\d+)$/.exec(sig);
18
+ if (!m)
19
+ return false;
20
+ const max = Number(m[1]);
21
+ return Number.isInteger(max) && max > 0 && m[1] === m[3] && Number(m[2]) === mergeMinTokens(max);
22
+ }
23
+ function splitLines(text) {
24
+ const lines = [];
25
+ let start = 0;
26
+ for (let i = 0; i < text.length; i++) {
27
+ if (text[i] === '\n') {
28
+ lines.push(text.slice(start, i + 1));
29
+ start = i + 1;
30
+ }
31
+ }
32
+ if (start < text.length)
33
+ lines.push(text.slice(start));
34
+ return lines;
35
+ }
36
+ const FENCE_RE = /^\s{0,3}(`{3,}|~{3,})/;
37
+ const HEADING_RE = /^#{1,4}[ \t]/;
38
+ const BULLET_RE = /^(\s*)([-*+]|\d+[.)])[ \t]/;
39
+ const BLANK_RE = /^\s*$/;
40
+ export function buildBlocks(lines) {
41
+ const blocks = [];
42
+ let i = 0;
43
+ const attachTrailingBlanks = (blk) => {
44
+ while (i < lines.length && BLANK_RE.test(lines[i]))
45
+ blk.lines.push(lines[i++]);
46
+ };
47
+ while (i < lines.length) {
48
+ const line = lines[i];
49
+ const fm = line.match(FENCE_RE);
50
+ if (fm) {
51
+ const marker = fm[1][0];
52
+ const openLen = fm[1].length;
53
+ const blk = { lines: [lines[i++]], heading: false };
54
+ const closeRe = new RegExp(`^\\s{0,3}\\${marker}{${openLen},}\\s*$`);
55
+ while (i < lines.length) {
56
+ const l = lines[i];
57
+ blk.lines.push(l);
58
+ i++;
59
+ if (closeRe.test(l.replace(/\r?\n$/, '')))
60
+ break; // unclosed -> EOF
61
+ }
62
+ attachTrailingBlanks(blk);
63
+ blocks.push(blk);
64
+ }
65
+ else if (HEADING_RE.test(line)) {
66
+ const blk = { lines: [lines[i++]], heading: true };
67
+ attachTrailingBlanks(blk);
68
+ blocks.push(blk);
69
+ }
70
+ else if (BULLET_RE.test(line)) {
71
+ const indent = line.match(BULLET_RE)[1].length;
72
+ const blk = { lines: [lines[i++]], heading: false };
73
+ while (i < lines.length) {
74
+ const l = lines[i];
75
+ if (BLANK_RE.test(l) || HEADING_RE.test(l) || FENCE_RE.test(l))
76
+ break;
77
+ const bm = l.match(BULLET_RE);
78
+ if (bm && bm[1].length <= indent)
79
+ break; // sibling/outer bullet
80
+ const li = l.match(/^(\s*)/)[1].length;
81
+ if (!bm && li <= indent)
82
+ break; // dedented prose
83
+ blk.lines.push(l);
84
+ i++;
85
+ }
86
+ attachTrailingBlanks(blk);
87
+ blocks.push(blk);
88
+ }
89
+ else if (BLANK_RE.test(line)) {
90
+ const blk = { lines: [], heading: false };
91
+ while (i < lines.length && BLANK_RE.test(lines[i]))
92
+ blk.lines.push(lines[i++]);
93
+ blocks.push(blk);
94
+ }
95
+ else {
96
+ const blk = { lines: [lines[i++]], heading: false };
97
+ while (i < lines.length && !BLANK_RE.test(lines[i]) && !HEADING_RE.test(lines[i])
98
+ && !FENCE_RE.test(lines[i]) && !BULLET_RE.test(lines[i])) {
99
+ blk.lines.push(lines[i++]);
100
+ }
101
+ attachTrailingBlanks(blk);
102
+ blocks.push(blk);
103
+ }
104
+ }
105
+ return blocks;
106
+ }
107
+ // cp-exact oversize split, non-monotonicity-safe (spec §4.3-6, r5-5).
108
+ // BPE prefix token counts are NOT monotonic ('/sdkX' -> 1,1,2,1,2), so binary
109
+ // search is invalid. Scan prefixes linearly, remember the last fitting one, and
110
+ // keep probing LOOKAHEAD candidates past a miss to recover dips. If not even
111
+ // one codepoint fits, throw — never emit an over-budget chunk.
112
+ const LOOKAHEAD = 16;
113
+ function splitOversize(block, enc, maxTokens) {
114
+ const cps = [...block];
115
+ const out = [];
116
+ let start = 0;
117
+ while (start < cps.length) {
118
+ let lastFit = 0;
119
+ let missesSinceFit = 0;
120
+ for (let probe = start + 1; probe <= cps.length && missesSinceFit < LOOKAHEAD; probe++) {
121
+ const t = enc.encode(cps.slice(start, probe).join('')).length;
122
+ if (t <= maxTokens) {
123
+ lastFit = probe - start;
124
+ missesSinceFit = 0;
125
+ }
126
+ else
127
+ missesSinceFit++;
128
+ }
129
+ if (lastFit === 0) {
130
+ throw new Error(`chunker c1: codepoint at offset ${start} exceeds maxTokens=${maxTokens} on its own — cannot honor the token budget`);
131
+ }
132
+ out.push(cps.slice(start, start + lastFit).join(''));
133
+ start += lastFit;
134
+ }
135
+ return out;
136
+ }
137
+ export function chunkStructured(text, enc, maxTokens = DEFAULT_MAX_TOKENS) {
138
+ if (text.length === 0)
139
+ return [];
140
+ const blocks = buildBlocks(splitLines(text));
141
+ // r11 section pass: a section = one heading block through the next heading
142
+ // (preamble = blocks before the first heading). Cut decisions are precomputed
143
+ // per section so they depend only on section sizes, not packing state.
144
+ const minSection = mergeMinTokens(maxTokens);
145
+ const sectionOfBlock = new Array(blocks.length);
146
+ const sectionTexts = [];
147
+ let sec = -1;
148
+ blocks.forEach((b, bi) => {
149
+ if (b.heading || sec === -1) {
150
+ sec++;
151
+ sectionTexts[sec] = '';
152
+ }
153
+ sectionOfBlock[bi] = sec;
154
+ sectionTexts[sec] += b.lines.join('');
155
+ });
156
+ const secTokens = sectionTexts.map(t => enc.encode(t).length);
157
+ // Cut at section s iff the group accumulated since the last cut and section s
158
+ // are BOTH >= minSection. Group size = sum of section token counts (token
159
+ // counts are not additive across joins; the sum is a deterministic threshold
160
+ // proxy, never used as a budget).
161
+ const cutAtSection = new Array(secTokens.length).fill(false);
162
+ {
163
+ let groupTokens = 0;
164
+ for (let s = 0; s < secTokens.length; s++) {
165
+ if (s > 0 && groupTokens >= minSection && secTokens[s] >= minSection) {
166
+ cutAtSection[s] = true;
167
+ groupTokens = 0;
168
+ }
169
+ groupTokens += secTokens[s];
170
+ }
171
+ }
172
+ const pieces = [];
173
+ let acc = '';
174
+ const flush = () => { if (acc.length > 0) {
175
+ pieces.push(acc);
176
+ acc = '';
177
+ } };
178
+ for (let bi = 0; bi < blocks.length; bi++) {
179
+ const b = blocks[bi];
180
+ const btext = b.lines.join('');
181
+ if (btext.length === 0)
182
+ continue;
183
+ if (b.heading) {
184
+ const s = sectionOfBlock[bi];
185
+ // Preferred cut: honor the precomputed section cut. Merged headings still
186
+ // flush when their whole section cannot join the open chunk — splitting
187
+ // at the heading beats orphaning the heading line at a budget flush.
188
+ if (cutAtSection[s])
189
+ flush();
190
+ else if (acc.length > 0 && enc.encode(acc + sectionTexts[s]).length > maxTokens)
191
+ flush();
192
+ }
193
+ if (acc.length > 0 && enc.encode(acc + btext).length > maxTokens)
194
+ flush();
195
+ if (acc.length === 0 && enc.encode(btext).length > maxTokens) {
196
+ pieces.push(...splitOversize(btext, enc, maxTokens)); // mega-block fallback
197
+ continue;
198
+ }
199
+ acc += btext;
200
+ }
201
+ flush();
202
+ const segments = [];
203
+ let cursor = 0;
204
+ for (const p of pieces) {
205
+ const len = [...p].length;
206
+ segments.push({ text: p, start_pos: cursor, end_pos: cursor + len });
207
+ cursor += len;
208
+ }
209
+ return segments;
210
+ }
@@ -1,2 +1,4 @@
1
1
  import { Migration } from './migration-manager.js';
2
+ export type MigrationFaultPoint = 'preflight' | 'roots' | 'revisions' | 'sources' | 'gate';
3
+ export declare function setMigrationFaultPoint(point: MigrationFaultPoint | null): void;
2
4
  export declare const migrations: Migration[];
@@ -1,3 +1,9 @@
1
+ import { OBSERVATION_SCHEMA_SQL } from '../observations/schema.js';
2
+ import { randomUUID } from 'node:crypto';
3
+ let faultPoint = null;
4
+ export function setMigrationFaultPoint(point) {
5
+ faultPoint = point;
6
+ }
1
7
  export const migrations = [
2
8
  {
3
9
  version: 1,
@@ -641,5 +647,138 @@ export const migrations = [
641
647
  db.exec(`DROP TABLE IF EXISTS embedding_profiles`);
642
648
  db.exec(`DROP TABLE IF EXISTS server_meta`);
643
649
  }
650
+ },
651
+ // v13 (spec 2026-07-30 observation-lifecycle §4): 관찰 생애주기 정규화.
652
+ // 이 항목은 DDL 만 만든다. 데이터 변환은 후속 단계에서 같은 항목에 덧붙인다.
653
+ {
654
+ version: 13,
655
+ description: 'Observation lifecycle: roots/revisions/sources/events + immutability triggers',
656
+ up: (db) => {
657
+ db.exec(OBSERVATION_SCHEMA_SQL);
658
+ // ---- spec §5.3 변환 ----
659
+ // MIGRATION_TS = 이 DB 의 v12->v13 트랜잭션 시작시각 하나 (전 행 공유).
660
+ // entities.created_at 을 복사하지 않는다 — 그것은 entity 생성시각을
661
+ // 관찰 기록시각으로 단정하는 것이고, import event 는 실제로 지금 발생했다.
662
+ // 원래 관찰 시각은 unknown 으로 둔다 (v13 은 기록시간 축만 다룬다).
663
+ const MIGRATION_TS = new Date().toISOString();
664
+ const BATCH_ID = randomUUID();
665
+ // 단계 경계 fault injection. 주입은 setMigrationFaultPoint() 로만 하며
666
+ // 환경변수를 보지 않는다 — 환경변수로 두면 프로덕션에 "마이그레이션을 깨는
667
+ // 스위치"가 상시 존재하고, 오설정 한 줄이 .bak 을 남겨 재시작을 막는다
668
+ // (advisor beta 자기의심 2 = "더 나쁘다").
669
+ const fault = (point) => {
670
+ if (faultPoint === point)
671
+ throw new Error(`injected fault at '${point}' (test-only)`);
672
+ };
673
+ // 1) 읽기 전용 preflight: array<string> 검증.
674
+ // JSON.parse 만으로는 객체·숫자·null 요소가 통과한다.
675
+ const rows = db.prepare(`SELECT id, observations FROM entities`).all();
676
+ const parsed = new Map();
677
+ for (const r of rows) {
678
+ let val;
679
+ try {
680
+ val = JSON.parse(r.observations ?? '[]');
681
+ }
682
+ catch {
683
+ throw new Error(`v13 preflight: entity ${r.id} observations is not JSON`);
684
+ }
685
+ if (!Array.isArray(val))
686
+ throw new Error(`v13 preflight: entity ${r.id} observations is not an array<string>`);
687
+ for (const el of val) {
688
+ if (typeof el !== 'string')
689
+ throw new Error(`v13 preflight: entity ${r.id} observations contains a non-string ` +
690
+ `element (${el === null ? 'null' : typeof el}) — array<string> required`);
691
+ }
692
+ parsed.set(r.id, val);
693
+ }
694
+ fault('preflight');
695
+ // 2~4) roots -> revisions -> sources/events.
696
+ // 이 순서가 계약이다: trg_obs_matches_root 가 root 선행을 요구한다.
697
+ // 3패스로 나눈 이유 = 지점별 fault injection 을 검증 가능하게 하려면
698
+ // 단계 경계가 실제로 존재해야 한다(T11).
699
+ const insRoot = db.prepare(`INSERT INTO observation_roots
700
+ (root_id, entity_id, projection_order, created_at) VALUES (?, ?, ?, ?)`);
701
+ const insRev = db.prepare(`INSERT INTO entity_observations
702
+ (observation_id, root_id, entity_id, revision_no, projection_order,
703
+ content, status, supersedes_id, recorded_at, superseded_at)
704
+ VALUES (?, ?, ?, 1, ?, ?, 'active', NULL, ?, NULL)`);
705
+ const insSrc = db.prepare(`INSERT INTO observation_sources
706
+ (observation_id, source_kind, source_ref, source_hash, recorded_at)
707
+ VALUES (?, 'import', 'v12-migration', NULL, ?)`);
708
+ const insEv = db.prepare(`INSERT INTO observation_events
709
+ (event_id, root_id, from_id, to_id, event, change_kind, reason, actor, batch_id, recorded_at)
710
+ VALUES (?, ?, NULL, ?, 'import', NULL, NULL, 'v12-migration', ?, ?)`);
711
+ const plan = [];
712
+ for (const [entityId, arr] of parsed) {
713
+ arr.forEach((content, order) => {
714
+ const rootId = randomUUID();
715
+ insRoot.run(rootId, entityId, order, MIGRATION_TS);
716
+ plan.push({ entityId, rootId, obsId: randomUUID(), order, content });
717
+ });
718
+ }
719
+ fault('roots');
720
+ // pass 2: revisions
721
+ for (const p of plan)
722
+ insRev.run(p.obsId, p.rootId, p.entityId, p.order, p.content, MIGRATION_TS);
723
+ fault('revisions');
724
+ // pass 3: sources + events
725
+ for (const p of plan) {
726
+ insSrc.run(p.obsId, MIGRATION_TS);
727
+ insEv.run(randomUUID(), p.rootId, p.obsId, BATCH_ID, MIGRATION_TS);
728
+ }
729
+ fault('sources');
730
+ // 6) 검증 게이트 (a): FK 무결성.
731
+ // FK 가 켜져 있어도 방어층으로 확인한다.
732
+ const fkBad = db.prepare(`PRAGMA foreign_key_check`).all();
733
+ if (fkBad.length > 0)
734
+ throw new Error(`v13 gate: foreign_key_check reported ${fkBad.length} violation(s)`);
735
+ fault('gate');
736
+ // 7) 검증 게이트 (b): 합성 배열 == 원본 배열 (중복·순서 포함 byte 동일)
737
+ const synthStmt = db.prepare(`SELECT content FROM entity_observations
738
+ WHERE entity_id = ? AND status = 'active' ORDER BY projection_order`);
739
+ for (const [entityId, arr] of parsed) {
740
+ const rebuilt = JSON.stringify(synthStmt.all(entityId).map(x => x.content));
741
+ const original = JSON.stringify(arr);
742
+ if (rebuilt !== original)
743
+ throw new Error(`v13 gate: projection mismatch for entity ${entityId}\n` +
744
+ ` original: ${original}\n rebuilt: ${rebuilt}`);
745
+ }
746
+ },
747
+ down: (db) => {
748
+ // 역순 삭제 (FK 의존 순서). 트리거는 테이블과 함께 사라진다.
749
+ db.exec(`DROP TABLE IF EXISTS observation_events`);
750
+ db.exec(`DROP TABLE IF EXISTS observation_sources`);
751
+ db.exec(`DROP TABLE IF EXISTS entity_observations`);
752
+ db.exec(`DROP TABLE IF EXISTS observation_roots`);
753
+ }
754
+ },
755
+ {
756
+ version: 14,
757
+ description: 'Chunking signature: documents.chunking_signature (schema-only; backfill legacy-unknown)',
758
+ up: (db) => {
759
+ // spec §6.1: schema-only. 전환은 sync 의 content 변경 시에만 (spec §5.1, r4 D3).
760
+ // 백필값은 'legacy-unknown', NOT 'bpe-800-160': custom chunkDocument 파라미터가
761
+ // 기록된 적 없어 단정하면 거짓 표기가 된다 (advisor r1).
762
+ const before = db.prepare(`SELECT count(*) AS n FROM documents`).get().n;
763
+ db.exec(`ALTER TABLE documents ADD COLUMN chunking_signature TEXT NOT NULL DEFAULT 'legacy-unknown'`);
764
+ // 리터럴 고정 — 마이그레이션은 동결된 역사다. 런타임 기본값 진화는 boot upsert 소관.
765
+ db.prepare(`INSERT INTO server_meta (key, value) VALUES ('current_default_chunker', ?)
766
+ ON CONFLICT(key) DO UPDATE SET value = excluded.value`)
767
+ .run('c1:enc=cl100k_base:max=800:overlap=0:fence=on:fallback=cp-exact-800');
768
+ const after = db.prepare(`SELECT count(*) AS n FROM documents`).get().n;
769
+ if (after !== before)
770
+ throw new Error(`v14 gate: documents rows changed ${before} -> ${after}`);
771
+ const cols = db.prepare(`PRAGMA table_info(documents)`).all();
772
+ if (!cols.some(c => c.name === 'chunking_signature'))
773
+ throw new Error('v14 gate: column missing after ALTER');
774
+ const fk = db.prepare(`PRAGMA foreign_key_check`).all();
775
+ if (fk.length > 0)
776
+ throw new Error(`v14 gate: foreign_key_check reported ${fk.length} violations`);
777
+ },
778
+ down: (db) => {
779
+ // 호환성 rollback 뿐 (spec §6.3): chunk 경계는 복원하지 않는다. c1 행은 v13 코드가 읽는다.
780
+ db.exec(`ALTER TABLE documents DROP COLUMN chunking_signature`);
781
+ db.prepare(`DELETE FROM server_meta WHERE key = 'current_default_chunker'`).run();
782
+ }
644
783
  }
645
784
  ];
@@ -0,0 +1,8 @@
1
+ import type Database from 'better-sqlite3';
2
+ export declare function getObservationHistory(db: Database.Database, sel: {
3
+ entity_name?: string;
4
+ observation_id?: string;
5
+ root_id?: string;
6
+ }): {
7
+ roots: any[];
8
+ };
@@ -0,0 +1,64 @@
1
+ // spec §6.2: history 의 유일한 창구.
2
+ // 응답은 항상 roots 배열이다 (entity_name 이면 N개, root_id/observation_id 면 1개).
3
+ // 단일 객체로 두면 entity_name 선택자와 cardinality 가 모순된다.
4
+ export function getObservationHistory(db, sel) {
5
+ // 정확히 하나만 받는다. 우선순위를 조용히 적용하면 두 개를 넘긴 호출자가
6
+ // *다른* 선택자의 답을 받고 그 사실을 모른다 — 실측으로 entity_name 이 유효한데
7
+ // 존재하지 않는 root_id 가 이겨서 `{roots: []}` 가 나갔다(advisor beta 발견 4-2).
8
+ const given = ['entity_name', 'observation_id', 'root_id']
9
+ .filter(k => sel[k] !== undefined && sel[k] !== null && sel[k] !== '');
10
+ if (given.length === 0) {
11
+ throw new Error('getObservationHistory requires one of: entity_name, observation_id, root_id');
12
+ }
13
+ if (given.length > 1) {
14
+ throw new Error(`getObservationHistory takes exactly one selector, got ${given.length} (${given.join(', ')}). ` +
15
+ `Applying a precedence order would silently answer a different question than the one asked.`);
16
+ }
17
+ let rootIds;
18
+ if (sel.root_id) {
19
+ rootIds = [sel.root_id];
20
+ }
21
+ else if (sel.observation_id) {
22
+ const r = db.prepare(`SELECT root_id FROM entity_observations WHERE observation_id = ?`)
23
+ .get(sel.observation_id);
24
+ rootIds = r ? [r.root_id] : [];
25
+ }
26
+ else if (sel.entity_name) {
27
+ const entityId = `entity_${sel.entity_name.toLowerCase().replace(/[^\p{L}\p{N}]/gu, '_')}`;
28
+ rootIds = db.prepare(`SELECT root_id FROM observation_roots
29
+ WHERE entity_id = ? ORDER BY projection_order`)
30
+ .all(entityId).map(r => r.root_id);
31
+ }
32
+ else {
33
+ throw new Error('getObservationHistory requires one of: entity_name, observation_id, root_id');
34
+ }
35
+ const roots = rootIds.map(rid => {
36
+ const root = db.prepare(`SELECT * FROM observation_roots WHERE root_id = ?`).get(rid);
37
+ if (!root)
38
+ return null;
39
+ const revs = db.prepare(`SELECT * FROM entity_observations WHERE root_id = ? ORDER BY revision_no`).all(rid);
40
+ // 정렬 = recorded_at, 동시각이면 event_id 사전순 (결정론)
41
+ const events = db.prepare(`SELECT * FROM observation_events WHERE root_id = ? ORDER BY recorded_at, event_id`).all(rid);
42
+ return {
43
+ root_id: root.root_id,
44
+ entity_id: root.entity_id,
45
+ projection_order: root.projection_order,
46
+ revisions: revs.map(v => ({
47
+ observation_id: v.observation_id,
48
+ revision_no: v.revision_no,
49
+ content: v.content,
50
+ status: v.status,
51
+ supersedes_id: v.supersedes_id,
52
+ recorded_at: v.recorded_at,
53
+ superseded_at: v.superseded_at,
54
+ sources: db.prepare(`SELECT source_kind, source_ref, source_hash, recorded_at
55
+ FROM observation_sources WHERE observation_id = ?
56
+ ORDER BY recorded_at, source_kind, source_ref`).all(v.observation_id),
57
+ // 이 revision 을 from/to 로 지목한 event 만
58
+ events: events.filter(e => e.from_id === v.observation_id || e.to_id === v.observation_id),
59
+ })),
60
+ };
61
+ }).filter(Boolean);
62
+ roots.sort((a, b) => a.projection_order - b.projection_order);
63
+ return { roots };
64
+ }