sphica 0.0.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,3 @@
1
+ -- Deletes the requirements and design sources. Their children (document sections, their files, and full-text index entries) are cleaned up
2
+ -- by the foreign keys' on delete cascade and the knowledge delete trigger. 0003 removes the kinds from the CHECK.
3
+ delete from source_item where kind in ('requirements', 'design');
@@ -0,0 +1,45 @@
1
+ -- sphica: foreign_keys=off
2
+ -- Removes requirements / design from source_item's kind CHECK. SQLite cannot alter a CHECK, so the table is rebuilt.
3
+ -- Dropping with foreign keys on would delete child rows (conversation, knowledge) by cascade, so the runner turns them off.
4
+ -- drop loses the autoincrement maximum, so it is saved and restored (deleted ids are never reused).
5
+ create temp table source_item_seq as select seq from sqlite_sequence where name = 'source_item';
6
+ create table "source_item_new" (
7
+ id integer primary key autoincrement not null,
8
+ connector_id integer not null references connector (id) on delete cascade,
9
+ external_id text not null,
10
+ kind text not null check (kind in ('pull_request', 'issue', 'document')),
11
+ title text not null check (title <> ''),
12
+ state text,
13
+ url text,
14
+ path text check (path <> '' and path not glob '/*' and path not glob '*[/]..[/]*' and path not glob '..[/]*'
15
+ and path not glob '*[/]..' and path <> '..'),
16
+ body text,
17
+ author_identity_id integer references person_identity (id) on delete set null,
18
+ source_created_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', source_created_at) is source_created_at),
19
+ source_updated_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', source_updated_at) is source_updated_at),
20
+ -- For a PR, the merge time (or the close time if closed without merging); for an issue, the close time. null for open items and documents.
21
+ closed_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', closed_at) is closed_at),
22
+ content_hash blob not null check (length(content_hash) = 32),
23
+ metadata text not null default '{}' check (json_valid(metadata) and json_type(metadata) = 'object'),
24
+ synced_at text not null default (strftime('%Y-%m-%dT%H:%M:%fZ', 'now'))
25
+ check (strftime('%Y-%m-%dT%H:%M:%fZ', synced_at) is synced_at),
26
+ unique (connector_id, external_id),
27
+ check (
28
+ case
29
+ when kind = 'document' then path is not null and body is not null and state is null
30
+ and closed_at is null
31
+ else path is null and body is null and state in ('open', 'merged', 'closed') and (state = 'open') = (closed_at is null)
32
+ end
33
+ ),
34
+ -- With a NULL state the CHECK above evaluates to NULL and passes, and the pairing with closed_at is not enforced either.
35
+ constraint source_item_state_required check (kind = 'document' or state is not null)
36
+ ) strict;
37
+ insert into "source_item_new" select * from source_item;
38
+ drop table source_item;
39
+ alter table "source_item_new" rename to "source_item";
40
+ create index source_item_listing on source_item (connector_id, kind, state, source_updated_at desc);
41
+ update sqlite_sequence set seq = (select seq from source_item_seq)
42
+ where name = 'source_item' and seq < (select seq from source_item_seq);
43
+ insert into sqlite_sequence (name, seq) select 'source_item', seq from source_item_seq
44
+ where not exists (select 1 from sqlite_sequence where name = 'source_item');
45
+ drop table source_item_seq;
@@ -0,0 +1,46 @@
1
+ -- sphica: foreign_keys=off
2
+ -- Adds extra search words per record (knowledge_terms) and rebuilds the knowledge index with a third column for them.
3
+ create table knowledge_terms (
4
+ knowledge_id integer primary key not null references knowledge (id) on delete cascade,
5
+ terms text not null check (terms <> '' and length(terms) <= 400),
6
+ content_hash blob not null check (length(content_hash) = 32),
7
+ source text not null check (source in ('trace', 'pr', 'import')),
8
+ written_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', written_at) is written_at)
9
+ ) strict;
10
+
11
+ create view knowledge_search_text as
12
+ select k.id,
13
+ sphica_terms(coalesce(k.heading, '')) as h,
14
+ sphica_terms(k.body || char(10) || coalesce(k.reason, '')) as b,
15
+ sphica_terms(coalesce(t.terms, '')) as e
16
+ from knowledge k
17
+ left join knowledge_terms t on t.knowledge_id = k.id and t.content_hash = k.content_hash;
18
+
19
+ drop trigger knowledge_fts_ai;
20
+ drop trigger knowledge_fts_ad;
21
+ drop trigger knowledge_fts_au;
22
+ drop table knowledge_fts;
23
+ create virtual table knowledge_fts using fts5(h, b, e, content='', contentless_delete=1);
24
+ create trigger knowledge_fts_ai after insert on knowledge begin
25
+ insert into knowledge_fts (rowid, h, b, e) select id, h, b, e from knowledge_search_text where id = new.id;
26
+ end;
27
+ create trigger knowledge_fts_ad after delete on knowledge begin
28
+ delete from knowledge_fts where rowid = old.id;
29
+ end;
30
+ create trigger knowledge_fts_au after update of heading, body, reason, content_hash on knowledge begin
31
+ delete from knowledge_fts where rowid = old.id;
32
+ insert into knowledge_fts (rowid, h, b, e) select id, h, b, e from knowledge_search_text where id = new.id;
33
+ end;
34
+ create trigger knowledge_terms_ai after insert on knowledge_terms begin
35
+ delete from knowledge_fts where rowid = new.knowledge_id;
36
+ insert into knowledge_fts (rowid, h, b, e) select id, h, b, e from knowledge_search_text where id = new.knowledge_id;
37
+ end;
38
+ create trigger knowledge_terms_au after update of terms, content_hash on knowledge_terms begin
39
+ delete from knowledge_fts where rowid = new.knowledge_id;
40
+ insert into knowledge_fts (rowid, h, b, e) select id, h, b, e from knowledge_search_text where id = new.knowledge_id;
41
+ end;
42
+ create trigger knowledge_terms_ad after delete on knowledge_terms begin
43
+ delete from knowledge_fts where rowid = old.knowledge_id;
44
+ insert into knowledge_fts (rowid, h, b, e) select id, h, b, e from knowledge_search_text where id = old.knowledge_id;
45
+ end;
46
+ insert into knowledge_fts (rowid, h, b, e) select id, h, b, e from knowledge_search_text;
@@ -0,0 +1,20 @@
1
+ -- Recreates the triggers and the view that call the tokenizer function, so they call sphica_terms. The index keeps its rows
2
+ -- (the tokenizer is unchanged), and the old function need not be registered: dropping a trigger or view does not call it.
3
+ drop trigger message_fts_ai;
4
+ drop trigger message_fts_au;
5
+ drop view knowledge_search_text;
6
+
7
+ create trigger message_fts_ai after insert on message when new.indexed = 1 begin
8
+ insert into message_fts (rowid, lexemes) values (new.seq, sphica_terms(new.body));
9
+ end;
10
+ create trigger message_fts_au after update of body, indexed on message begin
11
+ delete from message_fts where rowid = old.seq and old.indexed = 1;
12
+ insert into message_fts (rowid, lexemes) select new.seq, sphica_terms(new.body) where new.indexed = 1;
13
+ end;
14
+ create view knowledge_search_text as
15
+ select k.id,
16
+ sphica_terms(coalesce(k.heading, '')) as h,
17
+ sphica_terms(k.body || char(10) || coalesce(k.reason, '')) as b,
18
+ sphica_terms(coalesce(t.terms, '')) as e
19
+ from knowledge k
20
+ left join knowledge_terms t on t.knowledge_id = k.id and t.content_hash = k.content_hash;
package/db/schema.sql ADDED
@@ -0,0 +1,349 @@
1
+ -- The source of truth for sphica's database (SQLite, `node:sqlite`). It lets one owner look up decisions and conversations on that machine.
2
+ -- **Each machine is independent and shares no records.** One file (~/.sphica/sphica.db) is one database, with no schema qualifiers.
3
+ --
4
+ -- Three boundaries: the current state of sources (connector / source_item), verbatim conversations (conversation / message),
5
+ -- and searchable knowledge (knowledge). Work status (work_item) is state that gets updated, so it has its own table.
6
+ --
7
+ -- The version is `pragma user_version` at the end. MCP and the CLI compare it with SCHEMA_REVISION in server/src/db.ts
8
+ -- when opening, and stop on a mismatch. `sphica init` creates an empty database (server/src/admin.ts).
9
+ -- Every table is STRICT (rejects type mismatches). Every primary key says not null (SQLite allows NULL in non-integer primary keys).
10
+ -- server/src/sqlite.ts sets journal_mode and foreign_keys per connection (not here).
11
+ --
12
+ -- Times are ISO 8601 UTC strings (the `Date#toISOString()` form), so lexical order is chronological order.
13
+ -- `strftime(...) is column` rejects values not in normal form (`...:00Z` without milliseconds, offsets, dates not on the calendar).
14
+ -- Mixed forms break ordering within a second and make date filters miss at the boundaries.
15
+
16
+ create table project (
17
+ id integer primary key autoincrement not null,
18
+ -- A key from the normalized git remote (`git:github.com/owner/repo`), or a key set per machine for a project without a remote.
19
+ -- Local paths are not stored. Locations differ per machine.
20
+ key text not null unique check (
21
+ (key glob 'git:*' and key not glob '*[ ' || char(9) || '-' || char(13) || ']*' and length(key) > 4)
22
+ or (key glob 'local:[a-z0-9]*' and substr(key, 7) not glob '*[^a-z0-9._-]*')),
23
+ name text not null check (name <> ''),
24
+ created_at text not null default (strftime('%Y-%m-%dT%H:%M:%fZ', 'now'))
25
+ check (strftime('%Y-%m-%dT%H:%M:%fZ', created_at) is created_at)
26
+ ) strict;
27
+
28
+ create table person (
29
+ id integer primary key autoincrement not null,
30
+ display_name text not null unique check (display_name <> ''),
31
+ is_self integer not null default 0 check (is_self in (0, 1))
32
+ ) strict;
33
+ -- Exactly one person is the owner who asks. This decides who "I" is in "what did I say?".
34
+ create unique index person_one_self on person (is_self) where is_self = 1;
35
+
36
+ -- Identifiers at a source. A GitHub user id stays the same when the login changes, so it goes in external_id.
37
+ create table person_identity (
38
+ id integer primary key autoincrement not null,
39
+ person_id integer references person (id) on delete set null,
40
+ provider text not null check (provider in ('github')),
41
+ external_id text not null,
42
+ handle text not null,
43
+ unique (provider, external_id)
44
+ ) strict;
45
+ create index person_identity_handle on person_identity (provider, lower(handle));
46
+
47
+ -- Per source, the last imported version and the latest result. No secrets (they live in the syncing machine's environment).
48
+ -- For documents, the imported commit (head_oid). The next sync imports automatically only commits that fast-forward from it.
49
+ -- For GitHub, the time the fetch started (snapshot_at). A fetch that started earlier is not written, even if it commits later.
50
+ create table connector (
51
+ id integer primary key autoincrement not null,
52
+ project_id integer not null references project (id) on delete cascade,
53
+ provider text not null check (provider in ('github', 'docs')),
54
+ head_oid text check (head_oid is null or (provider = 'docs'
55
+ and (length(head_oid) = 40 or length(head_oid) = 64) and head_oid not glob '*[^0-9a-f]*')),
56
+ snapshot_at text check (snapshot_at is null or provider = 'github')
57
+ check (strftime('%Y-%m-%dT%H:%M:%fZ', snapshot_at) is snapshot_at),
58
+ last_success_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', last_success_at) is last_success_at),
59
+ last_error text,
60
+ unique (project_id, provider)
61
+ ) strict;
62
+
63
+ -- Paths the docs sync does not import. Not every tracked Markdown file states facts (such as audit fixtures).
64
+ -- **This is importer-side configuration.** It must work for read-only projects too, so it is not a manifest in the repository.
65
+ -- file matches the path exactly, and directory matches paths starting with `<path>/`. Only created for the docs connector.
66
+ create table docs_exclude (
67
+ connector_id integer not null references connector (id) on delete cascade,
68
+ kind text not null check (kind in ('file', 'directory')),
69
+ path text not null check (
70
+ path <> '' and path not glob '/*' and path not glob '*[/]..[/]*' and path not glob '..[/]*'
71
+ and path not glob '*[/]..' and path <> '..' and path not glob '*/'
72
+ and path not glob '*[' || char(1) || '-' || char(31) || char(127) || ']*'),
73
+ primary key (connector_id, kind, path)
74
+ ) strict;
75
+
76
+ -- The current state of a source. Items confirmed gone by a complete listing are deleted with their rows (no tombstones).
77
+ -- Documents keep their original text in body. Search uses the knowledge sections, and joining sections never restores the original.
78
+ create table "source_item" (
79
+ id integer primary key autoincrement not null,
80
+ connector_id integer not null references connector (id) on delete cascade,
81
+ external_id text not null,
82
+ kind text not null check (kind in ('pull_request', 'issue', 'document')),
83
+ title text not null check (title <> ''),
84
+ state text,
85
+ url text,
86
+ path text check (path <> '' and path not glob '/*' and path not glob '*[/]..[/]*' and path not glob '..[/]*'
87
+ and path not glob '*[/]..' and path <> '..'),
88
+ body text,
89
+ author_identity_id integer references person_identity (id) on delete set null,
90
+ source_created_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', source_created_at) is source_created_at),
91
+ source_updated_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', source_updated_at) is source_updated_at),
92
+ -- For a PR, the merge time (or the close time if closed without merging); for an issue, the close time. null for open items and documents.
93
+ closed_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', closed_at) is closed_at),
94
+ content_hash blob not null check (length(content_hash) = 32),
95
+ metadata text not null default '{}' check (json_valid(metadata) and json_type(metadata) = 'object'),
96
+ synced_at text not null default (strftime('%Y-%m-%dT%H:%M:%fZ', 'now'))
97
+ check (strftime('%Y-%m-%dT%H:%M:%fZ', synced_at) is synced_at),
98
+ unique (connector_id, external_id),
99
+ check (
100
+ case
101
+ when kind = 'document' then path is not null and body is not null and state is null
102
+ and closed_at is null
103
+ else path is null and body is null and state in ('open', 'merged', 'closed') and (state = 'open') = (closed_at is null)
104
+ end
105
+ ),
106
+ -- With a NULL state the CHECK above evaluates to NULL and passes, and the pairing with closed_at is not enforced either.
107
+ constraint source_item_state_required check (kind = 'document' or state is not null)
108
+ ) strict;
109
+ create index source_item_listing on source_item (connector_id, kind, state, source_updated_at desc);
110
+
111
+ -- A conversation: one coding session, or one GitHub PR or issue.
112
+ -- The id is a uuid derived deterministically from (project, origin, external_id). Sending the same session twice adds no rows.
113
+ create table conversation (
114
+ id text primary key not null,
115
+ project_id integer not null references project (id) on delete cascade,
116
+ source_item_id integer references source_item (id) on delete cascade,
117
+ origin text not null check (origin in ('claude-code', 'codex', 'github')),
118
+ external_id text not null,
119
+ branch text,
120
+ started_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', started_at) is started_at),
121
+ unique (project_id, origin, external_id),
122
+ check ((origin = 'github') = (source_item_id is not null))
123
+ ) strict;
124
+ create index conversation_recent on conversation (project_id, started_at desc);
125
+
126
+ -- One row per message. self is what the owner typed, assistant is the AI's last reply or an AI reviewer, and bot is an automated notice.
127
+ -- Oversized messages keep only their start and end, with truncated and the original size (UTF-8 bytes).
128
+ create table message (
129
+ -- seq is the FTS5 rowid. It is an explicit integer primary key rather than the implicit rowid, so VACUUM does not renumber it
130
+ seq integer primary key not null,
131
+ id text not null unique,
132
+ conversation_id text not null references conversation (id) on delete cascade,
133
+ external_id text not null,
134
+ turn_id text,
135
+ reply_to_id text references message (id) on delete set null,
136
+ speaker_kind text not null check (speaker_kind in ('self', 'person', 'assistant', 'bot')),
137
+ identity_id integer references person_identity (id) on delete set null,
138
+ body text not null check (body <> ''),
139
+ truncated integer not null default 0 check (truncated in (0, 1)),
140
+ original_bytes integer not null check (original_bytes > 0),
141
+ url text,
142
+ sent_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', sent_at) is sent_at),
143
+ content_hash blob not null check (length(content_hash) = 32),
144
+ -- Whether it goes into the full-text index. 0 for AI replies in coding sessions and automated notices (decided by indexesMessage in capture.ts / github.ts)
145
+ indexed integer not null check (indexed in (0, 1)),
146
+ unique (conversation_id, external_id),
147
+ check (truncated = 1 or original_bytes = length(cast(body as blob))),
148
+ check (truncated = 0 or original_bytes > length(cast(body as blob)))
149
+ ) strict;
150
+ create index message_order on message (conversation_id, sent_at);
151
+ create index message_by_identity on message (identity_id, sent_at desc) where identity_id is not null;
152
+ create index message_self on message (sent_at desc) where speaker_kind = 'self';
153
+
154
+ -- The full-text index. rowid = message.seq. Terms are split by sphica_terms() (terms() in server/src/text.ts, registered per connection).
155
+ -- Writes from a connection without the function fail with no such function (the index never silently misses rows).
156
+ create virtual table message_fts using fts5(lexemes, content='', contentless_delete=1);
157
+ create trigger message_fts_ai after insert on message when new.indexed = 1 begin
158
+ insert into message_fts (rowid, lexemes) values (new.seq, sphica_terms(new.body));
159
+ end;
160
+ create trigger message_fts_ad after delete on message when old.indexed = 1 begin
161
+ delete from message_fts where rowid = old.seq;
162
+ end;
163
+ create trigger message_fts_au after update of body, indexed on message begin
164
+ delete from message_fts where rowid = old.seq and old.indexed = 1;
165
+ insert into message_fts (rowid, lexemes) select new.seq, sphica_terms(new.body) where new.indexed = 1;
166
+ end;
167
+
168
+ -- Files linked to messages. Capture links an edited file (edit) to the owner's last message before the edit.
169
+ -- The GitHub sync links a file pointed to in a review (review) to that review message.
170
+ -- read records requirements and design documents read in the past, and is no longer written. path is relative to the project root.
171
+ create table message_file (
172
+ message_id text not null references message (id) on delete cascade,
173
+ path text not null check (path <> '' and path not glob '/*' and path not glob '*[/]..[/]*' and path not glob '..[/]*'
174
+ and path not glob '*[/]..' and path <> '..'),
175
+ action text not null check (action in ('edit', 'read', 'review')),
176
+ line_start integer check (line_start > 0),
177
+ line_end integer check (line_end >= line_start),
178
+ primary key (message_id, path, action)
179
+ ) strict;
180
+ create index message_file_path on message_file (path);
181
+
182
+ -- Work status, updated by trace. active / blocked / paused are candidates for continuing work.
183
+ create table work_item (
184
+ id integer primary key autoincrement not null,
185
+ project_id integer not null references project (id) on delete cascade,
186
+ source_key text not null,
187
+ title text not null check (title <> ''),
188
+ goal text not null check (goal <> ''),
189
+ current text not null check (current <> ''),
190
+ next text not null default '[]' check (json_valid(next) and json_type(next) = 'array'),
191
+ status text not null check (status in ('active', 'blocked', 'paused', 'done', 'abandoned')),
192
+ conversation_id text references conversation (id) on delete set null,
193
+ updated_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', updated_at) is updated_at),
194
+ unique (project_id, source_key)
195
+ ) strict;
196
+ create index work_item_open on work_item (project_id, updated_at desc) where status in ('active', 'blocked', 'paused');
197
+
198
+ -- A unit of searchable knowledge: decisions trace picked from conversations, and document sections.
199
+ -- Overturned decisions are not deleted (deleting them gets them proposed again). They become superseded and point to the successor.
200
+ -- stance follows from kind and status, and filters searches for paths not to take.
201
+ create table knowledge (
202
+ id integer primary key autoincrement not null,
203
+ project_id integer not null references project (id) on delete cascade,
204
+ source_item_id integer references source_item (id) on delete cascade,
205
+ conversation_id text references conversation (id) on delete cascade,
206
+ work_item_id integer references work_item (id) on delete set null,
207
+ source_key text not null,
208
+ kind text not null check (kind in ('decision', 'option', 'constraint', 'non_goal', 'dead_end', 'finding', 'debt',
209
+ 'verification', 'question', 'document')),
210
+ status text,
211
+ stance text not null generated always as (
212
+ case
213
+ when kind in ('constraint', 'non_goal', 'debt') then (case status when 'active' then 'dont' else 'neutral' end)
214
+ when kind = 'dead_end' then 'dont'
215
+ when kind = 'option' then (case status when 'chosen' then 'do' else 'dont' end)
216
+ when kind = 'decision' then (case status when 'accepted' then 'do' when 'proposed' then 'neutral' else 'dont' end)
217
+ when kind = 'verification' then (case status when 'failed' then 'dont' else 'neutral' end)
218
+ else 'neutral'
219
+ end
220
+ ) stored,
221
+ confidence text check (confidence in ('fact', 'inference', 'opinion')),
222
+ decision_id integer references knowledge (id) on delete cascade,
223
+ superseded_by_id integer references knowledge (id) on delete set null,
224
+ heading text,
225
+ body text not null check (body <> ''),
226
+ reason text,
227
+ confirmation text,
228
+ command text,
229
+ downsides text not null default '[]' check (json_valid(downsides) and json_type(downsides) = 'array'),
230
+ refs text not null default '[]' check (json_valid(refs) and json_type(refs) = 'array'),
231
+ occurred_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', occurred_at) is occurred_at),
232
+ content_hash blob not null check (length(content_hash) = 32),
233
+ unique (project_id, source_key),
234
+ check (source_item_id is not null or conversation_id is not null),
235
+ check (kind <> 'document' or (source_item_id is not null and heading is not null)),
236
+ check (
237
+ case kind
238
+ when 'decision' then status is not null and status in ('proposed', 'accepted', 'rejected', 'superseded')
239
+ when 'option' then status is not null and status in ('chosen', 'rejected', 'was_chosen')
240
+ when 'verification' then status is not null and status in ('passed', 'failed', 'not_run')
241
+ when 'question' then status is not null and status in ('open', 'blocking', 'resolved')
242
+ when 'constraint' then status is not null and status in ('active', 'retired')
243
+ when 'non_goal' then status is not null and status in ('active', 'retired')
244
+ when 'debt' then status is not null and status in ('active', 'retired')
245
+ else status is null
246
+ end
247
+ ),
248
+ check (case kind when 'option' then decision_id is not null when 'verification' then 1 else decision_id is null end),
249
+ check ((kind = 'decision' and status = 'superseded') = (superseded_by_id is not null)),
250
+ check (superseded_by_id is null or superseded_by_id <> id),
251
+ check (confirmation is null or kind = 'decision'),
252
+ check (command is null or kind = 'verification'),
253
+ check (json_array_length(downsides) = 0 or kind = 'decision'),
254
+ check (kind <> 'document' or work_item_id is null)
255
+ ) strict;
256
+ create index knowledge_listing on knowledge (project_id, kind, status, occurred_at desc);
257
+ create index knowledge_work on knowledge (work_item_id) where work_item_id is not null;
258
+
259
+ -- Extra search words for a record (synonyms, abbreviations, English equivalents of its words). **Search only**: no search result, read,
260
+ -- or CLI output shows them. content_hash is the record's hash when they were written; they are indexed only while it
261
+ -- still matches, so a record whose text changed stops being found by words written for its old text. source says who wrote them.
262
+ create table knowledge_terms (
263
+ knowledge_id integer primary key not null references knowledge (id) on delete cascade,
264
+ terms text not null check (terms <> '' and length(terms) <= 400),
265
+ content_hash blob not null check (length(content_hash) = 32),
266
+ source text not null check (source in ('trace', 'pr', 'import')),
267
+ written_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', written_at) is written_at)
268
+ ) strict;
269
+
270
+ -- What the knowledge index holds for each record: heading (h), body plus reason (b), and the extra search words whose hash matches (e).
271
+ -- The triggers and `sphica db reindex` all insert from here, so the rule lives in one place.
272
+ create view knowledge_search_text as
273
+ select k.id,
274
+ sphica_terms(coalesce(k.heading, '')) as h,
275
+ sphica_terms(k.body || char(10) || coalesce(k.reason, '')) as b,
276
+ sphica_terms(coalesce(t.terms, '')) as e
277
+ from knowledge k
278
+ left join knowledge_terms t on t.knowledge_id = k.id and t.content_hash = k.content_hash;
279
+
280
+ -- The full-text index. rowid = knowledge.id. Search uses bm25(knowledge_fts, 3, 1, 1).
281
+ -- Trace headings hold the work title, and document section headings hold the path and heading levels.
282
+ create virtual table knowledge_fts using fts5(h, b, e, content='', contentless_delete=1);
283
+ create trigger knowledge_fts_ai after insert on knowledge begin
284
+ insert into knowledge_fts (rowid, h, b, e) select id, h, b, e from knowledge_search_text where id = new.id;
285
+ end;
286
+ create trigger knowledge_fts_ad after delete on knowledge begin
287
+ delete from knowledge_fts where rowid = old.id;
288
+ end;
289
+ create trigger knowledge_fts_au after update of heading, body, reason, content_hash on knowledge begin
290
+ delete from knowledge_fts where rowid = old.id;
291
+ insert into knowledge_fts (rowid, h, b, e) select id, h, b, e from knowledge_search_text where id = new.id;
292
+ end;
293
+ create trigger knowledge_terms_ai after insert on knowledge_terms begin
294
+ delete from knowledge_fts where rowid = new.knowledge_id;
295
+ insert into knowledge_fts (rowid, h, b, e) select id, h, b, e from knowledge_search_text where id = new.knowledge_id;
296
+ end;
297
+ create trigger knowledge_terms_au after update of terms, content_hash on knowledge_terms begin
298
+ delete from knowledge_fts where rowid = new.knowledge_id;
299
+ insert into knowledge_fts (rowid, h, b, e) select id, h, b, e from knowledge_search_text where id = new.knowledge_id;
300
+ end;
301
+ create trigger knowledge_terms_ad after delete on knowledge_terms begin
302
+ delete from knowledge_fts where rowid = old.knowledge_id;
303
+ insert into knowledge_fts (rowid, h, b, e) select id, h, b, e from knowledge_search_text where id = old.knowledge_id;
304
+ end;
305
+
306
+ -- Direct links between decisions and files. applies_to is a constraint shown before editing, and evidence is a file cited as grounds.
307
+ create table knowledge_file (
308
+ knowledge_id integer not null references knowledge (id) on delete cascade,
309
+ path text not null check (path <> '' and path not glob '/*' and path not glob '*[/]..[/]*' and path not glob '..[/]*'
310
+ and path not glob '*[/]..' and path <> '..'),
311
+ role text not null check (role in ('applies_to', 'evidence')),
312
+ line_start integer check (line_start > 0),
313
+ line_end integer check (line_end >= line_start),
314
+ primary key (knowledge_id, path, role)
315
+ ) strict;
316
+ create index knowledge_file_path on knowledge_file (path, role);
317
+
318
+ -- The 3 views capture (the capture connection) can write. The authorizer in server/src/sqlite.ts allows capture only inserts into these views
319
+ -- and the writes inside the triggers below. source_item_id, identity_id, reply_to_id, and url are not in the views, so capture
320
+ -- can neither create GitHub conversations nor claim someone else's identity. Conversation ids can be computed deterministically, so adding messages
321
+ -- to an existing conversation is not blocked (the remaining surface if the capture path is abused).
322
+ create view capture_conversation as
323
+ select id, project_id, origin, external_id, branch, started_at from conversation;
324
+ create trigger capture_conversation_insert instead of insert on capture_conversation begin
325
+ insert into conversation (id, project_id, origin, external_id, branch, started_at)
326
+ values (new.id, new.project_id, new.origin, new.external_id, new.branch, new.started_at)
327
+ on conflict do nothing;
328
+ end;
329
+
330
+ create view capture_message as
331
+ select id, conversation_id, external_id, turn_id, speaker_kind, body, truncated, original_bytes, sent_at, content_hash, indexed
332
+ from message;
333
+ create trigger capture_message_insert instead of insert on capture_message begin
334
+ insert into message (id, conversation_id, external_id, turn_id, speaker_kind, body, truncated, original_bytes,
335
+ sent_at, content_hash, indexed)
336
+ values (new.id, new.conversation_id, new.external_id, new.turn_id, new.speaker_kind, new.body, new.truncated,
337
+ new.original_bytes, new.sent_at, new.content_hash, new.indexed)
338
+ on conflict do nothing;
339
+ end;
340
+
341
+ -- Link to the owner's last message before the edit. If that message is not in this database (such as a session that moved to another project midway), drop it.
342
+ create view capture_message_file as select message_id, path, action from message_file;
343
+ create trigger capture_message_file_insert instead of insert on capture_message_file begin
344
+ insert into message_file (message_id, path, action)
345
+ select new.message_id, new.path, new.action where exists (select 1 from message where id = new.message_id)
346
+ on conflict do nothing;
347
+ end;
348
+
349
+ pragma user_version = 5;