sphica 0.3.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/db/schema.sql CHANGED
@@ -1,22 +1,20 @@
1
- -- The source of truth for sphica's database (SQLite, `node:sqlite`). It lets one owner look up decisions and conversations on that machine.
2
- -- **Each machine is independent and shares no records.** One file (~/.sphica/sphica.db) is one database, with no schema qualifiers.
1
+ -- The source of truth for Sphica's database (SQLite, `node:sqlite`): the memory of past implementation and decisions for one owner on one machine.
2
+ -- Generation 2 (the 0.5.0 rebuild). `sphica_generation` holds the generation; `pragma user_version` is the revision within it.
3
+ -- A database of another generation is refused without being changed.
3
4
  --
4
- -- Three boundaries: the current state of sources (connector / source_item), verbatim conversations (conversation / message),
5
- -- and searchable knowledge (knowledge). Work status (work_item) is state that gets updated, so it has its own table.
6
- --
7
- -- The version is `pragma user_version` at the end. MCP and the CLI compare it with SCHEMA_REVISION in server/src/db.ts
8
- -- when opening, and stop on a mismatch. `sphica init` creates an empty database (server/src/admin.ts).
9
- -- Every table is STRICT (rejects type mismatches). Every primary key says not null (SQLite allows NULL in non-integer primary keys).
10
- -- server/src/sqlite.ts sets journal_mode and foreign_keys per connection (not here).
11
- --
12
- -- Times are ISO 8601 UTC strings (the `Date#toISOString()` form), so lexical order is chronological order.
13
- -- `strftime(...) is column` rejects values not in normal form (`...:00Z` without milliseconds, offsets, dates not on the calendar).
14
- -- Mixed forms break ordering within a second and make date filters miss at the boundaries.
5
+ -- Four boundaries:
6
+ -- captured sources session, source, artifact_link, edit_observation, external_reference: what was said or written, never rewritten
7
+ -- extracted units unit and its option, evidence, adoption, link, state, anchor, alias tables: what was decided or implemented
8
+ -- processing extraction_run, source_processing: what has been looked at and saved, so gaps are counted
9
+ -- work and delivery work, delivery, delivery_unit: the current work status and what the hooks injected
10
+ -- Every table is STRICT and every primary key is not null. Times are ISO 8601 UTC (`Date#toISOString()`); `strftime(...) is column` rejects others.
11
+ -- Byte offsets are into the UTF-8 bytes of source.text. Project consistency across tables is enforced by triggers, not only by code.
12
+
13
+ create table sphica_generation (generation integer not null check (generation = 2)) strict;
14
+ insert into sphica_generation values (2);
15
15
 
16
16
  create table project (
17
17
  id integer primary key autoincrement not null,
18
- -- A key from the normalized git remote (`git:github.com/owner/repo`), or a key set per machine for a project without a remote.
19
- -- Local paths are not stored. Locations differ per machine.
20
18
  key text not null unique check (
21
19
  (key glob 'git:*' and key not glob '*[ ' || char(9) || '-' || char(13) || ']*' and length(key) > 4)
22
20
  or (key glob 'local:[a-z0-9]*' and substr(key, 7) not glob '*[^a-z0-9._-]*')),
@@ -25,325 +23,674 @@ create table project (
25
23
  check (strftime('%Y-%m-%dT%H:%M:%fZ', created_at) is created_at)
26
24
  ) strict;
27
25
 
28
- create table person (
29
- id integer primary key autoincrement not null,
30
- display_name text not null unique check (display_name <> ''),
31
- is_self integer not null default 0 check (is_self in (0, 1))
32
- ) strict;
33
- -- Exactly one person is the owner who asks. This decides who "I" is in "what did I say?".
34
- create unique index person_one_self on person (is_self) where is_self = 1;
35
-
36
- -- Identifiers at a source. A GitHub user id stays the same when the login changes, so it goes in external_id.
37
- create table person_identity (
38
- id integer primary key autoincrement not null,
39
- person_id integer references person (id) on delete set null,
26
+ -- Identities bound to the owner of this machine (the owner's GitHub account id), set by the owner through the CLI.
27
+ -- An external source counts as the owner's words only when its author id matches; a login name alone never does.
28
+ create table owner_identity (
40
29
  provider text not null check (provider in ('github')),
41
- external_id text not null,
42
- handle text not null,
43
- unique (provider, external_id)
30
+ external_id text not null check (external_id <> ''),
31
+ login text,
32
+ bound_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', bound_at) is bound_at),
33
+ primary key (provider, external_id)
44
34
  ) strict;
45
- create index person_identity_handle on person_identity (provider, lower(handle));
46
35
 
47
- -- Per source, the last imported version and the latest result. No secrets (they live in the syncing machine's environment).
48
- -- For documents, the imported commit (head_oid). The next sync imports automatically only commits that fast-forward from it.
49
- -- For GitHub, the time the fetch started (snapshot_at). A fetch that started earlier is not written, even if it commits later.
50
- create table connector (
51
- id integer primary key autoincrement not null,
52
- project_id integer not null references project (id) on delete cascade,
53
- provider text not null check (provider in ('github', 'docs')),
54
- head_oid text check (head_oid is null or (provider = 'docs'
55
- and (length(head_oid) = 40 or length(head_oid) = 64) and head_oid not glob '*[^0-9a-f]*')),
56
- snapshot_at text check (snapshot_at is null or provider = 'github')
57
- check (strftime('%Y-%m-%dT%H:%M:%fZ', snapshot_at) is snapshot_at),
58
- last_success_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', last_success_at) is last_success_at),
59
- last_error text,
60
- unique (project_id, provider)
61
- ) strict;
62
-
63
- -- Paths the docs sync does not import. Not every tracked Markdown file states facts (such as audit fixtures).
64
- -- **This is importer-side configuration.** It must work for read-only projects too, so it is not a manifest in the repository.
65
- -- file matches the path exactly, and directory matches paths starting with `<path>/`. Only created for the docs connector.
66
- create table docs_exclude (
67
- connector_id integer not null references connector (id) on delete cascade,
68
- kind text not null check (kind in ('file', 'directory')),
69
- path text not null check (
70
- path <> '' and path not glob '/*' and path not glob '*[/]..[/]*' and path not glob '..[/]*'
71
- and path not glob '*[/]..' and path <> '..' and path not glob '*/'
72
- and path not glob '*[' || char(1) || '-' || char(31) || char(127) || ']*'),
73
- primary key (connector_id, kind, path)
74
- ) strict;
75
-
76
- -- The current state of a source. Items confirmed gone by a complete listing are deleted with their rows (no tombstones).
77
- -- Documents keep their original text in body. Search uses the knowledge sections, and joining sections never restores the original.
78
- create table "source_item" (
79
- id integer primary key autoincrement not null,
80
- connector_id integer not null references connector (id) on delete cascade,
81
- external_id text not null,
82
- kind text not null check (kind in ('pull_request', 'issue', 'document')),
83
- title text not null check (title <> ''),
84
- state text,
85
- url text,
86
- path text check (path <> '' and path not glob '/*' and path not glob '*[/]..[/]*' and path not glob '..[/]*'
87
- and path not glob '*[/]..' and path <> '..'),
88
- body text,
89
- author_identity_id integer references person_identity (id) on delete set null,
90
- source_created_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', source_created_at) is source_created_at),
91
- source_updated_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', source_updated_at) is source_updated_at),
92
- -- For a PR, the merge time (or the close time if closed without merging); for an issue, the close time. null for open items and documents.
93
- closed_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', closed_at) is closed_at),
94
- content_hash blob not null check (length(content_hash) = 32),
95
- metadata text not null default '{}' check (json_valid(metadata) and json_type(metadata) = 'object'),
96
- synced_at text not null default (strftime('%Y-%m-%dT%H:%M:%fZ', 'now'))
97
- check (strftime('%Y-%m-%dT%H:%M:%fZ', synced_at) is synced_at),
98
- unique (connector_id, external_id),
99
- check (
100
- case
101
- when kind = 'document' then path is not null and body is not null and state is null
102
- and closed_at is null
103
- else path is null and body is null and state in ('open', 'merged', 'closed') and (state = 'open') = (closed_at is null)
104
- end
105
- ),
106
- -- With a NULL state the CHECK above evaluates to NULL and passes, and the pairing with closed_at is not enforced either.
107
- constraint source_item_state_required check (kind = 'document' or state is not null)
108
- ) strict;
109
- create index source_item_listing on source_item (connector_id, kind, state, source_updated_at desc);
110
-
111
- -- A conversation: one coding session, or one GitHub PR or issue.
112
- -- The id is a uuid derived deterministically from (project, origin, external_id). Sending the same session twice adds no rows.
113
- create table conversation (
36
+ -- One coding session. The id is a uuid derived from (project, host, external_id), so resending is idempotent.
37
+ create table session (
114
38
  id text primary key not null,
115
39
  project_id integer not null references project (id) on delete cascade,
116
- source_item_id integer references source_item (id) on delete cascade,
117
- origin text not null check (origin in ('claude-code', 'codex', 'github')),
40
+ host text not null check (host in ('claude-code', 'codex')),
118
41
  external_id text not null,
119
42
  branch text,
120
43
  started_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', started_at) is started_at),
121
- unique (project_id, origin, external_id),
122
- check ((origin = 'github') = (source_item_id is not null))
44
+ unique (project_id, host, external_id)
123
45
  ) strict;
124
- create index conversation_recent on conversation (project_id, started_at desc);
125
46
 
126
- -- One row per message. self is what the owner typed, assistant is the AI's last reply or an AI reviewer, and bot is an automated notice.
127
- -- Oversized messages keep only their start and end, with truncated and the original size (UTF-8 bytes).
128
- create table message (
129
- -- seq is the FTS5 rowid. It is an explicit integer primary key rather than the implicit rowid, so VACUUM does not renumber it
130
- seq integer primary key not null,
131
- id text not null unique,
132
- conversation_id text not null references conversation (id) on delete cascade,
133
- external_id text not null,
47
+ -- A captured source, kept as retained and never updated. A changed external item (an edited PR body) is a new revision.
48
+ -- author_kind: owner (a session prompt from the host's own user, or an author matching owner_identity), assistant (a session reply),
49
+ -- person (anyone else), bot.
50
+ -- available_at: when this revision became visible where it lives (a PR body revision's edit time), verified from the provider;
51
+ -- null when that cannot be established, and then no as-of snapshot may include this revision.
52
+ create table source (
53
+ id integer primary key autoincrement not null,
54
+ project_id integer not null references project (id) on delete cascade,
55
+ kind text not null check (kind in ('session_message', 'pr_body', 'issue_body', 'pr_comment', 'issue_comment', 'review',
56
+ 'review_comment', 'commit_message', 'pr_event', 'file_excerpt')),
57
+ -- The artifact it belongs to: `session:<uuid>`, `pr:<n>`, `issue:<n>`, `commit:<sha>`, `file:<path>`
58
+ artifact text not null check (artifact <> ''),
59
+ external_id text not null check (external_id <> ''),
60
+ revision integer not null check (revision > 0),
61
+ session_id text references session (id) on delete cascade,
134
62
  turn_id text,
135
- reply_to_id text references message (id) on delete set null,
136
- speaker_kind text not null check (speaker_kind in ('self', 'person', 'assistant', 'bot')),
137
- identity_id integer references person_identity (id) on delete set null,
138
- body text not null check (body <> ''),
139
- truncated integer not null default 0 check (truncated in (0, 1)),
140
- original_bytes integer not null check (original_bytes > 0),
63
+ author_kind text not null check (author_kind in ('owner', 'assistant', 'person', 'bot')),
64
+ author_login text,
65
+ author_external_id text,
66
+ author_association text,
67
+ -- The comment or thread this replies to, or the thread a resolution event closed
68
+ parent_external_id text,
69
+ -- For pr_event: merged | closed | reopened | thread_resolved
70
+ event_kind text check (event_kind in ('merged', 'closed', 'reopened', 'thread_resolved')),
141
71
  url text,
142
- sent_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', sent_at) is sent_at),
72
+ created_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', created_at) is created_at),
73
+ available_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', available_at) is available_at),
74
+ captured_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', captured_at) is captured_at),
75
+ text text not null,
76
+ truncated integer not null default 0 check (truncated in (0, 1)),
77
+ redacted integer not null default 0 check (redacted in (0, 1)),
78
+ original_bytes integer not null check (original_bytes >= 0),
143
79
  content_hash blob not null check (length(content_hash) = 32),
144
- -- Whether it goes into the full-text index. 0 for AI replies in coding sessions and automated notices (decided by indexesMessage in capture.ts / github.ts)
80
+ -- Code position of a review comment or a file excerpt: a normalized repository-relative path with forward slashes
81
+ path text check (path is null or (path <> '' and path not glob '/*' and path not glob '*[/]..[/]*' and path not glob '..[/]*'
82
+ and path not glob '*[/]..' and path <> '..' and path not glob '*\*' and path not glob '[A-Za-z]:*' and path not glob '*//*'
83
+ and path not glob './*' and path not glob '*[/].[/]*')),
84
+ line_start integer check (line_start > 0),
85
+ line_end integer check (line_end >= line_start),
86
+ diff_hunk text,
87
+ commit_sha text check (commit_sha is null or (length(commit_sha) = 40 and commit_sha not glob '*[^0-9a-f]*')),
88
+ blob_sha text check (blob_sha is null or (length(blob_sha) = 40 and blob_sha not glob '*[^0-9a-f]*')),
89
+ -- 1 when searched in the source index (owner words and third-party text; assistant replies are not)
145
90
  indexed integer not null check (indexed in (0, 1)),
146
- unique (conversation_id, external_id),
147
- check (truncated = 1 or original_bytes = length(cast(body as blob))),
148
- check (truncated = 0 or original_bytes > length(cast(body as blob)))
91
+ check ((kind = 'session_message') = (session_id is not null)),
92
+ check ((kind = 'pr_event') = (event_kind is not null)),
93
+ check (kind <> 'session_message' or author_kind in ('owner', 'assistant')),
94
+ check (kind = 'session_message' or author_kind <> 'assistant'),
95
+ check (kind <> 'file_excerpt' or (path is not null and commit_sha is not null and blob_sha is not null
96
+ and line_start is not null and line_end is not null)),
97
+ check (truncated = 1 or redacted = 1 or original_bytes = length(cast(text as blob)))
149
98
  ) strict;
150
- create index message_order on message (conversation_id, sent_at);
151
- create index message_by_identity on message (identity_id, sent_at desc) where identity_id is not null;
152
- create index message_self on message (sent_at desc) where speaker_kind = 'self';
99
+ create index source_artifact on source (project_id, artifact, created_at);
100
+ create index source_session on source (session_id, created_at) where session_id is not null;
101
+ -- A captured message id is unique within its session; everything else within its project and kind
102
+ create unique index source_message_once on source (session_id, external_id, revision) where session_id is not null;
103
+ create unique index source_item_once on source (project_id, kind, external_id, revision) where session_id is null;
153
104
 
154
- -- The full-text index. rowid = message.seq. Terms are split by sphica_terms() (terms() in server/src/text.ts, registered per connection).
155
- -- Writes from a connection without the function fail with no such function (the index never silently misses rows).
156
- create virtual table message_fts using fts5(lexemes, content='', contentless_delete=1);
157
- create trigger message_fts_ai after insert on message when new.indexed = 1 begin
158
- insert into message_fts (rowid, lexemes) values (new.seq, sphica_terms(new.body));
105
+ -- An external source may claim the owner only through a bound identity; a retry with different bytes is refused, not silently dropped
106
+ create trigger source_owner_bound before insert on source
107
+ when new.kind <> 'session_message' and new.author_kind = 'owner'
108
+ and not exists (select 1 from owner_identity where provider = 'github' and external_id = new.author_external_id) begin
109
+ select raise(abort, 'owner authorship needs a bound owner identity');
159
110
  end;
160
- create trigger message_fts_ad after delete on message when old.indexed = 1 begin
161
- delete from message_fts where rowid = old.seq;
111
+ create trigger source_session_project before insert on source when new.session_id is not null
112
+ and not exists (select 1 from session where id = new.session_id and project_id = new.project_id) begin
113
+ select raise(abort, 'source and session belong to different projects');
162
114
  end;
163
- create trigger message_fts_au after update of body, indexed on message begin
164
- delete from message_fts where rowid = old.seq and old.indexed = 1;
165
- insert into message_fts (rowid, lexemes) select new.seq, sphica_terms(new.body) where new.indexed = 1;
115
+ create trigger source_no_update before update on source begin
116
+ select raise(abort, 'sources are never rewritten; capture a new revision');
166
117
  end;
167
118
 
168
- -- Files linked to messages. Capture links an edited file (edit) to the owner's last message before the edit.
169
- -- The GitHub sync links a file pointed to in a review (review) to that review message.
170
- -- read records requirements and design documents read in the past, and is no longer written. path is relative to the project root.
171
- create table message_file (
172
- message_id text not null references message (id) on delete cascade,
173
- path text not null check (path <> '' and path not glob '/*' and path not glob '*[/]..[/]*' and path not glob '..[/]*'
174
- and path not glob '*[/]..' and path <> '..'),
175
- action text not null check (action in ('edit', 'read', 'review')),
176
- line_start integer check (line_start > 0),
177
- line_end integer check (line_end >= line_start),
178
- primary key (message_id, path, action)
119
+ create virtual table source_fts using fts5(lexemes, content='', contentless_delete=1);
120
+ create trigger source_fts_ai after insert on source when new.indexed = 1 begin
121
+ insert into source_fts (rowid, lexemes) values (new.id, sphica_terms(new.text));
122
+ end;
123
+ create trigger source_fts_ad after delete on source when old.indexed = 1 begin
124
+ delete from source_fts where rowid = old.id;
125
+ end;
126
+
127
+ -- PR ↔ issue and other artifact references (by artifact, not by a particular revision)
128
+ create table artifact_link (
129
+ project_id integer not null references project (id) on delete cascade,
130
+ from_artifact text not null,
131
+ to_artifact text not null,
132
+ kind text not null check (kind in ('references', 'closes')),
133
+ primary key (project_id, from_artifact, to_artifact, kind)
179
134
  ) strict;
180
- create index message_file_path on message_file (path);
181
135
 
182
- -- Work status, updated by trace. active / blocked / paused are candidates for continuing work.
183
- create table work_item (
136
+ -- A reference the owner gave that Sphica could not fetch (a meeting-notes URL). It never counts as a fetched source:
137
+ -- a claim supported only by it stays unsourced until the text is fetched and checked.
138
+ create table external_reference (
184
139
  id integer primary key autoincrement not null,
185
140
  project_id integer not null references project (id) on delete cascade,
186
- source_key text not null,
187
- title text not null check (title <> ''),
188
- goal text not null check (goal <> ''),
189
- current text not null check (current <> ''),
190
- next text not null default '[]' check (json_valid(next) and json_type(next) = 'array'),
191
- status text not null check (status in ('active', 'blocked', 'paused', 'done', 'abandoned')),
192
- conversation_id text references conversation (id) on delete set null,
193
- updated_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', updated_at) is updated_at),
194
- unique (project_id, source_key)
141
+ url text not null check (url <> ''),
142
+ owner_source_id integer not null references source (id),
143
+ span_start integer not null check (span_start >= 0),
144
+ span_end integer not null check (span_end > span_start),
145
+ added_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', added_at) is added_at)
195
146
  ) strict;
196
- create index work_item_open on work_item (project_id, updated_at desc) where status in ('active', 'blocked', 'paused');
197
147
 
198
- -- A unit of searchable knowledge: decisions trace picked from conversations, and document sections.
199
- -- Overturned decisions are not deleted (deleting them gets them proposed again). They become superseded and point to the successor.
200
- -- stance follows from kind and status, and filters searches for paths not to take.
201
- create table knowledge (
148
+ -- A file path an edit tool reported, or that a turn-boundary git status snapshot found. It is not an implementation record.
149
+ create table edit_observation (
150
+ id integer primary key autoincrement not null,
151
+ session_id text not null references session (id) on delete cascade,
152
+ turn_id text,
153
+ tool_event_id text,
154
+ path text not null check (path <> '' and path not glob '/*' and path not glob '*[/]..[/]*' and path not glob '..[/]*'
155
+ and path not glob '*[/]..' and path <> '..' and path not glob '*\*' and path not glob '[A-Za-z]:*'),
156
+ via text not null check (via in ('tool', 'status')),
157
+ observed_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', observed_at) is observed_at),
158
+ unique (session_id, turn_id, path, via)
159
+ ) strict;
160
+ create index edit_observation_path on edit_observation (path);
161
+
162
+ -- One extraction by trace, harvest, or glean. target names what it read: `session:<uuid>`, `pr:<n>`, or `glean`.
163
+ create table extraction_run (
202
164
  id integer primary key autoincrement not null,
203
165
  project_id integer not null references project (id) on delete cascade,
204
- source_item_id integer references source_item (id) on delete cascade,
205
- conversation_id text references conversation (id) on delete cascade,
206
- work_item_id integer references work_item (id) on delete set null,
207
- source_key text not null,
208
- kind text not null check (kind in ('decision', 'option', 'constraint', 'non_goal', 'dead_end', 'finding', 'debt',
209
- 'verification', 'question', 'document')),
210
- status text,
211
- stance text not null generated always as (
212
- case
213
- when kind in ('constraint', 'non_goal', 'debt') then (case status when 'active' then 'dont' else 'neutral' end)
214
- when kind = 'dead_end' then 'dont'
215
- when kind = 'option' then (case status when 'chosen' then 'do' else 'dont' end)
216
- when kind = 'decision' then (case status when 'accepted' then 'do' when 'proposed' then 'neutral' else 'dont' end)
217
- when kind = 'verification' then (case status when 'failed' then 'dont' else 'neutral' end)
218
- else 'neutral'
219
- end
220
- ) stored,
221
- confidence text check (confidence in ('fact', 'inference', 'opinion')),
222
- decision_id integer references knowledge (id) on delete cascade,
223
- superseded_by_id integer references knowledge (id) on delete set null,
224
- heading text,
225
- body text not null check (body <> ''),
166
+ origin text not null check (origin in ('trace', 'harvest', 'glean')),
167
+ target text not null,
168
+ session_id text references session (id) on delete set null,
169
+ status text not null check (status in ('running', 'saved', 'failed', 'capped')),
226
170
  reason text,
227
- confirmation text,
228
- command text,
229
- downsides text not null default '[]' check (json_valid(downsides) and json_type(downsides) = 'array'),
230
- refs text not null default '[]' check (json_valid(refs) and json_type(refs) = 'array'),
231
- occurred_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', occurred_at) is occurred_at),
232
- content_hash blob not null check (length(content_hash) = 32),
233
- unique (project_id, source_key),
234
- check (source_item_id is not null or conversation_id is not null),
235
- check (kind <> 'document' or (source_item_id is not null and heading is not null)),
236
- check (
237
- case kind
238
- when 'decision' then status is not null and status in ('proposed', 'accepted', 'rejected', 'superseded')
239
- when 'option' then status is not null and status in ('chosen', 'rejected', 'was_chosen')
240
- when 'verification' then status is not null and status in ('passed', 'failed', 'not_run')
241
- when 'question' then status is not null and status in ('open', 'blocking', 'resolved')
242
- when 'constraint' then status is not null and status in ('active', 'retired')
243
- when 'non_goal' then status is not null and status in ('active', 'retired')
244
- when 'debt' then status is not null and status in ('active', 'retired')
245
- else status is null
246
- end
247
- ),
248
- check (case kind when 'option' then decision_id is not null when 'verification' then 1 else decision_id is null end),
249
- check ((kind = 'decision' and status = 'superseded') = (superseded_by_id is not null)),
250
- check (superseded_by_id is null or superseded_by_id <> id),
251
- check (confirmation is null or kind = 'decision'),
252
- check (command is null or kind = 'verification'),
253
- check (json_array_length(downsides) = 0 or kind = 'decision'),
254
- check (kind <> 'document' or work_item_id is null)
171
+ input_bytes integer check (input_bytes >= 0),
172
+ -- The CLI-issued draft this run saves. A saved run's draft saves nothing again; the draft is bound to this run's project and target
173
+ draft_id text unique,
174
+ started_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', started_at) is started_at),
175
+ finished_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', finished_at) is finished_at),
176
+ check (status in ('running', 'saved') or reason is not null)
177
+ ) strict;
178
+
179
+ -- Which sources an extraction looked at, and what came of them
180
+ create table source_processing (
181
+ source_id integer not null references source (id) on delete cascade,
182
+ run_id integer not null references extraction_run (id) on delete cascade,
183
+ outcome text not null check (outcome in ('units', 'no_unit', 'failed', 'capped')),
184
+ primary key (source_id, run_id)
255
185
  ) strict;
256
- create index knowledge_listing on knowledge (project_id, kind, status, occurred_at desc);
257
- create index knowledge_work on knowledge (work_item_id) where work_item_id is not null;
258
186
 
259
- -- Extra search words for a record (synonyms, abbreviations, English equivalents of its words). **Search only**: no search result, read,
260
- -- or CLI output shows them. content_hash is the record's hash when they were written; they are indexed only while it
261
- -- still matches, so a record whose text changed stops being found by words written for its old text. source says who wrote them.
262
- create table knowledge_terms (
263
- knowledge_id integer primary key not null references knowledge (id) on delete cascade,
264
- terms text not null check (terms <> '' and length(terms) <= 400),
187
+ -- An extracted unit. Its text is never rewritten: corrections are successors, withdrawals, retractions, and anchor replacements.
188
+ -- extraction: supported (every evidence span was found in retained text) or quarantined (with reason).
189
+ -- lifecycle changes only through unit_state (its trigger sets this column). revision rises with every change to the unit's relations,
190
+ -- so a draft made against an older revision is refused.
191
+ create table unit (
192
+ id integer primary key autoincrement not null,
193
+ project_id integer not null references project (id) on delete cascade,
194
+ -- `<origin>:<target>/<key>` for trace and harvest, `glean:<key>` for glean
195
+ key text not null,
196
+ kind text not null check (kind in ('decision', 'implementation', 'finding', 'dead_end', 'question', 'constraint')),
197
+ stance text check (stance in ('do', 'dont', 'defer')),
198
+ text text not null check (text <> ''),
199
+ why text,
200
+ scope_note text,
201
+ revisit_when text,
202
+ no_code_surface text,
203
+ extraction text not null check (extraction in ('supported', 'quarantined')),
204
+ extraction_reason text,
205
+ lifecycle text not null default 'candidate' check (lifecycle in ('candidate', 'active', 'superseded', 'withdrawn')),
206
+ unsourced integer not null default 0 check (unsourced in (0, 1)),
207
+ revision integer not null default 1 check (revision > 0),
208
+ run_id integer not null references extraction_run (id),
209
+ created_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', created_at) is created_at),
210
+ -- Hash of text, why, scope_note, revisit_when, and the options (in position order), computed by the save path
265
211
  content_hash blob not null check (length(content_hash) = 32),
266
- source text not null check (source in ('trace', 'pr', 'import')),
267
- written_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', written_at) is written_at)
212
+ unique (project_id, key),
213
+ check ((stance is null) = (kind not in ('decision', 'constraint'))),
214
+ check (revisit_when is null or stance = 'defer'),
215
+ check ((extraction = 'quarantined') = (extraction_reason is not null)),
216
+ check (extraction = 'supported' or lifecycle = 'candidate'),
217
+ check (unsourced = 0 or lifecycle <> 'active')
218
+ ) strict;
219
+ create index unit_live on unit (project_id, lifecycle, kind);
220
+ create trigger unit_insert_candidate before insert on unit when new.lifecycle <> 'candidate' begin
221
+ select raise(abort, 'units start as candidates');
222
+ end;
223
+ create trigger unit_run_project before insert on unit
224
+ when not exists (select 1 from extraction_run where id = new.run_id and project_id = new.project_id) begin
225
+ select raise(abort, 'unit and run belong to different projects');
226
+ end;
227
+ create trigger unit_text_frozen before update of project_id, key, kind, stance, text, why, scope_note, revisit_when, content_hash, run_id
228
+ on unit begin
229
+ select raise(abort, 'unit text is never rewritten; record a successor');
230
+ end;
231
+ create trigger unit_lifecycle_via_state before update of lifecycle on unit
232
+ when new.lifecycle is not (select to_state from unit_state where unit_id = new.id order by id desc limit 1) begin
233
+ select raise(abort, 'lifecycle changes only through unit_state');
234
+ end;
235
+
236
+ -- Options are part of the unit's text: written with it, never updated or removed on their own
237
+ create table unit_option (
238
+ id integer primary key autoincrement not null,
239
+ unit_id integer not null references unit (id) on delete cascade,
240
+ position integer not null check (position > 0),
241
+ text text not null check (text <> ''),
242
+ outcome text not null check (outcome in ('chosen', 'rejected', 'deferred', 'proposed')),
243
+ why text,
244
+ unique (unit_id, position),
245
+ unique (unit_id, id)
268
246
  ) strict;
247
+ create trigger unit_option_sealed before insert on unit_option
248
+ when exists (select 1 from unit_state where unit_id = new.unit_id) or exists (select 1 from unit_alias where unit_id = new.unit_id) begin
249
+ select raise(abort, 'options are written with the unit, before its first state; record a successor instead');
250
+ end;
251
+ create trigger unit_option_frozen before update on unit_option begin
252
+ select raise(abort, 'options are never rewritten; record a successor');
253
+ end;
254
+ create trigger unit_option_no_delete before delete on unit_option
255
+ when exists (select 1 from unit where id = old.unit_id) begin
256
+ select raise(abort, 'options are never removed on their own');
257
+ end;
269
258
 
270
- -- What the knowledge index holds for each record: heading (h), body plus reason (b), and the extra search words whose hash matches (e).
271
- -- The triggers and `sphica db reindex` all insert from here, so the rule lives in one place.
272
- create view knowledge_search_text as
273
- select k.id,
274
- sphica_terms(coalesce(k.heading, '')) as h,
275
- sphica_terms(k.body || char(10) || coalesce(k.reason, '')) as b,
276
- sphica_terms(coalesce(t.terms, '')) as e
277
- from knowledge k
278
- left join knowledge_terms t on t.knowledge_id = k.id and t.content_hash = k.content_hash;
259
+ -- A span of a source supporting a unit or one of its options. Retraction marks it mistaken without deleting it.
260
+ create table unit_evidence (
261
+ id integer primary key autoincrement not null,
262
+ unit_id integer not null references unit (id) on delete cascade,
263
+ option_id integer,
264
+ source_id integer not null references source (id) on delete cascade,
265
+ span_start integer not null check (span_start >= 0),
266
+ span_end integer not null check (span_end > span_start),
267
+ role text not null check (role in ('states', 'proposes', 'rejects', 'explains', 'implements')),
268
+ -- A third party the owner reported ("X said ..."): hearsay by the owner, never X's own statement
269
+ reported_speaker text,
270
+ run_id integer not null references extraction_run (id),
271
+ added_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', added_at) is added_at),
272
+ retracted_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', retracted_at) is retracted_at),
273
+ retraction_reason text,
274
+ retraction_source_id integer references source (id),
275
+ retraction_span_start integer,
276
+ retraction_span_end integer,
277
+ foreign key (unit_id, option_id) references unit_option (unit_id, id) on delete cascade,
278
+ check ((retracted_at is null) = (retraction_reason is null)),
279
+ check ((retracted_at is null) = (retraction_source_id is null)),
280
+ check ((retraction_source_id is null) = (retraction_span_start is null)),
281
+ check ((retraction_source_id is null) = (retraction_span_end is null)),
282
+ check (retraction_span_end is null or retraction_span_end > retraction_span_start)
283
+ ) strict;
284
+ create unique index unit_evidence_unit_once on unit_evidence (unit_id, source_id, span_start, span_end, role) where option_id is null;
285
+ create unique index unit_evidence_option_once on unit_evidence (option_id, source_id, span_start, span_end, role) where option_id is not null;
286
+ create index unit_evidence_source on unit_evidence (source_id);
279
287
 
280
- -- The full-text index. rowid = knowledge.id. Search uses bm25(knowledge_fts, 3, 1, 1).
281
- -- Trace headings hold the work title, and document section headings hold the path and heading levels.
282
- create virtual table knowledge_fts using fts5(h, b, e, content='', contentless_delete=1);
283
- create trigger knowledge_fts_ai after insert on knowledge begin
284
- insert into knowledge_fts (rowid, h, b, e) select id, h, b, e from knowledge_search_text where id = new.id;
288
+ -- Evidence that the project adopted a decision or constraint. route: owner_statement (an owner-kind source span) or
289
+ -- explicit (an explicit disposition in a source, such as a maintainer's reply saying it is adopted). A merge or a resolved thread is never adoption.
290
+ create table unit_adoption (
291
+ id integer primary key autoincrement not null,
292
+ unit_id integer not null references unit (id) on delete cascade,
293
+ route text not null check (route in ('owner_statement', 'explicit')),
294
+ source_id integer not null references source (id) on delete cascade,
295
+ span_start integer not null check (span_start >= 0),
296
+ span_end integer not null check (span_end > span_start),
297
+ run_id integer not null references extraction_run (id),
298
+ added_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', added_at) is added_at),
299
+ retracted_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', retracted_at) is retracted_at),
300
+ retraction_reason text,
301
+ retraction_source_id integer references source (id),
302
+ retraction_span_start integer,
303
+ retraction_span_end integer,
304
+ unique (unit_id, source_id, span_start, span_end),
305
+ check ((retracted_at is null) = (retraction_reason is null)),
306
+ check ((retracted_at is null) = (retraction_source_id is null)),
307
+ check ((retraction_source_id is null) = (retraction_span_start is null)),
308
+ check ((retraction_source_id is null) = (retraction_span_end is null)),
309
+ check (retraction_span_end is null or retraction_span_end > retraction_span_start)
310
+ ) strict;
311
+ create trigger unit_adoption_route before insert on unit_adoption begin
312
+ select raise(abort, 'owner_statement adoption needs an owner-authored source')
313
+ where new.route = 'owner_statement' and not exists (select 1 from source where id = new.source_id and author_kind = 'owner');
314
+ select raise(abort, 'explicit adoption needs the owner or a maintainer (OWNER, MEMBER, COLLABORATOR association)')
315
+ where new.route = 'explicit' and not exists (select 1 from source where id = new.source_id
316
+ and (author_kind = 'owner' or author_association in ('OWNER', 'MEMBER', 'COLLABORATOR')));
317
+ select raise(abort, 'merge and thread resolution events are not adoption')
318
+ where exists (select 1 from source where id = new.source_id and kind = 'pr_event');
319
+ select raise(abort, 'adoption applies to decisions and constraints')
320
+ where not exists (select 1 from unit where id = new.unit_id and kind in ('decision', 'constraint'));
321
+ end;
322
+
323
+ create table unit_link (
324
+ from_unit integer not null references unit (id) on delete cascade,
325
+ to_unit integer not null references unit (id) on delete cascade,
326
+ kind text not null check (kind in ('supersedes', 'implements', 'conflicts')),
327
+ run_id integer not null references extraction_run (id),
328
+ added_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', added_at) is added_at),
329
+ -- A conflict stays unresolved (and suppresses automatic delivery of both) until resolved with a reason
330
+ resolved_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', resolved_at) is resolved_at),
331
+ resolution text,
332
+ primary key (from_unit, to_unit, kind),
333
+ check (from_unit <> to_unit),
334
+ check ((resolved_at is null) = (resolution is null)),
335
+ check (kind = 'conflicts' or resolved_at is null)
336
+ ) strict;
337
+ create trigger unit_link_frozen before update on unit_link begin
338
+ select raise(abort, 'links are frozen; only an unresolved conflict can be resolved, once')
339
+ where new.from_unit is not old.from_unit or new.to_unit is not old.to_unit or new.kind is not old.kind
340
+ or new.run_id is not old.run_id or new.added_at is not old.added_at or old.resolved_at is not null or old.kind <> 'conflicts';
285
341
  end;
286
- create trigger knowledge_fts_ad after delete on knowledge begin
287
- delete from knowledge_fts where rowid = old.id;
342
+ create trigger unit_link_no_delete before delete on unit_link
343
+ when exists (select 1 from unit where id = old.from_unit) and exists (select 1 from unit where id = old.to_unit) begin
344
+ select raise(abort, 'links are never removed on their own');
288
345
  end;
289
- create trigger knowledge_fts_au after update of heading, body, reason, content_hash on knowledge begin
290
- delete from knowledge_fts where rowid = old.id;
291
- insert into knowledge_fts (rowid, h, b, e) select id, h, b, e from knowledge_search_text where id = new.id;
346
+ create trigger unit_link_supersedes_acyclic before insert on unit_link when new.kind = 'supersedes' begin
347
+ select raise(abort, 'supersedes links cannot form a cycle')
348
+ where exists (
349
+ with recursive chain(id) as (
350
+ select new.to_unit union select l.to_unit from unit_link l join chain on l.from_unit = chain.id where l.kind = 'supersedes')
351
+ select 1 from chain where id = new.from_unit);
352
+ end;
353
+
354
+ -- Lifecycle history and the only route for lifecycle changes. The trigger checks the rules and then sets unit.lifecycle.
355
+ create table unit_state (
356
+ id integer primary key autoincrement not null,
357
+ unit_id integer not null references unit (id) on delete cascade,
358
+ from_state text check (from_state in ('candidate', 'active', 'superseded', 'withdrawn')),
359
+ to_state text not null check (to_state in ('candidate', 'active', 'superseded', 'withdrawn')),
360
+ at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', at) is at),
361
+ reason text not null check (reason <> ''),
362
+ source_id integer references source (id),
363
+ run_id integer not null references extraction_run (id)
364
+ ) strict;
365
+ create index unit_state_order on unit_state (unit_id, id);
366
+ create trigger unit_state_rules before insert on unit_state begin
367
+ select raise(abort, 'from_state must be the current lifecycle')
368
+ where new.from_state is not (select lifecycle from unit where id = new.unit_id)
369
+ and exists (select 1 from unit_state where unit_id = new.unit_id);
370
+ select raise(abort, 'a quarantined or unsourced unit cannot become active')
371
+ where new.to_state = 'active' and exists (select 1 from unit where id = new.unit_id and (extraction <> 'supported' or unsourced = 1));
372
+ select raise(abort, 'an active decision or constraint needs unretracted evidence and adoption')
373
+ where new.to_state = 'active' and exists (select 1 from unit u where u.id = new.unit_id and u.kind in ('decision', 'constraint') and (
374
+ not exists (select 1 from unit_evidence e where e.unit_id = u.id and e.option_id is null and e.retracted_at is null)
375
+ or not exists (select 1 from unit_adoption a where a.unit_id = u.id and a.retracted_at is null)));
376
+ select raise(abort, 'an active implementation needs code or commit evidence')
377
+ where new.to_state = 'active' and exists (select 1 from unit u where u.id = new.unit_id and u.kind = 'implementation' and not (
378
+ exists (select 1 from unit_evidence e join source s on s.id = e.source_id where e.unit_id = u.id and e.option_id is null
379
+ and e.retracted_at is null and e.role = 'implements' and s.kind in ('commit_message', 'file_excerpt'))
380
+ or exists (select 1 from unit_anchor a where a.unit_id = u.id and a.retired_at is null and a.role = 'evidence'
381
+ and (a.commit_sha is not null or (a.edit_observation_id is not null and exists (select 1 from unit_evidence e
382
+ join source s on s.id = e.source_id join edit_observation o on o.id = a.edit_observation_id
383
+ where e.unit_id = u.id and e.option_id is null and e.retracted_at is null and e.role = 'implements'
384
+ and s.session_id = o.session_id))))));
385
+ select raise(abort, 'an active unit needs unretracted evidence')
386
+ where new.to_state = 'active' and exists (select 1 from unit u where u.id = new.unit_id and u.kind in ('finding', 'dead_end', 'question')
387
+ and not exists (select 1 from unit_evidence e where e.unit_id = u.id and e.option_id is null and e.retracted_at is null));
388
+ select raise(abort, 'superseded needs a supersedes link from its successor')
389
+ where new.to_state = 'superseded' and not exists (select 1 from unit_link where to_unit = new.unit_id and kind = 'supersedes');
292
390
  end;
293
- create trigger knowledge_terms_ai after insert on knowledge_terms begin
294
- delete from knowledge_fts where rowid = new.knowledge_id;
295
- insert into knowledge_fts (rowid, h, b, e) select id, h, b, e from knowledge_search_text where id = new.knowledge_id;
391
+ create trigger unit_state_append_only before update on unit_state begin
392
+ select raise(abort, 'state history is append-only');
296
393
  end;
297
- create trigger knowledge_terms_au after update of terms, content_hash on knowledge_terms begin
298
- delete from knowledge_fts where rowid = new.knowledge_id;
299
- insert into knowledge_fts (rowid, h, b, e) select id, h, b, e from knowledge_search_text where id = new.knowledge_id;
394
+ create trigger unit_state_no_delete before delete on unit_state when exists (select 1 from unit where id = old.unit_id) begin
395
+ select raise(abort, 'state history is append-only');
300
396
  end;
301
- create trigger knowledge_terms_ad after delete on knowledge_terms begin
302
- delete from knowledge_fts where rowid = old.knowledge_id;
303
- insert into knowledge_fts (rowid, h, b, e) select id, h, b, e from knowledge_search_text where id = old.knowledge_id;
397
+ create trigger unit_state_apply after insert on unit_state begin
398
+ update unit set lifecycle = new.to_state, revision = revision + 1 where id = new.unit_id;
304
399
  end;
305
400
 
306
- -- Direct links between decisions and files. applies_to is a constraint shown before editing, and evidence is a file cited as grounds.
307
- create table knowledge_file (
308
- knowledge_id integer not null references knowledge (id) on delete cascade,
401
+ -- Where a unit applies in code, or code cited as evidence. Validated against the working tree when served, never cached here.
402
+ create table unit_anchor (
403
+ id integer primary key autoincrement not null,
404
+ unit_id integer not null references unit (id) on delete cascade,
309
405
  path text not null check (path <> '' and path not glob '/*' and path not glob '*[/]..[/]*' and path not glob '..[/]*'
310
- and path not glob '*[/]..' and path <> '..'),
311
- role text not null check (role in ('applies_to', 'evidence')),
406
+ and path not glob '*[/]..' and path <> '..' and path not glob '*\*' and path not glob '[A-Za-z]:*'),
407
+ symbol text,
408
+ commit_sha text check (commit_sha is null or (length(commit_sha) = 40 and commit_sha not glob '*[^0-9a-f]*')),
312
409
  line_start integer check (line_start > 0),
313
410
  line_end integer check (line_end >= line_start),
314
- primary key (knowledge_id, path, role)
411
+ excerpt text,
412
+ role text not null check (role in ('applies_to', 'evidence')),
413
+ -- For work recorded before a commit: the edit observation of this path in the session, checked against the working tree when saved
414
+ edit_observation_id integer references edit_observation (id),
415
+ run_id integer not null references extraction_run (id),
416
+ added_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', added_at) is added_at),
417
+ retired_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', retired_at) is retired_at),
418
+ replaced_by integer references unit_anchor (id)
419
+ ) strict;
420
+ create index unit_anchor_path on unit_anchor (path, role) where retired_at is null;
421
+ create index unit_anchor_unit on unit_anchor (unit_id, retired_at);
422
+ -- Anchors are retired and replaced, never edited in place (except setting retired_at and replaced_by once)
423
+ create trigger unit_anchor_frozen before update on unit_anchor begin
424
+ select raise(abort, 'anchors are replaced, not edited; retirement happens once')
425
+ where new.unit_id is not old.unit_id or new.path is not old.path or new.symbol is not old.symbol or new.commit_sha is not old.commit_sha
426
+ or new.line_start is not old.line_start or new.line_end is not old.line_end or new.excerpt is not old.excerpt or new.role is not old.role
427
+ or new.edit_observation_id is not old.edit_observation_id or new.run_id is not old.run_id or new.added_at is not old.added_at
428
+ or old.retired_at is not null or new.retired_at is null;
429
+ end;
430
+ create trigger unit_anchor_no_delete before delete on unit_anchor when exists (select 1 from unit where id = old.unit_id) begin
431
+ select raise(abort, 'anchors are retired, never deleted');
432
+ end;
433
+
434
+ -- Search-only aliases in Japanese and English, written by the agent with the unit. Each set is bound to the unit's content_hash when written;
435
+ -- only the newest set whose hash matches is indexed. Older sets stay for as-of snapshots. Never evidence, never shown as something said.
436
+ create table unit_alias (
437
+ id integer primary key autoincrement not null,
438
+ unit_id integer not null references unit (id) on delete cascade,
439
+ terms text not null check (json_valid(terms) and json_type(terms) = 'array' and json_array_length(terms) between 0 and 12),
440
+ content_hash blob not null check (length(content_hash) = 32),
441
+ run_id integer not null references extraction_run (id),
442
+ added_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', added_at) is added_at)
315
443
  ) strict;
316
- create index knowledge_file_path on knowledge_file (path, role);
444
+ create index unit_alias_unit on unit_alias (unit_id, id);
445
+ create trigger unit_alias_terms before insert on unit_alias begin
446
+ select raise(abort, 'each alias is a non-empty string of at most 40 characters')
447
+ where exists (select 1 from json_each(new.terms) where type <> 'text' or length(trim(value)) = 0 or length(value) > 40);
448
+ end;
449
+ create trigger unit_alias_frozen before update on unit_alias begin
450
+ select raise(abort, 'alias sets are replaced by a newer set, not edited');
451
+ end;
452
+ create trigger unit_alias_no_delete before delete on unit_alias when exists (select 1 from unit where id = old.unit_id) begin
453
+ select raise(abort, 'alias sets are append-only; write an empty set to clear');
454
+ end;
455
+
456
+ -- Cross-project and span checks for everything that points at a unit, a source, or a run
457
+ create trigger unit_evidence_check before insert on unit_evidence begin
458
+ select raise(abort, 'evidence and unit belong to different projects')
459
+ where (select project_id from unit where id = new.unit_id) is not (select project_id from source where id = new.source_id)
460
+ or (select project_id from unit where id = new.unit_id) is not (select project_id from extraction_run where id = new.run_id);
461
+ select raise(abort, 'evidence span is outside the source text')
462
+ where new.span_end > (select length(cast(text as blob)) from source where id = new.source_id);
463
+ select raise(abort, 'a reported speaker is the owner reporting someone else, so it must cite an owner session message')
464
+ where new.reported_speaker is not null and (trim(new.reported_speaker) = '' or not exists (select 1 from source
465
+ where id = new.source_id and kind = 'session_message' and author_kind = 'owner'));
466
+ end;
467
+ create trigger unit_evidence_retract before update on unit_evidence begin
468
+ select raise(abort, 'evidence is only ever retracted, once')
469
+ where old.retracted_at is not null or new.unit_id is not old.unit_id or new.option_id is not old.option_id
470
+ or new.source_id is not old.source_id or new.span_start is not old.span_start or new.span_end is not old.span_end
471
+ or new.role is not old.role or new.reported_speaker is not old.reported_speaker or new.run_id is not old.run_id
472
+ or new.added_at is not old.added_at or new.retracted_at is null;
473
+ select raise(abort, 'the retraction must cite an owner span of the same project')
474
+ where not exists (select 1 from source s where s.id = new.retraction_source_id and s.author_kind = 'owner'
475
+ and s.project_id = (select project_id from unit where id = new.unit_id)
476
+ and new.retraction_span_end <= length(cast(s.text as blob)));
477
+ end;
478
+ -- Deleting a project (or its unit or source) cascades; only a direct delete of a live link is refused
479
+ create trigger unit_evidence_no_delete before delete on unit_evidence
480
+ when exists (select 1 from unit where id = old.unit_id) and exists (select 1 from source where id = old.source_id) begin
481
+ select raise(abort, 'evidence is retracted, never deleted');
482
+ end;
483
+ create trigger unit_adoption_no_delete before delete on unit_adoption
484
+ when exists (select 1 from unit where id = old.unit_id) and exists (select 1 from source where id = old.source_id) begin
485
+ select raise(abort, 'adoption is retracted, never deleted');
486
+ end;
487
+ -- A retraction that would leave an active unit without its required support must first move it back to candidate
488
+ create trigger unit_evidence_retract_support after update of retracted_at on unit_evidence
489
+ when exists (select 1 from unit where id = new.unit_id and lifecycle = 'active')
490
+ and not exists (select 1 from unit_evidence where unit_id = new.unit_id and retracted_at is null) begin
491
+ select raise(abort, 'move the unit back to candidate before retracting its last evidence');
492
+ end;
493
+ create trigger unit_adoption_retract_support after update of retracted_at on unit_adoption
494
+ when exists (select 1 from unit where id = new.unit_id and lifecycle = 'active')
495
+ and not exists (select 1 from unit_adoption where unit_id = new.unit_id and retracted_at is null) begin
496
+ select raise(abort, 'move the unit back to candidate before retracting its last adoption');
497
+ end;
498
+ create trigger unit_adoption_check before insert on unit_adoption begin
499
+ select raise(abort, 'adoption and unit belong to different projects')
500
+ where (select project_id from unit where id = new.unit_id) is not (select project_id from source where id = new.source_id)
501
+ or (select project_id from unit where id = new.unit_id) is not (select project_id from extraction_run where id = new.run_id);
502
+ select raise(abort, 'adoption span is outside the source text')
503
+ where new.span_end > (select length(cast(text as blob)) from source where id = new.source_id);
504
+ end;
505
+ create trigger unit_adoption_retract before update on unit_adoption begin
506
+ select raise(abort, 'adoption is only ever retracted, once')
507
+ where old.retracted_at is not null or new.unit_id is not old.unit_id or new.route is not old.route
508
+ or new.source_id is not old.source_id or new.span_start is not old.span_start or new.span_end is not old.span_end
509
+ or new.run_id is not old.run_id or new.added_at is not old.added_at or new.retracted_at is null;
510
+ select raise(abort, 'the retraction must cite an owner span of the same project')
511
+ where not exists (select 1 from source s where s.id = new.retraction_source_id and s.author_kind = 'owner'
512
+ and s.project_id = (select project_id from unit where id = new.unit_id)
513
+ and new.retraction_span_end <= length(cast(s.text as blob)));
514
+ end;
515
+ create trigger unit_link_check before insert on unit_link begin
516
+ select raise(abort, 'linked units belong to different projects')
517
+ where (select project_id from unit where id = new.from_unit) is not (select project_id from unit where id = new.to_unit)
518
+ or (select project_id from unit where id = new.from_unit) is not (select project_id from extraction_run where id = new.run_id);
519
+ end;
520
+ create trigger unit_state_project before insert on unit_state begin
521
+ select raise(abort, 'state and unit belong to different projects')
522
+ where (select project_id from unit where id = new.unit_id) is not (select project_id from extraction_run where id = new.run_id)
523
+ or (new.source_id is not null
524
+ and (select project_id from unit where id = new.unit_id) is not (select project_id from source where id = new.source_id));
525
+ end;
526
+ create trigger unit_anchor_project before insert on unit_anchor begin
527
+ select raise(abort, 'anchor and unit belong to different projects')
528
+ where (select project_id from unit where id = new.unit_id) is not (select project_id from extraction_run where id = new.run_id);
529
+ select raise(abort, 'the edit observation must be of this path in a session of the same project')
530
+ where new.edit_observation_id is not null and not exists (select 1 from edit_observation o join session s on s.id = o.session_id
531
+ where o.id = new.edit_observation_id and o.path = new.path and s.project_id = (select project_id from unit where id = new.unit_id));
532
+ end;
533
+ create trigger unit_alias_project before insert on unit_alias begin
534
+ select raise(abort, 'alias and unit belong to different projects')
535
+ where (select project_id from unit where id = new.unit_id) is not (select project_id from extraction_run where id = new.run_id);
536
+ end;
537
+ create trigger source_processing_project before insert on source_processing begin
538
+ select raise(abort, 'source and run belong to different projects')
539
+ where (select project_id from source where id = new.source_id) is not (select project_id from extraction_run where id = new.run_id);
540
+ end;
541
+ create trigger external_reference_check before insert on external_reference begin
542
+ select raise(abort, 'an external reference needs an owner span of the same project')
543
+ where not exists (select 1 from source s where s.id = new.owner_source_id and s.author_kind = 'owner' and s.project_id = new.project_id
544
+ and new.span_end <= length(cast(s.text as blob)));
545
+ end;
546
+
547
+ -- Every change to a unit's relations raises its revision (stale drafts are refused against it)
548
+ create trigger unit_rev_evidence_i after insert on unit_evidence begin update unit set revision = revision + 1 where id = new.unit_id; end;
549
+ create trigger unit_rev_evidence_u after update on unit_evidence begin update unit set revision = revision + 1 where id = new.unit_id; end;
550
+ create trigger unit_rev_adoption_i after insert on unit_adoption begin update unit set revision = revision + 1 where id = new.unit_id; end;
551
+ create trigger unit_rev_adoption_u after update on unit_adoption begin update unit set revision = revision + 1 where id = new.unit_id; end;
552
+ create trigger unit_rev_link_i after insert on unit_link begin
553
+ update unit set revision = revision + 1 where id in (new.from_unit, new.to_unit);
554
+ end;
555
+ create trigger unit_rev_link_u after update on unit_link begin
556
+ update unit set revision = revision + 1 where id in (new.from_unit, new.to_unit);
557
+ end;
558
+ create trigger unit_rev_anchor_i after insert on unit_anchor begin update unit set revision = revision + 1 where id = new.unit_id; end;
559
+ create trigger unit_rev_anchor_u after update on unit_anchor begin update unit set revision = revision + 1 where id = new.unit_id; end;
560
+ create trigger unit_rev_alias_i after insert on unit_alias begin update unit set revision = revision + 1 where id = new.unit_id; end;
317
561
 
318
- -- The 3 views capture (the capture connection) can write. The authorizer in server/src/sqlite.ts allows capture only inserts into these views
319
- -- and the writes inside the triggers below. source_item_id, identity_id, reply_to_id, and url are not in the views, so capture
320
- -- can neither create GitHub conversations nor claim someone else's identity. Conversation ids can be computed deterministically, so adding messages
321
- -- to an existing conversation is not blocked (the remaining surface if the capture path is abused).
322
- create view capture_conversation as
323
- select id, project_id, origin, external_id, branch, started_at from conversation;
324
- create trigger capture_conversation_insert instead of insert on capture_conversation begin
325
- insert into conversation (id, project_id, origin, external_id, branch, started_at)
326
- values (new.id, new.project_id, new.origin, new.external_id, new.branch, new.started_at)
327
- on conflict do nothing;
562
+ -- Unit search text: body (text, reason, scope, revisit condition, options), identifiers (live anchors), and the newest matching alias set.
563
+ -- search.ts weighs body and identifiers above aliases. Lifecycle is not indexed; queries filter it.
564
+ create view unit_search_text as
565
+ select u.id,
566
+ sphica_terms(u.text || char(10) || coalesce(u.why, '') || char(10) || coalesce(u.scope_note, '') || char(10)
567
+ || coalesce(u.revisit_when, '') || char(10)
568
+ || coalesce((select group_concat(o.text || ' ' || coalesce(o.why, ''), char(10))
569
+ from (select text, why from unit_option where unit_id = u.id order by position) o), '')) as body,
570
+ sphica_terms(coalesce((select group_concat(a.path || ' ' || coalesce(a.symbol, ''), char(10))
571
+ from (select path, symbol from unit_anchor where unit_id = u.id and retired_at is null order by id) a), '')) as ident,
572
+ sphica_terms(coalesce((select group_concat(j.value, ' ')
573
+ from json_each((select terms from unit_alias where unit_id = u.id and content_hash = u.content_hash order by id desc limit 1)) j), ''))
574
+ as alias
575
+ from unit u;
576
+ create virtual table unit_fts using fts5(body, ident, alias, content='', contentless_delete=1);
577
+ create trigger unit_fts_ai after insert on unit begin
578
+ insert into unit_fts (rowid, body, ident, alias) select id, body, ident, alias from unit_search_text where id = new.id;
579
+ end;
580
+ create trigger unit_fts_ad after delete on unit begin
581
+ delete from unit_fts where rowid = old.id;
582
+ end;
583
+ -- Children are written after the unit in the same transaction; every child change reindexes its unit
584
+ create trigger unit_fts_option_i after insert on unit_option begin
585
+ delete from unit_fts where rowid = new.unit_id;
586
+ insert into unit_fts (rowid, body, ident, alias) select id, body, ident, alias from unit_search_text where id = new.unit_id;
328
587
  end;
588
+ create trigger unit_fts_anchor_i after insert on unit_anchor begin
589
+ delete from unit_fts where rowid = new.unit_id;
590
+ insert into unit_fts (rowid, body, ident, alias) select id, body, ident, alias from unit_search_text where id = new.unit_id;
591
+ end;
592
+ create trigger unit_fts_anchor_u after update on unit_anchor begin
593
+ delete from unit_fts where rowid = new.unit_id;
594
+ insert into unit_fts (rowid, body, ident, alias) select id, body, ident, alias from unit_search_text where id = new.unit_id;
595
+ end;
596
+ create trigger unit_fts_anchor_d after delete on unit_anchor when exists (select 1 from unit where id = old.unit_id) begin
597
+ delete from unit_fts where rowid = old.unit_id;
598
+ insert into unit_fts (rowid, body, ident, alias) select id, body, ident, alias from unit_search_text where id = old.unit_id;
599
+ end;
600
+ create trigger unit_fts_alias_i after insert on unit_alias begin
601
+ delete from unit_fts where rowid = new.unit_id;
602
+ insert into unit_fts (rowid, body, ident, alias) select id, body, ident, alias from unit_search_text where id = new.unit_id;
603
+ end;
604
+ create trigger unit_fts_alias_d after delete on unit_alias when exists (select 1 from unit where id = old.unit_id) begin
605
+ delete from unit_fts where rowid = old.unit_id;
606
+ insert into unit_fts (rowid, body, ident, alias) select id, body, ident, alias from unit_search_text where id = old.unit_id;
607
+ end;
608
+
609
+ -- Current work status, updated by trace
610
+ create table work (
611
+ id integer primary key autoincrement not null,
612
+ project_id integer not null references project (id) on delete cascade,
613
+ key text not null,
614
+ title text not null check (title <> ''),
615
+ goal text not null check (goal <> ''),
616
+ current text not null check (current <> ''),
617
+ next text not null default '[]' check (json_valid(next) and json_type(next) = 'array'),
618
+ status text not null check (status in ('active', 'blocked', 'paused', 'done', 'abandoned')),
619
+ branch text,
620
+ run_id integer references extraction_run (id),
621
+ updated_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', updated_at) is updated_at),
622
+ unique (project_id, key)
623
+ ) strict;
624
+ create index work_open on work (project_id, updated_at desc) where status in ('active', 'blocked', 'paused');
329
625
 
626
+ -- What a delivery hook emitted or suppressed, and how many eligible units it left out. No source text is copied here.
627
+ create table delivery (
628
+ id integer primary key autoincrement not null,
629
+ session_id text references session (id) on delete cascade,
630
+ event text not null check (event in ('session_start', 'pre_edit', 'pre_read', 'prompt', 'review')),
631
+ outcome text not null check (outcome in ('emitted', 'nothing', 'unavailable', 'suppressed')),
632
+ reason text,
633
+ path text,
634
+ eligible integer not null default 0 check (eligible >= 0),
635
+ omitted integer not null default 0 check (omitted >= 0 and omitted <= eligible),
636
+ chars integer not null default 0 check (chars >= 0),
637
+ at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', at) is at)
638
+ ) strict;
639
+ create index delivery_session on delivery (session_id, at);
640
+ create table delivery_unit (
641
+ delivery_id integer not null references delivery (id) on delete cascade,
642
+ unit_id integer not null references unit (id) on delete cascade,
643
+ primary key (delivery_id, unit_id)
644
+ ) strict;
645
+
646
+ -- The views the capture connection may write. The capture authorizer allows inserts into these views only; the triggers derive
647
+ -- project, artifact, and indexing from the session and the speaker, so capture cannot write another project's rows or third-party text.
648
+ create view capture_session as select id, project_id, host, external_id, branch, started_at from session;
649
+ create trigger capture_session_insert instead of insert on capture_session begin
650
+ select raise(abort, 'the session already exists with different details')
651
+ where exists (select 1 from session where id = new.id and (project_id <> new.project_id or host <> new.host or external_id <> new.external_id));
652
+ insert into session (id, project_id, host, external_id, branch, started_at)
653
+ select new.id, new.project_id, new.host, new.external_id, new.branch, new.started_at
654
+ where not exists (select 1 from session where id = new.id);
655
+ end;
656
+ -- speaker: owner (the host's user typed it or answered AskUserQuestion) or assistant (the host's final reply, or the questions it asked)
330
657
  create view capture_message as
331
- select id, conversation_id, external_id, turn_id, speaker_kind, body, truncated, original_bytes, sent_at, content_hash, indexed
332
- from message;
658
+ select external_id, session_id, turn_id, author_kind as speaker, created_at, captured_at, text, truncated, redacted, original_bytes,
659
+ content_hash from source where kind = 'session_message';
333
660
  create trigger capture_message_insert instead of insert on capture_message begin
334
- insert into message (id, conversation_id, external_id, turn_id, speaker_kind, body, truncated, original_bytes,
335
- sent_at, content_hash, indexed)
336
- values (new.id, new.conversation_id, new.external_id, new.turn_id, new.speaker_kind, new.body, new.truncated,
337
- new.original_bytes, new.sent_at, new.content_hash, new.indexed)
338
- on conflict do nothing;
661
+ select raise(abort, 'unknown session') where not exists (select 1 from session where id = new.session_id);
662
+ select raise(abort, 'speaker must be owner or assistant') where new.speaker not in ('owner', 'assistant');
663
+ select raise(abort, 'the message already exists with different content')
664
+ where exists (select 1 from source where kind = 'session_message' and session_id = new.session_id and external_id = new.external_id
665
+ and (text is not new.text or author_kind is not new.speaker or turn_id is not new.turn_id or created_at is not new.created_at
666
+ or truncated is not new.truncated or redacted is not new.redacted or original_bytes is not new.original_bytes
667
+ or content_hash is not new.content_hash));
668
+ insert into source (project_id, kind, artifact, external_id, revision, session_id, turn_id, author_kind, created_at, available_at,
669
+ captured_at, text, truncated, redacted, original_bytes, content_hash, indexed)
670
+ select s.project_id, 'session_message', 'session:' || s.id, new.external_id, 1, s.id, new.turn_id, new.speaker, new.created_at,
671
+ new.created_at, new.captured_at, new.text, new.truncated, new.redacted, new.original_bytes, new.content_hash,
672
+ new.speaker = 'owner'
673
+ from session s where s.id = new.session_id
674
+ and not exists (select 1 from source where kind = 'session_message' and session_id = new.session_id and external_id = new.external_id);
339
675
  end;
340
-
341
- -- Link to the owner's last message before the edit. If that message is not in this database (such as a session that moved to another project midway), drop it.
342
- create view capture_message_file as select message_id, path, action from message_file;
343
- create trigger capture_message_file_insert instead of insert on capture_message_file begin
344
- insert into message_file (message_id, path, action)
345
- select new.message_id, new.path, new.action where exists (select 1 from message where id = new.message_id)
346
- on conflict do nothing;
676
+ create view capture_edit as select session_id, turn_id, tool_event_id, path, via, observed_at from edit_observation;
677
+ create trigger capture_edit_insert instead of insert on capture_edit begin
678
+ insert into edit_observation (session_id, turn_id, tool_event_id, path, via, observed_at)
679
+ select new.session_id, new.turn_id, new.tool_event_id, new.path, new.via, new.observed_at
680
+ where exists (select 1 from session where id = new.session_id) on conflict do nothing;
681
+ end;
682
+ -- units is a JSON array of the unit ids delivered; each must belong to the delivered session's project
683
+ create view capture_delivery as
684
+ select session_id, event, outcome, reason, path, eligible, omitted, chars, at, null as units from delivery;
685
+ create trigger capture_delivery_insert instead of insert on capture_delivery begin
686
+ select raise(abort, 'the unit and the delivered session belong to different projects')
687
+ where exists (select 1 from json_each(coalesce(new.units, '[]')) j
688
+ where (select project_id from unit where id = j.value) is not (select project_id from session where id = new.session_id));
689
+ insert into delivery (session_id, event, outcome, reason, path, eligible, omitted, chars, at)
690
+ values (new.session_id, new.event, new.outcome, new.reason, new.path, coalesce(new.eligible, 0), coalesce(new.omitted, 0),
691
+ coalesce(new.chars, 0), new.at);
692
+ insert into delivery_unit (delivery_id, unit_id)
693
+ select last_insert_rowid(), j.value from json_each(coalesce(new.units, '[]')) j where true on conflict do nothing;
347
694
  end;
348
695
 
349
- pragma user_version = 5;
696
+ pragma user_version = 1;