sphica 0.3.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/db/schema.sql CHANGED
@@ -1,7 +1,7 @@
1
1
  -- The source of truth for sphica's database (SQLite, `node:sqlite`). It lets one owner look up decisions and conversations on that machine.
2
2
  -- **Each machine is independent and shares no records.** One file (~/.sphica/sphica.db) is one database, with no schema qualifiers.
3
3
  --
4
- -- Three boundaries: the current state of sources (connector / source_item), verbatim conversations (conversation / message),
4
+ -- Three boundaries: verbatim conversations (conversation / message), the pull requests harvested into knowledge (pull_request),
5
5
  -- and searchable knowledge (knowledge). Work status (work_item) is state that gets updated, so it has its own table.
6
6
  --
7
7
  -- The version is `pragma user_version` at the end. MCP and the CLI compare it with SCHEMA_REVISION in server/src/db.ts
@@ -25,130 +25,56 @@ create table project (
25
25
  check (strftime('%Y-%m-%dT%H:%M:%fZ', created_at) is created_at)
26
26
  ) strict;
27
27
 
28
- create table person (
29
- id integer primary key autoincrement not null,
30
- display_name text not null unique check (display_name <> ''),
31
- is_self integer not null default 0 check (is_self in (0, 1))
32
- ) strict;
33
- -- Exactly one person is the owner who asks. This decides who "I" is in "what did I say?".
34
- create unique index person_one_self on person (is_self) where is_self = 1;
35
-
36
- -- Identifiers at a source. A GitHub user id stays the same when the login changes, so it goes in external_id.
37
- create table person_identity (
38
- id integer primary key autoincrement not null,
39
- person_id integer references person (id) on delete set null,
40
- provider text not null check (provider in ('github')),
41
- external_id text not null,
42
- handle text not null,
43
- unique (provider, external_id)
44
- ) strict;
45
- create index person_identity_handle on person_identity (provider, lower(handle));
46
-
47
- -- Per source, the last imported version and the latest result. No secrets (they live in the syncing machine's environment).
48
- -- For documents, the imported commit (head_oid). The next sync imports automatically only commits that fast-forward from it.
49
- -- For GitHub, the time the fetch started (snapshot_at). A fetch that started earlier is not written, even if it commits later.
50
- create table connector (
28
+ -- A pull request harvested into knowledge (the harvest Skill). github_id is GitHub's id for the PR: a project moved to another repository
29
+ -- can reuse a number, and the save command refuses a number whose id differs. harvested_at is the last successful save
30
+ -- (null for decisions moved here from the old PR-body extraction).
31
+ create table pull_request (
51
32
  id integer primary key autoincrement not null,
52
33
  project_id integer not null references project (id) on delete cascade,
53
- provider text not null check (provider in ('github', 'docs')),
54
- head_oid text check (head_oid is null or (provider = 'docs'
55
- and (length(head_oid) = 40 or length(head_oid) = 64) and head_oid not glob '*[^0-9a-f]*')),
56
- snapshot_at text check (snapshot_at is null or provider = 'github')
57
- check (strftime('%Y-%m-%dT%H:%M:%fZ', snapshot_at) is snapshot_at),
58
- last_success_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', last_success_at) is last_success_at),
59
- last_error text,
60
- unique (project_id, provider)
61
- ) strict;
62
-
63
- -- Paths the docs sync does not import. Not every tracked Markdown file states facts (such as audit fixtures).
64
- -- **This is importer-side configuration.** It must work for read-only projects too, so it is not a manifest in the repository.
65
- -- file matches the path exactly, and directory matches paths starting with `<path>/`. Only created for the docs connector.
66
- create table docs_exclude (
67
- connector_id integer not null references connector (id) on delete cascade,
68
- kind text not null check (kind in ('file', 'directory')),
69
- path text not null check (
70
- path <> '' and path not glob '/*' and path not glob '*[/]..[/]*' and path not glob '..[/]*'
71
- and path not glob '*[/]..' and path <> '..' and path not glob '*/'
72
- and path not glob '*[' || char(1) || '-' || char(31) || char(127) || ']*'),
73
- primary key (connector_id, kind, path)
74
- ) strict;
75
-
76
- -- The current state of a source. Items confirmed gone by a complete listing are deleted with their rows (no tombstones).
77
- -- Documents keep their original text in body. Search uses the knowledge sections, and joining sections never restores the original.
78
- create table "source_item" (
79
- id integer primary key autoincrement not null,
80
- connector_id integer not null references connector (id) on delete cascade,
81
- external_id text not null,
82
- kind text not null check (kind in ('pull_request', 'issue', 'document')),
34
+ number integer not null check (number > 0),
35
+ github_id integer check (github_id > 0),
83
36
  title text not null check (title <> ''),
84
- state text,
85
37
  url text,
86
- path text check (path <> '' and path not glob '/*' and path not glob '*[/]..[/]*' and path not glob '..[/]*'
87
- and path not glob '*[/]..' and path <> '..'),
88
- body text,
89
- author_identity_id integer references person_identity (id) on delete set null,
90
- source_created_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', source_created_at) is source_created_at),
91
- source_updated_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', source_updated_at) is source_updated_at),
92
- -- For a PR, the merge time (or the close time if closed without merging); for an issue, the close time. null for open items and documents.
93
- closed_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', closed_at) is closed_at),
94
- content_hash blob not null check (length(content_hash) = 32),
95
- metadata text not null default '{}' check (json_valid(metadata) and json_type(metadata) = 'object'),
96
- synced_at text not null default (strftime('%Y-%m-%dT%H:%M:%fZ', 'now'))
97
- check (strftime('%Y-%m-%dT%H:%M:%fZ', synced_at) is synced_at),
98
- unique (connector_id, external_id),
99
- check (
100
- case
101
- when kind = 'document' then path is not null and body is not null and state is null
102
- and closed_at is null
103
- else path is null and body is null and state in ('open', 'merged', 'closed') and (state = 'open') = (closed_at is null)
104
- end
105
- ),
106
- -- With a NULL state the CHECK above evaluates to NULL and passes, and the pairing with closed_at is not enforced either.
107
- constraint source_item_state_required check (kind = 'document' or state is not null)
38
+ state text not null check (state in ('open', 'merged', 'closed')),
39
+ harvested_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', harvested_at) is harvested_at),
40
+ unique (project_id, number)
108
41
  ) strict;
109
- create index source_item_listing on source_item (connector_id, kind, state, source_updated_at desc);
110
42
 
111
- -- A conversation: one coding session, or one GitHub PR or issue.
43
+ -- A conversation: one coding session.
112
44
  -- The id is a uuid derived deterministically from (project, origin, external_id). Sending the same session twice adds no rows.
113
- create table conversation (
45
+ create table "conversation" (
114
46
  id text primary key not null,
115
47
  project_id integer not null references project (id) on delete cascade,
116
- source_item_id integer references source_item (id) on delete cascade,
117
- origin text not null check (origin in ('claude-code', 'codex', 'github')),
48
+ origin text not null check (origin in ('claude-code', 'codex')),
118
49
  external_id text not null,
119
50
  branch text,
120
51
  started_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', started_at) is started_at),
121
- unique (project_id, origin, external_id),
122
- check ((origin = 'github') = (source_item_id is not null))
52
+ unique (project_id, origin, external_id)
123
53
  ) strict;
124
54
  create index conversation_recent on conversation (project_id, started_at desc);
125
55
 
126
- -- One row per message. self is what the owner typed, assistant is the AI's last reply or an AI reviewer, and bot is an automated notice.
56
+ -- One row per message. self is what the owner typed, and assistant is the AI's last reply.
127
57
  -- Oversized messages keep only their start and end, with truncated and the original size (UTF-8 bytes).
128
- create table message (
58
+ create table "message" (
129
59
  -- seq is the FTS5 rowid. It is an explicit integer primary key rather than the implicit rowid, so VACUUM does not renumber it
130
60
  seq integer primary key not null,
131
61
  id text not null unique,
132
62
  conversation_id text not null references conversation (id) on delete cascade,
133
63
  external_id text not null,
134
64
  turn_id text,
135
- reply_to_id text references message (id) on delete set null,
136
- speaker_kind text not null check (speaker_kind in ('self', 'person', 'assistant', 'bot')),
137
- identity_id integer references person_identity (id) on delete set null,
65
+ speaker_kind text not null check (speaker_kind in ('self', 'assistant')),
138
66
  body text not null check (body <> ''),
139
67
  truncated integer not null default 0 check (truncated in (0, 1)),
140
68
  original_bytes integer not null check (original_bytes > 0),
141
- url text,
142
69
  sent_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', sent_at) is sent_at),
143
70
  content_hash blob not null check (length(content_hash) = 32),
144
- -- Whether it goes into the full-text index. 0 for AI replies in coding sessions and automated notices (decided by indexesMessage in capture.ts / github.ts)
71
+ -- Whether it goes into the full-text index. 0 for AI replies (decided by indexesMessage in knowledge.ts)
145
72
  indexed integer not null check (indexed in (0, 1)),
146
73
  unique (conversation_id, external_id),
147
74
  check (truncated = 1 or original_bytes = length(cast(body as blob))),
148
75
  check (truncated = 0 or original_bytes > length(cast(body as blob)))
149
76
  ) strict;
150
77
  create index message_order on message (conversation_id, sent_at);
151
- create index message_by_identity on message (identity_id, sent_at desc) where identity_id is not null;
152
78
  create index message_self on message (sent_at desc) where speaker_kind = 'self';
153
79
 
154
80
  -- The full-text index. rowid = message.seq. Terms are split by sphica_terms() (terms() in server/src/text.ts, registered per connection).
@@ -166,13 +92,12 @@ create trigger message_fts_au after update of body, indexed on message begin
166
92
  end;
167
93
 
168
94
  -- Files linked to messages. Capture links an edited file (edit) to the owner's last message before the edit.
169
- -- The GitHub sync links a file pointed to in a review (review) to that review message.
170
95
  -- read records requirements and design documents read in the past, and is no longer written. path is relative to the project root.
171
- create table message_file (
96
+ create table "message_file" (
172
97
  message_id text not null references message (id) on delete cascade,
173
98
  path text not null check (path <> '' and path not glob '/*' and path not glob '*[/]..[/]*' and path not glob '..[/]*'
174
99
  and path not glob '*[/]..' and path <> '..'),
175
- action text not null check (action in ('edit', 'read', 'review')),
100
+ action text not null check (action in ('edit', 'read')),
176
101
  line_start integer check (line_start > 0),
177
102
  line_end integer check (line_end >= line_start),
178
103
  primary key (message_id, path, action)
@@ -195,18 +120,18 @@ create table work_item (
195
120
  ) strict;
196
121
  create index work_item_open on work_item (project_id, updated_at desc) where status in ('active', 'blocked', 'paused');
197
122
 
198
- -- A unit of searchable knowledge: decisions trace picked from conversations, and document sections.
123
+ -- A unit of searchable knowledge: decisions trace picked from a session, or harvest picked from a pull request.
199
124
  -- Overturned decisions are not deleted (deleting them gets them proposed again). They become superseded and point to the successor.
200
125
  -- stance follows from kind and status, and filters searches for paths not to take.
201
- create table knowledge (
126
+ create table "knowledge" (
202
127
  id integer primary key autoincrement not null,
203
128
  project_id integer not null references project (id) on delete cascade,
204
- source_item_id integer references source_item (id) on delete cascade,
205
129
  conversation_id text references conversation (id) on delete cascade,
130
+ pull_request_id integer references pull_request (id) on delete cascade,
206
131
  work_item_id integer references work_item (id) on delete set null,
207
132
  source_key text not null,
208
133
  kind text not null check (kind in ('decision', 'option', 'constraint', 'non_goal', 'dead_end', 'finding', 'debt',
209
- 'verification', 'question', 'document')),
134
+ 'verification', 'question')),
210
135
  status text,
211
136
  stance text not null generated always as (
212
137
  case
@@ -231,8 +156,8 @@ create table knowledge (
231
156
  occurred_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', occurred_at) is occurred_at),
232
157
  content_hash blob not null check (length(content_hash) = 32),
233
158
  unique (project_id, source_key),
234
- check (source_item_id is not null or conversation_id is not null),
235
- check (kind <> 'document' or (source_item_id is not null and heading is not null)),
159
+ -- Exactly one provenance: the session trace read, or the pull request harvest read
160
+ check ((conversation_id is null) <> (pull_request_id is null)),
236
161
  check (
237
162
  case kind
238
163
  when 'decision' then status is not null and status in ('proposed', 'accepted', 'rejected', 'superseded')
@@ -250,8 +175,7 @@ create table knowledge (
250
175
  check (superseded_by_id is null or superseded_by_id <> id),
251
176
  check (confirmation is null or kind = 'decision'),
252
177
  check (command is null or kind = 'verification'),
253
- check (json_array_length(downsides) = 0 or kind = 'decision'),
254
- check (kind <> 'document' or work_item_id is null)
178
+ check (json_array_length(downsides) = 0 or kind = 'decision')
255
179
  ) strict;
256
180
  create index knowledge_listing on knowledge (project_id, kind, status, occurred_at desc);
257
181
  create index knowledge_work on knowledge (work_item_id) where work_item_id is not null;
@@ -259,26 +183,27 @@ create index knowledge_work on knowledge (work_item_id) where work_item_id is no
259
183
  -- Extra search words for a record (synonyms, abbreviations, English equivalents of its words). **Search only**: no search result, read,
260
184
  -- or CLI output shows them. content_hash is the record's hash when they were written; they are indexed only while it
261
185
  -- still matches, so a record whose text changed stops being found by words written for its old text. source says who wrote them.
262
- create table knowledge_terms (
186
+ create table "knowledge_terms" (
263
187
  knowledge_id integer primary key not null references knowledge (id) on delete cascade,
264
188
  terms text not null check (terms <> '' and length(terms) <= 400),
265
189
  content_hash blob not null check (length(content_hash) = 32),
266
- source text not null check (source in ('trace', 'pr', 'import')),
190
+ source text not null check (source in ('trace', 'harvest', 'import')),
267
191
  written_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', written_at) is written_at)
268
192
  ) strict;
269
193
 
270
- -- What the knowledge index holds for each record: heading (h), body plus reason (b), and the extra search words whose hash matches (e).
194
+ -- What the knowledge index holds for each record: heading (h), body plus reason (b), and the extra search words whose hash matches plus the
195
+ -- refs (e), so a PR or issue number (pr:#12) finds the decisions that cite it.
271
196
  -- The triggers and `sphica db reindex` all insert from here, so the rule lives in one place.
272
197
  create view knowledge_search_text as
273
198
  select k.id,
274
199
  sphica_terms(coalesce(k.heading, '')) as h,
275
200
  sphica_terms(k.body || char(10) || coalesce(k.reason, '')) as b,
276
- sphica_terms(coalesce(t.terms, '')) as e
201
+ sphica_terms(coalesce(t.terms, '') || char(10) || k.refs) as e
277
202
  from knowledge k
278
203
  left join knowledge_terms t on t.knowledge_id = k.id and t.content_hash = k.content_hash;
279
204
 
280
205
  -- The full-text index. rowid = knowledge.id. Search uses bm25(knowledge_fts, 3, 1, 1).
281
- -- Trace headings hold the work title, and document section headings hold the path and heading levels.
206
+ -- Trace headings hold the work title, and harvest headings hold the pull request title.
282
207
  create virtual table knowledge_fts using fts5(h, b, e, content='', contentless_delete=1);
283
208
  create trigger knowledge_fts_ai after insert on knowledge begin
284
209
  insert into knowledge_fts (rowid, h, b, e) select id, h, b, e from knowledge_search_text where id = new.id;
@@ -316,8 +241,7 @@ create table knowledge_file (
316
241
  create index knowledge_file_path on knowledge_file (path, role);
317
242
 
318
243
  -- The 3 views capture (the capture connection) can write. The authorizer in server/src/sqlite.ts allows capture only inserts into these views
319
- -- and the writes inside the triggers below. source_item_id, identity_id, reply_to_id, and url are not in the views, so capture
320
- -- can neither create GitHub conversations nor claim someone else's identity. Conversation ids can be computed deterministically, so adding messages
244
+ -- and the writes inside the triggers below. Conversation ids can be computed deterministically, so adding messages
321
245
  -- to an existing conversation is not blocked (the remaining surface if the capture path is abused).
322
246
  create view capture_conversation as
323
247
  select id, project_id, origin, external_id, branch, started_at from conversation;
@@ -346,4 +270,4 @@ create trigger capture_message_file_insert instead of insert on capture_message_
346
270
  on conflict do nothing;
347
271
  end;
348
272
 
349
- pragma user_version = 5;
273
+ pragma user_version = 7;
package/dist/capture.js CHANGED
@@ -10813,7 +10813,7 @@ import fs from "node:fs";
10813
10813
  import os from "node:os";
10814
10814
  import path from "node:path";
10815
10815
  import { constants as C, DatabaseSync } from "node:sqlite";
10816
- var SCHEMA_REVISION = 5;
10816
+ var SCHEMA_REVISION = 7;
10817
10817
  var dbFile = () => process.env.SPHICA_DB || path.join(os.homedir(), ".sphica", "sphica.db");
10818
10818
  function requireRuntime() {
10819
10819
  const proto = DatabaseSync.prototype;
@@ -10856,16 +10856,7 @@ var READER_FUNCTIONS = new Set([
10856
10856
  var SHADOW = /^(knowledge|message)_fts_(data|idx|docsize|config)$/;
10857
10857
 
10858
10858
  // server/src/db.ts
10859
- var JSON_COLUMNS = new Set([
10860
- "refs",
10861
- "downsides",
10862
- "next",
10863
- "metadata",
10864
- "connectors",
10865
- "files",
10866
- "handles",
10867
- "paths"
10868
- ]);
10859
+ var JSON_COLUMNS = new Set(["refs", "downsides", "next", "files", "paths"]);
10869
10860
  var TOP_LEVEL = /^\$\[\d+\]\."([^"]+)"$/;
10870
10861
  var parseJson = new ParseJSONResultsPlugin({
10871
10862
  shouldParse: (_value, jsonPath) => JSON_COLUMNS.has(jsonPath.match(TOP_LEVEL)?.[1] ?? "")
@@ -11199,7 +11190,7 @@ function openWriter(role, file = dbFile()) {
11199
11190
  }
11200
11191
 
11201
11192
  // server/src/knowledge.ts
11202
- var indexesMessage = (origin, speakerKind) => speakerKind !== "bot" && !(origin !== "github" && speakerKind === "assistant");
11193
+ var indexesMessage = (speakerKind) => speakerKind === "self";
11203
11194
  var conversationId = (projectId, origin, externalId) => uuidFrom(String(projectId), origin, externalId);
11204
11195
 
11205
11196
  // server/src/panel.ts
@@ -11636,7 +11627,7 @@ async function write(db, batch, projects) {
11636
11627
  original_bytes: x.m.originalBytes,
11637
11628
  sent_at: iso(x.m.at),
11638
11629
  content_hash: sha256(x.m.body),
11639
- indexed: indexesMessage(x.m.host, x.m.speaker) ? 1 : 0
11630
+ indexed: indexesMessage(x.m.speaker) ? 1 : 0
11640
11631
  }))).execute();
11641
11632
  }
11642
11633
  let after = 0;