sphica 0.2.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/.codex-plugin/plugin.json +1 -1
- package/README.md +15 -30
- package/THIRD_PARTY_NOTICES.md +0 -53
- package/db/migrations/0006_harvest_provenance.sql +45 -0
- package/db/migrations/0007_drop_bulk_import.sql +238 -0
- package/db/schema.sql +35 -111
- package/dist/capture.js +4 -13
- package/dist/cli.js +1230 -3648
- package/dist/mcp.js +50 -163
- package/package.json +1 -1
- package/skills/harvest/SKILL.md +114 -0
- package/skills/harvest/agents/openai.yaml +2 -0
- package/skills/review/SKILL.md +1 -1
- package/skills/trace/SKILL.md +28 -18
package/db/schema.sql
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
-- The source of truth for sphica's database (SQLite, `node:sqlite`). It lets one owner look up decisions and conversations on that machine.
|
|
2
2
|
-- **Each machine is independent and shares no records.** One file (~/.sphica/sphica.db) is one database, with no schema qualifiers.
|
|
3
3
|
--
|
|
4
|
-
-- Three boundaries:
|
|
4
|
+
-- Three boundaries: verbatim conversations (conversation / message), the pull requests harvested into knowledge (pull_request),
|
|
5
5
|
-- and searchable knowledge (knowledge). Work status (work_item) is state that gets updated, so it has its own table.
|
|
6
6
|
--
|
|
7
7
|
-- The version is `pragma user_version` at the end. MCP and the CLI compare it with SCHEMA_REVISION in server/src/db.ts
|
|
@@ -25,130 +25,56 @@ create table project (
|
|
|
25
25
|
check (strftime('%Y-%m-%dT%H:%M:%fZ', created_at) is created_at)
|
|
26
26
|
) strict;
|
|
27
27
|
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
) strict;
|
|
33
|
-
-- Exactly one person is the owner who asks. This decides who "I" is in "what did I say?".
|
|
34
|
-
create unique index person_one_self on person (is_self) where is_self = 1;
|
|
35
|
-
|
|
36
|
-
-- Identifiers at a source. A GitHub user id stays the same when the login changes, so it goes in external_id.
|
|
37
|
-
create table person_identity (
|
|
38
|
-
id integer primary key autoincrement not null,
|
|
39
|
-
person_id integer references person (id) on delete set null,
|
|
40
|
-
provider text not null check (provider in ('github')),
|
|
41
|
-
external_id text not null,
|
|
42
|
-
handle text not null,
|
|
43
|
-
unique (provider, external_id)
|
|
44
|
-
) strict;
|
|
45
|
-
create index person_identity_handle on person_identity (provider, lower(handle));
|
|
46
|
-
|
|
47
|
-
-- Per source, the last imported version and the latest result. No secrets (they live in the syncing machine's environment).
|
|
48
|
-
-- For documents, the imported commit (head_oid). The next sync imports automatically only commits that fast-forward from it.
|
|
49
|
-
-- For GitHub, the time the fetch started (snapshot_at). A fetch that started earlier is not written, even if it commits later.
|
|
50
|
-
create table connector (
|
|
28
|
+
-- A pull request harvested into knowledge (the harvest Skill). github_id is GitHub's id for the PR: a project moved to another repository
|
|
29
|
+
-- can reuse a number, and the save command refuses a number whose id differs. harvested_at is the last successful save
|
|
30
|
+
-- (null for decisions moved here from the old PR-body extraction).
|
|
31
|
+
create table pull_request (
|
|
51
32
|
id integer primary key autoincrement not null,
|
|
52
33
|
project_id integer not null references project (id) on delete cascade,
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
and (length(head_oid) = 40 or length(head_oid) = 64) and head_oid not glob '*[^0-9a-f]*')),
|
|
56
|
-
snapshot_at text check (snapshot_at is null or provider = 'github')
|
|
57
|
-
check (strftime('%Y-%m-%dT%H:%M:%fZ', snapshot_at) is snapshot_at),
|
|
58
|
-
last_success_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', last_success_at) is last_success_at),
|
|
59
|
-
last_error text,
|
|
60
|
-
unique (project_id, provider)
|
|
61
|
-
) strict;
|
|
62
|
-
|
|
63
|
-
-- Paths the docs sync does not import. Not every tracked Markdown file states facts (such as audit fixtures).
|
|
64
|
-
-- **This is importer-side configuration.** It must work for read-only projects too, so it is not a manifest in the repository.
|
|
65
|
-
-- file matches the path exactly, and directory matches paths starting with `<path>/`. Only created for the docs connector.
|
|
66
|
-
create table docs_exclude (
|
|
67
|
-
connector_id integer not null references connector (id) on delete cascade,
|
|
68
|
-
kind text not null check (kind in ('file', 'directory')),
|
|
69
|
-
path text not null check (
|
|
70
|
-
path <> '' and path not glob '/*' and path not glob '*[/]..[/]*' and path not glob '..[/]*'
|
|
71
|
-
and path not glob '*[/]..' and path <> '..' and path not glob '*/'
|
|
72
|
-
and path not glob '*[' || char(1) || '-' || char(31) || char(127) || ']*'),
|
|
73
|
-
primary key (connector_id, kind, path)
|
|
74
|
-
) strict;
|
|
75
|
-
|
|
76
|
-
-- The current state of a source. Items confirmed gone by a complete listing are deleted with their rows (no tombstones).
|
|
77
|
-
-- Documents keep their original text in body. Search uses the knowledge sections, and joining sections never restores the original.
|
|
78
|
-
create table "source_item" (
|
|
79
|
-
id integer primary key autoincrement not null,
|
|
80
|
-
connector_id integer not null references connector (id) on delete cascade,
|
|
81
|
-
external_id text not null,
|
|
82
|
-
kind text not null check (kind in ('pull_request', 'issue', 'document')),
|
|
34
|
+
number integer not null check (number > 0),
|
|
35
|
+
github_id integer check (github_id > 0),
|
|
83
36
|
title text not null check (title <> ''),
|
|
84
|
-
state text,
|
|
85
37
|
url text,
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
author_identity_id integer references person_identity (id) on delete set null,
|
|
90
|
-
source_created_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', source_created_at) is source_created_at),
|
|
91
|
-
source_updated_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', source_updated_at) is source_updated_at),
|
|
92
|
-
-- For a PR, the merge time (or the close time if closed without merging); for an issue, the close time. null for open items and documents.
|
|
93
|
-
closed_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', closed_at) is closed_at),
|
|
94
|
-
content_hash blob not null check (length(content_hash) = 32),
|
|
95
|
-
metadata text not null default '{}' check (json_valid(metadata) and json_type(metadata) = 'object'),
|
|
96
|
-
synced_at text not null default (strftime('%Y-%m-%dT%H:%M:%fZ', 'now'))
|
|
97
|
-
check (strftime('%Y-%m-%dT%H:%M:%fZ', synced_at) is synced_at),
|
|
98
|
-
unique (connector_id, external_id),
|
|
99
|
-
check (
|
|
100
|
-
case
|
|
101
|
-
when kind = 'document' then path is not null and body is not null and state is null
|
|
102
|
-
and closed_at is null
|
|
103
|
-
else path is null and body is null and state in ('open', 'merged', 'closed') and (state = 'open') = (closed_at is null)
|
|
104
|
-
end
|
|
105
|
-
),
|
|
106
|
-
-- With a NULL state the CHECK above evaluates to NULL and passes, and the pairing with closed_at is not enforced either.
|
|
107
|
-
constraint source_item_state_required check (kind = 'document' or state is not null)
|
|
38
|
+
state text not null check (state in ('open', 'merged', 'closed')),
|
|
39
|
+
harvested_at text check (strftime('%Y-%m-%dT%H:%M:%fZ', harvested_at) is harvested_at),
|
|
40
|
+
unique (project_id, number)
|
|
108
41
|
) strict;
|
|
109
|
-
create index source_item_listing on source_item (connector_id, kind, state, source_updated_at desc);
|
|
110
42
|
|
|
111
|
-
-- A conversation: one coding session
|
|
43
|
+
-- A conversation: one coding session.
|
|
112
44
|
-- The id is a uuid derived deterministically from (project, origin, external_id). Sending the same session twice adds no rows.
|
|
113
|
-
create table conversation (
|
|
45
|
+
create table "conversation" (
|
|
114
46
|
id text primary key not null,
|
|
115
47
|
project_id integer not null references project (id) on delete cascade,
|
|
116
|
-
|
|
117
|
-
origin text not null check (origin in ('claude-code', 'codex', 'github')),
|
|
48
|
+
origin text not null check (origin in ('claude-code', 'codex')),
|
|
118
49
|
external_id text not null,
|
|
119
50
|
branch text,
|
|
120
51
|
started_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', started_at) is started_at),
|
|
121
|
-
unique (project_id, origin, external_id)
|
|
122
|
-
check ((origin = 'github') = (source_item_id is not null))
|
|
52
|
+
unique (project_id, origin, external_id)
|
|
123
53
|
) strict;
|
|
124
54
|
create index conversation_recent on conversation (project_id, started_at desc);
|
|
125
55
|
|
|
126
|
-
-- One row per message. self is what the owner typed, assistant is the AI's last reply
|
|
56
|
+
-- One row per message. self is what the owner typed, and assistant is the AI's last reply.
|
|
127
57
|
-- Oversized messages keep only their start and end, with truncated and the original size (UTF-8 bytes).
|
|
128
|
-
create table message (
|
|
58
|
+
create table "message" (
|
|
129
59
|
-- seq is the FTS5 rowid. It is an explicit integer primary key rather than the implicit rowid, so VACUUM does not renumber it
|
|
130
60
|
seq integer primary key not null,
|
|
131
61
|
id text not null unique,
|
|
132
62
|
conversation_id text not null references conversation (id) on delete cascade,
|
|
133
63
|
external_id text not null,
|
|
134
64
|
turn_id text,
|
|
135
|
-
|
|
136
|
-
speaker_kind text not null check (speaker_kind in ('self', 'person', 'assistant', 'bot')),
|
|
137
|
-
identity_id integer references person_identity (id) on delete set null,
|
|
65
|
+
speaker_kind text not null check (speaker_kind in ('self', 'assistant')),
|
|
138
66
|
body text not null check (body <> ''),
|
|
139
67
|
truncated integer not null default 0 check (truncated in (0, 1)),
|
|
140
68
|
original_bytes integer not null check (original_bytes > 0),
|
|
141
|
-
url text,
|
|
142
69
|
sent_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', sent_at) is sent_at),
|
|
143
70
|
content_hash blob not null check (length(content_hash) = 32),
|
|
144
|
-
-- Whether it goes into the full-text index. 0 for AI replies
|
|
71
|
+
-- Whether it goes into the full-text index. 0 for AI replies (decided by indexesMessage in knowledge.ts)
|
|
145
72
|
indexed integer not null check (indexed in (0, 1)),
|
|
146
73
|
unique (conversation_id, external_id),
|
|
147
74
|
check (truncated = 1 or original_bytes = length(cast(body as blob))),
|
|
148
75
|
check (truncated = 0 or original_bytes > length(cast(body as blob)))
|
|
149
76
|
) strict;
|
|
150
77
|
create index message_order on message (conversation_id, sent_at);
|
|
151
|
-
create index message_by_identity on message (identity_id, sent_at desc) where identity_id is not null;
|
|
152
78
|
create index message_self on message (sent_at desc) where speaker_kind = 'self';
|
|
153
79
|
|
|
154
80
|
-- The full-text index. rowid = message.seq. Terms are split by sphica_terms() (terms() in server/src/text.ts, registered per connection).
|
|
@@ -166,13 +92,12 @@ create trigger message_fts_au after update of body, indexed on message begin
|
|
|
166
92
|
end;
|
|
167
93
|
|
|
168
94
|
-- Files linked to messages. Capture links an edited file (edit) to the owner's last message before the edit.
|
|
169
|
-
-- The GitHub sync links a file pointed to in a review (review) to that review message.
|
|
170
95
|
-- read records requirements and design documents read in the past, and is no longer written. path is relative to the project root.
|
|
171
|
-
create table message_file (
|
|
96
|
+
create table "message_file" (
|
|
172
97
|
message_id text not null references message (id) on delete cascade,
|
|
173
98
|
path text not null check (path <> '' and path not glob '/*' and path not glob '*[/]..[/]*' and path not glob '..[/]*'
|
|
174
99
|
and path not glob '*[/]..' and path <> '..'),
|
|
175
|
-
action text not null check (action in ('edit', 'read'
|
|
100
|
+
action text not null check (action in ('edit', 'read')),
|
|
176
101
|
line_start integer check (line_start > 0),
|
|
177
102
|
line_end integer check (line_end >= line_start),
|
|
178
103
|
primary key (message_id, path, action)
|
|
@@ -195,18 +120,18 @@ create table work_item (
|
|
|
195
120
|
) strict;
|
|
196
121
|
create index work_item_open on work_item (project_id, updated_at desc) where status in ('active', 'blocked', 'paused');
|
|
197
122
|
|
|
198
|
-
-- A unit of searchable knowledge: decisions trace picked from
|
|
123
|
+
-- A unit of searchable knowledge: decisions trace picked from a session, or harvest picked from a pull request.
|
|
199
124
|
-- Overturned decisions are not deleted (deleting them gets them proposed again). They become superseded and point to the successor.
|
|
200
125
|
-- stance follows from kind and status, and filters searches for paths not to take.
|
|
201
|
-
create table knowledge (
|
|
126
|
+
create table "knowledge" (
|
|
202
127
|
id integer primary key autoincrement not null,
|
|
203
128
|
project_id integer not null references project (id) on delete cascade,
|
|
204
|
-
source_item_id integer references source_item (id) on delete cascade,
|
|
205
129
|
conversation_id text references conversation (id) on delete cascade,
|
|
130
|
+
pull_request_id integer references pull_request (id) on delete cascade,
|
|
206
131
|
work_item_id integer references work_item (id) on delete set null,
|
|
207
132
|
source_key text not null,
|
|
208
133
|
kind text not null check (kind in ('decision', 'option', 'constraint', 'non_goal', 'dead_end', 'finding', 'debt',
|
|
209
|
-
'verification', 'question'
|
|
134
|
+
'verification', 'question')),
|
|
210
135
|
status text,
|
|
211
136
|
stance text not null generated always as (
|
|
212
137
|
case
|
|
@@ -231,8 +156,8 @@ create table knowledge (
|
|
|
231
156
|
occurred_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', occurred_at) is occurred_at),
|
|
232
157
|
content_hash blob not null check (length(content_hash) = 32),
|
|
233
158
|
unique (project_id, source_key),
|
|
234
|
-
|
|
235
|
-
check (
|
|
159
|
+
-- Exactly one provenance: the session trace read, or the pull request harvest read
|
|
160
|
+
check ((conversation_id is null) <> (pull_request_id is null)),
|
|
236
161
|
check (
|
|
237
162
|
case kind
|
|
238
163
|
when 'decision' then status is not null and status in ('proposed', 'accepted', 'rejected', 'superseded')
|
|
@@ -250,8 +175,7 @@ create table knowledge (
|
|
|
250
175
|
check (superseded_by_id is null or superseded_by_id <> id),
|
|
251
176
|
check (confirmation is null or kind = 'decision'),
|
|
252
177
|
check (command is null or kind = 'verification'),
|
|
253
|
-
check (json_array_length(downsides) = 0 or kind = 'decision')
|
|
254
|
-
check (kind <> 'document' or work_item_id is null)
|
|
178
|
+
check (json_array_length(downsides) = 0 or kind = 'decision')
|
|
255
179
|
) strict;
|
|
256
180
|
create index knowledge_listing on knowledge (project_id, kind, status, occurred_at desc);
|
|
257
181
|
create index knowledge_work on knowledge (work_item_id) where work_item_id is not null;
|
|
@@ -259,26 +183,27 @@ create index knowledge_work on knowledge (work_item_id) where work_item_id is no
|
|
|
259
183
|
-- Extra search words for a record (synonyms, abbreviations, English equivalents of its words). **Search only**: no search result, read,
|
|
260
184
|
-- or CLI output shows them. content_hash is the record's hash when they were written; they are indexed only while it
|
|
261
185
|
-- still matches, so a record whose text changed stops being found by words written for its old text. source says who wrote them.
|
|
262
|
-
create table knowledge_terms (
|
|
186
|
+
create table "knowledge_terms" (
|
|
263
187
|
knowledge_id integer primary key not null references knowledge (id) on delete cascade,
|
|
264
188
|
terms text not null check (terms <> '' and length(terms) <= 400),
|
|
265
189
|
content_hash blob not null check (length(content_hash) = 32),
|
|
266
|
-
source text not null check (source in ('trace', '
|
|
190
|
+
source text not null check (source in ('trace', 'harvest', 'import')),
|
|
267
191
|
written_at text not null check (strftime('%Y-%m-%dT%H:%M:%fZ', written_at) is written_at)
|
|
268
192
|
) strict;
|
|
269
193
|
|
|
270
|
-
-- What the knowledge index holds for each record: heading (h), body plus reason (b), and the extra search words whose hash matches
|
|
194
|
+
-- What the knowledge index holds for each record: heading (h), body plus reason (b), and the extra search words whose hash matches plus the
|
|
195
|
+
-- refs (e), so a PR or issue number (pr:#12) finds the decisions that cite it.
|
|
271
196
|
-- The triggers and `sphica db reindex` all insert from here, so the rule lives in one place.
|
|
272
197
|
create view knowledge_search_text as
|
|
273
198
|
select k.id,
|
|
274
199
|
sphica_terms(coalesce(k.heading, '')) as h,
|
|
275
200
|
sphica_terms(k.body || char(10) || coalesce(k.reason, '')) as b,
|
|
276
|
-
sphica_terms(coalesce(t.terms, '')) as e
|
|
201
|
+
sphica_terms(coalesce(t.terms, '') || char(10) || k.refs) as e
|
|
277
202
|
from knowledge k
|
|
278
203
|
left join knowledge_terms t on t.knowledge_id = k.id and t.content_hash = k.content_hash;
|
|
279
204
|
|
|
280
205
|
-- The full-text index. rowid = knowledge.id. Search uses bm25(knowledge_fts, 3, 1, 1).
|
|
281
|
-
-- Trace headings hold the work title, and
|
|
206
|
+
-- Trace headings hold the work title, and harvest headings hold the pull request title.
|
|
282
207
|
create virtual table knowledge_fts using fts5(h, b, e, content='', contentless_delete=1);
|
|
283
208
|
create trigger knowledge_fts_ai after insert on knowledge begin
|
|
284
209
|
insert into knowledge_fts (rowid, h, b, e) select id, h, b, e from knowledge_search_text where id = new.id;
|
|
@@ -316,8 +241,7 @@ create table knowledge_file (
|
|
|
316
241
|
create index knowledge_file_path on knowledge_file (path, role);
|
|
317
242
|
|
|
318
243
|
-- The 3 views capture (the capture connection) can write. The authorizer in server/src/sqlite.ts allows capture only inserts into these views
|
|
319
|
-
-- and the writes inside the triggers below.
|
|
320
|
-
-- can neither create GitHub conversations nor claim someone else's identity. Conversation ids can be computed deterministically, so adding messages
|
|
244
|
+
-- and the writes inside the triggers below. Conversation ids can be computed deterministically, so adding messages
|
|
321
245
|
-- to an existing conversation is not blocked (the remaining surface if the capture path is abused).
|
|
322
246
|
create view capture_conversation as
|
|
323
247
|
select id, project_id, origin, external_id, branch, started_at from conversation;
|
|
@@ -346,4 +270,4 @@ create trigger capture_message_file_insert instead of insert on capture_message_
|
|
|
346
270
|
on conflict do nothing;
|
|
347
271
|
end;
|
|
348
272
|
|
|
349
|
-
pragma user_version =
|
|
273
|
+
pragma user_version = 7;
|
package/dist/capture.js
CHANGED
|
@@ -10813,7 +10813,7 @@ import fs from "node:fs";
|
|
|
10813
10813
|
import os from "node:os";
|
|
10814
10814
|
import path from "node:path";
|
|
10815
10815
|
import { constants as C, DatabaseSync } from "node:sqlite";
|
|
10816
|
-
var SCHEMA_REVISION =
|
|
10816
|
+
var SCHEMA_REVISION = 7;
|
|
10817
10817
|
var dbFile = () => process.env.SPHICA_DB || path.join(os.homedir(), ".sphica", "sphica.db");
|
|
10818
10818
|
function requireRuntime() {
|
|
10819
10819
|
const proto = DatabaseSync.prototype;
|
|
@@ -10856,16 +10856,7 @@ var READER_FUNCTIONS = new Set([
|
|
|
10856
10856
|
var SHADOW = /^(knowledge|message)_fts_(data|idx|docsize|config)$/;
|
|
10857
10857
|
|
|
10858
10858
|
// server/src/db.ts
|
|
10859
|
-
var JSON_COLUMNS = new Set([
|
|
10860
|
-
"refs",
|
|
10861
|
-
"downsides",
|
|
10862
|
-
"next",
|
|
10863
|
-
"metadata",
|
|
10864
|
-
"connectors",
|
|
10865
|
-
"files",
|
|
10866
|
-
"handles",
|
|
10867
|
-
"paths"
|
|
10868
|
-
]);
|
|
10859
|
+
var JSON_COLUMNS = new Set(["refs", "downsides", "next", "files", "paths"]);
|
|
10869
10860
|
var TOP_LEVEL = /^\$\[\d+\]\."([^"]+)"$/;
|
|
10870
10861
|
var parseJson = new ParseJSONResultsPlugin({
|
|
10871
10862
|
shouldParse: (_value, jsonPath) => JSON_COLUMNS.has(jsonPath.match(TOP_LEVEL)?.[1] ?? "")
|
|
@@ -11199,7 +11190,7 @@ function openWriter(role, file = dbFile()) {
|
|
|
11199
11190
|
}
|
|
11200
11191
|
|
|
11201
11192
|
// server/src/knowledge.ts
|
|
11202
|
-
var indexesMessage = (
|
|
11193
|
+
var indexesMessage = (speakerKind) => speakerKind === "self";
|
|
11203
11194
|
var conversationId = (projectId, origin, externalId) => uuidFrom(String(projectId), origin, externalId);
|
|
11204
11195
|
|
|
11205
11196
|
// server/src/panel.ts
|
|
@@ -11636,7 +11627,7 @@ async function write(db, batch, projects) {
|
|
|
11636
11627
|
original_bytes: x.m.originalBytes,
|
|
11637
11628
|
sent_at: iso(x.m.at),
|
|
11638
11629
|
content_hash: sha256(x.m.body),
|
|
11639
|
-
indexed: indexesMessage(x.m.
|
|
11630
|
+
indexed: indexesMessage(x.m.speaker) ? 1 : 0
|
|
11640
11631
|
}))).execute();
|
|
11641
11632
|
}
|
|
11642
11633
|
let after = 0;
|