@jungjaehoon/mama-core 1.6.0 → 1.8.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/db/migrations/037-create-context-packets.sql +32 -0
- package/db/migrations/038-create-vnext-operator-contracts.sql +57 -0
- package/db/migrations/039-add-connector-event-operator-ingest-seq.sql +96 -0
- package/db/migrations/040-create-operator-memory-commit-intents.sql +19 -0
- package/db/migrations/041-enforce-operator-memory-commit-claim-invariant.sql +74 -0
- package/db/migrations/042-embedding-prefix-scheme.sql +46 -0
- package/dist/cases/wiki-page-index.js +6 -2
- package/dist/context-compile/boundary-defaults.d.ts +12 -0
- package/dist/context-compile/boundary-defaults.js +105 -0
- package/dist/context-compile/compiler-policy.d.ts +39 -0
- package/dist/context-compile/compiler-policy.js +147 -0
- package/dist/context-compile/compiler.d.ts +18 -0
- package/dist/context-compile/compiler.js +401 -0
- package/dist/context-compile/index.d.ts +9 -0
- package/dist/context-compile/index.js +25 -0
- package/dist/context-compile/packet-store.d.ts +9 -0
- package/dist/context-compile/packet-store.js +255 -0
- package/dist/context-compile/ref.d.ts +7 -0
- package/dist/context-compile/ref.js +102 -0
- package/dist/context-compile/source-readers.d.ts +69 -0
- package/dist/context-compile/source-readers.js +707 -0
- package/dist/context-compile/types.d.ts +138 -0
- package/dist/context-compile/types.js +11 -0
- package/dist/context-compile/visibility.d.ts +21 -0
- package/dist/context-compile/visibility.js +200 -0
- package/dist/db-adapter/node-sqlite-adapter.d.ts +17 -1
- package/dist/db-adapter/node-sqlite-adapter.js +399 -6
- package/dist/db-manager.d.ts +9 -10
- package/dist/db-manager.js +59 -4
- package/dist/edges/ref-validation.js +55 -16
- package/dist/edges/types.d.ts +2 -0
- package/dist/embedding-client.d.ts +3 -1
- package/dist/embedding-client.js +3 -2
- package/dist/embedding-server/index.d.ts +2 -0
- package/dist/embedding-server/index.js +10 -2
- package/dist/embeddings.d.ts +8 -3
- package/dist/embeddings.js +48 -15
- package/dist/entities/store.d.ts +4 -1
- package/dist/entities/store.js +206 -78
- package/dist/entities/types.d.ts +5 -1
- package/dist/entities/types.js +1 -1
- package/dist/index.d.ts +4 -2
- package/dist/index.js +8 -4
- package/dist/mama-api.js +1 -1
- package/dist/memory/api.d.ts +6 -1
- package/dist/memory/api.js +223 -29
- package/dist/memory/provenance.d.ts +1 -0
- package/dist/memory/provenance.js +2 -0
- package/dist/memory/types.d.ts +1 -0
- package/dist/memory-inject.js +1 -1
- package/dist/provenance/source-ref.d.ts +29 -0
- package/dist/provenance/source-ref.js +204 -0
- package/package.json +4 -2
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
CREATE TABLE IF NOT EXISTS context_packets (
|
|
2
|
+
packet_id TEXT PRIMARY KEY,
|
|
3
|
+
task TEXT NOT NULL,
|
|
4
|
+
packet_json TEXT NOT NULL,
|
|
5
|
+
scope_json TEXT NOT NULL,
|
|
6
|
+
scope_hash TEXT NOT NULL,
|
|
7
|
+
envelope_hash TEXT NOT NULL,
|
|
8
|
+
model_run_id TEXT NOT NULL,
|
|
9
|
+
agent_id TEXT NOT NULL,
|
|
10
|
+
input_snapshot_ref TEXT NOT NULL,
|
|
11
|
+
source_refs_json TEXT NOT NULL DEFAULT '[]',
|
|
12
|
+
tenant_id TEXT NOT NULL DEFAULT 'default',
|
|
13
|
+
project_id TEXT NOT NULL,
|
|
14
|
+
memory_scope_kind TEXT NOT NULL CHECK (memory_scope_kind IN ('global', 'user', 'channel', 'project')),
|
|
15
|
+
memory_scope_id TEXT NOT NULL,
|
|
16
|
+
created_at INTEGER NOT NULL CHECK (created_at >= 0)
|
|
17
|
+
);
|
|
18
|
+
|
|
19
|
+
CREATE INDEX IF NOT EXISTS idx_context_packets_model_run
|
|
20
|
+
ON context_packets(model_run_id, created_at DESC);
|
|
21
|
+
|
|
22
|
+
CREATE INDEX IF NOT EXISTS idx_context_packets_envelope
|
|
23
|
+
ON context_packets(envelope_hash, created_at DESC);
|
|
24
|
+
|
|
25
|
+
CREATE INDEX IF NOT EXISTS idx_context_packets_scope
|
|
26
|
+
ON context_packets(tenant_id, project_id, memory_scope_kind, memory_scope_id, created_at DESC);
|
|
27
|
+
|
|
28
|
+
CREATE INDEX IF NOT EXISTS idx_context_packets_scope_hash
|
|
29
|
+
ON context_packets(scope_hash, created_at DESC);
|
|
30
|
+
|
|
31
|
+
INSERT OR IGNORE INTO schema_version (version, description)
|
|
32
|
+
VALUES (37, 'Create context packet append-only store');
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
CREATE TABLE IF NOT EXISTS vnext_operator_cursors (
|
|
2
|
+
cursor_name TEXT PRIMARY KEY,
|
|
3
|
+
last_change_seq INTEGER NOT NULL DEFAULT 0 CHECK (last_change_seq >= 0),
|
|
4
|
+
last_idempotency_key TEXT,
|
|
5
|
+
updated_at_ms INTEGER NOT NULL CHECK (updated_at_ms >= 0)
|
|
6
|
+
);
|
|
7
|
+
|
|
8
|
+
CREATE TABLE IF NOT EXISTS vnext_operator_commits (
|
|
9
|
+
commit_id TEXT PRIMARY KEY,
|
|
10
|
+
cursor_name TEXT NOT NULL,
|
|
11
|
+
idempotency_key TEXT NOT NULL UNIQUE,
|
|
12
|
+
first_change_seq INTEGER NOT NULL CHECK (first_change_seq >= 0),
|
|
13
|
+
last_change_seq INTEGER NOT NULL CHECK (last_change_seq >= first_change_seq),
|
|
14
|
+
status TEXT NOT NULL CHECK (status IN ('changed', 'no_update')),
|
|
15
|
+
changed_refs_json TEXT NOT NULL CHECK (json_valid(changed_refs_json)),
|
|
16
|
+
source_refs_json TEXT NOT NULL CHECK (json_valid(source_refs_json)),
|
|
17
|
+
created_at_ms INTEGER NOT NULL CHECK (created_at_ms >= 0),
|
|
18
|
+
FOREIGN KEY (cursor_name) REFERENCES vnext_operator_cursors(cursor_name)
|
|
19
|
+
);
|
|
20
|
+
|
|
21
|
+
CREATE INDEX IF NOT EXISTS idx_vnext_operator_commits_cursor_seq
|
|
22
|
+
ON vnext_operator_commits(cursor_name, last_change_seq);
|
|
23
|
+
|
|
24
|
+
CREATE TABLE IF NOT EXISTS operator_no_updates (
|
|
25
|
+
no_update_id TEXT PRIMARY KEY,
|
|
26
|
+
scope_key TEXT NOT NULL,
|
|
27
|
+
reason TEXT NOT NULL,
|
|
28
|
+
source_refs_json TEXT NOT NULL CHECK (json_valid(source_refs_json)),
|
|
29
|
+
idempotency_key TEXT NOT NULL UNIQUE,
|
|
30
|
+
created_at_ms INTEGER NOT NULL CHECK (created_at_ms >= 0)
|
|
31
|
+
);
|
|
32
|
+
|
|
33
|
+
CREATE INDEX IF NOT EXISTS idx_operator_no_updates_scope_created
|
|
34
|
+
ON operator_no_updates(scope_key, created_at_ms DESC);
|
|
35
|
+
|
|
36
|
+
CREATE TABLE IF NOT EXISTS worker_proposals (
|
|
37
|
+
proposal_id TEXT PRIMARY KEY,
|
|
38
|
+
worker_id TEXT NOT NULL,
|
|
39
|
+
kind TEXT NOT NULL,
|
|
40
|
+
payload_json TEXT NOT NULL CHECK (json_valid(payload_json)),
|
|
41
|
+
source_refs_json TEXT NOT NULL CHECK (json_valid(source_refs_json)),
|
|
42
|
+
confidence REAL NOT NULL CHECK (confidence >= 0 AND confidence <= 1),
|
|
43
|
+
status TEXT NOT NULL CHECK (status IN ('proposed', 'accepted', 'rejected', 'superseded')),
|
|
44
|
+
created_at_ms INTEGER NOT NULL CHECK (created_at_ms >= 0),
|
|
45
|
+
accepted_at_ms INTEGER CHECK (
|
|
46
|
+
(status = 'proposed' AND accepted_at_ms IS NULL) OR
|
|
47
|
+
(status = 'accepted' AND accepted_at_ms IS NOT NULL AND accepted_at_ms >= created_at_ms) OR
|
|
48
|
+
(status = 'rejected' AND accepted_at_ms IS NULL) OR
|
|
49
|
+
(status = 'superseded' AND (accepted_at_ms IS NULL OR accepted_at_ms >= created_at_ms))
|
|
50
|
+
)
|
|
51
|
+
);
|
|
52
|
+
|
|
53
|
+
CREATE INDEX IF NOT EXISTS idx_worker_proposals_status_kind
|
|
54
|
+
ON worker_proposals(status, kind, created_at_ms);
|
|
55
|
+
|
|
56
|
+
INSERT OR IGNORE INTO schema_version (version, description)
|
|
57
|
+
VALUES (38, 'Create vNext operator contracts');
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
ALTER TABLE connector_event_index
|
|
2
|
+
ADD COLUMN operator_ingest_seq INTEGER CHECK (
|
|
3
|
+
operator_ingest_seq IS NULL OR operator_ingest_seq >= 1
|
|
4
|
+
);
|
|
5
|
+
|
|
6
|
+
CREATE TABLE IF NOT EXISTS connector_event_index_operator_seq_cursors (
|
|
7
|
+
source_connector TEXT NOT NULL,
|
|
8
|
+
channel TEXT NOT NULL DEFAULT '',
|
|
9
|
+
next_seq INTEGER NOT NULL CHECK (next_seq >= 1),
|
|
10
|
+
PRIMARY KEY (source_connector, channel)
|
|
11
|
+
);
|
|
12
|
+
|
|
13
|
+
WITH ranked_events AS (
|
|
14
|
+
SELECT
|
|
15
|
+
event_index_id,
|
|
16
|
+
ROW_NUMBER() OVER (
|
|
17
|
+
PARTITION BY source_connector, COALESCE(channel, '')
|
|
18
|
+
ORDER BY rowid ASC
|
|
19
|
+
) AS operator_seq
|
|
20
|
+
FROM connector_event_index
|
|
21
|
+
)
|
|
22
|
+
UPDATE connector_event_index
|
|
23
|
+
SET operator_ingest_seq = (
|
|
24
|
+
SELECT operator_seq
|
|
25
|
+
FROM ranked_events
|
|
26
|
+
WHERE ranked_events.event_index_id = connector_event_index.event_index_id
|
|
27
|
+
)
|
|
28
|
+
WHERE operator_ingest_seq IS NULL;
|
|
29
|
+
|
|
30
|
+
INSERT OR IGNORE INTO connector_event_index_operator_seq_cursors (
|
|
31
|
+
source_connector,
|
|
32
|
+
channel,
|
|
33
|
+
next_seq
|
|
34
|
+
)
|
|
35
|
+
SELECT
|
|
36
|
+
source_connector,
|
|
37
|
+
COALESCE(channel, ''),
|
|
38
|
+
COALESCE(MAX(operator_ingest_seq), 0) + 1
|
|
39
|
+
FROM connector_event_index
|
|
40
|
+
GROUP BY source_connector, COALESCE(channel, '');
|
|
41
|
+
|
|
42
|
+
CREATE UNIQUE INDEX IF NOT EXISTS idx_connector_event_index_operator_scope_seq
|
|
43
|
+
ON connector_event_index(source_connector, COALESCE(channel, ''), operator_ingest_seq)
|
|
44
|
+
WHERE operator_ingest_seq IS NOT NULL;
|
|
45
|
+
|
|
46
|
+
CREATE INDEX IF NOT EXISTS idx_connector_event_index_operator_cursor_order
|
|
47
|
+
ON connector_event_index(source_connector, channel, operator_ingest_seq);
|
|
48
|
+
|
|
49
|
+
CREATE TRIGGER IF NOT EXISTS trg_connector_event_index_operator_ingest_seq_ai
|
|
50
|
+
AFTER INSERT ON connector_event_index
|
|
51
|
+
WHEN NEW.operator_ingest_seq IS NULL
|
|
52
|
+
BEGIN
|
|
53
|
+
INSERT OR IGNORE INTO connector_event_index_operator_seq_cursors (
|
|
54
|
+
source_connector,
|
|
55
|
+
channel,
|
|
56
|
+
next_seq
|
|
57
|
+
)
|
|
58
|
+
VALUES (NEW.source_connector, COALESCE(NEW.channel, ''), 1);
|
|
59
|
+
|
|
60
|
+
UPDATE connector_event_index
|
|
61
|
+
SET operator_ingest_seq = (
|
|
62
|
+
SELECT next_seq
|
|
63
|
+
FROM connector_event_index_operator_seq_cursors
|
|
64
|
+
WHERE source_connector = NEW.source_connector
|
|
65
|
+
AND channel = COALESCE(NEW.channel, '')
|
|
66
|
+
)
|
|
67
|
+
WHERE event_index_id = NEW.event_index_id;
|
|
68
|
+
|
|
69
|
+
UPDATE connector_event_index_operator_seq_cursors
|
|
70
|
+
SET next_seq = next_seq + 1
|
|
71
|
+
WHERE source_connector = NEW.source_connector
|
|
72
|
+
AND channel = COALESCE(NEW.channel, '');
|
|
73
|
+
END;
|
|
74
|
+
|
|
75
|
+
CREATE TRIGGER IF NOT EXISTS trg_connector_event_index_operator_ingest_seq_explicit_ai
|
|
76
|
+
AFTER INSERT ON connector_event_index
|
|
77
|
+
WHEN NEW.operator_ingest_seq IS NOT NULL
|
|
78
|
+
BEGIN
|
|
79
|
+
INSERT OR IGNORE INTO connector_event_index_operator_seq_cursors (
|
|
80
|
+
source_connector,
|
|
81
|
+
channel,
|
|
82
|
+
next_seq
|
|
83
|
+
)
|
|
84
|
+
VALUES (NEW.source_connector, COALESCE(NEW.channel, ''), 1);
|
|
85
|
+
|
|
86
|
+
UPDATE connector_event_index_operator_seq_cursors
|
|
87
|
+
SET next_seq = CASE
|
|
88
|
+
WHEN next_seq <= NEW.operator_ingest_seq THEN NEW.operator_ingest_seq + 1
|
|
89
|
+
ELSE next_seq
|
|
90
|
+
END
|
|
91
|
+
WHERE source_connector = NEW.source_connector
|
|
92
|
+
AND channel = COALESCE(NEW.channel, '');
|
|
93
|
+
END;
|
|
94
|
+
|
|
95
|
+
INSERT OR IGNORE INTO schema_version (version, description)
|
|
96
|
+
VALUES (39, 'Add connector event operator ingest sequence');
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
CREATE TABLE IF NOT EXISTS operator_memory_commit_intents (
|
|
2
|
+
intent_id TEXT PRIMARY KEY,
|
|
3
|
+
cursor_name TEXT NOT NULL,
|
|
4
|
+
idempotency_key TEXT NOT NULL UNIQUE,
|
|
5
|
+
expected_memory_count INTEGER NOT NULL CHECK (expected_memory_count > 0),
|
|
6
|
+
memory_payload_hash TEXT NOT NULL CHECK (memory_payload_hash LIKE 'sha256:%'),
|
|
7
|
+
memory_ids_json TEXT NOT NULL CHECK (json_valid(memory_ids_json)),
|
|
8
|
+
source_refs_json TEXT NOT NULL CHECK (json_valid(source_refs_json)),
|
|
9
|
+
status TEXT NOT NULL CHECK (status IN ('pending', 'saving', 'saved', 'promoted')),
|
|
10
|
+
claim_token TEXT,
|
|
11
|
+
created_at_ms INTEGER NOT NULL CHECK (created_at_ms >= 0),
|
|
12
|
+
updated_at_ms INTEGER NOT NULL CHECK (updated_at_ms >= created_at_ms)
|
|
13
|
+
);
|
|
14
|
+
|
|
15
|
+
CREATE INDEX IF NOT EXISTS idx_operator_memory_commit_intents_cursor_created
|
|
16
|
+
ON operator_memory_commit_intents(cursor_name, created_at_ms DESC);
|
|
17
|
+
|
|
18
|
+
INSERT OR IGNORE INTO schema_version (version, description)
|
|
19
|
+
VALUES (40, 'Create operator memory commit intents');
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
CREATE TABLE IF NOT EXISTS operator_memory_commit_intents (
|
|
2
|
+
intent_id TEXT PRIMARY KEY,
|
|
3
|
+
cursor_name TEXT NOT NULL,
|
|
4
|
+
idempotency_key TEXT NOT NULL UNIQUE,
|
|
5
|
+
expected_memory_count INTEGER NOT NULL CHECK (expected_memory_count > 0),
|
|
6
|
+
memory_payload_hash TEXT NOT NULL CHECK (memory_payload_hash LIKE 'sha256:%'),
|
|
7
|
+
memory_ids_json TEXT NOT NULL CHECK (json_valid(memory_ids_json)),
|
|
8
|
+
source_refs_json TEXT NOT NULL CHECK (json_valid(source_refs_json)),
|
|
9
|
+
status TEXT NOT NULL CHECK (status IN ('pending', 'saving', 'saved', 'promoted')),
|
|
10
|
+
claim_token TEXT,
|
|
11
|
+
created_at_ms INTEGER NOT NULL CHECK (created_at_ms >= 0),
|
|
12
|
+
updated_at_ms INTEGER NOT NULL CHECK (updated_at_ms >= created_at_ms)
|
|
13
|
+
);
|
|
14
|
+
|
|
15
|
+
DROP TABLE IF EXISTS operator_memory_commit_intents_v041;
|
|
16
|
+
|
|
17
|
+
CREATE TABLE operator_memory_commit_intents_v041 (
|
|
18
|
+
intent_id TEXT PRIMARY KEY,
|
|
19
|
+
cursor_name TEXT NOT NULL,
|
|
20
|
+
idempotency_key TEXT NOT NULL UNIQUE,
|
|
21
|
+
expected_memory_count INTEGER NOT NULL CHECK (expected_memory_count > 0),
|
|
22
|
+
memory_payload_hash TEXT NOT NULL CHECK (memory_payload_hash LIKE 'sha256:%'),
|
|
23
|
+
memory_ids_json TEXT NOT NULL CHECK (json_valid(memory_ids_json)),
|
|
24
|
+
source_refs_json TEXT NOT NULL CHECK (json_valid(source_refs_json)),
|
|
25
|
+
status TEXT NOT NULL CHECK (status IN ('pending', 'saving', 'saved', 'promoted')),
|
|
26
|
+
claim_token TEXT CHECK (
|
|
27
|
+
(status = 'saving' AND claim_token IS NOT NULL) OR
|
|
28
|
+
(status != 'saving' AND claim_token IS NULL)
|
|
29
|
+
),
|
|
30
|
+
created_at_ms INTEGER NOT NULL CHECK (created_at_ms >= 0),
|
|
31
|
+
updated_at_ms INTEGER NOT NULL CHECK (updated_at_ms >= created_at_ms)
|
|
32
|
+
);
|
|
33
|
+
|
|
34
|
+
INSERT INTO operator_memory_commit_intents_v041 (
|
|
35
|
+
intent_id,
|
|
36
|
+
cursor_name,
|
|
37
|
+
idempotency_key,
|
|
38
|
+
expected_memory_count,
|
|
39
|
+
memory_payload_hash,
|
|
40
|
+
memory_ids_json,
|
|
41
|
+
source_refs_json,
|
|
42
|
+
status,
|
|
43
|
+
claim_token,
|
|
44
|
+
created_at_ms,
|
|
45
|
+
updated_at_ms
|
|
46
|
+
)
|
|
47
|
+
SELECT
|
|
48
|
+
intent_id,
|
|
49
|
+
cursor_name,
|
|
50
|
+
idempotency_key,
|
|
51
|
+
expected_memory_count,
|
|
52
|
+
memory_payload_hash,
|
|
53
|
+
memory_ids_json,
|
|
54
|
+
source_refs_json,
|
|
55
|
+
CASE
|
|
56
|
+
WHEN status = 'saving' AND claim_token IS NULL THEN 'pending'
|
|
57
|
+
ELSE status
|
|
58
|
+
END,
|
|
59
|
+
CASE
|
|
60
|
+
WHEN status = 'saving' AND claim_token IS NOT NULL THEN claim_token
|
|
61
|
+
ELSE NULL
|
|
62
|
+
END,
|
|
63
|
+
created_at_ms,
|
|
64
|
+
updated_at_ms
|
|
65
|
+
FROM operator_memory_commit_intents;
|
|
66
|
+
|
|
67
|
+
DROP TABLE operator_memory_commit_intents;
|
|
68
|
+
ALTER TABLE operator_memory_commit_intents_v041 RENAME TO operator_memory_commit_intents;
|
|
69
|
+
|
|
70
|
+
CREATE INDEX IF NOT EXISTS idx_operator_memory_commit_intents_cursor_created
|
|
71
|
+
ON operator_memory_commit_intents(cursor_name, created_at_ms DESC);
|
|
72
|
+
|
|
73
|
+
INSERT OR IGNORE INTO schema_version (version, description)
|
|
74
|
+
VALUES (41, 'Enforce operator memory commit claim invariant');
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
-- Migration 042: track the embedding instruction-prefix scheme (e5 query/passage).
|
|
2
|
+
-- Fresh DB (no stored vectors) starts current; an upgraded DB with pre-existing
|
|
3
|
+
-- unprefixed vectors (in EITHER vector store) is marked legacy so the runtime guard
|
|
4
|
+
-- forces a re-embed.
|
|
5
|
+
--
|
|
6
|
+
-- Robustness: a SQLite statement that references a missing table fails at prepare
|
|
7
|
+
-- time, so a sqlite_master-gated WHERE clause alone cannot protect a reference to an
|
|
8
|
+
-- absent table. Some legacy-recovery paths reach 042 with a partial schema (e.g. a DB
|
|
9
|
+
-- whose schema_version was fast-forwarded past migration 030, so wiki_page_embeddings
|
|
10
|
+
-- does not exist yet). We therefore CREATE IF NOT EXISTS both vector-store tables with
|
|
11
|
+
-- the exact shapes of migrations 013/030 (no-op on any normally-migrated DB, empty
|
|
12
|
+
-- shells on partial ones) before the marker statement references them. The runtime
|
|
13
|
+
-- guard in db-manager (assertEmbeddingSchemeCurrent) independently counts both tables
|
|
14
|
+
-- (try/catch-wrapped), so the two layers agree.
|
|
15
|
+
|
|
16
|
+
CREATE TABLE IF NOT EXISTS embedding_meta (
|
|
17
|
+
key TEXT PRIMARY KEY,
|
|
18
|
+
value TEXT NOT NULL,
|
|
19
|
+
updated_at INTEGER NOT NULL DEFAULT (unixepoch() * 1000)
|
|
20
|
+
);
|
|
21
|
+
|
|
22
|
+
-- Shape of migration 013 (embeddings) - no-op unless the schema is partial.
|
|
23
|
+
CREATE TABLE IF NOT EXISTS embeddings (
|
|
24
|
+
rowid INTEGER PRIMARY KEY,
|
|
25
|
+
embedding BLOB NOT NULL
|
|
26
|
+
);
|
|
27
|
+
|
|
28
|
+
-- Shape of migration 030 (wiki_page_embeddings) - no-op unless the schema is partial.
|
|
29
|
+
-- The FK to wiki_page_index resolves lazily in SQLite, so this is safe even when
|
|
30
|
+
-- wiki_page_index itself is absent.
|
|
31
|
+
CREATE TABLE IF NOT EXISTS wiki_page_embeddings (
|
|
32
|
+
page_id TEXT PRIMARY KEY REFERENCES wiki_page_index(page_id) ON DELETE CASCADE,
|
|
33
|
+
embedding BLOB NOT NULL
|
|
34
|
+
);
|
|
35
|
+
|
|
36
|
+
INSERT OR IGNORE INTO embedding_meta (key, value)
|
|
37
|
+
SELECT 'embedding_prefix_scheme',
|
|
38
|
+
CASE
|
|
39
|
+
WHEN EXISTS (SELECT 1 FROM embeddings)
|
|
40
|
+
OR EXISTS (SELECT 1 FROM wiki_page_embeddings)
|
|
41
|
+
THEN 'legacy-unprefixed'
|
|
42
|
+
ELSE 'e5-prefixed-v1'
|
|
43
|
+
END;
|
|
44
|
+
|
|
45
|
+
INSERT OR IGNORE INTO schema_version (version, description)
|
|
46
|
+
VALUES (42, 'Track embedding prefix scheme (e5 query/passage)');
|
|
@@ -153,7 +153,7 @@ function buildFtsQuery(query) {
|
|
|
153
153
|
.match(/[\p{L}\p{N}_]+/gu)
|
|
154
154
|
?.filter((token) => token.length > 0);
|
|
155
155
|
if (!tokens || tokens.length === 0) {
|
|
156
|
-
return
|
|
156
|
+
return '';
|
|
157
157
|
}
|
|
158
158
|
return tokens.map((token) => `"${token.replace(/"/g, '""')}"`).join(' OR ');
|
|
159
159
|
}
|
|
@@ -264,6 +264,10 @@ function ftsSearchWikiPages(adapter, query, limit) {
|
|
|
264
264
|
if (!searchTableExists(adapter, 'wiki_pages_fts')) {
|
|
265
265
|
return [];
|
|
266
266
|
}
|
|
267
|
+
const ftsQuery = buildFtsQuery(query);
|
|
268
|
+
if (!ftsQuery) {
|
|
269
|
+
return [];
|
|
270
|
+
}
|
|
267
271
|
const schema = readSchema(adapter);
|
|
268
272
|
const rows = adapter
|
|
269
273
|
.prepare(`
|
|
@@ -273,7 +277,7 @@ function ftsSearchWikiPages(adapter, query, limit) {
|
|
|
273
277
|
ORDER BY raw
|
|
274
278
|
LIMIT ?
|
|
275
279
|
`)
|
|
276
|
-
.all(
|
|
280
|
+
.all(ftsQuery, boundedLimit);
|
|
277
281
|
const hits = [];
|
|
278
282
|
for (const row of rows) {
|
|
279
283
|
const record = selectByFtsPageId(adapter, schema, row.page_id);
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
import type { ContextBoundary, ContextProjectRef, ContextRange } from './types.js';
|
|
2
|
+
import type { MemoryScopeRef } from '../memory/types.js';
|
|
3
|
+
export type ContextBoundaryReadableInput = {
|
|
4
|
+
scopes?: MemoryScopeRef[];
|
|
5
|
+
connectors?: string[];
|
|
6
|
+
project_refs?: ContextProjectRef[];
|
|
7
|
+
tenant_id?: string | null;
|
|
8
|
+
range?: ContextRange;
|
|
9
|
+
as_of?: string | number | null;
|
|
10
|
+
};
|
|
11
|
+
export declare function applyContextBoundaryReadDefaults<T extends ContextBoundaryReadableInput>(input: T, boundary: ContextBoundary | undefined): T;
|
|
12
|
+
//# sourceMappingURL=boundary-defaults.d.ts.map
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.applyContextBoundaryReadDefaults = applyContextBoundaryReadDefaults;
|
|
4
|
+
function parseBoundaryTimeMs(value, field) {
|
|
5
|
+
if (value === null || value === undefined) {
|
|
6
|
+
return null;
|
|
7
|
+
}
|
|
8
|
+
if (typeof value === 'number' && Number.isFinite(value)) {
|
|
9
|
+
return Math.floor(value);
|
|
10
|
+
}
|
|
11
|
+
if (typeof value === 'string') {
|
|
12
|
+
const trimmed = value.trim();
|
|
13
|
+
if (trimmed.length === 0) {
|
|
14
|
+
throw new Error(`Invalid context boundary ${field}: ${String(value)}`);
|
|
15
|
+
}
|
|
16
|
+
const numeric = Number(trimmed);
|
|
17
|
+
if (Number.isFinite(numeric)) {
|
|
18
|
+
return Math.floor(numeric);
|
|
19
|
+
}
|
|
20
|
+
const parsed = Date.parse(trimmed);
|
|
21
|
+
if (Number.isFinite(parsed)) {
|
|
22
|
+
return parsed;
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
throw new Error(`Invalid context boundary ${field}: ${String(value)}`);
|
|
26
|
+
}
|
|
27
|
+
function rangeBoundaryMs(value, field) {
|
|
28
|
+
if (value === undefined) {
|
|
29
|
+
return null;
|
|
30
|
+
}
|
|
31
|
+
if (typeof value === 'number' && Number.isFinite(value)) {
|
|
32
|
+
return Math.floor(value);
|
|
33
|
+
}
|
|
34
|
+
throw new Error(`Invalid context boundary ${field}: ${String(value)}`);
|
|
35
|
+
}
|
|
36
|
+
function intersectRange(range, boundaryRange) {
|
|
37
|
+
if (!boundaryRange) {
|
|
38
|
+
return range;
|
|
39
|
+
}
|
|
40
|
+
const requestedStartMs = rangeBoundaryMs(range?.start_ms, 'range.start_ms');
|
|
41
|
+
const requestedEndMs = rangeBoundaryMs(range?.end_ms, 'range.end_ms');
|
|
42
|
+
const boundaryStartMs = rangeBoundaryMs(boundaryRange.start_ms, 'boundary.range.start_ms');
|
|
43
|
+
const boundaryEndMs = rangeBoundaryMs(boundaryRange.end_ms, 'boundary.range.end_ms');
|
|
44
|
+
const startMs = requestedStartMs === null
|
|
45
|
+
? boundaryStartMs
|
|
46
|
+
: boundaryStartMs === null
|
|
47
|
+
? requestedStartMs
|
|
48
|
+
: Math.max(requestedStartMs, boundaryStartMs);
|
|
49
|
+
const endMs = requestedEndMs === null
|
|
50
|
+
? boundaryEndMs
|
|
51
|
+
: boundaryEndMs === null
|
|
52
|
+
? requestedEndMs
|
|
53
|
+
: Math.min(requestedEndMs, boundaryEndMs);
|
|
54
|
+
if (startMs === null && endMs === null) {
|
|
55
|
+
return undefined;
|
|
56
|
+
}
|
|
57
|
+
if (startMs !== null && endMs !== null && startMs > endMs) {
|
|
58
|
+
throw new RangeError(`Context boundary range is empty after intersection: start_ms ${startMs} > end_ms ${endMs}`);
|
|
59
|
+
}
|
|
60
|
+
return {
|
|
61
|
+
...(startMs !== null ? { start_ms: startMs } : {}),
|
|
62
|
+
...(endMs !== null ? { end_ms: endMs } : {}),
|
|
63
|
+
};
|
|
64
|
+
}
|
|
65
|
+
function clampAsOf(requested, boundaryAsOf) {
|
|
66
|
+
const requestedMs = parseBoundaryTimeMs(requested, 'as_of');
|
|
67
|
+
const boundaryMs = parseBoundaryTimeMs(boundaryAsOf, 'boundary.as_of');
|
|
68
|
+
if (requestedMs === null) {
|
|
69
|
+
return boundaryAsOf ?? null;
|
|
70
|
+
}
|
|
71
|
+
if (boundaryMs === null) {
|
|
72
|
+
return requested ?? null;
|
|
73
|
+
}
|
|
74
|
+
return requestedMs <= boundaryMs ? (requested ?? null) : (boundaryAsOf ?? null);
|
|
75
|
+
}
|
|
76
|
+
function applyContextBoundaryReadDefaults(input, boundary) {
|
|
77
|
+
if (!boundary) {
|
|
78
|
+
return input;
|
|
79
|
+
}
|
|
80
|
+
const requestedTenantId = input.tenant_id;
|
|
81
|
+
const boundaryTenantId = boundary.tenant_id;
|
|
82
|
+
if (requestedTenantId !== undefined &&
|
|
83
|
+
requestedTenantId !== null &&
|
|
84
|
+
boundaryTenantId !== undefined &&
|
|
85
|
+
boundaryTenantId !== null &&
|
|
86
|
+
requestedTenantId !== boundaryTenantId) {
|
|
87
|
+
throw new Error('Requested tenant is outside the context boundary');
|
|
88
|
+
}
|
|
89
|
+
const scopedInput = {
|
|
90
|
+
...input,
|
|
91
|
+
scopes: input.scopes === undefined ? boundary.scopes : input.scopes,
|
|
92
|
+
connectors: input.connectors === undefined ? boundary.connectors : input.connectors,
|
|
93
|
+
project_refs: input.project_refs === undefined ? boundary.project_refs : input.project_refs,
|
|
94
|
+
range: intersectRange(input.range, boundary.range),
|
|
95
|
+
as_of: clampAsOf(input.as_of, boundary.as_of),
|
|
96
|
+
};
|
|
97
|
+
if (boundaryTenantId !== undefined || requestedTenantId !== undefined) {
|
|
98
|
+
return {
|
|
99
|
+
...scopedInput,
|
|
100
|
+
tenant_id: boundaryTenantId !== undefined ? boundaryTenantId : requestedTenantId,
|
|
101
|
+
};
|
|
102
|
+
}
|
|
103
|
+
return scopedInput;
|
|
104
|
+
}
|
|
105
|
+
//# sourceMappingURL=boundary-defaults.js.map
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
import type { ContextEvidence, ContextRef } from './types.js';
|
|
2
|
+
import type { ContextCandidate, HiddenCandidateAggregate } from './source-readers.js';
|
|
3
|
+
export interface ContextCompilerPolicyInput {
|
|
4
|
+
task: string;
|
|
5
|
+
candidates: ContextCandidate[];
|
|
6
|
+
hidden: HiddenCandidateAggregate;
|
|
7
|
+
limit?: number;
|
|
8
|
+
strictness?: 'recall' | 'balanced' | 'strict' | 'low' | 'medium' | 'high';
|
|
9
|
+
max_tokens?: number;
|
|
10
|
+
}
|
|
11
|
+
export interface ContextCompilerPolicyResult {
|
|
12
|
+
selected_evidence: ContextEvidence[];
|
|
13
|
+
source_refs: ContextRef[];
|
|
14
|
+
evidence_clusters: unknown[];
|
|
15
|
+
related_decisions: Array<{
|
|
16
|
+
memory_id: string;
|
|
17
|
+
title: string;
|
|
18
|
+
}>;
|
|
19
|
+
rejected_refs: ContextRef[];
|
|
20
|
+
rejected_summary: string[];
|
|
21
|
+
missing_context: string[];
|
|
22
|
+
caveats: string[];
|
|
23
|
+
retrieval_diagnostics: {
|
|
24
|
+
candidate_count: number;
|
|
25
|
+
selected_count: number;
|
|
26
|
+
rejected_count: number;
|
|
27
|
+
hidden: HiddenCandidateAggregate;
|
|
28
|
+
deduplicated_count: number;
|
|
29
|
+
strict_vector_only_rejected_count: number;
|
|
30
|
+
token_budget_rejected_count: number;
|
|
31
|
+
limit_rejected_count: number;
|
|
32
|
+
truncated_by_tokens: boolean;
|
|
33
|
+
estimated_tokens: number;
|
|
34
|
+
};
|
|
35
|
+
estimated_tokens: number;
|
|
36
|
+
}
|
|
37
|
+
export declare function estimateEvidenceTokens(evidence: Pick<ContextEvidence, 'title' | 'excerpt'>): number;
|
|
38
|
+
export declare function applyContextCompilerPolicy(input: ContextCompilerPolicyInput): ContextCompilerPolicyResult;
|
|
39
|
+
//# sourceMappingURL=compiler-policy.d.ts.map
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.estimateEvidenceTokens = estimateEvidenceTokens;
|
|
4
|
+
exports.applyContextCompilerPolicy = applyContextCompilerPolicy;
|
|
5
|
+
const ref_js_1 = require("./ref.js");
|
|
6
|
+
function normalizeLimit(limit) {
|
|
7
|
+
return Math.max(0, Math.min(100, Math.floor(limit ?? 10)));
|
|
8
|
+
}
|
|
9
|
+
function estimateEvidenceTokens(evidence) {
|
|
10
|
+
const text = `${evidence.title ?? ''}\n${evidence.excerpt ?? ''}`;
|
|
11
|
+
return Math.max(1, Math.ceil(text.length / 4));
|
|
12
|
+
}
|
|
13
|
+
function sortedCandidates(candidates) {
|
|
14
|
+
return [...candidates].sort((left, right) => {
|
|
15
|
+
const scoreDiff = right.score - left.score;
|
|
16
|
+
if (scoreDiff !== 0) {
|
|
17
|
+
return scoreDiff;
|
|
18
|
+
}
|
|
19
|
+
return (right.timestamp_ms ?? 0) - (left.timestamp_ms ?? 0);
|
|
20
|
+
});
|
|
21
|
+
}
|
|
22
|
+
function shouldRejectVectorOnly(candidate, strictness) {
|
|
23
|
+
return (strictness === 'strict' &&
|
|
24
|
+
candidate.support.is_vector_only === true &&
|
|
25
|
+
candidate.support.confirmation_signals.length === 0);
|
|
26
|
+
}
|
|
27
|
+
function evidenceFromCandidate(candidate) {
|
|
28
|
+
return {
|
|
29
|
+
ref: candidate.ref,
|
|
30
|
+
title: candidate.title,
|
|
31
|
+
excerpt: candidate.excerpt,
|
|
32
|
+
score: candidate.score,
|
|
33
|
+
reasons: [
|
|
34
|
+
...(candidate.support.confirmation_signals ?? []),
|
|
35
|
+
...(candidate.support.graph_expanded ? ['graph_expanded'] : []),
|
|
36
|
+
],
|
|
37
|
+
...(candidate.retrieval_diagnostics
|
|
38
|
+
? { retrieval_diagnostics: candidate.retrieval_diagnostics }
|
|
39
|
+
: {}),
|
|
40
|
+
};
|
|
41
|
+
}
|
|
42
|
+
function addCountSummary(parts, label, count) {
|
|
43
|
+
if (count > 0) {
|
|
44
|
+
parts.push(`${label}: ${count}`);
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
function normalizeStrictness(strictness) {
|
|
48
|
+
switch (strictness) {
|
|
49
|
+
case 'low':
|
|
50
|
+
case 'recall':
|
|
51
|
+
return 'recall';
|
|
52
|
+
case 'high':
|
|
53
|
+
case 'strict':
|
|
54
|
+
return 'strict';
|
|
55
|
+
case 'medium':
|
|
56
|
+
case 'balanced':
|
|
57
|
+
case undefined:
|
|
58
|
+
return 'balanced';
|
|
59
|
+
}
|
|
60
|
+
return 'balanced';
|
|
61
|
+
}
|
|
62
|
+
function applyContextCompilerPolicy(input) {
|
|
63
|
+
const limit = normalizeLimit(input.limit);
|
|
64
|
+
const strictness = normalizeStrictness(input.strictness);
|
|
65
|
+
const maxTokens = typeof input.max_tokens === 'number' && Number.isFinite(input.max_tokens)
|
|
66
|
+
? Math.max(0, Math.floor(input.max_tokens))
|
|
67
|
+
: null;
|
|
68
|
+
const seen = new Set();
|
|
69
|
+
const selected = [];
|
|
70
|
+
const sourceRefs = [];
|
|
71
|
+
const rejectedRefs = [];
|
|
72
|
+
const counters = {
|
|
73
|
+
deduplicated: 0,
|
|
74
|
+
strictVectorOnly: 0,
|
|
75
|
+
tokenBudget: 0,
|
|
76
|
+
limit: 0,
|
|
77
|
+
};
|
|
78
|
+
let estimatedTokens = 0;
|
|
79
|
+
for (const candidate of sortedCandidates(input.candidates)) {
|
|
80
|
+
if (!candidate.visible) {
|
|
81
|
+
continue;
|
|
82
|
+
}
|
|
83
|
+
const key = (0, ref_js_1.serializeContextRefForProvenance)(candidate.ref);
|
|
84
|
+
if (seen.has(key)) {
|
|
85
|
+
counters.deduplicated += 1;
|
|
86
|
+
rejectedRefs.push(candidate.ref);
|
|
87
|
+
continue;
|
|
88
|
+
}
|
|
89
|
+
if (shouldRejectVectorOnly(candidate, strictness)) {
|
|
90
|
+
counters.strictVectorOnly += 1;
|
|
91
|
+
rejectedRefs.push(candidate.ref);
|
|
92
|
+
continue;
|
|
93
|
+
}
|
|
94
|
+
if (selected.length >= limit) {
|
|
95
|
+
counters.limit += 1;
|
|
96
|
+
rejectedRefs.push(candidate.ref);
|
|
97
|
+
continue;
|
|
98
|
+
}
|
|
99
|
+
const evidence = evidenceFromCandidate(candidate);
|
|
100
|
+
const nextTokens = estimateEvidenceTokens(evidence);
|
|
101
|
+
if (maxTokens !== null && estimatedTokens + nextTokens > maxTokens) {
|
|
102
|
+
counters.tokenBudget += 1;
|
|
103
|
+
rejectedRefs.push(candidate.ref);
|
|
104
|
+
continue;
|
|
105
|
+
}
|
|
106
|
+
selected.push(evidence);
|
|
107
|
+
sourceRefs.push(candidate.ref);
|
|
108
|
+
seen.add(key);
|
|
109
|
+
estimatedTokens += nextTokens;
|
|
110
|
+
}
|
|
111
|
+
const rejectedSummary = [];
|
|
112
|
+
addCountSummary(rejectedSummary, 'deduplicated duplicate candidates', counters.deduplicated);
|
|
113
|
+
addCountSummary(rejectedSummary, 'strictness rejected vector-only candidates', counters.strictVectorOnly);
|
|
114
|
+
addCountSummary(rejectedSummary, 'token budget rejected candidates', counters.tokenBudget);
|
|
115
|
+
addCountSummary(rejectedSummary, 'limit rejected candidates', counters.limit);
|
|
116
|
+
const caveats = [];
|
|
117
|
+
if (input.hidden.total > 0) {
|
|
118
|
+
caveats.push(`hidden candidates omitted: ${input.hidden.total}`);
|
|
119
|
+
}
|
|
120
|
+
const relatedDecisions = selected.flatMap((evidence) => evidence.ref.kind === 'memory'
|
|
121
|
+
? [{ memory_id: evidence.ref.id, title: evidence.title ?? evidence.ref.id }]
|
|
122
|
+
: []);
|
|
123
|
+
return {
|
|
124
|
+
selected_evidence: selected,
|
|
125
|
+
source_refs: sourceRefs,
|
|
126
|
+
evidence_clusters: [],
|
|
127
|
+
related_decisions: relatedDecisions,
|
|
128
|
+
rejected_refs: rejectedRefs,
|
|
129
|
+
rejected_summary: rejectedSummary,
|
|
130
|
+
missing_context: selected.length === 0 ? [`No visible evidence selected for task: ${input.task}`] : [],
|
|
131
|
+
caveats,
|
|
132
|
+
retrieval_diagnostics: {
|
|
133
|
+
candidate_count: input.candidates.length,
|
|
134
|
+
selected_count: selected.length,
|
|
135
|
+
rejected_count: rejectedRefs.length,
|
|
136
|
+
hidden: input.hidden,
|
|
137
|
+
deduplicated_count: counters.deduplicated,
|
|
138
|
+
strict_vector_only_rejected_count: counters.strictVectorOnly,
|
|
139
|
+
token_budget_rejected_count: counters.tokenBudget,
|
|
140
|
+
limit_rejected_count: counters.limit,
|
|
141
|
+
truncated_by_tokens: counters.tokenBudget > 0,
|
|
142
|
+
estimated_tokens: estimatedTokens,
|
|
143
|
+
},
|
|
144
|
+
estimated_tokens: estimatedTokens,
|
|
145
|
+
};
|
|
146
|
+
}
|
|
147
|
+
//# sourceMappingURL=compiler-policy.js.map
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
import type { DatabaseAdapter } from '../db-manager.js';
|
|
2
|
+
import type { ContextBoundary, ContextCompileInput, ContextPacket, ContextRef } from './types.js';
|
|
3
|
+
import { type ContextSourceReadInput, type ContextSourceReadResult } from './source-readers.js';
|
|
4
|
+
type ContextCompilerAdapter = Pick<DatabaseAdapter, 'prepare'>;
|
|
5
|
+
export interface ContextCompilerDeps {
|
|
6
|
+
adapter?: ContextCompilerAdapter;
|
|
7
|
+
boundary?: ContextBoundary;
|
|
8
|
+
signal?: AbortSignal;
|
|
9
|
+
deadlineMs?: number;
|
|
10
|
+
now?: () => number;
|
|
11
|
+
packetId?: () => string;
|
|
12
|
+
readMemoryCandidates?: (input: ContextSourceReadInput) => Promise<ContextSourceReadResult> | ContextSourceReadResult;
|
|
13
|
+
readRawCandidates?: (input: ContextSourceReadInput) => Promise<ContextSourceReadResult> | ContextSourceReadResult;
|
|
14
|
+
readGraphCandidates?: (input: ContextSourceReadInput, visibleRefs: readonly ContextRef[]) => Promise<ContextSourceReadResult> | ContextSourceReadResult;
|
|
15
|
+
}
|
|
16
|
+
export declare function compileContext(input: ContextCompileInput, deps?: ContextCompilerDeps): Promise<ContextPacket>;
|
|
17
|
+
export {};
|
|
18
|
+
//# sourceMappingURL=compiler.d.ts.map
|