zikaron 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- zikaron/__init__.py +1 -0
- zikaron/cli/__init__.py +1 -0
- zikaron/cli/main.py +114 -0
- zikaron/core/__init__.py +1 -0
- zikaron/core/clock.py +78 -0
- zikaron/core/config/__init__.py +1 -0
- zikaron/core/config/keys.py +395 -0
- zikaron/core/config/resolution.py +267 -0
- zikaron/core/consolidation/__init__.py +1 -0
- zikaron/core/consolidation/authorization.py +316 -0
- zikaron/core/consolidation/candidates.py +147 -0
- zikaron/core/consolidation/context.py +166 -0
- zikaron/core/consolidation/grouping.py +383 -0
- zikaron/core/consolidation/groups.py +490 -0
- zikaron/core/consolidation/payload.py +246 -0
- zikaron/core/consolidation/planning.py +192 -0
- zikaron/core/consolidation/rowstate.py +68 -0
- zikaron/core/consolidation/runs.py +306 -0
- zikaron/core/consolidation/serving.py +462 -0
- zikaron/core/consolidation/verbs.py +500 -0
- zikaron/core/errors.py +355 -0
- zikaron/core/events.py +748 -0
- zikaron/core/indexing/__init__.py +1 -0
- zikaron/core/indexing/acquisition.py +255 -0
- zikaron/core/indexing/chunking.py +368 -0
- zikaron/core/indexing/encoder.py +537 -0
- zikaron/core/indexing/lexical.py +86 -0
- zikaron/core/indexing/model_cache.py +93 -0
- zikaron/core/indexing/model_pin.py +89 -0
- zikaron/core/indexing/vectors.py +223 -0
- zikaron/core/indexing/writes.py +461 -0
- zikaron/core/knowledge/__init__.py +5 -0
- zikaron/core/knowledge/arms.py +104 -0
- zikaron/core/knowledge/builds.py +204 -0
- zikaron/core/knowledge/candidates.py +130 -0
- zikaron/core/knowledge/changes.py +175 -0
- zikaron/core/knowledge/chunking.py +376 -0
- zikaron/core/knowledge/counters.py +228 -0
- zikaron/core/knowledge/database.py +380 -0
- zikaron/core/knowledge/ddl.py +196 -0
- zikaron/core/knowledge/disposal.py +213 -0
- zikaron/core/knowledge/errors.py +166 -0
- zikaron/core/knowledge/files.py +202 -0
- zikaron/core/knowledge/git.py +385 -0
- zikaron/core/knowledge/groups.py +450 -0
- zikaron/core/knowledge/lexical.py +64 -0
- zikaron/core/knowledge/lifecycle.py +418 -0
- zikaron/core/knowledge/lock.py +277 -0
- zikaron/core/knowledge/meta.py +393 -0
- zikaron/core/knowledge/paths.py +55 -0
- zikaron/core/knowledge/pending.py +59 -0
- zikaron/core/knowledge/registry.py +264 -0
- zikaron/core/knowledge/repair.py +152 -0
- zikaron/core/knowledge/reporting.py +436 -0
- zikaron/core/knowledge/roots.py +91 -0
- zikaron/core/knowledge/scan.py +429 -0
- zikaron/core/knowledge/search.py +346 -0
- zikaron/core/knowledge/state.py +174 -0
- zikaron/core/knowledge/text.py +166 -0
- zikaron/core/knowledge/vectors.py +102 -0
- zikaron/core/knowledge/walk.py +264 -0
- zikaron/core/knowledge/writes.py +127 -0
- zikaron/core/records/__init__.py +1 -0
- zikaron/core/records/memory.py +961 -0
- zikaron/core/records/receipts.py +161 -0
- zikaron/core/records/supersession.py +221 -0
- zikaron/core/retrieval/__init__.py +1 -0
- zikaron/core/retrieval/arms.py +318 -0
- zikaron/core/retrieval/block.py +107 -0
- zikaron/core/retrieval/eligibility.py +164 -0
- zikaron/core/retrieval/query.py +327 -0
- zikaron/core/retrieval/ranking.py +260 -0
- zikaron/core/retrieval/reads.py +294 -0
- zikaron/core/retrieval/retrieve.py +158 -0
- zikaron/core/retrieval/similarity.py +87 -0
- zikaron/core/signals/__init__.py +34 -0
- zikaron/core/signals/contention.py +106 -0
- zikaron/core/signals/dedup.py +201 -0
- zikaron/core/signals/horizon.py +47 -0
- zikaron/core/signals/repair.py +161 -0
- zikaron/core/signals/retirement.py +83 -0
- zikaron/core/signals/sessions.py +105 -0
- zikaron/core/signals/writes.py +200 -0
- zikaron/core/store/__init__.py +1 -0
- zikaron/core/store/connection.py +202 -0
- zikaron/core/store/ddl.py +215 -0
- zikaron/core/store/embedder.py +45 -0
- zikaron/core/store/meta.py +152 -0
- zikaron/core/store/permissions.py +160 -0
- zikaron/core/store/store.py +408 -0
- zikaron/core/store/transactions.py +181 -0
- zikaron/core/write/__init__.py +33 -0
- zikaron/core/write/dedup.py +145 -0
- zikaron/core/write/tools.py +290 -0
- zikaron/doctor/__init__.py +1 -0
- zikaron/doctor/checks.py +220 -0
- zikaron/doctor/main.py +64 -0
- zikaron/harness/__init__.py +1 -0
- zikaron/harness/detect.py +92 -0
- zikaron/harness/spec.py +320 -0
- zikaron/hook/__init__.py +1 -0
- zikaron/hook/connect.py +379 -0
- zikaron/hook/envelope.py +106 -0
- zikaron/hook/failure.py +104 -0
- zikaron/hook/limits.py +61 -0
- zikaron/hook/main.py +118 -0
- zikaron/hook/push.py +183 -0
- zikaron/hook/rpc.py +85 -0
- zikaron/hook/spawn_warm.py +81 -0
- zikaron/hook/subagent_policy.py +57 -0
- zikaron/hook/tripwire.py +54 -0
- zikaron/hook/warm_helper.py +137 -0
- zikaron/hook/write_policy.py +319 -0
- zikaron/install/__init__.py +4 -0
- zikaron/install/__main__.py +18 -0
- zikaron/install/assets.py +394 -0
- zikaron/install/entries.py +370 -0
- zikaron/install/harness.py +185 -0
- zikaron/install/main.py +375 -0
- zikaron/install/targets.py +789 -0
- zikaron/install/writer.py +973 -0
- zikaron/knowledge/__init__.py +1 -0
- zikaron/knowledge/__main__.py +17 -0
- zikaron/knowledge/indexer/__init__.py +1 -0
- zikaron/knowledge/indexer/__main__.py +17 -0
- zikaron/knowledge/indexer/detach.py +83 -0
- zikaron/knowledge/indexer/main.py +187 -0
- zikaron/knowledge/main.py +466 -0
- zikaron/knowledge/scope.py +133 -0
- zikaron/mcp/__init__.py +6 -0
- zikaron/mcp/connection.py +583 -0
- zikaron/mcp/consolidator.py +316 -0
- zikaron/mcp/errors.py +73 -0
- zikaron/mcp/main.py +66 -0
- zikaron/mcp/primary.py +420 -0
- zikaron/mcp/server.py +96 -0
- zikaron/mcp/spill.py +328 -0
- zikaron/mcp/tool_names.py +67 -0
- zikaron/py.typed +0 -0
- zikaron/service/__init__.py +1 -0
- zikaron/service/asyncio_compat.py +126 -0
- zikaron/service/context.py +251 -0
- zikaron/service/dispatch.py +332 -0
- zikaron/service/dispatch_consolidation.py +397 -0
- zikaron/service/dispatch_knowledge.py +469 -0
- zikaron/service/envelope.py +166 -0
- zikaron/service/lifecycle.py +467 -0
- zikaron/service/log.py +96 -0
- zikaron/service/main.py +531 -0
- zikaron/service/params.py +168 -0
- zikaron/service/paths.py +181 -0
- zikaron/service/rpc.py +176 -0
- zikaron/service/security.py +156 -0
- zikaron/service/serialize.py +204 -0
- zikaron/service/serialize_knowledge.py +238 -0
- zikaron/service/server.py +416 -0
- zikaron-0.1.0.dist-info/METADATA +770 -0
- zikaron-0.1.0.dist-info/RECORD +162 -0
- zikaron-0.1.0.dist-info/WHEEL +5 -0
- zikaron-0.1.0.dist-info/entry_points.txt +4 -0
- zikaron-0.1.0.dist-info/licenses/LICENSE +21 -0
- zikaron-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,327 @@
|
|
|
1
|
+
"""Query construction: a prompt is not an FTS5 expression, and it is not always under 512 tokens.
|
|
2
|
+
|
|
3
|
+
`retrieval.md` §"Query construction" is normative. Two failures live here, one per arm, and both
|
|
4
|
+
were underspecified until that section was written:
|
|
5
|
+
|
|
6
|
+
- **Lexical.** A raw prompt is not a safe `MATCH` expression — a quote, a parenthesis or a bare
|
|
7
|
+
`OR` changes the query's meaning or raises `fts5: syntax error`, and a syntax error on the push
|
|
8
|
+
path would send every user message to the degraded path. The rule splits on boundaries only and
|
|
9
|
+
hands each fragment to FTS5 **as a quoted string literal**, so FTS5's own tokenizer does the case
|
|
10
|
+
folding and diacritic stripping. Reimplementing `unicode61` was the alternative and it is a
|
|
11
|
+
silent-divergence bug in the one component that must agree with the index.
|
|
12
|
+
- **Dense.** A long prompt, or consolidation's group-gist concatenation, can exceed the model's
|
|
13
|
+
512-token input, and BGE truncates silently. So the read side gets the same preflight the write
|
|
14
|
+
side has — through the same tokenizer — and resolves it the other way: it **truncates and records
|
|
15
|
+
it**, because a write can be handed back to an agent that still holds its text while refusing a
|
|
16
|
+
user's prompt would mean refusing to retrieve at all.
|
|
17
|
+
|
|
18
|
+
`retrieval.md` §"Two kinds of query" then makes both rules serve two callers. An **external** query
|
|
19
|
+
is text from outside the store; an **internal** query is a memory querying other memories, whose
|
|
20
|
+
dense side is its own stored first-chunk embedding and whose lexical side is its own `gist +
|
|
21
|
+
content` through the identical constructor. One rule called on stored prose instead of on a prompt,
|
|
22
|
+
never a second rule.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from dataclasses import dataclass
|
|
26
|
+
from enum import StrEnum
|
|
27
|
+
from typing import Final
|
|
28
|
+
|
|
29
|
+
import aiosqlite
|
|
30
|
+
|
|
31
|
+
from zikaron.core.errors import BadConfigSource, ErrorCode, ZikaronError
|
|
32
|
+
from zikaron.core.indexing.encoder import Encoder, assembled_tokens, token_head
|
|
33
|
+
from zikaron.core.indexing.vectors import serialize
|
|
34
|
+
|
|
35
|
+
#: What separates the two prose fields of an internal query's lexical side. Any non-alphanumeric
|
|
36
|
+
#: character would do — the constructor keeps only alphanumeric runs — so the choice is legibility.
|
|
37
|
+
_INTERNAL_JOIN: Final = "\n"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class QueryOrigin(StrEnum):
|
|
41
|
+
"""Where a query's dense vector came from, which decides whether a cosine may be read off it.
|
|
42
|
+
|
|
43
|
+
An `INTERNAL` query reuses a stored first-chunk embedding, which the write path L2-normalized,
|
|
44
|
+
so `cos = 1 - d^2/2` over `vec0`'s distance is exact and the three cosine cutoffs mean what
|
|
45
|
+
they say. An `EXTERNAL` query's vector is whatever the embedder returned for the prompt and is
|
|
46
|
+
deliberately **not** renormalized: ordering by distance to unit documents is monotone in cosine
|
|
47
|
+
for any fixed query vector, so the ranking is unaffected, and no external query is ever
|
|
48
|
+
thresholded — all three cutoffs are `s(X -> Y)` between two stored memories. Carrying the origin
|
|
49
|
+
is what keeps that from being a fact somebody has to remember: a cosine is offered only where it
|
|
50
|
+
is exact.
|
|
51
|
+
"""
|
|
52
|
+
|
|
53
|
+
EXTERNAL = "external"
|
|
54
|
+
INTERNAL = "internal"
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
@dataclass(frozen=True, slots=True)
|
|
58
|
+
class LexicalQuery:
|
|
59
|
+
"""One query's lexical side: the terms selected, and the expression bound into `MATCH`.
|
|
60
|
+
|
|
61
|
+
`expression` is `None` when no term survived, which is `retrieval.md`'s `lexical_skipped`: a
|
|
62
|
+
prompt of pure punctuation runs dense-only rather than raising or matching everything.
|
|
63
|
+
"""
|
|
64
|
+
|
|
65
|
+
terms: tuple[str, ...]
|
|
66
|
+
expression: str | None
|
|
67
|
+
|
|
68
|
+
@property
|
|
69
|
+
def skipped(self) -> bool:
|
|
70
|
+
"""Whether the lexical arm has nothing to run — zero surviving terms."""
|
|
71
|
+
return self.expression is None
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
@dataclass(frozen=True, slots=True)
|
|
75
|
+
class PreparedQuery:
|
|
76
|
+
"""Everything the retrieval algorithm needs of a query, whichever kind produced it.
|
|
77
|
+
|
|
78
|
+
The algorithm is identical for both kinds (`retrieval.md` §"Two kinds of query"), so this is
|
|
79
|
+
the seam: construction differs, retrieval does not.
|
|
80
|
+
"""
|
|
81
|
+
|
|
82
|
+
origin: QueryOrigin
|
|
83
|
+
vector: bytes
|
|
84
|
+
lexical: LexicalQuery
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
@dataclass(frozen=True, slots=True)
|
|
88
|
+
class ExternalQuery:
|
|
89
|
+
"""A prepared external query plus the two facts only its dense preflight can report.
|
|
90
|
+
|
|
91
|
+
`query_tokens` is the pre-truncation token count of the query text **alone**, excluding the BGE
|
|
92
|
+
prefix — the prefix is charged against the budget on the other side of the subtraction, so
|
|
93
|
+
counting it here would double it and make the recorded number disagree with the same field on a
|
|
94
|
+
store configured with an empty prefix. Both fields go on the `search`/`surface_call` event.
|
|
95
|
+
"""
|
|
96
|
+
|
|
97
|
+
prepared: PreparedQuery
|
|
98
|
+
query_tokens: int
|
|
99
|
+
query_truncated: bool
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _quote(term: str) -> str:
|
|
103
|
+
"""One term as an FTS5 string literal.
|
|
104
|
+
|
|
105
|
+
Quoting is what neutralizes operators — a quoted `OR` is the word "or", a quoted `-x` is "x" —
|
|
106
|
+
and, more importantly, what hands the term's *contents* to the table's own tokenizer, so case
|
|
107
|
+
folding and diacritic stripping happen on both sides by the same code. The internal-quote
|
|
108
|
+
doubling `retrieval.md` step 3 requires is unconditional rather than guarded: a term is a run of
|
|
109
|
+
alphanumerics and so cannot contain a quote today, and an unconditional `replace` costs nothing
|
|
110
|
+
while a conditional one would be a branch nothing can reach.
|
|
111
|
+
"""
|
|
112
|
+
escaped = term.replace('"', '""')
|
|
113
|
+
return f'"{escaped}"'
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _alphanumeric_runs(text: str) -> list[str]:
|
|
117
|
+
"""Every maximal run of Unicode alphanumerics in `text`, in order.
|
|
118
|
+
|
|
119
|
+
`str.isalnum()` per character rather than a tokenizer of our own: `retrieval.md` bounds the
|
|
120
|
+
residual risk of this being *our* boundary rule rather than SQLite's to one term missing one
|
|
121
|
+
match — checked against `fts5vocab` on nine technical strings with zero disagreements — which is
|
|
122
|
+
a strictly smaller exposure than reproducing `unicode61`'s folding tables.
|
|
123
|
+
"""
|
|
124
|
+
runs: list[str] = []
|
|
125
|
+
current: list[str] = []
|
|
126
|
+
for character in text:
|
|
127
|
+
if character.isalnum():
|
|
128
|
+
current.append(character)
|
|
129
|
+
elif current:
|
|
130
|
+
runs.append("".join(current))
|
|
131
|
+
current = []
|
|
132
|
+
if current:
|
|
133
|
+
runs.append("".join(current))
|
|
134
|
+
return runs
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def lexical_terms(text: str, *, max_terms: int) -> tuple[str, ...]:
|
|
138
|
+
"""The terms `text` contributes to a lexical query, in the order they will be joined.
|
|
139
|
+
|
|
140
|
+
Deduplicated on the exact string, then sorted **longest-first with ties broken by first
|
|
141
|
+
occurrence**, then cut to `max_terms`. Longest-first is the right truncation for this corpus
|
|
142
|
+
rather than an arbitrary one: in prose about code the long tokens are the identifiers, error
|
|
143
|
+
strings and paths, while the terms it drops are short common words BM25 would have scored near
|
|
144
|
+
zero anyway.
|
|
145
|
+
|
|
146
|
+
Deduplication compares the text as written, and deliberately does not fold case: folding is
|
|
147
|
+
FTS5's, on both sides, and a second folding rule here is exactly the silent divergence the
|
|
148
|
+
quoted-literal design avoids. Two differently-cased occurrences of one word therefore spend two
|
|
149
|
+
of the `max_terms` slots, which costs a slot and cannot cost correctness.
|
|
150
|
+
"""
|
|
151
|
+
seen: dict[str, int] = {}
|
|
152
|
+
for index, run in enumerate(_alphanumeric_runs(text)):
|
|
153
|
+
seen.setdefault(run, index)
|
|
154
|
+
ranked = sorted(seen.items(), key=lambda pair: (-len(pair[0]), pair[1]))
|
|
155
|
+
return tuple(term for term, _ in ranked[:max_terms])
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def lexical_query(text: str, *, max_terms: int) -> LexicalQuery:
|
|
159
|
+
"""`text` as an FTS5 `MATCH` expression, or a skipped arm if it yields no term.
|
|
160
|
+
|
|
161
|
+
The expression is `OR` over quoted terms, in selection order, and is always **bound as a
|
|
162
|
+
parameter** rather than formatted into SQL text. BM25 already ranks a document matching several
|
|
163
|
+
terms above one matching a single term, so `OR` loses nothing a phrase would gain — while a
|
|
164
|
+
phrase would require adjacency and so would silently drop every non-adjacent match on dotted
|
|
165
|
+
identifiers, paths and error strings, which is precisely the query shape this store serves.
|
|
166
|
+
"""
|
|
167
|
+
terms = lexical_terms(text, max_terms=max_terms)
|
|
168
|
+
expression = " OR ".join(_quote(term) for term in terms) if terms else None
|
|
169
|
+
return LexicalQuery(terms=terms, expression=expression)
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def internal_lexical_text(gist: str, content: str) -> str:
|
|
173
|
+
"""The prose an internal query's lexical side is built from: the memory's own `gist + content`.
|
|
174
|
+
|
|
175
|
+
A function rather than a call-site concatenation because the 64-term cap does real work here —
|
|
176
|
+
a memory has far more terms than a prompt does — so which text the cap is applied to is part of
|
|
177
|
+
the contract rather than an incidental detail of one caller.
|
|
178
|
+
"""
|
|
179
|
+
return f"{gist}{_INTERNAL_JOIN}{content}"
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def _prefix_too_long(prefix_tokens: int, cap: int) -> ZikaronError:
|
|
183
|
+
return ZikaronError(
|
|
184
|
+
ErrorCode.BAD_CONFIG,
|
|
185
|
+
source=BadConfigSource.FILE,
|
|
186
|
+
file="<effective config>",
|
|
187
|
+
key="embedding.embed_prefix_query",
|
|
188
|
+
value=f"{prefix_tokens} tokens",
|
|
189
|
+
expected=f"a prefix leaving at least one token of the model's {cap}-token input for the "
|
|
190
|
+
"query itself — a longer prefix leaves no query to embed, so the store cannot be read",
|
|
191
|
+
)
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def _prefix_leaves_no_query_token(prefix_tokens: int, cap: int) -> ZikaronError:
|
|
195
|
+
"""`bad_config` for a prefix beside which not even this query's first token fits.
|
|
196
|
+
|
|
197
|
+
Distinct from `_prefix_too_long` only in how it was discovered: that one is arithmetic on the
|
|
198
|
+
two counts, this one is what the assembled sequence turned out to cost once the boundary between
|
|
199
|
+
them was tokenized. The claim is deliberately about *this* query rather than about every query,
|
|
200
|
+
because the boundary's cost depends on the text either side of it and a different first token
|
|
201
|
+
could fuse more cheaply. Same key, same remedy, both reachable only by configuration.
|
|
202
|
+
"""
|
|
203
|
+
return ZikaronError(
|
|
204
|
+
ErrorCode.BAD_CONFIG,
|
|
205
|
+
source=BadConfigSource.FILE,
|
|
206
|
+
file="<effective config>",
|
|
207
|
+
key="embedding.embed_prefix_query",
|
|
208
|
+
value=f"{prefix_tokens} tokens",
|
|
209
|
+
expected="a prefix beside which this query's first token still fits once the two are "
|
|
210
|
+
f"tokenized together — assembled with even one of its tokens the sequence exceeds the "
|
|
211
|
+
f"model's {cap}-token input",
|
|
212
|
+
)
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def _fit_to_cap(text: str, *, prefix: str, budget: int, encoder: Encoder) -> str:
|
|
216
|
+
"""The nominal-budget head of `text`, shrunk until its **assembled** sequence fits the cap.
|
|
217
|
+
|
|
218
|
+
Called only once the whole text is known not to fit. The budget subtraction chooses the first
|
|
219
|
+
candidate; this is what settles it, by counting `prefix + head` as one string at each step
|
|
220
|
+
rather than adding the two pieces' counts. Each step drops one query token, so the loop
|
|
221
|
+
terminates, and its floor is one token: a prefix beside which not even this query's first token
|
|
222
|
+
fits is a configuration error rather than a query silently reduced to the prefix alone.
|
|
223
|
+
|
|
224
|
+
Not "the longest head that fits", and the distinction is worth stating because retokenization at
|
|
225
|
+
the boundary can *reduce* a count as well as raise it — so a head one token past the nominal
|
|
226
|
+
budget could in principle fit. The search deliberately does not look there: `retrieval.md` sets
|
|
227
|
+
the budget as what the query may keep, and going above it to reclaim a token the boundary
|
|
228
|
+
happened to absorb would make the kept length depend on the prefix in a way no field records.
|
|
229
|
+
|
|
230
|
+
Raises:
|
|
231
|
+
ZikaronError: `BAD_CONFIG` naming `embedding.embed_prefix_query` if not even this query's
|
|
232
|
+
first token fits beside the prefix.
|
|
233
|
+
"""
|
|
234
|
+
for tokens in range(min(budget, encoder.count_tokens(text)), 0, -1):
|
|
235
|
+
kept = token_head(text, tokens=tokens, encoder=encoder)
|
|
236
|
+
if assembled_tokens(prefix, kept, encoder=encoder) <= encoder.max_sequence_tokens:
|
|
237
|
+
return kept
|
|
238
|
+
raise _prefix_leaves_no_query_token(encoder.count_tokens(prefix), encoder.max_sequence_tokens)
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def external_query(text: str, *, encoder: Encoder, prefix: str, max_terms: int) -> ExternalQuery:
|
|
242
|
+
"""Build both arms of a query from text outside the store, embedding it under the token budget.
|
|
243
|
+
|
|
244
|
+
The dense side is embedded through the deployed artifact with D20's BGE query-instruction
|
|
245
|
+
prefix; the lexical side is the term constructor above, over the **whole** text whether or not
|
|
246
|
+
the dense side had to be truncated.
|
|
247
|
+
|
|
248
|
+
Blocking work — one embedding call — so a caller on an event loop is responsible for keeping
|
|
249
|
+
this off it, exactly as the write path's `embed_chunks` is.
|
|
250
|
+
|
|
251
|
+
Args:
|
|
252
|
+
text: the query as the caller received it: a user prompt, an agent's `search` string, or
|
|
253
|
+
consolidation's group-gist concatenation.
|
|
254
|
+
encoder: the deployed artifact. Its tokenizer counts, and its model embeds.
|
|
255
|
+
prefix: `embed_prefix_query`. Query-side only, per BGE's asymmetric convention; the empty
|
|
256
|
+
string disables it.
|
|
257
|
+
max_terms: `fts_query_max_terms`.
|
|
258
|
+
|
|
259
|
+
Raises:
|
|
260
|
+
ZikaronError: `BAD_CONFIG` if the configured prefix leaves no room for the query — either by
|
|
261
|
+
the budget arithmetic or once the two are tokenized as one string — or if the artifact's
|
|
262
|
+
token count and token spans disagree.
|
|
263
|
+
"""
|
|
264
|
+
cap = encoder.max_sequence_tokens
|
|
265
|
+
budget = cap - encoder.n_special_tokens - encoder.count_tokens(prefix)
|
|
266
|
+
if budget < 1:
|
|
267
|
+
raise _prefix_too_long(encoder.count_tokens(prefix), cap)
|
|
268
|
+
|
|
269
|
+
query_tokens = encoder.count_tokens(text)
|
|
270
|
+
# The assembled sequence is counted first, so the common path pays one tokenizer call and the
|
|
271
|
+
# search below runs only when the input genuinely overflows. Checking it this way also keeps
|
|
272
|
+
# `query_truncated` exact: a head slices from the first token's start to the last token's end,
|
|
273
|
+
# so calling it on a query that already fits would drop leading or trailing punctuation and
|
|
274
|
+
# report a truncation that did not happen.
|
|
275
|
+
fits_whole = assembled_tokens(prefix, text, encoder=encoder) <= cap
|
|
276
|
+
kept = text if fits_whole else _fit_to_cap(text, prefix=prefix, budget=budget, encoder=encoder)
|
|
277
|
+
(vector,) = encoder.embed([f"{prefix}{kept}"])
|
|
278
|
+
return ExternalQuery(
|
|
279
|
+
prepared=PreparedQuery(
|
|
280
|
+
origin=QueryOrigin.EXTERNAL,
|
|
281
|
+
vector=serialize(vector),
|
|
282
|
+
lexical=lexical_query(text, max_terms=max_terms),
|
|
283
|
+
),
|
|
284
|
+
query_tokens=query_tokens,
|
|
285
|
+
# What was dropped, not what the budget subtraction predicted would be dropped: the two
|
|
286
|
+
# differ exactly when the prefix boundary retokenizes, which is the case this preflight now
|
|
287
|
+
# settles by counting the assembled string instead of the sum of its parts.
|
|
288
|
+
query_truncated=kept != text,
|
|
289
|
+
)
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
async def internal_query(
|
|
293
|
+
db: aiosqlite.Connection, *, memory_uuid: str, max_terms: int
|
|
294
|
+
) -> PreparedQuery:
|
|
295
|
+
"""Build both arms of one memory's query against the others, reading no model at all.
|
|
296
|
+
|
|
297
|
+
The dense side is the memory's **stored first-chunk embedding** (`part_index = 0`): it already
|
|
298
|
+
exists, it is already normalized, it cannot overflow the model's input because the write-side
|
|
299
|
+
preflight proved it fits, and reusing it is what makes planning a pure function of the store
|
|
300
|
+
rather than of a second inference. The lexical side is the memory's own `gist + content` through
|
|
301
|
+
the identical constructor an external query uses, where the 64-term cap does real work.
|
|
302
|
+
|
|
303
|
+
Assumes the caller's own open transaction: an internal query's whole point is to run inside the
|
|
304
|
+
write or planning transaction that produced the row it is querying from.
|
|
305
|
+
|
|
306
|
+
Raises:
|
|
307
|
+
ZikaronError: `NOT_FOUND` if `memory_uuid` names no row, or names one with no `part_index=0`
|
|
308
|
+
chunk. Invariant 12 makes the second impossible for a row with content, so it is
|
|
309
|
+
reported rather than papered over with a second embed call.
|
|
310
|
+
"""
|
|
311
|
+
rows = await db.execute_fetchall(
|
|
312
|
+
"SELECT m.gist, m.content, v.embedding "
|
|
313
|
+
"FROM memory m "
|
|
314
|
+
"JOIN memory_chunk c ON c.memory_uuid = m.uuid AND c.part_index = 0 "
|
|
315
|
+
"JOIN memory_vec v ON v.rowid = c.chunk_id "
|
|
316
|
+
"WHERE m.uuid = ?",
|
|
317
|
+
(memory_uuid,),
|
|
318
|
+
)
|
|
319
|
+
found = list(rows)
|
|
320
|
+
if not found:
|
|
321
|
+
raise ZikaronError(ErrorCode.NOT_FOUND, uuid=memory_uuid)
|
|
322
|
+
gist, content, embedding = found[0]
|
|
323
|
+
return PreparedQuery(
|
|
324
|
+
origin=QueryOrigin.INTERNAL,
|
|
325
|
+
vector=bytes(embedding),
|
|
326
|
+
lexical=lexical_query(internal_lexical_text(str(gist), str(content)), max_terms=max_terms),
|
|
327
|
+
)
|
|
@@ -0,0 +1,260 @@
|
|
|
1
|
+
"""Fusion, demotion, the total order, and the supersession repair — all pure, no store.
|
|
2
|
+
|
|
3
|
+
`retrieval.md` §"The fused score, written out", §"Total order" and §"Supersession: eligible,
|
|
4
|
+
demoted, and labelled" are normative. Four steps, in this order, and the order is observable:
|
|
5
|
+
|
|
6
|
+
1. **RRF.** `Σ 1 / (rrf_k + rank)` over the arms that returned the row, ranks 1-based. An arm that
|
|
7
|
+
did not return it contributes nothing — not a term at a notional worst rank.
|
|
8
|
+
2. **Penalties.** A demoted row's fused score is multiplied by `supersession_penalty` or
|
|
9
|
+
`retired_penalty`. On the *immediate* edge, not on graph depth: nothing measured says a
|
|
10
|
+
twice-superseded record is more dangerous than a once-superseded one. The two states are mutually
|
|
11
|
+
exclusive, so the penalties never compound.
|
|
12
|
+
3. **The total order**, five steps deep, because fused ties are pervasive rather than rare —
|
|
13
|
+
measured: every hybrid query has at least one exact tie, and 12.5% have one *inside the top
|
|
14
|
+
five*, where order fell to whichever arm happened to be appended first.
|
|
15
|
+
4. **The supersession repair**, a stable topological pass that puts every retrieved replacement
|
|
16
|
+
ahead of every record it replaced.
|
|
17
|
+
|
|
18
|
+
The caller cuts to `limit` **after** all four, which is what lets the repair promote a replacement
|
|
19
|
+
into the returned set rather than merely reshuffling what already made the cut. That is the case the
|
|
20
|
+
measurement is about: a superseded record outranks its own replacement 28-50% of the time on the
|
|
21
|
+
polarity stubs, and in those cases the correct answer is provably present and losing.
|
|
22
|
+
|
|
23
|
+
Nothing here touches the store, so determinism is testable by calling it twice.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from collections.abc import Iterable, Mapping, Sequence
|
|
27
|
+
from dataclasses import dataclass
|
|
28
|
+
from typing import Final
|
|
29
|
+
|
|
30
|
+
from zikaron.core.errors import ErrorCode, RowState, ZikaronError
|
|
31
|
+
from zikaron.core.events import Demotion
|
|
32
|
+
from zikaron.core.records.memory import Tier
|
|
33
|
+
from zikaron.core.records.supersession import BadSupersessionReason
|
|
34
|
+
from zikaron.core.retrieval.arms import ArmOutcome, ArmRow
|
|
35
|
+
|
|
36
|
+
#: Which tier wins a fused tie: consolidated prose has been reviewed by the consolidator, and
|
|
37
|
+
#: `~/Memory` reached the same tiebreak independently ("cohesive memory over raw journal line").
|
|
38
|
+
_TIER_ORDER: Final[Mapping[Tier, int]] = {Tier.LONG_TERM: 0, Tier.JOURNAL: 1}
|
|
39
|
+
|
|
40
|
+
#: The recency component for a long-term row. Step 4 is D27's *journal-local* recency tiebreak, so a
|
|
41
|
+
#: long-term row skips it and falls through to `uuid` — nothing about a long-term record's age is a
|
|
42
|
+
#: claim about its truth. A constant is well defined here precisely because step 3 has already
|
|
43
|
+
#: partitioned the remaining ties by tier: two rows still tied at step 4 share a tier, so this value
|
|
44
|
+
#: is only ever compared against another long-term row's copy of it.
|
|
45
|
+
_NO_RECENCY: Final = -1
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@dataclass(frozen=True, slots=True)
|
|
49
|
+
class PoolRow:
|
|
50
|
+
"""One candidate's row facts — exactly the columns the read path needs, and no more.
|
|
51
|
+
|
|
52
|
+
`content` and `version` are deliberately absent. `search` returns neither (a version would imply
|
|
53
|
+
a licence to write that the agent has not earned), the injected block shows only the gist, and
|
|
54
|
+
selecting long text for every row of a fused pool would spend I/O on the latency-critical path
|
|
55
|
+
for prose nobody reads.
|
|
56
|
+
"""
|
|
57
|
+
|
|
58
|
+
uuid: str
|
|
59
|
+
tier: Tier
|
|
60
|
+
gist: str
|
|
61
|
+
active: bool
|
|
62
|
+
superseded_by: str | None
|
|
63
|
+
created_at: str
|
|
64
|
+
updated_at: str
|
|
65
|
+
|
|
66
|
+
@property
|
|
67
|
+
def state(self) -> RowState:
|
|
68
|
+
"""This row's display state, per `schema.md` §"Retrieval eligibility"'s three states."""
|
|
69
|
+
if self.active:
|
|
70
|
+
return RowState.LIVE
|
|
71
|
+
if self.superseded_by is not None:
|
|
72
|
+
return RowState.SUPERSEDED
|
|
73
|
+
return RowState.RETIRED
|
|
74
|
+
|
|
75
|
+
@property
|
|
76
|
+
def demotion(self) -> Demotion | None:
|
|
77
|
+
"""Why this row is demoted, or `None` for a live row that is not."""
|
|
78
|
+
if self.active:
|
|
79
|
+
return None
|
|
80
|
+
return Demotion.SUPERSEDED if self.superseded_by is not None else Demotion.RETIRED
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
@dataclass(frozen=True, slots=True)
|
|
84
|
+
class Penalties:
|
|
85
|
+
"""The two demotion multipliers, which are separate config keys with equal defaults.
|
|
86
|
+
|
|
87
|
+
Equal by honesty rather than by symmetry: nothing measured distinguishes a superseded row from
|
|
88
|
+
an outright-retired one, so they start equal and can diverge once something does.
|
|
89
|
+
"""
|
|
90
|
+
|
|
91
|
+
superseded: float
|
|
92
|
+
retired: float
|
|
93
|
+
|
|
94
|
+
def factor(self, demotion: Demotion | None) -> float:
|
|
95
|
+
"""The multiplier for one row's demotion state — 1.0 for a row that is not demoted."""
|
|
96
|
+
if demotion is None:
|
|
97
|
+
return 1.0
|
|
98
|
+
return self.superseded if demotion is Demotion.SUPERSEDED else self.retired
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
@dataclass(frozen=True, slots=True)
|
|
102
|
+
class RankedMemory:
|
|
103
|
+
"""One candidate, scored and placed: the row, its fused score, and where each arm found it.
|
|
104
|
+
|
|
105
|
+
`best_rank` is the minimum over the arms that returned it, so a row only one arm found is ranked
|
|
106
|
+
on that arm's number rather than penalized for the other's silence. `best_distance` is the dense
|
|
107
|
+
arm's rolled-up minimum, kept because the three cosine cutoffs are read off it — and only
|
|
108
|
+
meaningful for a query whose vector is unit length, which `query.QueryOrigin` is what tracks.
|
|
109
|
+
"""
|
|
110
|
+
|
|
111
|
+
row: PoolRow
|
|
112
|
+
fused_score: float
|
|
113
|
+
dense_rank: int | None
|
|
114
|
+
lexical_rank: int | None
|
|
115
|
+
best_distance: float | None
|
|
116
|
+
|
|
117
|
+
@property
|
|
118
|
+
def best_rank(self) -> int:
|
|
119
|
+
"""The best rank any arm gave this row. Both arms cannot be absent: it would not be here."""
|
|
120
|
+
ranks = [rank for rank in (self.dense_rank, self.lexical_rank) if rank is not None]
|
|
121
|
+
return min(ranks)
|
|
122
|
+
|
|
123
|
+
@property
|
|
124
|
+
def demoted(self) -> bool:
|
|
125
|
+
"""Whether this row was demoted, which is exactly whether it is inactive."""
|
|
126
|
+
return self.row.demotion is not None
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def rrf_score(ranks: Iterable[int], *, rrf_k: int) -> float:
|
|
130
|
+
"""Reciprocal Rank Fusion over the ranks one row achieved, in the arms that returned it.
|
|
131
|
+
|
|
132
|
+
`Σ 1 / (rrf_k + rank)`, ranks **1-based** — the formula every quality figure in `retrieval.md`
|
|
133
|
+
was measured through. Rebasing to 0 would leave `rrf_k = 60` naming a different denominator, so
|
|
134
|
+
the config default and the measurements would silently stop describing the same pipeline.
|
|
135
|
+
"""
|
|
136
|
+
return sum(1.0 / (rrf_k + rank) for rank in ranks)
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _by_uuid(outcome: ArmOutcome) -> Mapping[str, ArmRow]:
|
|
140
|
+
return {row.uuid: row for row in outcome.rows}
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _journal_recency(rows: Iterable[PoolRow]) -> Mapping[str, int]:
|
|
144
|
+
"""Each journal row's position when the pool's timestamps are ordered newest first.
|
|
145
|
+
|
|
146
|
+
An index rather than the timestamp itself, so step 4 needs no string inversion and no timestamp
|
|
147
|
+
parsing — `created_at` is opaque text this layer never has to interpret, and a mapping from
|
|
148
|
+
distinct value to descending position is monotone, so comparing indices ascending *is* comparing
|
|
149
|
+
timestamps descending. Ties on the timestamp share an index and fall through to `uuid`, which is
|
|
150
|
+
what step 5 is for.
|
|
151
|
+
"""
|
|
152
|
+
journal = [row for row in rows if row.tier is Tier.JOURNAL]
|
|
153
|
+
descending = sorted({row.created_at for row in journal}, reverse=True)
|
|
154
|
+
position = {value: index for index, value in enumerate(descending)}
|
|
155
|
+
return {row.uuid: position[row.created_at] for row in journal}
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _order_key(
|
|
159
|
+
ranked: RankedMemory, recency: Mapping[str, int]
|
|
160
|
+
) -> tuple[float, int, int, int, str]:
|
|
161
|
+
"""`retrieval.md`'s five-step total order, as one comparable tuple.
|
|
162
|
+
|
|
163
|
+
Ascending on every component, so the fused score is negated. The components are, in order: the
|
|
164
|
+
fused score descending; the best rank any arm gave it ascending, since a document one arm ranked
|
|
165
|
+
first is a better bet than one both arms ranked tenth; tier, long-term first; `created_at`
|
|
166
|
+
descending **inside a journal block only**; and `uuid`, which is arbitrary and is the point —
|
|
167
|
+
it guarantees a total order, so the same store and query always produce the same list.
|
|
168
|
+
"""
|
|
169
|
+
return (
|
|
170
|
+
-ranked.fused_score,
|
|
171
|
+
ranked.best_rank,
|
|
172
|
+
_TIER_ORDER[ranked.row.tier],
|
|
173
|
+
recency.get(ranked.row.uuid, _NO_RECENCY),
|
|
174
|
+
ranked.row.uuid,
|
|
175
|
+
)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _repair_supersession(ordered: Sequence[RankedMemory]) -> tuple[RankedMemory, ...]:
|
|
179
|
+
"""Reorder so every retrieved replacement precedes every record it replaced.
|
|
180
|
+
|
|
181
|
+
Not "immediately above" — that rule is unsatisfiable, because `merge` absorbing `A` and `B` into
|
|
182
|
+
`C` writes `A→C` *and* `B→C`, so no linear order can put `C` immediately above both. The
|
|
183
|
+
realizable rule is precedence, enforced as a stable topological pass: scan in order and emit the
|
|
184
|
+
first not-yet-emitted row whose `superseded_by` target is either absent from the pool or already
|
|
185
|
+
emitted. It handles `A→B→C` transitively, giving `C, B, A`.
|
|
186
|
+
|
|
187
|
+
Adjacency is neither preserved nor needed: the injected block gives the agent the replacement's
|
|
188
|
+
uuid explicitly, which is a stronger cue than proximity and is the part that survives a group of
|
|
189
|
+
rows sharing one replacement.
|
|
190
|
+
|
|
191
|
+
Raises:
|
|
192
|
+
ZikaronError: `BAD_SUPERSESSION` with `reason='cycle'` if no row is emittable, which
|
|
193
|
+
invariant 6 makes impossible — every row waits on at most one other and the graph is
|
|
194
|
+
acyclic, so the wait-for relation is a forest. Refused rather than worked around: a
|
|
195
|
+
fallback to the pre-repair order would answer from a corrupted graph with a *plausible*
|
|
196
|
+
list, and putting the replacement first is the whole point of the pass.
|
|
197
|
+
"""
|
|
198
|
+
in_pool = {ranked.row.uuid for ranked in ordered}
|
|
199
|
+
waiting = list(ordered)
|
|
200
|
+
emitted: list[RankedMemory] = []
|
|
201
|
+
emitted_uuids: set[str] = set()
|
|
202
|
+
while waiting:
|
|
203
|
+
for index, ranked in enumerate(waiting):
|
|
204
|
+
target = ranked.row.superseded_by
|
|
205
|
+
if target is None or target not in in_pool or target in emitted_uuids:
|
|
206
|
+
emitted.append(waiting.pop(index))
|
|
207
|
+
emitted_uuids.add(ranked.row.uuid)
|
|
208
|
+
break
|
|
209
|
+
else:
|
|
210
|
+
stalled = waiting[0].row
|
|
211
|
+
raise ZikaronError(
|
|
212
|
+
ErrorCode.BAD_SUPERSESSION,
|
|
213
|
+
uuid=stalled.uuid,
|
|
214
|
+
target=stalled.superseded_by,
|
|
215
|
+
reason=BadSupersessionReason.CYCLE,
|
|
216
|
+
)
|
|
217
|
+
return tuple(emitted)
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def rank(
|
|
221
|
+
*,
|
|
222
|
+
dense: ArmOutcome,
|
|
223
|
+
lexical: ArmOutcome,
|
|
224
|
+
rows: Mapping[str, PoolRow],
|
|
225
|
+
rrf_k: int,
|
|
226
|
+
penalties: Penalties,
|
|
227
|
+
) -> tuple[RankedMemory, ...]:
|
|
228
|
+
"""Fuse both arms into one ordered pool: RRF, penalties, the total order, then the repair.
|
|
229
|
+
|
|
230
|
+
The returned pool is **not** cut to any output budget; the caller cuts it, because the repair
|
|
231
|
+
must run first if it is to promote a replacement into the returned set.
|
|
232
|
+
|
|
233
|
+
Args:
|
|
234
|
+
rows: the row facts for every uuid either arm returned. A uuid missing from this mapping is
|
|
235
|
+
a row that vanished between the arm query and the pool load, which one read transaction
|
|
236
|
+
makes impossible; it is dropped rather than guessed at.
|
|
237
|
+
"""
|
|
238
|
+
dense_rows = _by_uuid(dense)
|
|
239
|
+
lexical_rows = _by_uuid(lexical)
|
|
240
|
+
scored: list[RankedMemory] = []
|
|
241
|
+
for uuid in [*dense_rows, *(uuid for uuid in lexical_rows if uuid not in dense_rows)]:
|
|
242
|
+
row = rows.get(uuid)
|
|
243
|
+
if row is None:
|
|
244
|
+
continue
|
|
245
|
+
found = [arm[uuid] for arm in (dense_rows, lexical_rows) if uuid in arm]
|
|
246
|
+
fused = rrf_score((arm_row.rank for arm_row in found), rrf_k=rrf_k)
|
|
247
|
+
dense_row = dense_rows.get(uuid)
|
|
248
|
+
lexical_row = lexical_rows.get(uuid)
|
|
249
|
+
scored.append(
|
|
250
|
+
RankedMemory(
|
|
251
|
+
row=row,
|
|
252
|
+
fused_score=fused * penalties.factor(row.demotion),
|
|
253
|
+
dense_rank=None if dense_row is None else dense_row.rank,
|
|
254
|
+
lexical_rank=None if lexical_row is None else lexical_row.rank,
|
|
255
|
+
best_distance=None if dense_row is None else dense_row.distance,
|
|
256
|
+
)
|
|
257
|
+
)
|
|
258
|
+
recency = _journal_recency(ranked.row for ranked in scored)
|
|
259
|
+
ordered = sorted(scored, key=lambda ranked: _order_key(ranked, recency))
|
|
260
|
+
return _repair_supersession(ordered)
|