zikaron 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (162) hide show
  1. zikaron/__init__.py +1 -0
  2. zikaron/cli/__init__.py +1 -0
  3. zikaron/cli/main.py +114 -0
  4. zikaron/core/__init__.py +1 -0
  5. zikaron/core/clock.py +78 -0
  6. zikaron/core/config/__init__.py +1 -0
  7. zikaron/core/config/keys.py +395 -0
  8. zikaron/core/config/resolution.py +267 -0
  9. zikaron/core/consolidation/__init__.py +1 -0
  10. zikaron/core/consolidation/authorization.py +316 -0
  11. zikaron/core/consolidation/candidates.py +147 -0
  12. zikaron/core/consolidation/context.py +166 -0
  13. zikaron/core/consolidation/grouping.py +383 -0
  14. zikaron/core/consolidation/groups.py +490 -0
  15. zikaron/core/consolidation/payload.py +246 -0
  16. zikaron/core/consolidation/planning.py +192 -0
  17. zikaron/core/consolidation/rowstate.py +68 -0
  18. zikaron/core/consolidation/runs.py +306 -0
  19. zikaron/core/consolidation/serving.py +462 -0
  20. zikaron/core/consolidation/verbs.py +500 -0
  21. zikaron/core/errors.py +355 -0
  22. zikaron/core/events.py +748 -0
  23. zikaron/core/indexing/__init__.py +1 -0
  24. zikaron/core/indexing/acquisition.py +255 -0
  25. zikaron/core/indexing/chunking.py +368 -0
  26. zikaron/core/indexing/encoder.py +537 -0
  27. zikaron/core/indexing/lexical.py +86 -0
  28. zikaron/core/indexing/model_cache.py +93 -0
  29. zikaron/core/indexing/model_pin.py +89 -0
  30. zikaron/core/indexing/vectors.py +223 -0
  31. zikaron/core/indexing/writes.py +461 -0
  32. zikaron/core/knowledge/__init__.py +5 -0
  33. zikaron/core/knowledge/arms.py +104 -0
  34. zikaron/core/knowledge/builds.py +204 -0
  35. zikaron/core/knowledge/candidates.py +130 -0
  36. zikaron/core/knowledge/changes.py +175 -0
  37. zikaron/core/knowledge/chunking.py +376 -0
  38. zikaron/core/knowledge/counters.py +228 -0
  39. zikaron/core/knowledge/database.py +380 -0
  40. zikaron/core/knowledge/ddl.py +196 -0
  41. zikaron/core/knowledge/disposal.py +213 -0
  42. zikaron/core/knowledge/errors.py +166 -0
  43. zikaron/core/knowledge/files.py +202 -0
  44. zikaron/core/knowledge/git.py +385 -0
  45. zikaron/core/knowledge/groups.py +450 -0
  46. zikaron/core/knowledge/lexical.py +64 -0
  47. zikaron/core/knowledge/lifecycle.py +418 -0
  48. zikaron/core/knowledge/lock.py +277 -0
  49. zikaron/core/knowledge/meta.py +393 -0
  50. zikaron/core/knowledge/paths.py +55 -0
  51. zikaron/core/knowledge/pending.py +59 -0
  52. zikaron/core/knowledge/registry.py +264 -0
  53. zikaron/core/knowledge/repair.py +152 -0
  54. zikaron/core/knowledge/reporting.py +436 -0
  55. zikaron/core/knowledge/roots.py +91 -0
  56. zikaron/core/knowledge/scan.py +429 -0
  57. zikaron/core/knowledge/search.py +346 -0
  58. zikaron/core/knowledge/state.py +174 -0
  59. zikaron/core/knowledge/text.py +166 -0
  60. zikaron/core/knowledge/vectors.py +102 -0
  61. zikaron/core/knowledge/walk.py +264 -0
  62. zikaron/core/knowledge/writes.py +127 -0
  63. zikaron/core/records/__init__.py +1 -0
  64. zikaron/core/records/memory.py +961 -0
  65. zikaron/core/records/receipts.py +161 -0
  66. zikaron/core/records/supersession.py +221 -0
  67. zikaron/core/retrieval/__init__.py +1 -0
  68. zikaron/core/retrieval/arms.py +318 -0
  69. zikaron/core/retrieval/block.py +107 -0
  70. zikaron/core/retrieval/eligibility.py +164 -0
  71. zikaron/core/retrieval/query.py +327 -0
  72. zikaron/core/retrieval/ranking.py +260 -0
  73. zikaron/core/retrieval/reads.py +294 -0
  74. zikaron/core/retrieval/retrieve.py +158 -0
  75. zikaron/core/retrieval/similarity.py +87 -0
  76. zikaron/core/signals/__init__.py +34 -0
  77. zikaron/core/signals/contention.py +106 -0
  78. zikaron/core/signals/dedup.py +201 -0
  79. zikaron/core/signals/horizon.py +47 -0
  80. zikaron/core/signals/repair.py +161 -0
  81. zikaron/core/signals/retirement.py +83 -0
  82. zikaron/core/signals/sessions.py +105 -0
  83. zikaron/core/signals/writes.py +200 -0
  84. zikaron/core/store/__init__.py +1 -0
  85. zikaron/core/store/connection.py +202 -0
  86. zikaron/core/store/ddl.py +215 -0
  87. zikaron/core/store/embedder.py +45 -0
  88. zikaron/core/store/meta.py +152 -0
  89. zikaron/core/store/permissions.py +160 -0
  90. zikaron/core/store/store.py +408 -0
  91. zikaron/core/store/transactions.py +181 -0
  92. zikaron/core/write/__init__.py +33 -0
  93. zikaron/core/write/dedup.py +145 -0
  94. zikaron/core/write/tools.py +290 -0
  95. zikaron/doctor/__init__.py +1 -0
  96. zikaron/doctor/checks.py +220 -0
  97. zikaron/doctor/main.py +64 -0
  98. zikaron/harness/__init__.py +1 -0
  99. zikaron/harness/detect.py +92 -0
  100. zikaron/harness/spec.py +320 -0
  101. zikaron/hook/__init__.py +1 -0
  102. zikaron/hook/connect.py +379 -0
  103. zikaron/hook/envelope.py +106 -0
  104. zikaron/hook/failure.py +104 -0
  105. zikaron/hook/limits.py +61 -0
  106. zikaron/hook/main.py +118 -0
  107. zikaron/hook/push.py +183 -0
  108. zikaron/hook/rpc.py +85 -0
  109. zikaron/hook/spawn_warm.py +81 -0
  110. zikaron/hook/subagent_policy.py +57 -0
  111. zikaron/hook/tripwire.py +54 -0
  112. zikaron/hook/warm_helper.py +137 -0
  113. zikaron/hook/write_policy.py +319 -0
  114. zikaron/install/__init__.py +4 -0
  115. zikaron/install/__main__.py +18 -0
  116. zikaron/install/assets.py +394 -0
  117. zikaron/install/entries.py +370 -0
  118. zikaron/install/harness.py +185 -0
  119. zikaron/install/main.py +375 -0
  120. zikaron/install/targets.py +789 -0
  121. zikaron/install/writer.py +973 -0
  122. zikaron/knowledge/__init__.py +1 -0
  123. zikaron/knowledge/__main__.py +17 -0
  124. zikaron/knowledge/indexer/__init__.py +1 -0
  125. zikaron/knowledge/indexer/__main__.py +17 -0
  126. zikaron/knowledge/indexer/detach.py +83 -0
  127. zikaron/knowledge/indexer/main.py +187 -0
  128. zikaron/knowledge/main.py +466 -0
  129. zikaron/knowledge/scope.py +133 -0
  130. zikaron/mcp/__init__.py +6 -0
  131. zikaron/mcp/connection.py +583 -0
  132. zikaron/mcp/consolidator.py +316 -0
  133. zikaron/mcp/errors.py +73 -0
  134. zikaron/mcp/main.py +66 -0
  135. zikaron/mcp/primary.py +420 -0
  136. zikaron/mcp/server.py +96 -0
  137. zikaron/mcp/spill.py +328 -0
  138. zikaron/mcp/tool_names.py +67 -0
  139. zikaron/py.typed +0 -0
  140. zikaron/service/__init__.py +1 -0
  141. zikaron/service/asyncio_compat.py +126 -0
  142. zikaron/service/context.py +251 -0
  143. zikaron/service/dispatch.py +332 -0
  144. zikaron/service/dispatch_consolidation.py +397 -0
  145. zikaron/service/dispatch_knowledge.py +469 -0
  146. zikaron/service/envelope.py +166 -0
  147. zikaron/service/lifecycle.py +467 -0
  148. zikaron/service/log.py +96 -0
  149. zikaron/service/main.py +531 -0
  150. zikaron/service/params.py +168 -0
  151. zikaron/service/paths.py +181 -0
  152. zikaron/service/rpc.py +176 -0
  153. zikaron/service/security.py +156 -0
  154. zikaron/service/serialize.py +204 -0
  155. zikaron/service/serialize_knowledge.py +238 -0
  156. zikaron/service/server.py +416 -0
  157. zikaron-0.1.0.dist-info/METADATA +770 -0
  158. zikaron-0.1.0.dist-info/RECORD +162 -0
  159. zikaron-0.1.0.dist-info/WHEEL +5 -0
  160. zikaron-0.1.0.dist-info/entry_points.txt +4 -0
  161. zikaron-0.1.0.dist-info/licenses/LICENSE +21 -0
  162. zikaron-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,327 @@
1
+ """Query construction: a prompt is not an FTS5 expression, and it is not always under 512 tokens.
2
+
3
+ `retrieval.md` §"Query construction" is normative. Two failures live here, one per arm, and both
4
+ were underspecified until that section was written:
5
+
6
+ - **Lexical.** A raw prompt is not a safe `MATCH` expression — a quote, a parenthesis or a bare
7
+ `OR` changes the query's meaning or raises `fts5: syntax error`, and a syntax error on the push
8
+ path would send every user message to the degraded path. The rule splits on boundaries only and
9
+ hands each fragment to FTS5 **as a quoted string literal**, so FTS5's own tokenizer does the case
10
+ folding and diacritic stripping. Reimplementing `unicode61` was the alternative and it is a
11
+ silent-divergence bug in the one component that must agree with the index.
12
+ - **Dense.** A long prompt, or consolidation's group-gist concatenation, can exceed the model's
13
+ 512-token input, and BGE truncates silently. So the read side gets the same preflight the write
14
+ side has — through the same tokenizer — and resolves it the other way: it **truncates and records
15
+ it**, because a write can be handed back to an agent that still holds its text while refusing a
16
+ user's prompt would mean refusing to retrieve at all.
17
+
18
+ `retrieval.md` §"Two kinds of query" then makes both rules serve two callers. An **external** query
19
+ is text from outside the store; an **internal** query is a memory querying other memories, whose
20
+ dense side is its own stored first-chunk embedding and whose lexical side is its own `gist +
21
+ content` through the identical constructor. One rule called on stored prose instead of on a prompt,
22
+ never a second rule.
23
+ """
24
+
25
+ from dataclasses import dataclass
26
+ from enum import StrEnum
27
+ from typing import Final
28
+
29
+ import aiosqlite
30
+
31
+ from zikaron.core.errors import BadConfigSource, ErrorCode, ZikaronError
32
+ from zikaron.core.indexing.encoder import Encoder, assembled_tokens, token_head
33
+ from zikaron.core.indexing.vectors import serialize
34
+
35
+ #: What separates the two prose fields of an internal query's lexical side. Any non-alphanumeric
36
+ #: character would do — the constructor keeps only alphanumeric runs — so the choice is legibility.
37
+ _INTERNAL_JOIN: Final = "\n"
38
+
39
+
40
+ class QueryOrigin(StrEnum):
41
+ """Where a query's dense vector came from, which decides whether a cosine may be read off it.
42
+
43
+ An `INTERNAL` query reuses a stored first-chunk embedding, which the write path L2-normalized,
44
+ so `cos = 1 - d^2/2` over `vec0`'s distance is exact and the three cosine cutoffs mean what
45
+ they say. An `EXTERNAL` query's vector is whatever the embedder returned for the prompt and is
46
+ deliberately **not** renormalized: ordering by distance to unit documents is monotone in cosine
47
+ for any fixed query vector, so the ranking is unaffected, and no external query is ever
48
+ thresholded — all three cutoffs are `s(X -> Y)` between two stored memories. Carrying the origin
49
+ is what keeps that from being a fact somebody has to remember: a cosine is offered only where it
50
+ is exact.
51
+ """
52
+
53
+ EXTERNAL = "external"
54
+ INTERNAL = "internal"
55
+
56
+
57
+ @dataclass(frozen=True, slots=True)
58
+ class LexicalQuery:
59
+ """One query's lexical side: the terms selected, and the expression bound into `MATCH`.
60
+
61
+ `expression` is `None` when no term survived, which is `retrieval.md`'s `lexical_skipped`: a
62
+ prompt of pure punctuation runs dense-only rather than raising or matching everything.
63
+ """
64
+
65
+ terms: tuple[str, ...]
66
+ expression: str | None
67
+
68
+ @property
69
+ def skipped(self) -> bool:
70
+ """Whether the lexical arm has nothing to run — zero surviving terms."""
71
+ return self.expression is None
72
+
73
+
74
+ @dataclass(frozen=True, slots=True)
75
+ class PreparedQuery:
76
+ """Everything the retrieval algorithm needs of a query, whichever kind produced it.
77
+
78
+ The algorithm is identical for both kinds (`retrieval.md` §"Two kinds of query"), so this is
79
+ the seam: construction differs, retrieval does not.
80
+ """
81
+
82
+ origin: QueryOrigin
83
+ vector: bytes
84
+ lexical: LexicalQuery
85
+
86
+
87
+ @dataclass(frozen=True, slots=True)
88
+ class ExternalQuery:
89
+ """A prepared external query plus the two facts only its dense preflight can report.
90
+
91
+ `query_tokens` is the pre-truncation token count of the query text **alone**, excluding the BGE
92
+ prefix — the prefix is charged against the budget on the other side of the subtraction, so
93
+ counting it here would double it and make the recorded number disagree with the same field on a
94
+ store configured with an empty prefix. Both fields go on the `search`/`surface_call` event.
95
+ """
96
+
97
+ prepared: PreparedQuery
98
+ query_tokens: int
99
+ query_truncated: bool
100
+
101
+
102
+ def _quote(term: str) -> str:
103
+ """One term as an FTS5 string literal.
104
+
105
+ Quoting is what neutralizes operators — a quoted `OR` is the word "or", a quoted `-x` is "x" —
106
+ and, more importantly, what hands the term's *contents* to the table's own tokenizer, so case
107
+ folding and diacritic stripping happen on both sides by the same code. The internal-quote
108
+ doubling `retrieval.md` step 3 requires is unconditional rather than guarded: a term is a run of
109
+ alphanumerics and so cannot contain a quote today, and an unconditional `replace` costs nothing
110
+ while a conditional one would be a branch nothing can reach.
111
+ """
112
+ escaped = term.replace('"', '""')
113
+ return f'"{escaped}"'
114
+
115
+
116
+ def _alphanumeric_runs(text: str) -> list[str]:
117
+ """Every maximal run of Unicode alphanumerics in `text`, in order.
118
+
119
+ `str.isalnum()` per character rather than a tokenizer of our own: `retrieval.md` bounds the
120
+ residual risk of this being *our* boundary rule rather than SQLite's to one term missing one
121
+ match — checked against `fts5vocab` on nine technical strings with zero disagreements — which is
122
+ a strictly smaller exposure than reproducing `unicode61`'s folding tables.
123
+ """
124
+ runs: list[str] = []
125
+ current: list[str] = []
126
+ for character in text:
127
+ if character.isalnum():
128
+ current.append(character)
129
+ elif current:
130
+ runs.append("".join(current))
131
+ current = []
132
+ if current:
133
+ runs.append("".join(current))
134
+ return runs
135
+
136
+
137
+ def lexical_terms(text: str, *, max_terms: int) -> tuple[str, ...]:
138
+ """The terms `text` contributes to a lexical query, in the order they will be joined.
139
+
140
+ Deduplicated on the exact string, then sorted **longest-first with ties broken by first
141
+ occurrence**, then cut to `max_terms`. Longest-first is the right truncation for this corpus
142
+ rather than an arbitrary one: in prose about code the long tokens are the identifiers, error
143
+ strings and paths, while the terms it drops are short common words BM25 would have scored near
144
+ zero anyway.
145
+
146
+ Deduplication compares the text as written, and deliberately does not fold case: folding is
147
+ FTS5's, on both sides, and a second folding rule here is exactly the silent divergence the
148
+ quoted-literal design avoids. Two differently-cased occurrences of one word therefore spend two
149
+ of the `max_terms` slots, which costs a slot and cannot cost correctness.
150
+ """
151
+ seen: dict[str, int] = {}
152
+ for index, run in enumerate(_alphanumeric_runs(text)):
153
+ seen.setdefault(run, index)
154
+ ranked = sorted(seen.items(), key=lambda pair: (-len(pair[0]), pair[1]))
155
+ return tuple(term for term, _ in ranked[:max_terms])
156
+
157
+
158
+ def lexical_query(text: str, *, max_terms: int) -> LexicalQuery:
159
+ """`text` as an FTS5 `MATCH` expression, or a skipped arm if it yields no term.
160
+
161
+ The expression is `OR` over quoted terms, in selection order, and is always **bound as a
162
+ parameter** rather than formatted into SQL text. BM25 already ranks a document matching several
163
+ terms above one matching a single term, so `OR` loses nothing a phrase would gain — while a
164
+ phrase would require adjacency and so would silently drop every non-adjacent match on dotted
165
+ identifiers, paths and error strings, which is precisely the query shape this store serves.
166
+ """
167
+ terms = lexical_terms(text, max_terms=max_terms)
168
+ expression = " OR ".join(_quote(term) for term in terms) if terms else None
169
+ return LexicalQuery(terms=terms, expression=expression)
170
+
171
+
172
+ def internal_lexical_text(gist: str, content: str) -> str:
173
+ """The prose an internal query's lexical side is built from: the memory's own `gist + content`.
174
+
175
+ A function rather than a call-site concatenation because the 64-term cap does real work here —
176
+ a memory has far more terms than a prompt does — so which text the cap is applied to is part of
177
+ the contract rather than an incidental detail of one caller.
178
+ """
179
+ return f"{gist}{_INTERNAL_JOIN}{content}"
180
+
181
+
182
+ def _prefix_too_long(prefix_tokens: int, cap: int) -> ZikaronError:
183
+ return ZikaronError(
184
+ ErrorCode.BAD_CONFIG,
185
+ source=BadConfigSource.FILE,
186
+ file="<effective config>",
187
+ key="embedding.embed_prefix_query",
188
+ value=f"{prefix_tokens} tokens",
189
+ expected=f"a prefix leaving at least one token of the model's {cap}-token input for the "
190
+ "query itself — a longer prefix leaves no query to embed, so the store cannot be read",
191
+ )
192
+
193
+
194
+ def _prefix_leaves_no_query_token(prefix_tokens: int, cap: int) -> ZikaronError:
195
+ """`bad_config` for a prefix beside which not even this query's first token fits.
196
+
197
+ Distinct from `_prefix_too_long` only in how it was discovered: that one is arithmetic on the
198
+ two counts, this one is what the assembled sequence turned out to cost once the boundary between
199
+ them was tokenized. The claim is deliberately about *this* query rather than about every query,
200
+ because the boundary's cost depends on the text either side of it and a different first token
201
+ could fuse more cheaply. Same key, same remedy, both reachable only by configuration.
202
+ """
203
+ return ZikaronError(
204
+ ErrorCode.BAD_CONFIG,
205
+ source=BadConfigSource.FILE,
206
+ file="<effective config>",
207
+ key="embedding.embed_prefix_query",
208
+ value=f"{prefix_tokens} tokens",
209
+ expected="a prefix beside which this query's first token still fits once the two are "
210
+ f"tokenized together — assembled with even one of its tokens the sequence exceeds the "
211
+ f"model's {cap}-token input",
212
+ )
213
+
214
+
215
+ def _fit_to_cap(text: str, *, prefix: str, budget: int, encoder: Encoder) -> str:
216
+ """The nominal-budget head of `text`, shrunk until its **assembled** sequence fits the cap.
217
+
218
+ Called only once the whole text is known not to fit. The budget subtraction chooses the first
219
+ candidate; this is what settles it, by counting `prefix + head` as one string at each step
220
+ rather than adding the two pieces' counts. Each step drops one query token, so the loop
221
+ terminates, and its floor is one token: a prefix beside which not even this query's first token
222
+ fits is a configuration error rather than a query silently reduced to the prefix alone.
223
+
224
+ Not "the longest head that fits", and the distinction is worth stating because retokenization at
225
+ the boundary can *reduce* a count as well as raise it — so a head one token past the nominal
226
+ budget could in principle fit. The search deliberately does not look there: `retrieval.md` sets
227
+ the budget as what the query may keep, and going above it to reclaim a token the boundary
228
+ happened to absorb would make the kept length depend on the prefix in a way no field records.
229
+
230
+ Raises:
231
+ ZikaronError: `BAD_CONFIG` naming `embedding.embed_prefix_query` if not even this query's
232
+ first token fits beside the prefix.
233
+ """
234
+ for tokens in range(min(budget, encoder.count_tokens(text)), 0, -1):
235
+ kept = token_head(text, tokens=tokens, encoder=encoder)
236
+ if assembled_tokens(prefix, kept, encoder=encoder) <= encoder.max_sequence_tokens:
237
+ return kept
238
+ raise _prefix_leaves_no_query_token(encoder.count_tokens(prefix), encoder.max_sequence_tokens)
239
+
240
+
241
+ def external_query(text: str, *, encoder: Encoder, prefix: str, max_terms: int) -> ExternalQuery:
242
+ """Build both arms of a query from text outside the store, embedding it under the token budget.
243
+
244
+ The dense side is embedded through the deployed artifact with D20's BGE query-instruction
245
+ prefix; the lexical side is the term constructor above, over the **whole** text whether or not
246
+ the dense side had to be truncated.
247
+
248
+ Blocking work — one embedding call — so a caller on an event loop is responsible for keeping
249
+ this off it, exactly as the write path's `embed_chunks` is.
250
+
251
+ Args:
252
+ text: the query as the caller received it: a user prompt, an agent's `search` string, or
253
+ consolidation's group-gist concatenation.
254
+ encoder: the deployed artifact. Its tokenizer counts, and its model embeds.
255
+ prefix: `embed_prefix_query`. Query-side only, per BGE's asymmetric convention; the empty
256
+ string disables it.
257
+ max_terms: `fts_query_max_terms`.
258
+
259
+ Raises:
260
+ ZikaronError: `BAD_CONFIG` if the configured prefix leaves no room for the query — either by
261
+ the budget arithmetic or once the two are tokenized as one string — or if the artifact's
262
+ token count and token spans disagree.
263
+ """
264
+ cap = encoder.max_sequence_tokens
265
+ budget = cap - encoder.n_special_tokens - encoder.count_tokens(prefix)
266
+ if budget < 1:
267
+ raise _prefix_too_long(encoder.count_tokens(prefix), cap)
268
+
269
+ query_tokens = encoder.count_tokens(text)
270
+ # The assembled sequence is counted first, so the common path pays one tokenizer call and the
271
+ # search below runs only when the input genuinely overflows. Checking it this way also keeps
272
+ # `query_truncated` exact: a head slices from the first token's start to the last token's end,
273
+ # so calling it on a query that already fits would drop leading or trailing punctuation and
274
+ # report a truncation that did not happen.
275
+ fits_whole = assembled_tokens(prefix, text, encoder=encoder) <= cap
276
+ kept = text if fits_whole else _fit_to_cap(text, prefix=prefix, budget=budget, encoder=encoder)
277
+ (vector,) = encoder.embed([f"{prefix}{kept}"])
278
+ return ExternalQuery(
279
+ prepared=PreparedQuery(
280
+ origin=QueryOrigin.EXTERNAL,
281
+ vector=serialize(vector),
282
+ lexical=lexical_query(text, max_terms=max_terms),
283
+ ),
284
+ query_tokens=query_tokens,
285
+ # What was dropped, not what the budget subtraction predicted would be dropped: the two
286
+ # differ exactly when the prefix boundary retokenizes, which is the case this preflight now
287
+ # settles by counting the assembled string instead of the sum of its parts.
288
+ query_truncated=kept != text,
289
+ )
290
+
291
+
292
+ async def internal_query(
293
+ db: aiosqlite.Connection, *, memory_uuid: str, max_terms: int
294
+ ) -> PreparedQuery:
295
+ """Build both arms of one memory's query against the others, reading no model at all.
296
+
297
+ The dense side is the memory's **stored first-chunk embedding** (`part_index = 0`): it already
298
+ exists, it is already normalized, it cannot overflow the model's input because the write-side
299
+ preflight proved it fits, and reusing it is what makes planning a pure function of the store
300
+ rather than of a second inference. The lexical side is the memory's own `gist + content` through
301
+ the identical constructor an external query uses, where the 64-term cap does real work.
302
+
303
+ Assumes the caller's own open transaction: an internal query's whole point is to run inside the
304
+ write or planning transaction that produced the row it is querying from.
305
+
306
+ Raises:
307
+ ZikaronError: `NOT_FOUND` if `memory_uuid` names no row, or names one with no `part_index=0`
308
+ chunk. Invariant 12 makes the second impossible for a row with content, so it is
309
+ reported rather than papered over with a second embed call.
310
+ """
311
+ rows = await db.execute_fetchall(
312
+ "SELECT m.gist, m.content, v.embedding "
313
+ "FROM memory m "
314
+ "JOIN memory_chunk c ON c.memory_uuid = m.uuid AND c.part_index = 0 "
315
+ "JOIN memory_vec v ON v.rowid = c.chunk_id "
316
+ "WHERE m.uuid = ?",
317
+ (memory_uuid,),
318
+ )
319
+ found = list(rows)
320
+ if not found:
321
+ raise ZikaronError(ErrorCode.NOT_FOUND, uuid=memory_uuid)
322
+ gist, content, embedding = found[0]
323
+ return PreparedQuery(
324
+ origin=QueryOrigin.INTERNAL,
325
+ vector=bytes(embedding),
326
+ lexical=lexical_query(internal_lexical_text(str(gist), str(content)), max_terms=max_terms),
327
+ )
@@ -0,0 +1,260 @@
1
+ """Fusion, demotion, the total order, and the supersession repair — all pure, no store.
2
+
3
+ `retrieval.md` §"The fused score, written out", §"Total order" and §"Supersession: eligible,
4
+ demoted, and labelled" are normative. Four steps, in this order, and the order is observable:
5
+
6
+ 1. **RRF.** `Σ 1 / (rrf_k + rank)` over the arms that returned the row, ranks 1-based. An arm that
7
+ did not return it contributes nothing — not a term at a notional worst rank.
8
+ 2. **Penalties.** A demoted row's fused score is multiplied by `supersession_penalty` or
9
+ `retired_penalty`. On the *immediate* edge, not on graph depth: nothing measured says a
10
+ twice-superseded record is more dangerous than a once-superseded one. The two states are mutually
11
+ exclusive, so the penalties never compound.
12
+ 3. **The total order**, five steps deep, because fused ties are pervasive rather than rare —
13
+ measured: every hybrid query has at least one exact tie, and 12.5% have one *inside the top
14
+ five*, where order fell to whichever arm happened to be appended first.
15
+ 4. **The supersession repair**, a stable topological pass that puts every retrieved replacement
16
+ ahead of every record it replaced.
17
+
18
+ The caller cuts to `limit` **after** all four, which is what lets the repair promote a replacement
19
+ into the returned set rather than merely reshuffling what already made the cut. That is the case the
20
+ measurement is about: a superseded record outranks its own replacement 28-50% of the time on the
21
+ polarity stubs, and in those cases the correct answer is provably present and losing.
22
+
23
+ Nothing here touches the store, so determinism is testable by calling it twice.
24
+ """
25
+
26
+ from collections.abc import Iterable, Mapping, Sequence
27
+ from dataclasses import dataclass
28
+ from typing import Final
29
+
30
+ from zikaron.core.errors import ErrorCode, RowState, ZikaronError
31
+ from zikaron.core.events import Demotion
32
+ from zikaron.core.records.memory import Tier
33
+ from zikaron.core.records.supersession import BadSupersessionReason
34
+ from zikaron.core.retrieval.arms import ArmOutcome, ArmRow
35
+
36
+ #: Which tier wins a fused tie: consolidated prose has been reviewed by the consolidator, and
37
+ #: `~/Memory` reached the same tiebreak independently ("cohesive memory over raw journal line").
38
+ _TIER_ORDER: Final[Mapping[Tier, int]] = {Tier.LONG_TERM: 0, Tier.JOURNAL: 1}
39
+
40
+ #: The recency component for a long-term row. Step 4 is D27's *journal-local* recency tiebreak, so a
41
+ #: long-term row skips it and falls through to `uuid` — nothing about a long-term record's age is a
42
+ #: claim about its truth. A constant is well defined here precisely because step 3 has already
43
+ #: partitioned the remaining ties by tier: two rows still tied at step 4 share a tier, so this value
44
+ #: is only ever compared against another long-term row's copy of it.
45
+ _NO_RECENCY: Final = -1
46
+
47
+
48
+ @dataclass(frozen=True, slots=True)
49
+ class PoolRow:
50
+ """One candidate's row facts — exactly the columns the read path needs, and no more.
51
+
52
+ `content` and `version` are deliberately absent. `search` returns neither (a version would imply
53
+ a licence to write that the agent has not earned), the injected block shows only the gist, and
54
+ selecting long text for every row of a fused pool would spend I/O on the latency-critical path
55
+ for prose nobody reads.
56
+ """
57
+
58
+ uuid: str
59
+ tier: Tier
60
+ gist: str
61
+ active: bool
62
+ superseded_by: str | None
63
+ created_at: str
64
+ updated_at: str
65
+
66
+ @property
67
+ def state(self) -> RowState:
68
+ """This row's display state, per `schema.md` §"Retrieval eligibility"'s three states."""
69
+ if self.active:
70
+ return RowState.LIVE
71
+ if self.superseded_by is not None:
72
+ return RowState.SUPERSEDED
73
+ return RowState.RETIRED
74
+
75
+ @property
76
+ def demotion(self) -> Demotion | None:
77
+ """Why this row is demoted, or `None` for a live row that is not."""
78
+ if self.active:
79
+ return None
80
+ return Demotion.SUPERSEDED if self.superseded_by is not None else Demotion.RETIRED
81
+
82
+
83
+ @dataclass(frozen=True, slots=True)
84
+ class Penalties:
85
+ """The two demotion multipliers, which are separate config keys with equal defaults.
86
+
87
+ Equal by honesty rather than by symmetry: nothing measured distinguishes a superseded row from
88
+ an outright-retired one, so they start equal and can diverge once something does.
89
+ """
90
+
91
+ superseded: float
92
+ retired: float
93
+
94
+ def factor(self, demotion: Demotion | None) -> float:
95
+ """The multiplier for one row's demotion state — 1.0 for a row that is not demoted."""
96
+ if demotion is None:
97
+ return 1.0
98
+ return self.superseded if demotion is Demotion.SUPERSEDED else self.retired
99
+
100
+
101
+ @dataclass(frozen=True, slots=True)
102
+ class RankedMemory:
103
+ """One candidate, scored and placed: the row, its fused score, and where each arm found it.
104
+
105
+ `best_rank` is the minimum over the arms that returned it, so a row only one arm found is ranked
106
+ on that arm's number rather than penalized for the other's silence. `best_distance` is the dense
107
+ arm's rolled-up minimum, kept because the three cosine cutoffs are read off it — and only
108
+ meaningful for a query whose vector is unit length, which `query.QueryOrigin` is what tracks.
109
+ """
110
+
111
+ row: PoolRow
112
+ fused_score: float
113
+ dense_rank: int | None
114
+ lexical_rank: int | None
115
+ best_distance: float | None
116
+
117
+ @property
118
+ def best_rank(self) -> int:
119
+ """The best rank any arm gave this row. Both arms cannot be absent: it would not be here."""
120
+ ranks = [rank for rank in (self.dense_rank, self.lexical_rank) if rank is not None]
121
+ return min(ranks)
122
+
123
+ @property
124
+ def demoted(self) -> bool:
125
+ """Whether this row was demoted, which is exactly whether it is inactive."""
126
+ return self.row.demotion is not None
127
+
128
+
129
+ def rrf_score(ranks: Iterable[int], *, rrf_k: int) -> float:
130
+ """Reciprocal Rank Fusion over the ranks one row achieved, in the arms that returned it.
131
+
132
+ `Σ 1 / (rrf_k + rank)`, ranks **1-based** — the formula every quality figure in `retrieval.md`
133
+ was measured through. Rebasing to 0 would leave `rrf_k = 60` naming a different denominator, so
134
+ the config default and the measurements would silently stop describing the same pipeline.
135
+ """
136
+ return sum(1.0 / (rrf_k + rank) for rank in ranks)
137
+
138
+
139
+ def _by_uuid(outcome: ArmOutcome) -> Mapping[str, ArmRow]:
140
+ return {row.uuid: row for row in outcome.rows}
141
+
142
+
143
+ def _journal_recency(rows: Iterable[PoolRow]) -> Mapping[str, int]:
144
+ """Each journal row's position when the pool's timestamps are ordered newest first.
145
+
146
+ An index rather than the timestamp itself, so step 4 needs no string inversion and no timestamp
147
+ parsing — `created_at` is opaque text this layer never has to interpret, and a mapping from
148
+ distinct value to descending position is monotone, so comparing indices ascending *is* comparing
149
+ timestamps descending. Ties on the timestamp share an index and fall through to `uuid`, which is
150
+ what step 5 is for.
151
+ """
152
+ journal = [row for row in rows if row.tier is Tier.JOURNAL]
153
+ descending = sorted({row.created_at for row in journal}, reverse=True)
154
+ position = {value: index for index, value in enumerate(descending)}
155
+ return {row.uuid: position[row.created_at] for row in journal}
156
+
157
+
158
+ def _order_key(
159
+ ranked: RankedMemory, recency: Mapping[str, int]
160
+ ) -> tuple[float, int, int, int, str]:
161
+ """`retrieval.md`'s five-step total order, as one comparable tuple.
162
+
163
+ Ascending on every component, so the fused score is negated. The components are, in order: the
164
+ fused score descending; the best rank any arm gave it ascending, since a document one arm ranked
165
+ first is a better bet than one both arms ranked tenth; tier, long-term first; `created_at`
166
+ descending **inside a journal block only**; and `uuid`, which is arbitrary and is the point —
167
+ it guarantees a total order, so the same store and query always produce the same list.
168
+ """
169
+ return (
170
+ -ranked.fused_score,
171
+ ranked.best_rank,
172
+ _TIER_ORDER[ranked.row.tier],
173
+ recency.get(ranked.row.uuid, _NO_RECENCY),
174
+ ranked.row.uuid,
175
+ )
176
+
177
+
178
+ def _repair_supersession(ordered: Sequence[RankedMemory]) -> tuple[RankedMemory, ...]:
179
+ """Reorder so every retrieved replacement precedes every record it replaced.
180
+
181
+ Not "immediately above" — that rule is unsatisfiable, because `merge` absorbing `A` and `B` into
182
+ `C` writes `A→C` *and* `B→C`, so no linear order can put `C` immediately above both. The
183
+ realizable rule is precedence, enforced as a stable topological pass: scan in order and emit the
184
+ first not-yet-emitted row whose `superseded_by` target is either absent from the pool or already
185
+ emitted. It handles `A→B→C` transitively, giving `C, B, A`.
186
+
187
+ Adjacency is neither preserved nor needed: the injected block gives the agent the replacement's
188
+ uuid explicitly, which is a stronger cue than proximity and is the part that survives a group of
189
+ rows sharing one replacement.
190
+
191
+ Raises:
192
+ ZikaronError: `BAD_SUPERSESSION` with `reason='cycle'` if no row is emittable, which
193
+ invariant 6 makes impossible — every row waits on at most one other and the graph is
194
+ acyclic, so the wait-for relation is a forest. Refused rather than worked around: a
195
+ fallback to the pre-repair order would answer from a corrupted graph with a *plausible*
196
+ list, and putting the replacement first is the whole point of the pass.
197
+ """
198
+ in_pool = {ranked.row.uuid for ranked in ordered}
199
+ waiting = list(ordered)
200
+ emitted: list[RankedMemory] = []
201
+ emitted_uuids: set[str] = set()
202
+ while waiting:
203
+ for index, ranked in enumerate(waiting):
204
+ target = ranked.row.superseded_by
205
+ if target is None or target not in in_pool or target in emitted_uuids:
206
+ emitted.append(waiting.pop(index))
207
+ emitted_uuids.add(ranked.row.uuid)
208
+ break
209
+ else:
210
+ stalled = waiting[0].row
211
+ raise ZikaronError(
212
+ ErrorCode.BAD_SUPERSESSION,
213
+ uuid=stalled.uuid,
214
+ target=stalled.superseded_by,
215
+ reason=BadSupersessionReason.CYCLE,
216
+ )
217
+ return tuple(emitted)
218
+
219
+
220
+ def rank(
221
+ *,
222
+ dense: ArmOutcome,
223
+ lexical: ArmOutcome,
224
+ rows: Mapping[str, PoolRow],
225
+ rrf_k: int,
226
+ penalties: Penalties,
227
+ ) -> tuple[RankedMemory, ...]:
228
+ """Fuse both arms into one ordered pool: RRF, penalties, the total order, then the repair.
229
+
230
+ The returned pool is **not** cut to any output budget; the caller cuts it, because the repair
231
+ must run first if it is to promote a replacement into the returned set.
232
+
233
+ Args:
234
+ rows: the row facts for every uuid either arm returned. A uuid missing from this mapping is
235
+ a row that vanished between the arm query and the pool load, which one read transaction
236
+ makes impossible; it is dropped rather than guessed at.
237
+ """
238
+ dense_rows = _by_uuid(dense)
239
+ lexical_rows = _by_uuid(lexical)
240
+ scored: list[RankedMemory] = []
241
+ for uuid in [*dense_rows, *(uuid for uuid in lexical_rows if uuid not in dense_rows)]:
242
+ row = rows.get(uuid)
243
+ if row is None:
244
+ continue
245
+ found = [arm[uuid] for arm in (dense_rows, lexical_rows) if uuid in arm]
246
+ fused = rrf_score((arm_row.rank for arm_row in found), rrf_k=rrf_k)
247
+ dense_row = dense_rows.get(uuid)
248
+ lexical_row = lexical_rows.get(uuid)
249
+ scored.append(
250
+ RankedMemory(
251
+ row=row,
252
+ fused_score=fused * penalties.factor(row.demotion),
253
+ dense_rank=None if dense_row is None else dense_row.rank,
254
+ lexical_rank=None if lexical_row is None else lexical_row.rank,
255
+ best_distance=None if dense_row is None else dense_row.distance,
256
+ )
257
+ )
258
+ recency = _journal_recency(ranked.row for ranked in scored)
259
+ ordered = sorted(scored, key=lambda ranked: _order_key(ranked, recency))
260
+ return _repair_supersession(ordered)