zikaron 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- zikaron/__init__.py +1 -0
- zikaron/cli/__init__.py +1 -0
- zikaron/cli/main.py +114 -0
- zikaron/core/__init__.py +1 -0
- zikaron/core/clock.py +78 -0
- zikaron/core/config/__init__.py +1 -0
- zikaron/core/config/keys.py +395 -0
- zikaron/core/config/resolution.py +267 -0
- zikaron/core/consolidation/__init__.py +1 -0
- zikaron/core/consolidation/authorization.py +316 -0
- zikaron/core/consolidation/candidates.py +147 -0
- zikaron/core/consolidation/context.py +166 -0
- zikaron/core/consolidation/grouping.py +383 -0
- zikaron/core/consolidation/groups.py +490 -0
- zikaron/core/consolidation/payload.py +246 -0
- zikaron/core/consolidation/planning.py +192 -0
- zikaron/core/consolidation/rowstate.py +68 -0
- zikaron/core/consolidation/runs.py +306 -0
- zikaron/core/consolidation/serving.py +462 -0
- zikaron/core/consolidation/verbs.py +500 -0
- zikaron/core/errors.py +355 -0
- zikaron/core/events.py +748 -0
- zikaron/core/indexing/__init__.py +1 -0
- zikaron/core/indexing/acquisition.py +255 -0
- zikaron/core/indexing/chunking.py +368 -0
- zikaron/core/indexing/encoder.py +537 -0
- zikaron/core/indexing/lexical.py +86 -0
- zikaron/core/indexing/model_cache.py +93 -0
- zikaron/core/indexing/model_pin.py +89 -0
- zikaron/core/indexing/vectors.py +223 -0
- zikaron/core/indexing/writes.py +461 -0
- zikaron/core/knowledge/__init__.py +5 -0
- zikaron/core/knowledge/arms.py +104 -0
- zikaron/core/knowledge/builds.py +204 -0
- zikaron/core/knowledge/candidates.py +130 -0
- zikaron/core/knowledge/changes.py +175 -0
- zikaron/core/knowledge/chunking.py +376 -0
- zikaron/core/knowledge/counters.py +228 -0
- zikaron/core/knowledge/database.py +380 -0
- zikaron/core/knowledge/ddl.py +196 -0
- zikaron/core/knowledge/disposal.py +213 -0
- zikaron/core/knowledge/errors.py +166 -0
- zikaron/core/knowledge/files.py +202 -0
- zikaron/core/knowledge/git.py +385 -0
- zikaron/core/knowledge/groups.py +450 -0
- zikaron/core/knowledge/lexical.py +64 -0
- zikaron/core/knowledge/lifecycle.py +418 -0
- zikaron/core/knowledge/lock.py +277 -0
- zikaron/core/knowledge/meta.py +393 -0
- zikaron/core/knowledge/paths.py +55 -0
- zikaron/core/knowledge/pending.py +59 -0
- zikaron/core/knowledge/registry.py +264 -0
- zikaron/core/knowledge/repair.py +152 -0
- zikaron/core/knowledge/reporting.py +436 -0
- zikaron/core/knowledge/roots.py +91 -0
- zikaron/core/knowledge/scan.py +429 -0
- zikaron/core/knowledge/search.py +346 -0
- zikaron/core/knowledge/state.py +174 -0
- zikaron/core/knowledge/text.py +166 -0
- zikaron/core/knowledge/vectors.py +102 -0
- zikaron/core/knowledge/walk.py +264 -0
- zikaron/core/knowledge/writes.py +127 -0
- zikaron/core/records/__init__.py +1 -0
- zikaron/core/records/memory.py +961 -0
- zikaron/core/records/receipts.py +161 -0
- zikaron/core/records/supersession.py +221 -0
- zikaron/core/retrieval/__init__.py +1 -0
- zikaron/core/retrieval/arms.py +318 -0
- zikaron/core/retrieval/block.py +107 -0
- zikaron/core/retrieval/eligibility.py +164 -0
- zikaron/core/retrieval/query.py +327 -0
- zikaron/core/retrieval/ranking.py +260 -0
- zikaron/core/retrieval/reads.py +294 -0
- zikaron/core/retrieval/retrieve.py +158 -0
- zikaron/core/retrieval/similarity.py +87 -0
- zikaron/core/signals/__init__.py +34 -0
- zikaron/core/signals/contention.py +106 -0
- zikaron/core/signals/dedup.py +201 -0
- zikaron/core/signals/horizon.py +47 -0
- zikaron/core/signals/repair.py +161 -0
- zikaron/core/signals/retirement.py +83 -0
- zikaron/core/signals/sessions.py +105 -0
- zikaron/core/signals/writes.py +200 -0
- zikaron/core/store/__init__.py +1 -0
- zikaron/core/store/connection.py +202 -0
- zikaron/core/store/ddl.py +215 -0
- zikaron/core/store/embedder.py +45 -0
- zikaron/core/store/meta.py +152 -0
- zikaron/core/store/permissions.py +160 -0
- zikaron/core/store/store.py +408 -0
- zikaron/core/store/transactions.py +181 -0
- zikaron/core/write/__init__.py +33 -0
- zikaron/core/write/dedup.py +145 -0
- zikaron/core/write/tools.py +290 -0
- zikaron/doctor/__init__.py +1 -0
- zikaron/doctor/checks.py +220 -0
- zikaron/doctor/main.py +64 -0
- zikaron/harness/__init__.py +1 -0
- zikaron/harness/detect.py +92 -0
- zikaron/harness/spec.py +320 -0
- zikaron/hook/__init__.py +1 -0
- zikaron/hook/connect.py +379 -0
- zikaron/hook/envelope.py +106 -0
- zikaron/hook/failure.py +104 -0
- zikaron/hook/limits.py +61 -0
- zikaron/hook/main.py +118 -0
- zikaron/hook/push.py +183 -0
- zikaron/hook/rpc.py +85 -0
- zikaron/hook/spawn_warm.py +81 -0
- zikaron/hook/subagent_policy.py +57 -0
- zikaron/hook/tripwire.py +54 -0
- zikaron/hook/warm_helper.py +137 -0
- zikaron/hook/write_policy.py +319 -0
- zikaron/install/__init__.py +4 -0
- zikaron/install/__main__.py +18 -0
- zikaron/install/assets.py +394 -0
- zikaron/install/entries.py +370 -0
- zikaron/install/harness.py +185 -0
- zikaron/install/main.py +375 -0
- zikaron/install/targets.py +789 -0
- zikaron/install/writer.py +973 -0
- zikaron/knowledge/__init__.py +1 -0
- zikaron/knowledge/__main__.py +17 -0
- zikaron/knowledge/indexer/__init__.py +1 -0
- zikaron/knowledge/indexer/__main__.py +17 -0
- zikaron/knowledge/indexer/detach.py +83 -0
- zikaron/knowledge/indexer/main.py +187 -0
- zikaron/knowledge/main.py +466 -0
- zikaron/knowledge/scope.py +133 -0
- zikaron/mcp/__init__.py +6 -0
- zikaron/mcp/connection.py +583 -0
- zikaron/mcp/consolidator.py +316 -0
- zikaron/mcp/errors.py +73 -0
- zikaron/mcp/main.py +66 -0
- zikaron/mcp/primary.py +420 -0
- zikaron/mcp/server.py +96 -0
- zikaron/mcp/spill.py +328 -0
- zikaron/mcp/tool_names.py +67 -0
- zikaron/py.typed +0 -0
- zikaron/service/__init__.py +1 -0
- zikaron/service/asyncio_compat.py +126 -0
- zikaron/service/context.py +251 -0
- zikaron/service/dispatch.py +332 -0
- zikaron/service/dispatch_consolidation.py +397 -0
- zikaron/service/dispatch_knowledge.py +469 -0
- zikaron/service/envelope.py +166 -0
- zikaron/service/lifecycle.py +467 -0
- zikaron/service/log.py +96 -0
- zikaron/service/main.py +531 -0
- zikaron/service/params.py +168 -0
- zikaron/service/paths.py +181 -0
- zikaron/service/rpc.py +176 -0
- zikaron/service/security.py +156 -0
- zikaron/service/serialize.py +204 -0
- zikaron/service/serialize_knowledge.py +238 -0
- zikaron/service/server.py +416 -0
- zikaron-0.1.0.dist-info/METADATA +770 -0
- zikaron-0.1.0.dist-info/RECORD +162 -0
- zikaron-0.1.0.dist-info/WHEEL +5 -0
- zikaron-0.1.0.dist-info/entry_points.txt +4 -0
- zikaron-0.1.0.dist-info/licenses/LICENSE +21 -0
- zikaron-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,383 @@
|
|
|
1
|
+
"""D29's grouping algorithm: anchor by retrieval, cluster the orphans, then partition for cohesion.
|
|
2
|
+
|
|
3
|
+
`design/consolidation.md` §"Mechanism: retrieval is the adjacency function, everywhere" is
|
|
4
|
+
normative.
|
|
5
|
+
|
|
6
|
+
**Planning is a pure function of the store plus the effective config.** Same store, same parameters,
|
|
7
|
+
same groups, every time — which is not decoration: a nondeterministic partition changes what the
|
|
8
|
+
model is asked, so two runs could reach different long-term records from identical inputs and
|
|
9
|
+
neither would be reproducible. Everything here is therefore either a total order or a deterministic
|
|
10
|
+
scan, and no step consults a clock, a hash seed or an iteration order the store does not fix.
|
|
11
|
+
|
|
12
|
+
**The two halves are not symmetric, and that asymmetry is the design.** Journal → long-term
|
|
13
|
+
adjacency *is* retrieval, needing only a rank scan and a floor. Journal → journal grouping is
|
|
14
|
+
clustering, which needs a threshold, a linkage rule and an ordering — and is the half that gets
|
|
15
|
+
dropped. It cannot be avoided, only shrunk: as the long-term tier fills, most entries anchor and
|
|
16
|
+
never reach the clustering code at all.
|
|
17
|
+
|
|
18
|
+
**Mutual-K does not prevent chaining.** Connected components over mutual edges are still single
|
|
19
|
+
linkage over those edges, so `A↔B↔C` is one component with or without mutuality: mutuality raises
|
|
20
|
+
the bar for an *edge*, and a chain is built from edges that each clear it. The complete-linkage
|
|
21
|
+
pass is what prevents it, which is why the `A~B`, `B~C`, `A≁C` case is an acceptance test *of that
|
|
22
|
+
pass*. Complete linkage over-splits, deliberately: an over-split costs one extra consolidator call,
|
|
23
|
+
because groups are processed oldest-first against a store that updates as it goes and a later group
|
|
24
|
+
can still merge into what an earlier one created — while an under-split fuses two unrelated lessons
|
|
25
|
+
into one record that is false in both directions.
|
|
26
|
+
|
|
27
|
+
**No embedding call happens here.** Both arms of every query in this module are internal: the dense
|
|
28
|
+
side reuses the querying row's stored first-chunk vector and the lexical side is its own
|
|
29
|
+
`gist + content` through the same term constructor an external query uses. That is what lets
|
|
30
|
+
planning run inside one transaction and be a function of the store rather than of a second
|
|
31
|
+
inference.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
from collections.abc import Mapping, Sequence
|
|
35
|
+
from dataclasses import dataclass
|
|
36
|
+
from typing import Final
|
|
37
|
+
|
|
38
|
+
import aiosqlite
|
|
39
|
+
|
|
40
|
+
from zikaron.core.consolidation.context import ConsolidationCall
|
|
41
|
+
from zikaron.core.consolidation.groups import Shard
|
|
42
|
+
from zikaron.core.retrieval.eligibility import Consumer, Scope
|
|
43
|
+
from zikaron.core.retrieval.query import internal_query
|
|
44
|
+
from zikaron.core.retrieval.retrieve import retrieve
|
|
45
|
+
from zikaron.core.retrieval.similarity import directed_cosines
|
|
46
|
+
|
|
47
|
+
#: What separates the two halves of `consolidation_group.order_key`. Any character outside the
|
|
48
|
+
#: timestamp's own alphabet would do; `|` is the literal the schema's column comment names.
|
|
49
|
+
_ORDER_KEY_SEPARATOR: Final = "|"
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
@dataclass(frozen=True, slots=True)
|
|
53
|
+
class Entry:
|
|
54
|
+
"""One unconsolidated journal row as the planner sees it: its identity, order and version.
|
|
55
|
+
|
|
56
|
+
`version` is the plan-time version, which becomes `consolidation_group_member.version_seen` —
|
|
57
|
+
change detection only. The prose is deliberately absent: the planner decides *which* rows group
|
|
58
|
+
together, and the payload that carries prose is rebuilt at serve time from rows that may have
|
|
59
|
+
moved since.
|
|
60
|
+
"""
|
|
61
|
+
|
|
62
|
+
uuid: str
|
|
63
|
+
created_at: str
|
|
64
|
+
version: int
|
|
65
|
+
|
|
66
|
+
@property
|
|
67
|
+
def order(self) -> tuple[str, str]:
|
|
68
|
+
"""This entry's position in group order: `(created_at, uuid)`."""
|
|
69
|
+
return (self.created_at, self.uuid)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
@dataclass(frozen=True, slots=True)
|
|
73
|
+
class PlannedGroup:
|
|
74
|
+
"""One group as the planner produced it, before anything is written.
|
|
75
|
+
|
|
76
|
+
A value type rather than rows inserted as they are computed, so that the whole partition exists
|
|
77
|
+
before any of it is persisted — which is what lets invariant 19's set-level condition be a
|
|
78
|
+
property of a computed plan rather than of a half-written table.
|
|
79
|
+
"""
|
|
80
|
+
|
|
81
|
+
anchor_uuid: str | None
|
|
82
|
+
order_key: str
|
|
83
|
+
shard: Shard
|
|
84
|
+
members: tuple[Entry, ...]
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def order_key(members: Sequence[Entry]) -> str:
|
|
88
|
+
"""`'earliest_created_at|min_uuid'` of a **pre-shard** group, per the schema's column comment.
|
|
89
|
+
|
|
90
|
+
Computed over the pre-shard members so every shard of one group carries the same key, sorts
|
|
91
|
+
adjacently, and is ordered among its siblings by `shard_index` — which is the only reason step
|
|
92
|
+
3's total order needs a third component at all.
|
|
93
|
+
|
|
94
|
+
It also identifies the pre-shard group, which is what invariant 19 rests on: cohesive subgroups
|
|
95
|
+
and anchored member sets are disjoint, so no two of them in one run can share a minimum uuid.
|
|
96
|
+
|
|
97
|
+
Raises:
|
|
98
|
+
ValueError: `members` is empty. A group with no members has no order and could not be
|
|
99
|
+
served; refusing to name one is what keeps an empty group from being written at all.
|
|
100
|
+
"""
|
|
101
|
+
if not members:
|
|
102
|
+
raise ValueError("a group with no members has no order key")
|
|
103
|
+
earliest = min(member.created_at for member in members)
|
|
104
|
+
smallest = min(member.uuid for member in members)
|
|
105
|
+
return f"{earliest}{_ORDER_KEY_SEPARATOR}{smallest}"
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def shard(
|
|
109
|
+
members: Sequence[Entry], *, group_max: int
|
|
110
|
+
) -> tuple[tuple[Shard, tuple[Entry, ...]], ...]:
|
|
111
|
+
"""Split one pre-shard group into consecutive shards of at most `group_max` members.
|
|
112
|
+
|
|
113
|
+
"Consecutive in the group's own order" is the whole partition rule, so `members` must already be
|
|
114
|
+
in group order — the same order the caller computed `order_key` from — and the same group always
|
|
115
|
+
shards the same way. Indices are 1-based and `of` is the total, so an unsplit group is
|
|
116
|
+
`{index: 1, of: 1}` rather than a third convention meaning "not sharded".
|
|
117
|
+
|
|
118
|
+
Both numbers are returned, and both are persisted by the caller, because `of` cannot be
|
|
119
|
+
recovered later: recomputing it would mean replanning a group whose membership is frozen,
|
|
120
|
+
against a store that has since moved.
|
|
121
|
+
"""
|
|
122
|
+
total = -(-len(members) // group_max)
|
|
123
|
+
return tuple(
|
|
124
|
+
(
|
|
125
|
+
Shard(index=index, of=total),
|
|
126
|
+
tuple(members[start : start + group_max]),
|
|
127
|
+
)
|
|
128
|
+
for index, start in enumerate(range(0, len(members), group_max), start=1)
|
|
129
|
+
)
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
async def journal_entries(db: aiosqlite.Connection) -> tuple[Entry, ...]:
|
|
133
|
+
"""Every unconsolidated journal row, in group order — the universe this pass partitions.
|
|
134
|
+
|
|
135
|
+
Not one of `eligibility`'s five consumers, and deliberately not routed through it: those five
|
|
136
|
+
are *candidate pools* narrowed by the legality condition of the action they are retrieved for,
|
|
137
|
+
while this is the set of rows being *assigned* to groups. The predicate is nonetheless
|
|
138
|
+
identical to `Consumer.ORPHAN`'s narrowing, because both express the same fact — an
|
|
139
|
+
unconsolidated journal row is exactly what can become a group member — and a test asserts the
|
|
140
|
+
two agree on a fixture holding rows in every state, so the two statements cannot drift apart.
|
|
141
|
+
"""
|
|
142
|
+
rows = await db.execute_fetchall(
|
|
143
|
+
"SELECT uuid, created_at, version FROM memory "
|
|
144
|
+
"WHERE tier = 'journal' AND active = 1 "
|
|
145
|
+
"ORDER BY created_at ASC, uuid ASC"
|
|
146
|
+
)
|
|
147
|
+
return tuple(
|
|
148
|
+
Entry(uuid=str(uuid), created_at=str(created_at), version=int(str(version)))
|
|
149
|
+
for uuid, created_at, version in rows
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
async def anchor_for(
|
|
154
|
+
db: aiosqlite.Connection, entry: Entry, *, call: ConsolidationCall
|
|
155
|
+
) -> str | None:
|
|
156
|
+
"""The long-term record `entry` anchors to, or `None` if none clears `anchor_cutoff`.
|
|
157
|
+
|
|
158
|
+
The hybrid decides *which* records are candidates and in what order; the floor decides which of
|
|
159
|
+
them is close enough. The scan takes the **highest-ranked record that clears the floor** rather
|
|
160
|
+
than gating rank 1 alone — `consolidation.md` step 1 fixes that reading, because RRF fuses two
|
|
161
|
+
arms while `s` is a dense quantity, so a record the lexical arm alone surfaced can outrank one
|
|
162
|
+
with a far higher directed cosine, and the two readings give different partitions.
|
|
163
|
+
|
|
164
|
+
Candidates are restricted to `tier='long_term' AND active=1` (`Consumer.CONSOLIDATION`): every
|
|
165
|
+
record shown here is one `merge` may be asked to rewrite, and rewriting a historical row is
|
|
166
|
+
incoherent, since `superseded_by` is immutable once set.
|
|
167
|
+
|
|
168
|
+
Assumes the caller's own open transaction.
|
|
169
|
+
"""
|
|
170
|
+
query = await internal_query(
|
|
171
|
+
db, memory_uuid=entry.uuid, max_terms=call.retrieval.fts_query_max_terms
|
|
172
|
+
)
|
|
173
|
+
retrieved = await retrieve(
|
|
174
|
+
db, query=query, scope=Scope(Consumer.CONSOLIDATION), settings=call.retrieval
|
|
175
|
+
)
|
|
176
|
+
cosines = await directed_cosines(db, retrieved.pool, query=query)
|
|
177
|
+
for ranked in retrieved.pool:
|
|
178
|
+
cosine = cosines.get(ranked.row.uuid)
|
|
179
|
+
if cosine is not None and cosine >= call.settings.anchor_cutoff:
|
|
180
|
+
return ranked.row.uuid
|
|
181
|
+
return None
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
async def neighbours_of(
|
|
185
|
+
db: aiosqlite.Connection, entry: Entry, *, call: ConsolidationCall
|
|
186
|
+
) -> Mapping[str, float]:
|
|
187
|
+
"""`entry`'s top-`mutual_k` journal neighbours, each with `s(entry → neighbour)`.
|
|
188
|
+
|
|
189
|
+
The candidate pool is `tier='journal' AND active=1` excluding `entry` itself
|
|
190
|
+
(`Consumer.ORPHAN`) — **every** unconsolidated journal row, not only the orphans, because that
|
|
191
|
+
filter *is* the definition of a group member and adding a narrower one would be a filter the
|
|
192
|
+
design's own table does not list. The consequence is worth stating rather than discovering: an
|
|
193
|
+
entry that anchored can occupy one of the K ranks and can never become an edge, since edges are
|
|
194
|
+
drawn only between two orphans. Excluding self is not pedantry either — a memory is its own
|
|
195
|
+
nearest neighbour, so without it every row would spend a slot on itself.
|
|
196
|
+
|
|
197
|
+
Returns:
|
|
198
|
+
The top `mutual_k` of the fused pool, mapped to their directed cosines. Rank is what the cut
|
|
199
|
+
is by; the cosine is what the floor is against. A neighbour with no computable cosine — no
|
|
200
|
+
stored vector at all, which invariant 12 makes unreachable for an active row — is omitted
|
|
201
|
+
rather than admitted at an invented score.
|
|
202
|
+
"""
|
|
203
|
+
query = await internal_query(
|
|
204
|
+
db, memory_uuid=entry.uuid, max_terms=call.retrieval.fts_query_max_terms
|
|
205
|
+
)
|
|
206
|
+
retrieved = await retrieve(
|
|
207
|
+
db,
|
|
208
|
+
query=query,
|
|
209
|
+
scope=Scope(Consumer.ORPHAN, exclude_uuid=entry.uuid),
|
|
210
|
+
settings=call.retrieval,
|
|
211
|
+
)
|
|
212
|
+
cosines = await directed_cosines(db, retrieved.pool, query=query)
|
|
213
|
+
top = retrieved.pool[: call.settings.mutual_k]
|
|
214
|
+
return {
|
|
215
|
+
ranked.row.uuid: cosines[ranked.row.uuid] for ranked in top if ranked.row.uuid in cosines
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def mutual_edges(
|
|
220
|
+
neighbours: Mapping[str, Mapping[str, float]], *, cutoff: float
|
|
221
|
+
) -> Mapping[str, frozenset[str]]:
|
|
222
|
+
"""The undirected mutual-K graph over the orphans, as an adjacency map.
|
|
223
|
+
|
|
224
|
+
An edge `A↔B` exists only when **both** conditions hold: `A` is in `B`'s top-`mutual_k` *and*
|
|
225
|
+
`B` is in `A`'s, and `min(s(A → B), s(B → A)) ≥ cutoff`.
|
|
226
|
+
|
|
227
|
+
**The directionality is not pedantry.** `s(X → Y)` is X's first chunk against Y's *best* chunk,
|
|
228
|
+
so it is asymmetric whenever the two rows have different chunk counts — X offers one vector, Y
|
|
229
|
+
offers its best of many. A spec written as `cos(A, B)` would leave two conforming
|
|
230
|
+
implementations free to pick different directions, which changes which edges exist and
|
|
231
|
+
therefore which groups the consolidator is shown. `min` is the right symmetrization for the
|
|
232
|
+
same reason the test already demands mutuality: an edge should require agreement from both
|
|
233
|
+
endpoints, and the conservative direction is the safe one.
|
|
234
|
+
|
|
235
|
+
**The floor is not redundant with mutuality.** In a journal of six rows with `mutual_k = 5`
|
|
236
|
+
every row is in every other row's top five, so the mutual graph is complete and the whole
|
|
237
|
+
journal fuses into one group. Rank says *which* neighbours are closest; the floor says
|
|
238
|
+
*whether* they are close.
|
|
239
|
+
|
|
240
|
+
Args:
|
|
241
|
+
neighbours: each orphan's own top-K map, keyed by orphan uuid. A uuid appearing only as a
|
|
242
|
+
*value* is not an orphan — an entry that anchored, or one whose own map was never
|
|
243
|
+
computed — so it can never form an edge, which is what restricts the graph to the set
|
|
244
|
+
being clustered without a second filter.
|
|
245
|
+
"""
|
|
246
|
+
edges: dict[str, set[str]] = {uuid: set() for uuid in neighbours}
|
|
247
|
+
for source, reachable in neighbours.items():
|
|
248
|
+
for target, forward in reachable.items():
|
|
249
|
+
back = neighbours.get(target, {}).get(source)
|
|
250
|
+
if back is not None and min(forward, back) >= cutoff:
|
|
251
|
+
edges[source].add(target)
|
|
252
|
+
edges[target].add(source)
|
|
253
|
+
return {uuid: frozenset(adjacent) for uuid, adjacent in edges.items()}
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def components(
|
|
257
|
+
orphans: Sequence[Entry], edges: Mapping[str, frozenset[str]]
|
|
258
|
+
) -> tuple[tuple[Entry, ...], ...]:
|
|
259
|
+
"""Connected components of the mutual graph, each in group order, the components themselves too.
|
|
260
|
+
|
|
261
|
+
A cheap and correct *prefilter*: anything not in one component can never be grouped. It is not
|
|
262
|
+
the answer, because components are single linkage over the accepted edges — `A↔B` and `B↔C` put
|
|
263
|
+
`A` and `C` together even when they share no edge — which is what `cohesive_subgroups` then
|
|
264
|
+
partitions.
|
|
265
|
+
|
|
266
|
+
Deterministic by construction: the seed order is `orphans`' own group order and each component
|
|
267
|
+
is grown by scanning that same order, so no set iteration reaches the result.
|
|
268
|
+
"""
|
|
269
|
+
by_uuid = {entry.uuid: entry for entry in orphans}
|
|
270
|
+
seen: set[str] = set()
|
|
271
|
+
found: list[tuple[Entry, ...]] = []
|
|
272
|
+
for entry in orphans:
|
|
273
|
+
if entry.uuid in seen:
|
|
274
|
+
continue
|
|
275
|
+
member_uuids: list[str] = []
|
|
276
|
+
frontier = [entry.uuid]
|
|
277
|
+
seen.add(entry.uuid)
|
|
278
|
+
while frontier:
|
|
279
|
+
current = frontier.pop()
|
|
280
|
+
member_uuids.append(current)
|
|
281
|
+
for adjacent in sorted(edges.get(current, frozenset())):
|
|
282
|
+
if adjacent not in seen and adjacent in by_uuid:
|
|
283
|
+
seen.add(adjacent)
|
|
284
|
+
frontier.append(adjacent)
|
|
285
|
+
found.append(tuple(sorted((by_uuid[uuid] for uuid in member_uuids), key=lambda e: e.order)))
|
|
286
|
+
return tuple(found)
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
def cohesive_subgroups(
|
|
290
|
+
component: Sequence[Entry], edges: Mapping[str, frozenset[str]]
|
|
291
|
+
) -> tuple[tuple[Entry, ...], ...]:
|
|
292
|
+
"""Partition one component by **complete linkage**, deterministically.
|
|
293
|
+
|
|
294
|
+
Order members by `(created_at, uuid)`; seed a subgroup with the earliest unassigned member; add
|
|
295
|
+
a candidate — in that same order — only if it has a mutual edge to **every** member already in
|
|
296
|
+
the subgroup; when nothing more can be added, close it and seed the next subgroup from the
|
|
297
|
+
earliest remaining member. Repeat until every member is assigned.
|
|
298
|
+
|
|
299
|
+
One pass per subgroup is enough rather than an approximation: the requirement only ever tightens
|
|
300
|
+
as members are added, so a candidate rejected against a smaller subgroup cannot qualify against
|
|
301
|
+
a larger one.
|
|
302
|
+
|
|
303
|
+
On the `A↔B↔C` chain this yields `{A, B}` and `{C}`, which is the acceptance test. The chain has
|
|
304
|
+
to be *prevented* rather than merely detected: `A` and `C` share no edge, so a partition that
|
|
305
|
+
put them together would hand the consolidator one group asserting a relation nothing measured.
|
|
306
|
+
"""
|
|
307
|
+
remaining = sorted(component, key=lambda entry: entry.order)
|
|
308
|
+
subgroups: list[tuple[Entry, ...]] = []
|
|
309
|
+
while remaining:
|
|
310
|
+
subgroup = [remaining[0]]
|
|
311
|
+
deferred: list[Entry] = []
|
|
312
|
+
for candidate in remaining[1:]:
|
|
313
|
+
if all(candidate.uuid in edges.get(member.uuid, frozenset()) for member in subgroup):
|
|
314
|
+
subgroup.append(candidate)
|
|
315
|
+
else:
|
|
316
|
+
deferred.append(candidate)
|
|
317
|
+
subgroups.append(tuple(subgroup))
|
|
318
|
+
remaining = deferred
|
|
319
|
+
return tuple(subgroups)
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def _anchored_sets(
|
|
323
|
+
entries: Sequence[Entry], anchors: Mapping[str, str]
|
|
324
|
+
) -> tuple[tuple[str, tuple[Entry, ...]], ...]:
|
|
325
|
+
"""Entries sharing an anchor, as one pre-shard group each, in group order.
|
|
326
|
+
|
|
327
|
+
No cohesion pass runs over these, and that asymmetry is deliberate: the anchor *is* the cohesion
|
|
328
|
+
criterion, since every member cleared `anchor_cutoff` against the same record, so the group
|
|
329
|
+
already has a common centre. Complete linkage among the members would split that set on
|
|
330
|
+
member-to-member similarity the decision does not rest on. Step 2's pass exists precisely
|
|
331
|
+
because an orphan set has no common centre.
|
|
332
|
+
"""
|
|
333
|
+
grouped: dict[str, list[Entry]] = {}
|
|
334
|
+
for entry in entries:
|
|
335
|
+
anchor = anchors.get(entry.uuid)
|
|
336
|
+
if anchor is not None:
|
|
337
|
+
grouped.setdefault(anchor, []).append(entry)
|
|
338
|
+
return tuple((anchor, tuple(members)) for anchor, members in grouped.items())
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
async def plan(db: aiosqlite.Connection, *, call: ConsolidationCall) -> tuple[PlannedGroup, ...]:
|
|
342
|
+
"""Partition the whole journal into the groups one run will serve, in step 3's total order.
|
|
343
|
+
|
|
344
|
+
Anchored groups and cohesive orphan subgroups are computed independently and then sharded by the
|
|
345
|
+
same rule, because both are pre-shard groups. The result is sorted by `(order_key,
|
|
346
|
+
shard_index)`, which is D29's step-3 order: oldest first, with the uuid tiebreak load-bearing
|
|
347
|
+
rather than decorative — concurrent writes really do share a `created_at` string.
|
|
348
|
+
|
|
349
|
+
Assumes the caller's own open transaction, since every step reads the store and the plan must
|
|
350
|
+
describe one snapshot of it.
|
|
351
|
+
|
|
352
|
+
Returns:
|
|
353
|
+
Every group to be written, possibly empty — an empty journal plans no groups, which is a run
|
|
354
|
+
that is immediately complete rather than an error.
|
|
355
|
+
"""
|
|
356
|
+
entries = await journal_entries(db)
|
|
357
|
+
anchors: dict[str, str] = {}
|
|
358
|
+
for entry in entries:
|
|
359
|
+
anchor = await anchor_for(db, entry, call=call)
|
|
360
|
+
if anchor is not None:
|
|
361
|
+
anchors[entry.uuid] = anchor
|
|
362
|
+
|
|
363
|
+
orphans = tuple(entry for entry in entries if entry.uuid not in anchors)
|
|
364
|
+
neighbours = {orphan.uuid: await neighbours_of(db, orphan, call=call) for orphan in orphans}
|
|
365
|
+
edges = mutual_edges(neighbours, cutoff=call.settings.orphan_edge_cutoff)
|
|
366
|
+
|
|
367
|
+
pre_shard: list[tuple[str | None, tuple[Entry, ...]]] = [
|
|
368
|
+
(anchor, members) for anchor, members in _anchored_sets(entries, anchors)
|
|
369
|
+
]
|
|
370
|
+
for component in components(orphans, edges):
|
|
371
|
+
pre_shard.extend((None, subgroup) for subgroup in cohesive_subgroups(component, edges))
|
|
372
|
+
|
|
373
|
+
planned = [
|
|
374
|
+
PlannedGroup(
|
|
375
|
+
anchor_uuid=anchor,
|
|
376
|
+
order_key=order_key(members),
|
|
377
|
+
shard=shard_of,
|
|
378
|
+
members=shard_members,
|
|
379
|
+
)
|
|
380
|
+
for anchor, members in pre_shard
|
|
381
|
+
for shard_of, shard_members in shard(members, group_max=call.settings.group_max)
|
|
382
|
+
]
|
|
383
|
+
return tuple(sorted(planned, key=lambda group: (group.order_key, group.shard.index)))
|