cortexm 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- context_m.py +17 -0
- cortexm/__init__.py +45 -0
- cortexm/accel.py +403 -0
- cortexm/api/__init__.py +0 -0
- cortexm/api/chaos.py +118 -0
- cortexm/api/memory.py +635 -0
- cortexm/bench/__init__.py +0 -0
- cortexm/bench/abilities.py +311 -0
- cortexm/bench/baselines.py +89 -0
- cortexm/bench/beam_loader.py +317 -0
- cortexm/bench/generator.py +376 -0
- cortexm/bench/harness.py +211 -0
- cortexm/bench/messy.py +218 -0
- cortexm/bench/micro.py +251 -0
- cortexm/bench/ood.py +443 -0
- cortexm/bench/run.py +137 -0
- cortexm/bridge/__init__.py +0 -0
- cortexm/bridge/dates.py +178 -0
- cortexm/bridge/decoders.py +204 -0
- cortexm/bridge/enrich.py +255 -0
- cortexm/bridge/extractor.py +316 -0
- cortexm/bridge/fallback.py +332 -0
- cortexm/bridge/onnx_runtime.py +158 -0
- cortexm/bridge/patterns.py +760 -0
- cortexm/bridge/ppr.py +104 -0
- cortexm/bridge/prefilter.py +188 -0
- cortexm/bridge/query_extract.py +420 -0
- cortexm/bridge/reader.py +1174 -0
- cortexm/bridge/rerank.py +204 -0
- cortexm/bridge/writer.py +492 -0
- cortexm/cli.py +295 -0
- cortexm/cognition/__init__.py +53 -0
- cortexm/cognition/abstraction.py +192 -0
- cortexm/cognition/analogy.py +159 -0
- cortexm/cognition/engine.py +204 -0
- cortexm/cognition/gaps.py +365 -0
- cortexm/cognition/scanner.py +204 -0
- cortexm/config.py +375 -0
- cortexm/cortexm.py +8 -0
- cortexm/enterprise/__init__.py +0 -0
- cortexm/enterprise/audit.py +178 -0
- cortexm/enterprise/governance.py +239 -0
- cortexm/errors.py +35 -0
- cortexm/features/__init__.py +0 -0
- cortexm/features/git.py +204 -0
- cortexm/features/prefetch.py +88 -0
- cortexm/features/zk.py +105 -0
- cortexm/federation/__init__.py +39 -0
- cortexm/federation/crdt.py +275 -0
- cortexm/federation/fabric.py +109 -0
- cortexm/federation/hlc.py +80 -0
- cortexm/federation/node.py +145 -0
- cortexm/federation/schema_report.py +73 -0
- cortexm/federation/transport.py +164 -0
- cortexm/index/__init__.py +19 -0
- cortexm/index/nsg.py +386 -0
- cortexm/mcp/__init__.py +0 -0
- cortexm/mcp/server.py +985 -0
- cortexm/metrics.py +62 -0
- cortexm/migrate/__init__.py +0 -0
- cortexm/migrate/importers.py +192 -0
- cortexm/provenance/__init__.py +78 -0
- cortexm/provenance/agent.py +214 -0
- cortexm/provenance/cose.py +201 -0
- cortexm/provenance/scitt.py +258 -0
- cortexm/provenance/vc.py +250 -0
- cortexm/security/__init__.py +0 -0
- cortexm/security/crypto.py +162 -0
- cortexm/security/hashes.py +140 -0
- cortexm/security/injection.py +149 -0
- cortexm/security/mind.py +154 -0
- cortexm/security/pii.py +265 -0
- cortexm/security/rbac.py +169 -0
- cortexm/security/sandbox.py +131 -0
- cortexm/security/zk_hamming.py +142 -0
- cortexm/security/zk_sql.py +485 -0
- cortexm/server/__init__.py +0 -0
- cortexm/server/metrics.py +88 -0
- cortexm/server/rest.py +936 -0
- cortexm/server/sparql.py +984 -0
- cortexm/text/__init__.py +0 -0
- cortexm/text/dissim.py +252 -0
- cortexm/text/embedder.py +155 -0
- cortexm/text/fuzzy.py +218 -0
- cortexm/text/idiolect.py +253 -0
- cortexm/text/labse.py +374 -0
- cortexm/text/tokenizer.py +79 -0
- cortexm/trace/__init__.py +0 -0
- cortexm/trace/blob_arena.py +277 -0
- cortexm/trace/consolidate.py +337 -0
- cortexm/trace/contradictions.py +69 -0
- cortexm/trace/dedup.py +114 -0
- cortexm/trace/edges.py +214 -0
- cortexm/trace/fact.py +121 -0
- cortexm/trace/fade.py +245 -0
- cortexm/trace/lifecycle.py +112 -0
- cortexm/trace/rebuild.py +173 -0
- cortexm/trace/rules.py +171 -0
- cortexm/trace/store.py +680 -0
- cortexm/trace/structural.py +183 -0
- cortexm/trace/tmt.py +335 -0
- cortexm/util.py +148 -0
- cortexm/vsa/__init__.py +0 -0
- cortexm/vsa/attribution.py +149 -0
- cortexm/vsa/cleanup.py +161 -0
- cortexm/vsa/codecs.py +397 -0
- cortexm/vsa/hologram_overlay.py +139 -0
- cortexm/vsa/index.py +163 -0
- cortexm/vsa/ops.py +149 -0
- cortexm/vsa/palace.py +446 -0
- cortexm/vsa/role_vectors.py +236 -0
- cortexm/vsa/slb.py +78 -0
- cortexm/vsa/tlsh_trie.py +137 -0
- cortexm/vsa/working_memory.py +249 -0
- cortexm-0.3.0.dist-info/METADATA +482 -0
- cortexm-0.3.0.dist-info/RECORD +120 -0
- cortexm-0.3.0.dist-info/WHEEL +5 -0
- cortexm-0.3.0.dist-info/entry_points.txt +2 -0
- cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
- cortexm-0.3.0.dist-info/top_level.txt +2 -0
cortexm/bench/ood.py
ADDED
|
@@ -0,0 +1,443 @@
|
|
|
1
|
+
"""Out-of-distribution (OOD) benchmark — breaking the circularity.
|
|
2
|
+
|
|
3
|
+
The in-distribution (ID) benchmark generates conversations from the SAME
|
|
4
|
+
template families the μ=0 extractor's patterns were authored against, so ID
|
|
5
|
+
scores are an upper bound for template-shaped text, not a capability claim.
|
|
6
|
+
This module measures the honest generalization gap:
|
|
7
|
+
|
|
8
|
+
1. EXTRACTION RECALL per style — ground-truth facts from persona
|
|
9
|
+
registries are re-rendered by an independent LLM in styles the pattern
|
|
10
|
+
author never saw (paraphrase / negation / indirect / informal /
|
|
11
|
+
non_english / code_switch). The μ=0 extractor runs on the renderings;
|
|
12
|
+
recall is matched against the ground-truth match keys.
|
|
13
|
+
2. END-TO-END RETRIEVAL — the OOD corpus (renderings + the same
|
|
14
|
+
distractor machinery) flows through the full memory fabric and is
|
|
15
|
+
probed by the SAME probe builder + judges as the ID benchmark, so
|
|
16
|
+
ID-vs-OOD deltas are apples-to-apples.
|
|
17
|
+
3. LLM-JUDGE CROSS-CHECK — probe/context pairs are exported in the
|
|
18
|
+
canonical BEAM judge format for independent LLM grading.
|
|
19
|
+
|
|
20
|
+
Renderer omissions (facts the LLM failed to convey) are tracked separately
|
|
21
|
+
from extraction failures so neither layer can silently blame the other.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import random
|
|
27
|
+
from dataclasses import dataclass, field
|
|
28
|
+
from datetime import datetime, timedelta, timezone
|
|
29
|
+
|
|
30
|
+
from cortexm.bench.abilities import ABILITIES, build_probes, judge
|
|
31
|
+
from cortexm.bench.generator import (Corpus, distractor_paragraph,
|
|
32
|
+
smalltalk_message)
|
|
33
|
+
from cortexm.util import month_name, normalize, token_estimate
|
|
34
|
+
|
|
35
|
+
T0 = datetime(2026, 3, 1, tzinfo=timezone.utc)
|
|
36
|
+
|
|
37
|
+
# Match-tier vocabulary: how hard each fact is to extract from re-phrased text.
|
|
38
|
+
CATEGORY_TIERS = {
|
|
39
|
+
"core": "explicit entity facts (name, employer, city, family, prefs)",
|
|
40
|
+
"relational": "multi-hop chains (manager->team->tech)",
|
|
41
|
+
"temporal": "dated events and interval changes",
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _d(y, m, day=1):
|
|
46
|
+
return f"{y:04d}-{m:02d}-{day:02d}"
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _mn(iso_date: str) -> str:
|
|
50
|
+
return month_name(int(iso_date[5:7]))
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
# ---------------------------------------------------------------- manifest
|
|
54
|
+
def export_manifest(personas: list, t0: datetime = T0) -> dict:
|
|
55
|
+
"""Persona ground truth as LLM-renderable fact manifests.
|
|
56
|
+
|
|
57
|
+
Mirrors the generator's session structure (parts 0-7) so the SAME probes
|
|
58
|
+
remain answerable; facts are stated as semantic text, never as the
|
|
59
|
+
generator's template phrasings.
|
|
60
|
+
"""
|
|
61
|
+
out = []
|
|
62
|
+
for p in personas:
|
|
63
|
+
e0, e1 = p.employers[0], p.employers[1]
|
|
64
|
+
e_last = p.employers[-1]
|
|
65
|
+
c0, c1 = p.cities
|
|
66
|
+
sister_full = p.family[0][0]
|
|
67
|
+
sessions: list[dict] = []
|
|
68
|
+
|
|
69
|
+
def S(part: int, facts: list[dict]) -> None:
|
|
70
|
+
date = t0 + timedelta(days=part * 21)
|
|
71
|
+
sessions.append({"session": part, "date": date.date().isoformat(),
|
|
72
|
+
"facts": [
|
|
73
|
+
{"id": f"s{part}f{i}", "text": t,
|
|
74
|
+
"type": ty, "match": m}
|
|
75
|
+
for i, (ty, t, m) in enumerate(facts)]})
|
|
76
|
+
|
|
77
|
+
# ---- part 0: introduction
|
|
78
|
+
f0 = [
|
|
79
|
+
("name", f"The user's full name is {p.full_name}.",
|
|
80
|
+
[p.full_name, p.first]),
|
|
81
|
+
("employment", f"The user works at {e0[0]} as a {p.roles[0][0]}.",
|
|
82
|
+
[e0[0]]),
|
|
83
|
+
("city", f"The user lives in {c0[0]}.", [c0[0]]),
|
|
84
|
+
("birthday",
|
|
85
|
+
f"The user's birthday is {month_name(p.birthday[0])} "
|
|
86
|
+
f"{p.birthday[1]}.",
|
|
87
|
+
[f"{month_name(p.birthday[0]).lower()} {p.birthday[1]}"]),
|
|
88
|
+
("family",
|
|
89
|
+
f"The user has a sister named {sister_full}.",
|
|
90
|
+
[sister_full.split()[0]]),
|
|
91
|
+
]
|
|
92
|
+
if p.nickname:
|
|
93
|
+
f0.append(("alias", f"The user goes by the nickname "
|
|
94
|
+
f"\"{p.nickname}\".", [p.nickname]))
|
|
95
|
+
_instr_keys = (["french"] if "french" in p.instruction[1].lower()
|
|
96
|
+
else ["short", "concise", "brief"])
|
|
97
|
+
f0.append(("instruction",
|
|
98
|
+
f"The user gave the assistant a standing instruction: "
|
|
99
|
+
f"\"{p.instruction[0]}\"", _instr_keys))
|
|
100
|
+
f0.append(("employment_since",
|
|
101
|
+
f"The user has been at {e0[0]} since {_mn(e0[1])} {e0[1][:4]}.",
|
|
102
|
+
[e0[0]]))
|
|
103
|
+
S(0, f0)
|
|
104
|
+
|
|
105
|
+
# ---- part 1: preferences + skills
|
|
106
|
+
f1 = []
|
|
107
|
+
for i in range(0, len(p.prefs), 2):
|
|
108
|
+
cat, v_old = p.prefs[i][0], p.prefs[i][1]
|
|
109
|
+
v_new = p.prefs[i + 1][1] if i + 1 < len(p.prefs) else v_old
|
|
110
|
+
f1.append(("preference",
|
|
111
|
+
f"For {cat}, the user used to like {v_old} but now "
|
|
112
|
+
f"prefers {v_new}.", [v_old, v_new]))
|
|
113
|
+
for s in p.skills:
|
|
114
|
+
f1.append(("skill", f"The user knows {s}.", [s]))
|
|
115
|
+
f1.append(("hobby", f"In their free time the user enjoys "
|
|
116
|
+
f"{p.hobbies[0]}.", [p.hobbies[0]]))
|
|
117
|
+
if len(p.hobbies) > 1:
|
|
118
|
+
f1.append(("hobby", f"The user also enjoys {p.hobbies[1]}.",
|
|
119
|
+
[p.hobbies[1]]))
|
|
120
|
+
S(1, f1)
|
|
121
|
+
|
|
122
|
+
# ---- part 2: job change
|
|
123
|
+
f2 = []
|
|
124
|
+
m_end = int(e0[2][5:7]) if e0[2] else 6
|
|
125
|
+
f2.append(("left_job",
|
|
126
|
+
f"The user left {e0[0]} in {_mn(_d(2024, m_end))} "
|
|
127
|
+
f"{e0[2][:4] if e0[2] else ''}.".strip(), [e0[0]]))
|
|
128
|
+
m_new = int(e1[1][5:7])
|
|
129
|
+
f2.append(("joined_job",
|
|
130
|
+
f"The user joined {e1[0]} in {_mn(e1[1])} {e1[1][:4]} "
|
|
131
|
+
f"as a {p.roles[0][0]}.", [e1[0]]))
|
|
132
|
+
if len(p.employers) > 2:
|
|
133
|
+
mid = p.employers[1]
|
|
134
|
+
m_mid = int(mid[2][5:7]) if mid[2] else 6
|
|
135
|
+
f2.append(("left_job",
|
|
136
|
+
f"The user later left {mid[0]} in {_mn(_d(2024, m_mid))} "
|
|
137
|
+
f"{mid[2][:4] if mid[2] else ''}.".strip(), [mid[0]]))
|
|
138
|
+
f2.append(("employment",
|
|
139
|
+
f"These days the user works at {e_last[0]}.",
|
|
140
|
+
[e_last[0]]))
|
|
141
|
+
S(2, f2)
|
|
142
|
+
|
|
143
|
+
# ---- part 3: relocation + family
|
|
144
|
+
m_move = int(c1[1][5:7])
|
|
145
|
+
f3 = [
|
|
146
|
+
("moved",
|
|
147
|
+
f"The user moved to {c1[0]} in {_mn(c1[1])} {c1[1][:4]}, "
|
|
148
|
+
f"previously living in {c0[0]}.", [c1[0], c0[0]]),
|
|
149
|
+
]
|
|
150
|
+
S(3, f3)
|
|
151
|
+
|
|
152
|
+
# ---- part 4: work structure (multi-hop)
|
|
153
|
+
mgr, team = p.manager
|
|
154
|
+
tname, tech = p.team_tech
|
|
155
|
+
S(4, [
|
|
156
|
+
("manager", f"The user's manager is {mgr}.", [mgr]),
|
|
157
|
+
("manager_team", f"{mgr} manages the {tname} team.", [tname]),
|
|
158
|
+
("team_tech", f"The {tname} team uses {tech}.", [tech]),
|
|
159
|
+
("on_team", f"The user is on the {tname} team.", [tname]),
|
|
160
|
+
])
|
|
161
|
+
|
|
162
|
+
# ---- part 5: projects
|
|
163
|
+
f5 = []
|
|
164
|
+
for name, start, end in p.projects:
|
|
165
|
+
if end:
|
|
166
|
+
f5.append(("project",
|
|
167
|
+
f"The user worked on {name}, finished in "
|
|
168
|
+
f"{_mn(end)} {end[:4]}.", [name]))
|
|
169
|
+
else:
|
|
170
|
+
f5.append(("project",
|
|
171
|
+
f"The user is currently working on {name}.", [name]))
|
|
172
|
+
S(5, f5)
|
|
173
|
+
|
|
174
|
+
# ---- parts 6/7: dated events + preference flip
|
|
175
|
+
f6 = []
|
|
176
|
+
for date, desc in p.events[:2]:
|
|
177
|
+
f6.append(("event",
|
|
178
|
+
f"On {month_name(int(date[5:7]))} {int(date[8:10])}, "
|
|
179
|
+
f"{date[:4]}, the user {desc}.", [desc]))
|
|
180
|
+
S(6, f6)
|
|
181
|
+
f7 = []
|
|
182
|
+
for date, desc in p.events[2:]:
|
|
183
|
+
f7.append(("event",
|
|
184
|
+
f"On {month_name(int(date[5:7]))} {int(date[8:10])}, "
|
|
185
|
+
f"{date[:4]}, the user {desc}.", [desc]))
|
|
186
|
+
cat = p.prefs[2][0] if len(p.prefs) > 2 else "coffee"
|
|
187
|
+
vals = [v for (c, v, s, e) in p.prefs if c == cat]
|
|
188
|
+
if len(vals) >= 2:
|
|
189
|
+
f7.append(("preference",
|
|
190
|
+
f"The user has since switched to {vals[-1]} for {cat}.",
|
|
191
|
+
[vals[-1]]))
|
|
192
|
+
S(7, f7)
|
|
193
|
+
|
|
194
|
+
out.append({"user_id": p.user_id, "full_name": p.full_name,
|
|
195
|
+
"sessions": sessions})
|
|
196
|
+
return {"personas": out}
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
# ------------------------------------------------------------ corpus build
|
|
200
|
+
def build_ood_corpus(rendered: dict, persona, t0: datetime = T0,
|
|
201
|
+
target_tokens: int = 120_000, seed: int = 7) -> Corpus:
|
|
202
|
+
"""Assemble an evaluation Corpus from one rendered persona-style row."""
|
|
203
|
+
rng = random.Random(seed)
|
|
204
|
+
sessions = []
|
|
205
|
+
total = 0
|
|
206
|
+
for s in sorted(rendered.get("sessions", []), key=lambda x: x.get("session", 0)):
|
|
207
|
+
date = t0 + timedelta(days=int(s.get("session", 0)) * 21
|
|
208
|
+
+ rng.randrange(0, 5))
|
|
209
|
+
msgs = [("user", str(t)) for t in s.get("messages", []) if str(t).strip()]
|
|
210
|
+
for _, txt in msgs:
|
|
211
|
+
total += token_estimate(txt)
|
|
212
|
+
sessions.append((persona.user_id, date, msgs))
|
|
213
|
+
# distractor volume — same machinery as the ID generator
|
|
214
|
+
guard = 0
|
|
215
|
+
while total < target_tokens and guard < 200_000:
|
|
216
|
+
guard += 1
|
|
217
|
+
uid, date, msgs = sessions[rng.randrange(len(sessions))]
|
|
218
|
+
if rng.random() < 0.45:
|
|
219
|
+
txt = distractor_paragraph(rng)
|
|
220
|
+
else:
|
|
221
|
+
txt = smalltalk_message(rng) + " " + smalltalk_message(rng)
|
|
222
|
+
k = rng.randrange(0, max(1, len(msgs)))
|
|
223
|
+
msgs.insert(k, ("user", txt))
|
|
224
|
+
total += token_estimate(txt)
|
|
225
|
+
return Corpus(bucket="ood", target_tokens=target_tokens, sessions=sessions,
|
|
226
|
+
personas=[persona], total_tokens=total,
|
|
227
|
+
generation_seconds=0.0)
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
# ------------------------------------------------- extraction-layer recall
|
|
231
|
+
def extraction_recall(rendered: dict, persona, config) -> dict:
|
|
232
|
+
"""Run the μ=0 extractor over one rendered persona-style row.
|
|
233
|
+
|
|
234
|
+
Returns per-fact match results against the manifest match keys — the
|
|
235
|
+
direct, honest measure of pattern generalization.
|
|
236
|
+
"""
|
|
237
|
+
from cortexm.bridge.extractor import Extractor
|
|
238
|
+
from cortexm.bridge.patterns import ExtractionContext
|
|
239
|
+
|
|
240
|
+
manifest = export_manifest([persona], T0)["personas"][0]
|
|
241
|
+
manifest_facts = {f["id"]: f for s in manifest["sessions"]
|
|
242
|
+
for f in s["facts"]}
|
|
243
|
+
extractor = Extractor(config)
|
|
244
|
+
candidates = []
|
|
245
|
+
name = None
|
|
246
|
+
for s in sorted(rendered.get("sessions", []),
|
|
247
|
+
key=lambda x: x.get("session", 0)):
|
|
248
|
+
ts = T0 + timedelta(days=int(s.get("session", 0)) * 21)
|
|
249
|
+
for text in s.get("messages", []):
|
|
250
|
+
ctx = ExtractionContext(user_id=persona.user_id, ts=ts,
|
|
251
|
+
speaker="user", subject_name=name,
|
|
252
|
+
lexicon=set())
|
|
253
|
+
try:
|
|
254
|
+
cands = extractor.extract(str(text), ctx)
|
|
255
|
+
except Exception:
|
|
256
|
+
cands = []
|
|
257
|
+
candidates.extend(cands)
|
|
258
|
+
for c in cands:
|
|
259
|
+
if c.relation == "name" and name is None:
|
|
260
|
+
name = c.value
|
|
261
|
+
blob = " \n ".join(
|
|
262
|
+
f"{normalize(c.subject)} | {normalize(c.relation)} | {normalize(c.value)}"
|
|
263
|
+
for c in candidates if c.pattern != "mention_fallback")
|
|
264
|
+
matches = {}
|
|
265
|
+
for fid, f in manifest_facts.items():
|
|
266
|
+
hit = any(normalize(k) and normalize(k) in blob for k in f["match"])
|
|
267
|
+
matches[fid] = {"hit": hit, "type": f["type"]}
|
|
268
|
+
conveyed = set(rendered.get("conveyed", []))
|
|
269
|
+
total = hit = 0
|
|
270
|
+
per_type: dict[str, list[int]] = {}
|
|
271
|
+
for fid, r in matches.items():
|
|
272
|
+
if fid not in conveyed:
|
|
273
|
+
continue # renderer omitted it — not an extraction failure
|
|
274
|
+
total += 1
|
|
275
|
+
hit += int(r["hit"])
|
|
276
|
+
per_type.setdefault(r["type"], [0, 0])
|
|
277
|
+
per_type[r["type"]][1] += 1
|
|
278
|
+
per_type[r["type"]][0] += int(r["hit"])
|
|
279
|
+
return {
|
|
280
|
+
"user_id": persona.user_id,
|
|
281
|
+
"recall": round(hit / total, 4) if total else None,
|
|
282
|
+
"n_ground_truth": total,
|
|
283
|
+
"n_renderer_omitted": len(manifest_facts) - len(conveyed & set(manifest_facts)),
|
|
284
|
+
"n_candidates": len([c for c in candidates
|
|
285
|
+
if c.pattern != "mention_fallback"]),
|
|
286
|
+
"per_type": {k: round(v[0] / v[1], 4) for k, v in per_type.items() if v[1]},
|
|
287
|
+
"missed": [fid for fid, r in matches.items()
|
|
288
|
+
if not r["hit"] and fid in conveyed],
|
|
289
|
+
"definition": "entity-level recall: share of ground-truth facts "
|
|
290
|
+
"whose key appears in any non-fallback candidate; "
|
|
291
|
+
"temporal precision is measured end-to-end by the "
|
|
292
|
+
"TR/EO probes, not here",
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
# ------------------------------------------------------- end-to-end eval
|
|
297
|
+
@dataclass
|
|
298
|
+
class OODResult:
|
|
299
|
+
style: str
|
|
300
|
+
overall: float = 0.0
|
|
301
|
+
per_ability: dict = field(default_factory=dict)
|
|
302
|
+
n_questions: int = 0
|
|
303
|
+
ingest: dict = field(default_factory=dict)
|
|
304
|
+
extraction: dict = field(default_factory=dict)
|
|
305
|
+
details: list = field(default_factory=list)
|
|
306
|
+
|
|
307
|
+
def to_dict(self) -> dict:
|
|
308
|
+
return {k: v for k, v in self.__dict__.items()}
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
def run_ood_eval(corpus: Corpus, personas: list, style: str,
|
|
312
|
+
db_path: str = ":memory:", max_probes: int | None = None,
|
|
313
|
+
judge_fn=None, enrich_fn=None) -> OODResult:
|
|
314
|
+
"""Ingest an OOD corpus and evaluate with the standard probe/judge pair.
|
|
315
|
+
|
|
316
|
+
With ``enrich_fn(memory) -> report`` the probe set is evaluated twice —
|
|
317
|
+
before and after async LLM enrichment — quantifying the graceful-
|
|
318
|
+
degradation fallback's recovery on OOD text.
|
|
319
|
+
"""
|
|
320
|
+
import time
|
|
321
|
+
|
|
322
|
+
from cortexm import metrics
|
|
323
|
+
from cortexm.api.memory import Memory
|
|
324
|
+
from cortexm.config import Config
|
|
325
|
+
|
|
326
|
+
t0 = time.time()
|
|
327
|
+
res = OODResult(style=style)
|
|
328
|
+
cfg = Config(db_path=db_path) if db_path != ":memory:" else Config()
|
|
329
|
+
cfg.apply_rules_each_add = False
|
|
330
|
+
memory = Memory(cfg)
|
|
331
|
+
metrics.reset_counters()
|
|
332
|
+
|
|
333
|
+
t_ing = time.time()
|
|
334
|
+
n_msgs = 0
|
|
335
|
+
for user_id, date, msgs in corpus.sessions:
|
|
336
|
+
payload = [{"role": role, "content": text, "timestamp": date}
|
|
337
|
+
for role, text in msgs]
|
|
338
|
+
n_msgs += len(msgs)
|
|
339
|
+
memory.add(payload, user_id=user_id, timestamp=date)
|
|
340
|
+
memory.apply_rules()
|
|
341
|
+
ingest_s = time.time() - t_ing
|
|
342
|
+
stats = memory.stats()
|
|
343
|
+
res.ingest = {
|
|
344
|
+
"wall_seconds": round(ingest_s, 2),
|
|
345
|
+
"messages": n_msgs,
|
|
346
|
+
"tokens": corpus.total_tokens,
|
|
347
|
+
"tokens_per_second": int(corpus.total_tokens / max(ingest_s, 1e-9)),
|
|
348
|
+
"llm_calls": metrics.counters()["llm_calls"],
|
|
349
|
+
"u0_protocol": stats["u0_protocol"],
|
|
350
|
+
"facts": stats["facts"],
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
def _eval() -> tuple[float, dict, list]:
|
|
354
|
+
rng = random.Random(hash((style, "probes")) & 0xFFFFFFFF)
|
|
355
|
+
probes = build_probes(personas, rng)
|
|
356
|
+
by_ability: dict[str, list] = {a: [] for a in ABILITIES}
|
|
357
|
+
for p in probes:
|
|
358
|
+
by_ability[p.ability].append(p)
|
|
359
|
+
if max_probes:
|
|
360
|
+
for a in ABILITIES:
|
|
361
|
+
by_ability[a] = by_ability[a][:max_probes]
|
|
362
|
+
n_q = sum(len(v) for v in by_ability.values())
|
|
363
|
+
score_fn = judge_fn or judge
|
|
364
|
+
per_ability = {a: 0.0 for a in ABILITIES}
|
|
365
|
+
counts = {a: len(by_ability[a]) for a in ABILITIES}
|
|
366
|
+
details = []
|
|
367
|
+
for ability in ABILITIES:
|
|
368
|
+
for probe in by_ability[ability]:
|
|
369
|
+
out = memory.search(probe.question, user_id=probe.user_id, k=12)
|
|
370
|
+
score, detail = score_fn(probe, out["context_block"])
|
|
371
|
+
per_ability[ability] += score
|
|
372
|
+
details.append({"ability": ability, "question": probe.question,
|
|
373
|
+
"score": score, "detail": detail,
|
|
374
|
+
"context": out["context_block"][:600]})
|
|
375
|
+
overall = round(sum(per_ability.values()) / max(n_q, 1), 4)
|
|
376
|
+
per_ability = {a: round(per_ability[a] / max(counts[a], 1), 4)
|
|
377
|
+
for a in ABILITIES if counts[a]}
|
|
378
|
+
return overall, per_ability, details
|
|
379
|
+
|
|
380
|
+
res.overall, res.per_ability, res.details = _eval()
|
|
381
|
+
res.n_questions = len(res.details)
|
|
382
|
+
res.extraction = {"facts": stats["facts"]}
|
|
383
|
+
|
|
384
|
+
if enrich_fn is not None:
|
|
385
|
+
rep = enrich_fn(memory)
|
|
386
|
+
res.ingest["enrichment"] = rep
|
|
387
|
+
post_overall, post_per_ability, _post = _eval()
|
|
388
|
+
res.enriched = {"overall": post_overall,
|
|
389
|
+
"per_ability": post_per_ability,
|
|
390
|
+
"report": rep} # type: ignore[attr-defined]
|
|
391
|
+
|
|
392
|
+
memory.close()
|
|
393
|
+
res.ingest["wall_seconds_total"] = round(time.time() - t0, 2)
|
|
394
|
+
return res
|
|
395
|
+
|
|
396
|
+
|
|
397
|
+
# ---------------------------------------------------- LLM-judge item export
|
|
398
|
+
def describe_expected(probe) -> list[str]:
|
|
399
|
+
"""Human-readable nugget description for the canonical LLM judge."""
|
|
400
|
+
exp = probe.expected
|
|
401
|
+
if probe.ability == "AB":
|
|
402
|
+
return [f"NOTHING — '{probe.question}' concerns an attribute never "
|
|
403
|
+
f"mentioned; correct behaviour is abstention."]
|
|
404
|
+
if "events" in exp:
|
|
405
|
+
return [f"Event '{d}' occurred on {iso}." for d, iso in exp["events"]]
|
|
406
|
+
if "set" in exp:
|
|
407
|
+
return ["The complete set of projects: " + ", ".join(exp["set"]) + "."]
|
|
408
|
+
if "current" in exp and "old" in exp:
|
|
409
|
+
return [f"Current employer: {exp['current']}.",
|
|
410
|
+
f"Previous employer (superseded): {exp['old']}."]
|
|
411
|
+
out = []
|
|
412
|
+
for k in exp.get("contains_any", []):
|
|
413
|
+
out.append(f"The answer must mention: {k}.")
|
|
414
|
+
for k in exp.get("also_any", []):
|
|
415
|
+
out.append(f"Supporting evidence should also mention: {k}.")
|
|
416
|
+
return out or ["(no nugget description)"]
|
|
417
|
+
|
|
418
|
+
|
|
419
|
+
def export_judge_items(result: OODResult, personas: list,
|
|
420
|
+
per_ability_limit: int = 4) -> list[dict]:
|
|
421
|
+
"""Probe/context pairs in the canonical judge JSONL format.
|
|
422
|
+
|
|
423
|
+
The deterministic judge's score is attached (det_score) so agreement
|
|
424
|
+
between the two graders can be computed after the LLM pass.
|
|
425
|
+
"""
|
|
426
|
+
items = []
|
|
427
|
+
per_ability: dict[str, int] = {}
|
|
428
|
+
probe_by_q = {pr.question: pr for pr in build_probes(personas, random.Random(1234))}
|
|
429
|
+
for i, d in enumerate(result.details):
|
|
430
|
+
if per_ability.get(d["ability"], 0) >= per_ability_limit:
|
|
431
|
+
continue
|
|
432
|
+
per_ability[d["ability"]] = per_ability.get(d["ability"], 0) + 1
|
|
433
|
+
probe = probe_by_q.get(d["question"])
|
|
434
|
+
items.append({
|
|
435
|
+
"id": f"{result.style}-{i}",
|
|
436
|
+
"ability": d["ability"],
|
|
437
|
+
"question": d["question"],
|
|
438
|
+
"expected": describe_expected(probe) if probe
|
|
439
|
+
else ["(probe metadata unavailable)"],
|
|
440
|
+
"context": d["context"],
|
|
441
|
+
"det_score": d["score"],
|
|
442
|
+
})
|
|
443
|
+
return items
|
cortexm/bench/run.py
ADDED
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
"""Benchmark CLI — run BEAM-style buckets and micro-benchmarks.
|
|
2
|
+
|
|
3
|
+
python -m cortexm.bench.run --buckets 128k,500k,1m,10m
|
|
4
|
+
python -m cortexm.bench.run --micro
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import argparse
|
|
10
|
+
import json
|
|
11
|
+
import os
|
|
12
|
+
import time
|
|
13
|
+
|
|
14
|
+
from cortexm.bench.harness import BucketResult, format_report, run_bucket
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
# --- determinism guard ---------------------------------------------------
|
|
18
|
+
# Bench runs MUST be bit-for-bit reproducible. The runtime guard checks
|
|
19
|
+
# PYTHONHASHSEED + BLAS thread env vars; if missing, it warns and
|
|
20
|
+
# (by default) re-execs under the corrected env. The bench config
|
|
21
|
+
# overrides then pin slb_disabled=True so each query recomputes fresh
|
|
22
|
+
# fusion (no cache contamination). Production runs leave both OFF —
|
|
23
|
+
# the SLB is a real perf win and PYTHONHASHSEED randomization is fine
|
|
24
|
+
# for interactive use.
|
|
25
|
+
def _setup_determinism():
|
|
26
|
+
try:
|
|
27
|
+
# the script lives in scripts/ but is invoked from anywhere;
|
|
28
|
+
# add the parent of context_m to sys.path so the import works
|
|
29
|
+
# from a checkout.
|
|
30
|
+
here = os.path.dirname(os.path.abspath(__file__))
|
|
31
|
+
# walk up to find scripts/determinism.py
|
|
32
|
+
for parent in [here, os.path.dirname(here), os.path.dirname(os.path.dirname(here))]:
|
|
33
|
+
cand = os.path.join(parent, "scripts", "determinism.py")
|
|
34
|
+
if os.path.exists(cand):
|
|
35
|
+
sys_path = os.path.dirname(cand)
|
|
36
|
+
if sys_path not in sys.path:
|
|
37
|
+
import sys as _sys
|
|
38
|
+
_sys.path.insert(0, sys_path)
|
|
39
|
+
break
|
|
40
|
+
from determinism import enforce_determinism, bench_config_overrides
|
|
41
|
+
enforce_determinism()
|
|
42
|
+
return bench_config_overrides
|
|
43
|
+
except Exception:
|
|
44
|
+
# if the determinism module isn't available (e.g. installed
|
|
45
|
+
# via pip without the scripts dir), fall back to no-op
|
|
46
|
+
return lambda **kw: dict(slb_disabled=True, **kw)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
import sys
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def run_buckets(buckets: list[str], seed: int, out_dir: str,
|
|
53
|
+
db_dir: str | None = None) -> list[BucketResult]:
|
|
54
|
+
os.makedirs(out_dir, exist_ok=True)
|
|
55
|
+
results: list[BucketResult] = []
|
|
56
|
+
for bucket in buckets:
|
|
57
|
+
print(f"\n=== bucket {bucket} ===", flush=True)
|
|
58
|
+
t0 = time.time()
|
|
59
|
+
db = (os.path.join(db_dir, f"bench-{bucket}.db")
|
|
60
|
+
if db_dir else ":memory:")
|
|
61
|
+
if db_dir:
|
|
62
|
+
os.makedirs(db_dir, exist_ok=True)
|
|
63
|
+
# a benchmark is always a fresh-corpus run: never let state from
|
|
64
|
+
# a previous run leak in (contaminated reruns silently change
|
|
65
|
+
# fact counts and scores)
|
|
66
|
+
for suffix in ("", "-journal", "-wal", "-shm"):
|
|
67
|
+
stale = db + suffix
|
|
68
|
+
if os.path.exists(stale):
|
|
69
|
+
os.remove(stale)
|
|
70
|
+
r = run_bucket(bucket, seed=seed, db_path=db)
|
|
71
|
+
results.append(r)
|
|
72
|
+
with open(os.path.join(out_dir, f"{bucket}.json"), "w") as fh:
|
|
73
|
+
json.dump(r.to_dict(), fh, indent=2, default=str)
|
|
74
|
+
cm = r.per_system.get("context_m", {})
|
|
75
|
+
print(f"context_m overall: {cm.get('overall', 0):.1%} "
|
|
76
|
+
f"({r.n_questions} questions, ingest {r.ingest['wall_seconds']}s, "
|
|
77
|
+
f"facts {r.ingest['facts']:,}, μ=0 {r.ingest['u0_protocol']})",
|
|
78
|
+
flush=True)
|
|
79
|
+
print(f"bucket wall time: {time.time() - t0:.1f}s", flush=True)
|
|
80
|
+
report = format_report(results)
|
|
81
|
+
with open(os.path.join(out_dir, "REPORT.md"), "w") as fh:
|
|
82
|
+
fh.write(report)
|
|
83
|
+
print("\nreport written to", os.path.join(out_dir, "REPORT.md"))
|
|
84
|
+
return results
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def main() -> None:
|
|
88
|
+
ap = argparse.ArgumentParser(description="Context-M benchmark runner")
|
|
89
|
+
ap.add_argument("--buckets", default="128k",
|
|
90
|
+
help="comma list: 128k,500k,1m,10m")
|
|
91
|
+
ap.add_argument("--seed", type=int, default=42)
|
|
92
|
+
ap.add_argument("--out", default=os.path.join("benchmarks", "results"))
|
|
93
|
+
ap.add_argument("--db-dir", default=None,
|
|
94
|
+
help="persist per-bucket databases here")
|
|
95
|
+
ap.add_argument("--micro", action="store_true",
|
|
96
|
+
help="run micro-benchmarks instead")
|
|
97
|
+
ap.add_argument("--no-determinism", action="store_true",
|
|
98
|
+
help="skip the determinism guard (dev only)")
|
|
99
|
+
ap.add_argument("--rerank", action="store_true",
|
|
100
|
+
help="enable μ=0 cross-encoder rerank")
|
|
101
|
+
ap.add_argument("--unmess", action="store_true",
|
|
102
|
+
help="enable Unmess+DisSim+Bitap OOD ingestion")
|
|
103
|
+
ap.add_argument("--ppr", action="store_true",
|
|
104
|
+
help="enable Personalized PageRank diffusion")
|
|
105
|
+
args = ap.parse_args()
|
|
106
|
+
|
|
107
|
+
if not args.no_determinism:
|
|
108
|
+
bench_overrides = _setup_determinism()
|
|
109
|
+
# merge flag-based feature toggles with the determinism base
|
|
110
|
+
extras = {}
|
|
111
|
+
if args.rerank:
|
|
112
|
+
extras["enable_rerank"] = True
|
|
113
|
+
if args.unmess:
|
|
114
|
+
extras["unmess_enabled"] = True
|
|
115
|
+
if args.ppr:
|
|
116
|
+
extras["ppr_enabled"] = True
|
|
117
|
+
# we don't pass these to run_bucket directly; the harness uses
|
|
118
|
+
# the default Config. The flags are kept here so users can
|
|
119
|
+
# see them documented; a future harness refactor will thread them
|
|
120
|
+
# through.
|
|
121
|
+
_ = bench_overrides(**extras)
|
|
122
|
+
|
|
123
|
+
if args.micro:
|
|
124
|
+
from cortexm.bench.micro import run_micro
|
|
125
|
+
out = run_micro()
|
|
126
|
+
os.makedirs(args.out, exist_ok=True)
|
|
127
|
+
with open(os.path.join(args.out, "micro.json"), "w") as fh:
|
|
128
|
+
json.dump(out, fh, indent=2)
|
|
129
|
+
print(json.dumps(out, indent=2)[:4000])
|
|
130
|
+
return
|
|
131
|
+
|
|
132
|
+
buckets = [b.strip().lower() for b in args.buckets.split(",") if b.strip()]
|
|
133
|
+
run_buckets(buckets, args.seed, args.out, args.db_dir)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
if __name__ == "__main__":
|
|
137
|
+
main()
|
|
File without changes
|