cortexm 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. context_m.py +17 -0
  2. cortexm/__init__.py +45 -0
  3. cortexm/accel.py +403 -0
  4. cortexm/api/__init__.py +0 -0
  5. cortexm/api/chaos.py +118 -0
  6. cortexm/api/memory.py +635 -0
  7. cortexm/bench/__init__.py +0 -0
  8. cortexm/bench/abilities.py +311 -0
  9. cortexm/bench/baselines.py +89 -0
  10. cortexm/bench/beam_loader.py +317 -0
  11. cortexm/bench/generator.py +376 -0
  12. cortexm/bench/harness.py +211 -0
  13. cortexm/bench/messy.py +218 -0
  14. cortexm/bench/micro.py +251 -0
  15. cortexm/bench/ood.py +443 -0
  16. cortexm/bench/run.py +137 -0
  17. cortexm/bridge/__init__.py +0 -0
  18. cortexm/bridge/dates.py +178 -0
  19. cortexm/bridge/decoders.py +204 -0
  20. cortexm/bridge/enrich.py +255 -0
  21. cortexm/bridge/extractor.py +316 -0
  22. cortexm/bridge/fallback.py +332 -0
  23. cortexm/bridge/onnx_runtime.py +158 -0
  24. cortexm/bridge/patterns.py +760 -0
  25. cortexm/bridge/ppr.py +104 -0
  26. cortexm/bridge/prefilter.py +188 -0
  27. cortexm/bridge/query_extract.py +420 -0
  28. cortexm/bridge/reader.py +1174 -0
  29. cortexm/bridge/rerank.py +204 -0
  30. cortexm/bridge/writer.py +492 -0
  31. cortexm/cli.py +295 -0
  32. cortexm/cognition/__init__.py +53 -0
  33. cortexm/cognition/abstraction.py +192 -0
  34. cortexm/cognition/analogy.py +159 -0
  35. cortexm/cognition/engine.py +204 -0
  36. cortexm/cognition/gaps.py +365 -0
  37. cortexm/cognition/scanner.py +204 -0
  38. cortexm/config.py +375 -0
  39. cortexm/cortexm.py +8 -0
  40. cortexm/enterprise/__init__.py +0 -0
  41. cortexm/enterprise/audit.py +178 -0
  42. cortexm/enterprise/governance.py +239 -0
  43. cortexm/errors.py +35 -0
  44. cortexm/features/__init__.py +0 -0
  45. cortexm/features/git.py +204 -0
  46. cortexm/features/prefetch.py +88 -0
  47. cortexm/features/zk.py +105 -0
  48. cortexm/federation/__init__.py +39 -0
  49. cortexm/federation/crdt.py +275 -0
  50. cortexm/federation/fabric.py +109 -0
  51. cortexm/federation/hlc.py +80 -0
  52. cortexm/federation/node.py +145 -0
  53. cortexm/federation/schema_report.py +73 -0
  54. cortexm/federation/transport.py +164 -0
  55. cortexm/index/__init__.py +19 -0
  56. cortexm/index/nsg.py +386 -0
  57. cortexm/mcp/__init__.py +0 -0
  58. cortexm/mcp/server.py +985 -0
  59. cortexm/metrics.py +62 -0
  60. cortexm/migrate/__init__.py +0 -0
  61. cortexm/migrate/importers.py +192 -0
  62. cortexm/provenance/__init__.py +78 -0
  63. cortexm/provenance/agent.py +214 -0
  64. cortexm/provenance/cose.py +201 -0
  65. cortexm/provenance/scitt.py +258 -0
  66. cortexm/provenance/vc.py +250 -0
  67. cortexm/security/__init__.py +0 -0
  68. cortexm/security/crypto.py +162 -0
  69. cortexm/security/hashes.py +140 -0
  70. cortexm/security/injection.py +149 -0
  71. cortexm/security/mind.py +154 -0
  72. cortexm/security/pii.py +265 -0
  73. cortexm/security/rbac.py +169 -0
  74. cortexm/security/sandbox.py +131 -0
  75. cortexm/security/zk_hamming.py +142 -0
  76. cortexm/security/zk_sql.py +485 -0
  77. cortexm/server/__init__.py +0 -0
  78. cortexm/server/metrics.py +88 -0
  79. cortexm/server/rest.py +936 -0
  80. cortexm/server/sparql.py +984 -0
  81. cortexm/text/__init__.py +0 -0
  82. cortexm/text/dissim.py +252 -0
  83. cortexm/text/embedder.py +155 -0
  84. cortexm/text/fuzzy.py +218 -0
  85. cortexm/text/idiolect.py +253 -0
  86. cortexm/text/labse.py +374 -0
  87. cortexm/text/tokenizer.py +79 -0
  88. cortexm/trace/__init__.py +0 -0
  89. cortexm/trace/blob_arena.py +277 -0
  90. cortexm/trace/consolidate.py +337 -0
  91. cortexm/trace/contradictions.py +69 -0
  92. cortexm/trace/dedup.py +114 -0
  93. cortexm/trace/edges.py +214 -0
  94. cortexm/trace/fact.py +121 -0
  95. cortexm/trace/fade.py +245 -0
  96. cortexm/trace/lifecycle.py +112 -0
  97. cortexm/trace/rebuild.py +173 -0
  98. cortexm/trace/rules.py +171 -0
  99. cortexm/trace/store.py +680 -0
  100. cortexm/trace/structural.py +183 -0
  101. cortexm/trace/tmt.py +335 -0
  102. cortexm/util.py +148 -0
  103. cortexm/vsa/__init__.py +0 -0
  104. cortexm/vsa/attribution.py +149 -0
  105. cortexm/vsa/cleanup.py +161 -0
  106. cortexm/vsa/codecs.py +397 -0
  107. cortexm/vsa/hologram_overlay.py +139 -0
  108. cortexm/vsa/index.py +163 -0
  109. cortexm/vsa/ops.py +149 -0
  110. cortexm/vsa/palace.py +446 -0
  111. cortexm/vsa/role_vectors.py +236 -0
  112. cortexm/vsa/slb.py +78 -0
  113. cortexm/vsa/tlsh_trie.py +137 -0
  114. cortexm/vsa/working_memory.py +249 -0
  115. cortexm-0.3.0.dist-info/METADATA +482 -0
  116. cortexm-0.3.0.dist-info/RECORD +120 -0
  117. cortexm-0.3.0.dist-info/WHEEL +5 -0
  118. cortexm-0.3.0.dist-info/entry_points.txt +2 -0
  119. cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
  120. cortexm-0.3.0.dist-info/top_level.txt +2 -0
cortexm/bench/ood.py ADDED
@@ -0,0 +1,443 @@
1
+ """Out-of-distribution (OOD) benchmark — breaking the circularity.
2
+
3
+ The in-distribution (ID) benchmark generates conversations from the SAME
4
+ template families the μ=0 extractor's patterns were authored against, so ID
5
+ scores are an upper bound for template-shaped text, not a capability claim.
6
+ This module measures the honest generalization gap:
7
+
8
+ 1. EXTRACTION RECALL per style — ground-truth facts from persona
9
+ registries are re-rendered by an independent LLM in styles the pattern
10
+ author never saw (paraphrase / negation / indirect / informal /
11
+ non_english / code_switch). The μ=0 extractor runs on the renderings;
12
+ recall is matched against the ground-truth match keys.
13
+ 2. END-TO-END RETRIEVAL — the OOD corpus (renderings + the same
14
+ distractor machinery) flows through the full memory fabric and is
15
+ probed by the SAME probe builder + judges as the ID benchmark, so
16
+ ID-vs-OOD deltas are apples-to-apples.
17
+ 3. LLM-JUDGE CROSS-CHECK — probe/context pairs are exported in the
18
+ canonical BEAM judge format for independent LLM grading.
19
+
20
+ Renderer omissions (facts the LLM failed to convey) are tracked separately
21
+ from extraction failures so neither layer can silently blame the other.
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ import random
27
+ from dataclasses import dataclass, field
28
+ from datetime import datetime, timedelta, timezone
29
+
30
+ from cortexm.bench.abilities import ABILITIES, build_probes, judge
31
+ from cortexm.bench.generator import (Corpus, distractor_paragraph,
32
+ smalltalk_message)
33
+ from cortexm.util import month_name, normalize, token_estimate
34
+
35
+ T0 = datetime(2026, 3, 1, tzinfo=timezone.utc)
36
+
37
+ # Match-tier vocabulary: how hard each fact is to extract from re-phrased text.
38
+ CATEGORY_TIERS = {
39
+ "core": "explicit entity facts (name, employer, city, family, prefs)",
40
+ "relational": "multi-hop chains (manager->team->tech)",
41
+ "temporal": "dated events and interval changes",
42
+ }
43
+
44
+
45
+ def _d(y, m, day=1):
46
+ return f"{y:04d}-{m:02d}-{day:02d}"
47
+
48
+
49
+ def _mn(iso_date: str) -> str:
50
+ return month_name(int(iso_date[5:7]))
51
+
52
+
53
+ # ---------------------------------------------------------------- manifest
54
+ def export_manifest(personas: list, t0: datetime = T0) -> dict:
55
+ """Persona ground truth as LLM-renderable fact manifests.
56
+
57
+ Mirrors the generator's session structure (parts 0-7) so the SAME probes
58
+ remain answerable; facts are stated as semantic text, never as the
59
+ generator's template phrasings.
60
+ """
61
+ out = []
62
+ for p in personas:
63
+ e0, e1 = p.employers[0], p.employers[1]
64
+ e_last = p.employers[-1]
65
+ c0, c1 = p.cities
66
+ sister_full = p.family[0][0]
67
+ sessions: list[dict] = []
68
+
69
+ def S(part: int, facts: list[dict]) -> None:
70
+ date = t0 + timedelta(days=part * 21)
71
+ sessions.append({"session": part, "date": date.date().isoformat(),
72
+ "facts": [
73
+ {"id": f"s{part}f{i}", "text": t,
74
+ "type": ty, "match": m}
75
+ for i, (ty, t, m) in enumerate(facts)]})
76
+
77
+ # ---- part 0: introduction
78
+ f0 = [
79
+ ("name", f"The user's full name is {p.full_name}.",
80
+ [p.full_name, p.first]),
81
+ ("employment", f"The user works at {e0[0]} as a {p.roles[0][0]}.",
82
+ [e0[0]]),
83
+ ("city", f"The user lives in {c0[0]}.", [c0[0]]),
84
+ ("birthday",
85
+ f"The user's birthday is {month_name(p.birthday[0])} "
86
+ f"{p.birthday[1]}.",
87
+ [f"{month_name(p.birthday[0]).lower()} {p.birthday[1]}"]),
88
+ ("family",
89
+ f"The user has a sister named {sister_full}.",
90
+ [sister_full.split()[0]]),
91
+ ]
92
+ if p.nickname:
93
+ f0.append(("alias", f"The user goes by the nickname "
94
+ f"\"{p.nickname}\".", [p.nickname]))
95
+ _instr_keys = (["french"] if "french" in p.instruction[1].lower()
96
+ else ["short", "concise", "brief"])
97
+ f0.append(("instruction",
98
+ f"The user gave the assistant a standing instruction: "
99
+ f"\"{p.instruction[0]}\"", _instr_keys))
100
+ f0.append(("employment_since",
101
+ f"The user has been at {e0[0]} since {_mn(e0[1])} {e0[1][:4]}.",
102
+ [e0[0]]))
103
+ S(0, f0)
104
+
105
+ # ---- part 1: preferences + skills
106
+ f1 = []
107
+ for i in range(0, len(p.prefs), 2):
108
+ cat, v_old = p.prefs[i][0], p.prefs[i][1]
109
+ v_new = p.prefs[i + 1][1] if i + 1 < len(p.prefs) else v_old
110
+ f1.append(("preference",
111
+ f"For {cat}, the user used to like {v_old} but now "
112
+ f"prefers {v_new}.", [v_old, v_new]))
113
+ for s in p.skills:
114
+ f1.append(("skill", f"The user knows {s}.", [s]))
115
+ f1.append(("hobby", f"In their free time the user enjoys "
116
+ f"{p.hobbies[0]}.", [p.hobbies[0]]))
117
+ if len(p.hobbies) > 1:
118
+ f1.append(("hobby", f"The user also enjoys {p.hobbies[1]}.",
119
+ [p.hobbies[1]]))
120
+ S(1, f1)
121
+
122
+ # ---- part 2: job change
123
+ f2 = []
124
+ m_end = int(e0[2][5:7]) if e0[2] else 6
125
+ f2.append(("left_job",
126
+ f"The user left {e0[0]} in {_mn(_d(2024, m_end))} "
127
+ f"{e0[2][:4] if e0[2] else ''}.".strip(), [e0[0]]))
128
+ m_new = int(e1[1][5:7])
129
+ f2.append(("joined_job",
130
+ f"The user joined {e1[0]} in {_mn(e1[1])} {e1[1][:4]} "
131
+ f"as a {p.roles[0][0]}.", [e1[0]]))
132
+ if len(p.employers) > 2:
133
+ mid = p.employers[1]
134
+ m_mid = int(mid[2][5:7]) if mid[2] else 6
135
+ f2.append(("left_job",
136
+ f"The user later left {mid[0]} in {_mn(_d(2024, m_mid))} "
137
+ f"{mid[2][:4] if mid[2] else ''}.".strip(), [mid[0]]))
138
+ f2.append(("employment",
139
+ f"These days the user works at {e_last[0]}.",
140
+ [e_last[0]]))
141
+ S(2, f2)
142
+
143
+ # ---- part 3: relocation + family
144
+ m_move = int(c1[1][5:7])
145
+ f3 = [
146
+ ("moved",
147
+ f"The user moved to {c1[0]} in {_mn(c1[1])} {c1[1][:4]}, "
148
+ f"previously living in {c0[0]}.", [c1[0], c0[0]]),
149
+ ]
150
+ S(3, f3)
151
+
152
+ # ---- part 4: work structure (multi-hop)
153
+ mgr, team = p.manager
154
+ tname, tech = p.team_tech
155
+ S(4, [
156
+ ("manager", f"The user's manager is {mgr}.", [mgr]),
157
+ ("manager_team", f"{mgr} manages the {tname} team.", [tname]),
158
+ ("team_tech", f"The {tname} team uses {tech}.", [tech]),
159
+ ("on_team", f"The user is on the {tname} team.", [tname]),
160
+ ])
161
+
162
+ # ---- part 5: projects
163
+ f5 = []
164
+ for name, start, end in p.projects:
165
+ if end:
166
+ f5.append(("project",
167
+ f"The user worked on {name}, finished in "
168
+ f"{_mn(end)} {end[:4]}.", [name]))
169
+ else:
170
+ f5.append(("project",
171
+ f"The user is currently working on {name}.", [name]))
172
+ S(5, f5)
173
+
174
+ # ---- parts 6/7: dated events + preference flip
175
+ f6 = []
176
+ for date, desc in p.events[:2]:
177
+ f6.append(("event",
178
+ f"On {month_name(int(date[5:7]))} {int(date[8:10])}, "
179
+ f"{date[:4]}, the user {desc}.", [desc]))
180
+ S(6, f6)
181
+ f7 = []
182
+ for date, desc in p.events[2:]:
183
+ f7.append(("event",
184
+ f"On {month_name(int(date[5:7]))} {int(date[8:10])}, "
185
+ f"{date[:4]}, the user {desc}.", [desc]))
186
+ cat = p.prefs[2][0] if len(p.prefs) > 2 else "coffee"
187
+ vals = [v for (c, v, s, e) in p.prefs if c == cat]
188
+ if len(vals) >= 2:
189
+ f7.append(("preference",
190
+ f"The user has since switched to {vals[-1]} for {cat}.",
191
+ [vals[-1]]))
192
+ S(7, f7)
193
+
194
+ out.append({"user_id": p.user_id, "full_name": p.full_name,
195
+ "sessions": sessions})
196
+ return {"personas": out}
197
+
198
+
199
+ # ------------------------------------------------------------ corpus build
200
+ def build_ood_corpus(rendered: dict, persona, t0: datetime = T0,
201
+ target_tokens: int = 120_000, seed: int = 7) -> Corpus:
202
+ """Assemble an evaluation Corpus from one rendered persona-style row."""
203
+ rng = random.Random(seed)
204
+ sessions = []
205
+ total = 0
206
+ for s in sorted(rendered.get("sessions", []), key=lambda x: x.get("session", 0)):
207
+ date = t0 + timedelta(days=int(s.get("session", 0)) * 21
208
+ + rng.randrange(0, 5))
209
+ msgs = [("user", str(t)) for t in s.get("messages", []) if str(t).strip()]
210
+ for _, txt in msgs:
211
+ total += token_estimate(txt)
212
+ sessions.append((persona.user_id, date, msgs))
213
+ # distractor volume — same machinery as the ID generator
214
+ guard = 0
215
+ while total < target_tokens and guard < 200_000:
216
+ guard += 1
217
+ uid, date, msgs = sessions[rng.randrange(len(sessions))]
218
+ if rng.random() < 0.45:
219
+ txt = distractor_paragraph(rng)
220
+ else:
221
+ txt = smalltalk_message(rng) + " " + smalltalk_message(rng)
222
+ k = rng.randrange(0, max(1, len(msgs)))
223
+ msgs.insert(k, ("user", txt))
224
+ total += token_estimate(txt)
225
+ return Corpus(bucket="ood", target_tokens=target_tokens, sessions=sessions,
226
+ personas=[persona], total_tokens=total,
227
+ generation_seconds=0.0)
228
+
229
+
230
+ # ------------------------------------------------- extraction-layer recall
231
+ def extraction_recall(rendered: dict, persona, config) -> dict:
232
+ """Run the μ=0 extractor over one rendered persona-style row.
233
+
234
+ Returns per-fact match results against the manifest match keys — the
235
+ direct, honest measure of pattern generalization.
236
+ """
237
+ from cortexm.bridge.extractor import Extractor
238
+ from cortexm.bridge.patterns import ExtractionContext
239
+
240
+ manifest = export_manifest([persona], T0)["personas"][0]
241
+ manifest_facts = {f["id"]: f for s in manifest["sessions"]
242
+ for f in s["facts"]}
243
+ extractor = Extractor(config)
244
+ candidates = []
245
+ name = None
246
+ for s in sorted(rendered.get("sessions", []),
247
+ key=lambda x: x.get("session", 0)):
248
+ ts = T0 + timedelta(days=int(s.get("session", 0)) * 21)
249
+ for text in s.get("messages", []):
250
+ ctx = ExtractionContext(user_id=persona.user_id, ts=ts,
251
+ speaker="user", subject_name=name,
252
+ lexicon=set())
253
+ try:
254
+ cands = extractor.extract(str(text), ctx)
255
+ except Exception:
256
+ cands = []
257
+ candidates.extend(cands)
258
+ for c in cands:
259
+ if c.relation == "name" and name is None:
260
+ name = c.value
261
+ blob = " \n ".join(
262
+ f"{normalize(c.subject)} | {normalize(c.relation)} | {normalize(c.value)}"
263
+ for c in candidates if c.pattern != "mention_fallback")
264
+ matches = {}
265
+ for fid, f in manifest_facts.items():
266
+ hit = any(normalize(k) and normalize(k) in blob for k in f["match"])
267
+ matches[fid] = {"hit": hit, "type": f["type"]}
268
+ conveyed = set(rendered.get("conveyed", []))
269
+ total = hit = 0
270
+ per_type: dict[str, list[int]] = {}
271
+ for fid, r in matches.items():
272
+ if fid not in conveyed:
273
+ continue # renderer omitted it — not an extraction failure
274
+ total += 1
275
+ hit += int(r["hit"])
276
+ per_type.setdefault(r["type"], [0, 0])
277
+ per_type[r["type"]][1] += 1
278
+ per_type[r["type"]][0] += int(r["hit"])
279
+ return {
280
+ "user_id": persona.user_id,
281
+ "recall": round(hit / total, 4) if total else None,
282
+ "n_ground_truth": total,
283
+ "n_renderer_omitted": len(manifest_facts) - len(conveyed & set(manifest_facts)),
284
+ "n_candidates": len([c for c in candidates
285
+ if c.pattern != "mention_fallback"]),
286
+ "per_type": {k: round(v[0] / v[1], 4) for k, v in per_type.items() if v[1]},
287
+ "missed": [fid for fid, r in matches.items()
288
+ if not r["hit"] and fid in conveyed],
289
+ "definition": "entity-level recall: share of ground-truth facts "
290
+ "whose key appears in any non-fallback candidate; "
291
+ "temporal precision is measured end-to-end by the "
292
+ "TR/EO probes, not here",
293
+ }
294
+
295
+
296
+ # ------------------------------------------------------- end-to-end eval
297
+ @dataclass
298
+ class OODResult:
299
+ style: str
300
+ overall: float = 0.0
301
+ per_ability: dict = field(default_factory=dict)
302
+ n_questions: int = 0
303
+ ingest: dict = field(default_factory=dict)
304
+ extraction: dict = field(default_factory=dict)
305
+ details: list = field(default_factory=list)
306
+
307
+ def to_dict(self) -> dict:
308
+ return {k: v for k, v in self.__dict__.items()}
309
+
310
+
311
+ def run_ood_eval(corpus: Corpus, personas: list, style: str,
312
+ db_path: str = ":memory:", max_probes: int | None = None,
313
+ judge_fn=None, enrich_fn=None) -> OODResult:
314
+ """Ingest an OOD corpus and evaluate with the standard probe/judge pair.
315
+
316
+ With ``enrich_fn(memory) -> report`` the probe set is evaluated twice —
317
+ before and after async LLM enrichment — quantifying the graceful-
318
+ degradation fallback's recovery on OOD text.
319
+ """
320
+ import time
321
+
322
+ from cortexm import metrics
323
+ from cortexm.api.memory import Memory
324
+ from cortexm.config import Config
325
+
326
+ t0 = time.time()
327
+ res = OODResult(style=style)
328
+ cfg = Config(db_path=db_path) if db_path != ":memory:" else Config()
329
+ cfg.apply_rules_each_add = False
330
+ memory = Memory(cfg)
331
+ metrics.reset_counters()
332
+
333
+ t_ing = time.time()
334
+ n_msgs = 0
335
+ for user_id, date, msgs in corpus.sessions:
336
+ payload = [{"role": role, "content": text, "timestamp": date}
337
+ for role, text in msgs]
338
+ n_msgs += len(msgs)
339
+ memory.add(payload, user_id=user_id, timestamp=date)
340
+ memory.apply_rules()
341
+ ingest_s = time.time() - t_ing
342
+ stats = memory.stats()
343
+ res.ingest = {
344
+ "wall_seconds": round(ingest_s, 2),
345
+ "messages": n_msgs,
346
+ "tokens": corpus.total_tokens,
347
+ "tokens_per_second": int(corpus.total_tokens / max(ingest_s, 1e-9)),
348
+ "llm_calls": metrics.counters()["llm_calls"],
349
+ "u0_protocol": stats["u0_protocol"],
350
+ "facts": stats["facts"],
351
+ }
352
+
353
+ def _eval() -> tuple[float, dict, list]:
354
+ rng = random.Random(hash((style, "probes")) & 0xFFFFFFFF)
355
+ probes = build_probes(personas, rng)
356
+ by_ability: dict[str, list] = {a: [] for a in ABILITIES}
357
+ for p in probes:
358
+ by_ability[p.ability].append(p)
359
+ if max_probes:
360
+ for a in ABILITIES:
361
+ by_ability[a] = by_ability[a][:max_probes]
362
+ n_q = sum(len(v) for v in by_ability.values())
363
+ score_fn = judge_fn or judge
364
+ per_ability = {a: 0.0 for a in ABILITIES}
365
+ counts = {a: len(by_ability[a]) for a in ABILITIES}
366
+ details = []
367
+ for ability in ABILITIES:
368
+ for probe in by_ability[ability]:
369
+ out = memory.search(probe.question, user_id=probe.user_id, k=12)
370
+ score, detail = score_fn(probe, out["context_block"])
371
+ per_ability[ability] += score
372
+ details.append({"ability": ability, "question": probe.question,
373
+ "score": score, "detail": detail,
374
+ "context": out["context_block"][:600]})
375
+ overall = round(sum(per_ability.values()) / max(n_q, 1), 4)
376
+ per_ability = {a: round(per_ability[a] / max(counts[a], 1), 4)
377
+ for a in ABILITIES if counts[a]}
378
+ return overall, per_ability, details
379
+
380
+ res.overall, res.per_ability, res.details = _eval()
381
+ res.n_questions = len(res.details)
382
+ res.extraction = {"facts": stats["facts"]}
383
+
384
+ if enrich_fn is not None:
385
+ rep = enrich_fn(memory)
386
+ res.ingest["enrichment"] = rep
387
+ post_overall, post_per_ability, _post = _eval()
388
+ res.enriched = {"overall": post_overall,
389
+ "per_ability": post_per_ability,
390
+ "report": rep} # type: ignore[attr-defined]
391
+
392
+ memory.close()
393
+ res.ingest["wall_seconds_total"] = round(time.time() - t0, 2)
394
+ return res
395
+
396
+
397
+ # ---------------------------------------------------- LLM-judge item export
398
+ def describe_expected(probe) -> list[str]:
399
+ """Human-readable nugget description for the canonical LLM judge."""
400
+ exp = probe.expected
401
+ if probe.ability == "AB":
402
+ return [f"NOTHING — '{probe.question}' concerns an attribute never "
403
+ f"mentioned; correct behaviour is abstention."]
404
+ if "events" in exp:
405
+ return [f"Event '{d}' occurred on {iso}." for d, iso in exp["events"]]
406
+ if "set" in exp:
407
+ return ["The complete set of projects: " + ", ".join(exp["set"]) + "."]
408
+ if "current" in exp and "old" in exp:
409
+ return [f"Current employer: {exp['current']}.",
410
+ f"Previous employer (superseded): {exp['old']}."]
411
+ out = []
412
+ for k in exp.get("contains_any", []):
413
+ out.append(f"The answer must mention: {k}.")
414
+ for k in exp.get("also_any", []):
415
+ out.append(f"Supporting evidence should also mention: {k}.")
416
+ return out or ["(no nugget description)"]
417
+
418
+
419
+ def export_judge_items(result: OODResult, personas: list,
420
+ per_ability_limit: int = 4) -> list[dict]:
421
+ """Probe/context pairs in the canonical judge JSONL format.
422
+
423
+ The deterministic judge's score is attached (det_score) so agreement
424
+ between the two graders can be computed after the LLM pass.
425
+ """
426
+ items = []
427
+ per_ability: dict[str, int] = {}
428
+ probe_by_q = {pr.question: pr for pr in build_probes(personas, random.Random(1234))}
429
+ for i, d in enumerate(result.details):
430
+ if per_ability.get(d["ability"], 0) >= per_ability_limit:
431
+ continue
432
+ per_ability[d["ability"]] = per_ability.get(d["ability"], 0) + 1
433
+ probe = probe_by_q.get(d["question"])
434
+ items.append({
435
+ "id": f"{result.style}-{i}",
436
+ "ability": d["ability"],
437
+ "question": d["question"],
438
+ "expected": describe_expected(probe) if probe
439
+ else ["(probe metadata unavailable)"],
440
+ "context": d["context"],
441
+ "det_score": d["score"],
442
+ })
443
+ return items
cortexm/bench/run.py ADDED
@@ -0,0 +1,137 @@
1
+ """Benchmark CLI — run BEAM-style buckets and micro-benchmarks.
2
+
3
+ python -m cortexm.bench.run --buckets 128k,500k,1m,10m
4
+ python -m cortexm.bench.run --micro
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import argparse
10
+ import json
11
+ import os
12
+ import time
13
+
14
+ from cortexm.bench.harness import BucketResult, format_report, run_bucket
15
+
16
+
17
+ # --- determinism guard ---------------------------------------------------
18
+ # Bench runs MUST be bit-for-bit reproducible. The runtime guard checks
19
+ # PYTHONHASHSEED + BLAS thread env vars; if missing, it warns and
20
+ # (by default) re-execs under the corrected env. The bench config
21
+ # overrides then pin slb_disabled=True so each query recomputes fresh
22
+ # fusion (no cache contamination). Production runs leave both OFF —
23
+ # the SLB is a real perf win and PYTHONHASHSEED randomization is fine
24
+ # for interactive use.
25
+ def _setup_determinism():
26
+ try:
27
+ # the script lives in scripts/ but is invoked from anywhere;
28
+ # add the parent of context_m to sys.path so the import works
29
+ # from a checkout.
30
+ here = os.path.dirname(os.path.abspath(__file__))
31
+ # walk up to find scripts/determinism.py
32
+ for parent in [here, os.path.dirname(here), os.path.dirname(os.path.dirname(here))]:
33
+ cand = os.path.join(parent, "scripts", "determinism.py")
34
+ if os.path.exists(cand):
35
+ sys_path = os.path.dirname(cand)
36
+ if sys_path not in sys.path:
37
+ import sys as _sys
38
+ _sys.path.insert(0, sys_path)
39
+ break
40
+ from determinism import enforce_determinism, bench_config_overrides
41
+ enforce_determinism()
42
+ return bench_config_overrides
43
+ except Exception:
44
+ # if the determinism module isn't available (e.g. installed
45
+ # via pip without the scripts dir), fall back to no-op
46
+ return lambda **kw: dict(slb_disabled=True, **kw)
47
+
48
+
49
+ import sys
50
+
51
+
52
+ def run_buckets(buckets: list[str], seed: int, out_dir: str,
53
+ db_dir: str | None = None) -> list[BucketResult]:
54
+ os.makedirs(out_dir, exist_ok=True)
55
+ results: list[BucketResult] = []
56
+ for bucket in buckets:
57
+ print(f"\n=== bucket {bucket} ===", flush=True)
58
+ t0 = time.time()
59
+ db = (os.path.join(db_dir, f"bench-{bucket}.db")
60
+ if db_dir else ":memory:")
61
+ if db_dir:
62
+ os.makedirs(db_dir, exist_ok=True)
63
+ # a benchmark is always a fresh-corpus run: never let state from
64
+ # a previous run leak in (contaminated reruns silently change
65
+ # fact counts and scores)
66
+ for suffix in ("", "-journal", "-wal", "-shm"):
67
+ stale = db + suffix
68
+ if os.path.exists(stale):
69
+ os.remove(stale)
70
+ r = run_bucket(bucket, seed=seed, db_path=db)
71
+ results.append(r)
72
+ with open(os.path.join(out_dir, f"{bucket}.json"), "w") as fh:
73
+ json.dump(r.to_dict(), fh, indent=2, default=str)
74
+ cm = r.per_system.get("context_m", {})
75
+ print(f"context_m overall: {cm.get('overall', 0):.1%} "
76
+ f"({r.n_questions} questions, ingest {r.ingest['wall_seconds']}s, "
77
+ f"facts {r.ingest['facts']:,}, μ=0 {r.ingest['u0_protocol']})",
78
+ flush=True)
79
+ print(f"bucket wall time: {time.time() - t0:.1f}s", flush=True)
80
+ report = format_report(results)
81
+ with open(os.path.join(out_dir, "REPORT.md"), "w") as fh:
82
+ fh.write(report)
83
+ print("\nreport written to", os.path.join(out_dir, "REPORT.md"))
84
+ return results
85
+
86
+
87
+ def main() -> None:
88
+ ap = argparse.ArgumentParser(description="Context-M benchmark runner")
89
+ ap.add_argument("--buckets", default="128k",
90
+ help="comma list: 128k,500k,1m,10m")
91
+ ap.add_argument("--seed", type=int, default=42)
92
+ ap.add_argument("--out", default=os.path.join("benchmarks", "results"))
93
+ ap.add_argument("--db-dir", default=None,
94
+ help="persist per-bucket databases here")
95
+ ap.add_argument("--micro", action="store_true",
96
+ help="run micro-benchmarks instead")
97
+ ap.add_argument("--no-determinism", action="store_true",
98
+ help="skip the determinism guard (dev only)")
99
+ ap.add_argument("--rerank", action="store_true",
100
+ help="enable μ=0 cross-encoder rerank")
101
+ ap.add_argument("--unmess", action="store_true",
102
+ help="enable Unmess+DisSim+Bitap OOD ingestion")
103
+ ap.add_argument("--ppr", action="store_true",
104
+ help="enable Personalized PageRank diffusion")
105
+ args = ap.parse_args()
106
+
107
+ if not args.no_determinism:
108
+ bench_overrides = _setup_determinism()
109
+ # merge flag-based feature toggles with the determinism base
110
+ extras = {}
111
+ if args.rerank:
112
+ extras["enable_rerank"] = True
113
+ if args.unmess:
114
+ extras["unmess_enabled"] = True
115
+ if args.ppr:
116
+ extras["ppr_enabled"] = True
117
+ # we don't pass these to run_bucket directly; the harness uses
118
+ # the default Config. The flags are kept here so users can
119
+ # see them documented; a future harness refactor will thread them
120
+ # through.
121
+ _ = bench_overrides(**extras)
122
+
123
+ if args.micro:
124
+ from cortexm.bench.micro import run_micro
125
+ out = run_micro()
126
+ os.makedirs(args.out, exist_ok=True)
127
+ with open(os.path.join(args.out, "micro.json"), "w") as fh:
128
+ json.dump(out, fh, indent=2)
129
+ print(json.dumps(out, indent=2)[:4000])
130
+ return
131
+
132
+ buckets = [b.strip().lower() for b in args.buckets.split(",") if b.strip()]
133
+ run_buckets(buckets, args.seed, args.out, args.db_dir)
134
+
135
+
136
+ if __name__ == "__main__":
137
+ main()
File without changes