cortexm 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- context_m.py +17 -0
- cortexm/__init__.py +45 -0
- cortexm/accel.py +403 -0
- cortexm/api/__init__.py +0 -0
- cortexm/api/chaos.py +118 -0
- cortexm/api/memory.py +635 -0
- cortexm/bench/__init__.py +0 -0
- cortexm/bench/abilities.py +311 -0
- cortexm/bench/baselines.py +89 -0
- cortexm/bench/beam_loader.py +317 -0
- cortexm/bench/generator.py +376 -0
- cortexm/bench/harness.py +211 -0
- cortexm/bench/messy.py +218 -0
- cortexm/bench/micro.py +251 -0
- cortexm/bench/ood.py +443 -0
- cortexm/bench/run.py +137 -0
- cortexm/bridge/__init__.py +0 -0
- cortexm/bridge/dates.py +178 -0
- cortexm/bridge/decoders.py +204 -0
- cortexm/bridge/enrich.py +255 -0
- cortexm/bridge/extractor.py +316 -0
- cortexm/bridge/fallback.py +332 -0
- cortexm/bridge/onnx_runtime.py +158 -0
- cortexm/bridge/patterns.py +760 -0
- cortexm/bridge/ppr.py +104 -0
- cortexm/bridge/prefilter.py +188 -0
- cortexm/bridge/query_extract.py +420 -0
- cortexm/bridge/reader.py +1174 -0
- cortexm/bridge/rerank.py +204 -0
- cortexm/bridge/writer.py +492 -0
- cortexm/cli.py +295 -0
- cortexm/cognition/__init__.py +53 -0
- cortexm/cognition/abstraction.py +192 -0
- cortexm/cognition/analogy.py +159 -0
- cortexm/cognition/engine.py +204 -0
- cortexm/cognition/gaps.py +365 -0
- cortexm/cognition/scanner.py +204 -0
- cortexm/config.py +375 -0
- cortexm/cortexm.py +8 -0
- cortexm/enterprise/__init__.py +0 -0
- cortexm/enterprise/audit.py +178 -0
- cortexm/enterprise/governance.py +239 -0
- cortexm/errors.py +35 -0
- cortexm/features/__init__.py +0 -0
- cortexm/features/git.py +204 -0
- cortexm/features/prefetch.py +88 -0
- cortexm/features/zk.py +105 -0
- cortexm/federation/__init__.py +39 -0
- cortexm/federation/crdt.py +275 -0
- cortexm/federation/fabric.py +109 -0
- cortexm/federation/hlc.py +80 -0
- cortexm/federation/node.py +145 -0
- cortexm/federation/schema_report.py +73 -0
- cortexm/federation/transport.py +164 -0
- cortexm/index/__init__.py +19 -0
- cortexm/index/nsg.py +386 -0
- cortexm/mcp/__init__.py +0 -0
- cortexm/mcp/server.py +985 -0
- cortexm/metrics.py +62 -0
- cortexm/migrate/__init__.py +0 -0
- cortexm/migrate/importers.py +192 -0
- cortexm/provenance/__init__.py +78 -0
- cortexm/provenance/agent.py +214 -0
- cortexm/provenance/cose.py +201 -0
- cortexm/provenance/scitt.py +258 -0
- cortexm/provenance/vc.py +250 -0
- cortexm/security/__init__.py +0 -0
- cortexm/security/crypto.py +162 -0
- cortexm/security/hashes.py +140 -0
- cortexm/security/injection.py +149 -0
- cortexm/security/mind.py +154 -0
- cortexm/security/pii.py +265 -0
- cortexm/security/rbac.py +169 -0
- cortexm/security/sandbox.py +131 -0
- cortexm/security/zk_hamming.py +142 -0
- cortexm/security/zk_sql.py +485 -0
- cortexm/server/__init__.py +0 -0
- cortexm/server/metrics.py +88 -0
- cortexm/server/rest.py +936 -0
- cortexm/server/sparql.py +984 -0
- cortexm/text/__init__.py +0 -0
- cortexm/text/dissim.py +252 -0
- cortexm/text/embedder.py +155 -0
- cortexm/text/fuzzy.py +218 -0
- cortexm/text/idiolect.py +253 -0
- cortexm/text/labse.py +374 -0
- cortexm/text/tokenizer.py +79 -0
- cortexm/trace/__init__.py +0 -0
- cortexm/trace/blob_arena.py +277 -0
- cortexm/trace/consolidate.py +337 -0
- cortexm/trace/contradictions.py +69 -0
- cortexm/trace/dedup.py +114 -0
- cortexm/trace/edges.py +214 -0
- cortexm/trace/fact.py +121 -0
- cortexm/trace/fade.py +245 -0
- cortexm/trace/lifecycle.py +112 -0
- cortexm/trace/rebuild.py +173 -0
- cortexm/trace/rules.py +171 -0
- cortexm/trace/store.py +680 -0
- cortexm/trace/structural.py +183 -0
- cortexm/trace/tmt.py +335 -0
- cortexm/util.py +148 -0
- cortexm/vsa/__init__.py +0 -0
- cortexm/vsa/attribution.py +149 -0
- cortexm/vsa/cleanup.py +161 -0
- cortexm/vsa/codecs.py +397 -0
- cortexm/vsa/hologram_overlay.py +139 -0
- cortexm/vsa/index.py +163 -0
- cortexm/vsa/ops.py +149 -0
- cortexm/vsa/palace.py +446 -0
- cortexm/vsa/role_vectors.py +236 -0
- cortexm/vsa/slb.py +78 -0
- cortexm/vsa/tlsh_trie.py +137 -0
- cortexm/vsa/working_memory.py +249 -0
- cortexm-0.3.0.dist-info/METADATA +482 -0
- cortexm-0.3.0.dist-info/RECORD +120 -0
- cortexm-0.3.0.dist-info/WHEEL +5 -0
- cortexm-0.3.0.dist-info/entry_points.txt +2 -0
- cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
- cortexm-0.3.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,420 @@
|
|
|
1
|
+
"""Query-time extraction — store raw + extract when needed.
|
|
2
|
+
|
|
3
|
+
arxiv research (Con #3 from user's research): the agent memory literature
|
|
4
|
+
shows that forcing extraction at ingest time (μ=0) is the wrong target.
|
|
5
|
+
Better: store raw text chunks + embeddings at ingest; extract triples
|
|
6
|
+
lazily at query time when the query disambiguates context.
|
|
7
|
+
|
|
8
|
+
This module implements the hybrid via MemoryWriter as the single write
|
|
9
|
+
path. Previous versions had a bespoke `_store_fact` that wrote to the
|
|
10
|
+
Trace without quarantine / contradiction / lifecycle / palace encoding /
|
|
11
|
+
edge wiring — that bypassed MemoryWriter's pronoun resolution and
|
|
12
|
+
entity tracking, which is why the query-time path measured *worse* than
|
|
13
|
+
baseline (0.835 vs 1.017 recall). Now both ingest and query-time
|
|
14
|
+
extraction go through `MemoryWriter.add()` / `ingest_candidates()` so
|
|
15
|
+
the standard pipeline runs end-to-end.
|
|
16
|
+
|
|
17
|
+
The bi-temporal model is preserved: every fact has `valid_at` (when
|
|
18
|
+
the fact became true) AND `extracted_at` (when we materialized it).
|
|
19
|
+
The `extracted_at` timestamp is stored in the fact's `provenance` dict
|
|
20
|
+
under the `extracted_at` key, so the existing schema doesn't need a
|
|
21
|
+
migration. Both axes are queryable via `provenance ->> 'extracted_at'`.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import datetime as _dt
|
|
27
|
+
from dataclasses import dataclass
|
|
28
|
+
from typing import Any
|
|
29
|
+
|
|
30
|
+
from cortexm.util import iso, new_id
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass
|
|
34
|
+
class RawChunk:
|
|
35
|
+
"""A raw text chunk stored at ingest, awaiting query-time extraction.
|
|
36
|
+
|
|
37
|
+
This is now a logical handle only — the chunk itself lives in the
|
|
38
|
+
standard `chunks` table managed by TraceStore.add_chunk(). This
|
|
39
|
+
dataclass is kept for backward compatibility with callers that
|
|
40
|
+
inspect the return value of `QueryTimeExtractor.ingest()`.
|
|
41
|
+
"""
|
|
42
|
+
chunk_id: str
|
|
43
|
+
text: str
|
|
44
|
+
user_id: str
|
|
45
|
+
valid_at: str
|
|
46
|
+
extracted_at: str | None = None
|
|
47
|
+
source_hash: str = ""
|
|
48
|
+
embedding: Any | None = None
|
|
49
|
+
consumed: bool = False
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class QueryTimeExtractor:
|
|
53
|
+
"""Hybrid ingest + query-time extraction orchestrator.
|
|
54
|
+
|
|
55
|
+
NOW routes every write through `MemoryWriter` so query-time-extracted
|
|
56
|
+
facts go through the SAME quarantine, contradiction, lifecycle, palace
|
|
57
|
+
encoding, and edge wiring as ingest-time facts. No bespoke DB writes,
|
|
58
|
+
no separate `_store_fact` that bypasses the standard pipeline.
|
|
59
|
+
|
|
60
|
+
Usage (preferred — let the orchestrator build the writer):
|
|
61
|
+
extractor = QueryTimeExtractor(
|
|
62
|
+
palace, store, embedder,
|
|
63
|
+
writer=mem, # MemoryWriter — REQUIRED
|
|
64
|
+
dissim=dissim, idiolect=idiolect,
|
|
65
|
+
pattern_extractor=mem.extractor)
|
|
66
|
+
|
|
67
|
+
Ingest (delegates to MemoryWriter.add — does chunk + pattern + palace
|
|
68
|
+
+ lifecycle + edges, all μ=0):
|
|
69
|
+
out = extractor.ingest(text, user_id, valid_at="2026-03-01")
|
|
70
|
+
|
|
71
|
+
Query (palace search → for each un-extracted chunk, lazily extract
|
|
72
|
+
via dissim + pattern extractor + writer.ingest_candidates):
|
|
73
|
+
results = extractor.query("where does Alice work?", user_id)
|
|
74
|
+
"""
|
|
75
|
+
|
|
76
|
+
def __init__(self, palace, store, embedder, *, writer,
|
|
77
|
+
reader=None, dissim=None, idiolect=None,
|
|
78
|
+
pattern_extractor=None, cleanup=None, overlay=None) -> None:
|
|
79
|
+
# writer is now REQUIRED — every write path goes through it
|
|
80
|
+
if writer is None:
|
|
81
|
+
raise ValueError(
|
|
82
|
+
"QueryTimeExtractor now requires a `writer` (MemoryWriter) "
|
|
83
|
+
"argument so query-time-extracted facts route through the "
|
|
84
|
+
"standard pipeline. Previous versions bypassed MemoryWriter "
|
|
85
|
+
"and measured 0.835 vs 1.017 recall — this regression is "
|
|
86
|
+
"the reason for the refactor.")
|
|
87
|
+
self.palace = palace
|
|
88
|
+
self.store = store
|
|
89
|
+
self.embedder = embedder
|
|
90
|
+
self.writer = writer
|
|
91
|
+
# reader is OPTIONAL but recommended — when present, query()
|
|
92
|
+
# delegates primary retrieval to it so query-time path has
|
|
93
|
+
# parity with the ingest path's mem.search() (intent planner,
|
|
94
|
+
# fusion, entity-hop, mentioned-damping). Without a reader,
|
|
95
|
+
# query() falls back to raw palace.search() which is dumber
|
|
96
|
+
# and was the cause of the 0.835 recall regression.
|
|
97
|
+
self.reader = reader
|
|
98
|
+
self.dissim = dissim
|
|
99
|
+
self.idiolect = idiolect
|
|
100
|
+
# default to the writer's own extractor so the SAME patterns run
|
|
101
|
+
# at query time as at ingest time
|
|
102
|
+
self.pattern_extractor = pattern_extractor or writer.extractor
|
|
103
|
+
self.cleanup = cleanup
|
|
104
|
+
self.overlay = overlay
|
|
105
|
+
|
|
106
|
+
# ------------------------------------------------------------------
|
|
107
|
+
def ingest(self, text: str, user_id: str = "default",
|
|
108
|
+
valid_at: str | None = None,
|
|
109
|
+
run_pattern: bool = True) -> dict:
|
|
110
|
+
"""Store raw text chunk + embedding. Run pattern extractor as
|
|
111
|
+
best-effort (μ=0 path) by delegating to MemoryWriter.add().
|
|
112
|
+
|
|
113
|
+
- run_pattern=True (default): full MemoryWriter.add() — chunk
|
|
114
|
+
insert + pattern extraction + lifecycle + palace encoding +
|
|
115
|
+
edges + Datalog materialization. Returns the writer's result
|
|
116
|
+
dict.
|
|
117
|
+
- run_pattern=False: raw-only — just `store.add_chunk()` + palace
|
|
118
|
+
embed, no extraction. The chunk is queryable; facts will be
|
|
119
|
+
extracted lazily at query time. Returns a small handle dict.
|
|
120
|
+
|
|
121
|
+
CRITICAL: chunks are added to the palace so query-time search
|
|
122
|
+
can retrieve them by VSA similarity. This is done in BOTH paths
|
|
123
|
+
(writer.add() does it implicitly via encode_fact; raw-only path
|
|
124
|
+
does it explicitly here).
|
|
125
|
+
"""
|
|
126
|
+
# observe idiolect (mutates normalizer state for later queries)
|
|
127
|
+
if self.idiolect:
|
|
128
|
+
self.idiolect.observe(user_id, text)
|
|
129
|
+
|
|
130
|
+
if not run_pattern:
|
|
131
|
+
# raw-only: skip extraction, just store the chunk + embed it
|
|
132
|
+
ts = self._parse_ts(valid_at)
|
|
133
|
+
chunk_id = self.store.add_chunk(
|
|
134
|
+
text, user_id=user_id, ts=ts,
|
|
135
|
+
source="query_time_raw")
|
|
136
|
+
emb = self.embedder.embed(text)
|
|
137
|
+
try:
|
|
138
|
+
self.palace.add(chunk_id, emb)
|
|
139
|
+
except Exception:
|
|
140
|
+
pass
|
|
141
|
+
return {"event": "RAW_CHUNK", "chunk_id": chunk_id,
|
|
142
|
+
"results": [], "commit": None,
|
|
143
|
+
"stats": {"messages": 1,
|
|
144
|
+
"tokens": len(text) // 4,
|
|
145
|
+
"facts_inserted": 0, "llm_calls": 0}}
|
|
146
|
+
|
|
147
|
+
# full path: delegate to MemoryWriter.add() — this adds the
|
|
148
|
+
# chunk via store.add_chunk and encodes any extracted FACTS
|
|
149
|
+
# into the palace (via encode_fact on the bound triple vector).
|
|
150
|
+
# We ADDITIONALLY embed the raw chunk text into the palace so
|
|
151
|
+
# query-time palace.search() can retrieve the chunk itself by
|
|
152
|
+
# text similarity (fact vectors are bound role-fillers, not
|
|
153
|
+
# text embeddings, so they don't help find the source chunk).
|
|
154
|
+
# CRITICAL: normalize text via idiolect BEFORE writer.add() so
|
|
155
|
+
# the pattern extractor sees the same canonical text as the
|
|
156
|
+
# +unmess path — otherwise patterns miss on slang that the
|
|
157
|
+
# normalizer could fix (e.g. "I work @ Microsoft" → "I work
|
|
158
|
+
# at Microsoft" via the text-speak escape hatch).
|
|
159
|
+
ts = self._parse_ts(valid_at)
|
|
160
|
+
norm_text = text
|
|
161
|
+
if self.idiolect:
|
|
162
|
+
norm_text = self.idiolect.normalize(user_id, text)
|
|
163
|
+
out = self.writer.add(
|
|
164
|
+
[{"role": "user", "content": norm_text}],
|
|
165
|
+
user_id=user_id, ts=ts, source="query_time_ingest")
|
|
166
|
+
# pull the most-recent chunk for this user — that's the chunk
|
|
167
|
+
# writer.add just inserted — and embed it into the palace
|
|
168
|
+
try:
|
|
169
|
+
row = self.store.conn.execute(
|
|
170
|
+
"SELECT id FROM chunks WHERE user_id=? "
|
|
171
|
+
"ORDER BY ts DESC LIMIT 1", (user_id,)).fetchone()
|
|
172
|
+
if row:
|
|
173
|
+
emb = self.embedder.embed(norm_text)
|
|
174
|
+
self.palace.add(row[0], emb)
|
|
175
|
+
except Exception:
|
|
176
|
+
pass
|
|
177
|
+
return out
|
|
178
|
+
|
|
179
|
+
# ------------------------------------------------------------------
|
|
180
|
+
def query(self, query_text: str, user_id: str = "default",
|
|
181
|
+
k: int = 5, extract_fresh: bool = True
|
|
182
|
+
) -> list[dict]:
|
|
183
|
+
"""Retrieve facts relevant to query; lazily extract from
|
|
184
|
+
un-extracted chunks through MemoryWriter.ingest_candidates
|
|
185
|
+
so the standard pipeline runs end-to-end.
|
|
186
|
+
|
|
187
|
+
Two-pass retrieval:
|
|
188
|
+
PASS 1 — Delegate to the structured reader (mem.search) when
|
|
189
|
+
available. This gives the query-time path retrieval PARITY
|
|
190
|
+
with the ingest path — same intent planner, fusion,
|
|
191
|
+
entity-hop, mentioned-damping. On clean queries like
|
|
192
|
+
"What is the works_at of user0?" the reader returns the
|
|
193
|
+
fact directly via the Trace, no raw chunk search needed.
|
|
194
|
+
|
|
195
|
+
PASS 2 — If the reader missed (or no reader was supplied),
|
|
196
|
+
fall back to raw palace.search() over chunk text embeddings
|
|
197
|
+
and lazily extract from un-extracted chunks via dissim +
|
|
198
|
+
pattern extractor + writer.ingest_candidates. This is the
|
|
199
|
+
"query-time extraction" win — patterns that missed at ingest
|
|
200
|
+
on slangy text now match after idiolect normalization +
|
|
201
|
+
dissim simplification, and the resulting facts land in the
|
|
202
|
+
Trace through the same quarantine / lifecycle / palace /
|
|
203
|
+
edges pipeline as ingest-time facts.
|
|
204
|
+
|
|
205
|
+
Returns list of dicts (shape compatible with both the reader
|
|
206
|
+
result format and the raw-chunk fallback format):
|
|
207
|
+
{"fact": {...}|None, "retrieval_path": "...",
|
|
208
|
+
"source_chunk_id": "...", "extracted_at": "..."|None,
|
|
209
|
+
"memory": "...", "score": float,
|
|
210
|
+
"raw_text": "..." (only when fact is None)}
|
|
211
|
+
"""
|
|
212
|
+
# NOTE: we deliberately do NOT normalize the query at the top
|
|
213
|
+
# level — the reader's intent planner expects clean structured
|
|
214
|
+
# queries like "What is the works_at of user0?" and idiolect
|
|
215
|
+
# normalization (with the text-speak escape hatch) can rewrite
|
|
216
|
+
# "is" / "of" / short tokens in ways that confuse the planner.
|
|
217
|
+
# Normalization is only applied to the raw-chunk retrieval path
|
|
218
|
+
# (PASS 2) where it actually helps match chunk text.
|
|
219
|
+
q_text = query_text
|
|
220
|
+
now = iso(_dt.datetime.now(_dt.timezone.utc))
|
|
221
|
+
out: list[dict] = []
|
|
222
|
+
|
|
223
|
+
# --- PASS 1: reader (structured retrieval) -----------------------
|
|
224
|
+
if self.reader is not None:
|
|
225
|
+
try:
|
|
226
|
+
# reader.search() returns a RetrievalResult dataclass;
|
|
227
|
+
# .memories() converts to list of {id, memory, score, ...}
|
|
228
|
+
# dicts — same shape as Memory.search()'s `results`.
|
|
229
|
+
# We parse the "subject | relation | value" memory string
|
|
230
|
+
# back into structured fields so downstream checks (e.g.
|
|
231
|
+
# the BEAM benchmark's value-match) work the same as
|
|
232
|
+
# for query-time-extracted facts.
|
|
233
|
+
rr = self.reader.search(
|
|
234
|
+
q_text, user_id=user_id, k=k)
|
|
235
|
+
for r in rr.memories():
|
|
236
|
+
mem_str = r.get("memory", "")
|
|
237
|
+
parts = [p.strip() for p in mem_str.split("|")]
|
|
238
|
+
fact_dict = {**r,
|
|
239
|
+
"subject": parts[0] if len(parts) > 0 else "",
|
|
240
|
+
"relation": parts[1] if len(parts) > 1 else "",
|
|
241
|
+
"value": parts[2] if len(parts) > 2 else ""}
|
|
242
|
+
out.append({
|
|
243
|
+
"fact": fact_dict,
|
|
244
|
+
"retrieval_path": "reader",
|
|
245
|
+
"source_chunk_id": r.get("id", ""),
|
|
246
|
+
"extracted_at": None, # reader facts lack this
|
|
247
|
+
"memory": mem_str,
|
|
248
|
+
"score": r.get("score", 0.0),
|
|
249
|
+
})
|
|
250
|
+
except Exception:
|
|
251
|
+
pass
|
|
252
|
+
|
|
253
|
+
# --- PASS 2: raw chunk retrieval + lazy reextract ---------------
|
|
254
|
+
# Only run if the reader returned < k results, OR no reader.
|
|
255
|
+
if len(out) < k:
|
|
256
|
+
q_emb = self.embedder.embed(q_text)
|
|
257
|
+
results = self.palace.search(q_emb, k=k * 2)
|
|
258
|
+
if results:
|
|
259
|
+
chunks: list[dict] = []
|
|
260
|
+
seen_chunk_ids = {r.get("source_chunk_id", "")
|
|
261
|
+
for r in out if r.get("source_chunk_id")}
|
|
262
|
+
for chunk_id, score in results:
|
|
263
|
+
if chunk_id in seen_chunk_ids:
|
|
264
|
+
continue
|
|
265
|
+
chunk = (self.store.get_chunk(chunk_id)
|
|
266
|
+
if hasattr(self.store, "get_chunk") else None)
|
|
267
|
+
if chunk:
|
|
268
|
+
chunk["_score"] = float(score)
|
|
269
|
+
chunks.append(chunk)
|
|
270
|
+
for chunk in chunks[: k - len(out)]:
|
|
271
|
+
already_extracted = self._chunk_has_facts(chunk["id"])
|
|
272
|
+
if extract_fresh and not already_extracted:
|
|
273
|
+
self._reextract_chunk(chunk, user_id, now)
|
|
274
|
+
facts = self._facts_for_chunk(chunk["id"])
|
|
275
|
+
if facts:
|
|
276
|
+
for f in facts:
|
|
277
|
+
out.append({
|
|
278
|
+
"fact": f,
|
|
279
|
+
"retrieval_path": "query_time_pattern",
|
|
280
|
+
"source_chunk_id": chunk["id"],
|
|
281
|
+
"extracted_at": (f.get("provenance") or {})
|
|
282
|
+
.get("extracted_at") or now,
|
|
283
|
+
"memory": f"{f.get('subject','')} | "
|
|
284
|
+
f"{f.get('relation','')} | "
|
|
285
|
+
f"{f.get('value','')}",
|
|
286
|
+
"score": chunk.get("_score", 0.0),
|
|
287
|
+
})
|
|
288
|
+
else:
|
|
289
|
+
out.append({
|
|
290
|
+
"fact": None,
|
|
291
|
+
"retrieval_path": "raw_chunk",
|
|
292
|
+
"source_chunk_id": chunk["id"],
|
|
293
|
+
"extracted_at": None,
|
|
294
|
+
"raw_text": chunk["text"],
|
|
295
|
+
"memory": chunk["text"],
|
|
296
|
+
"score": chunk.get("_score", 0.0),
|
|
297
|
+
})
|
|
298
|
+
return out
|
|
299
|
+
|
|
300
|
+
# ------------------------------------------------------------------
|
|
301
|
+
# Internals — all writes go through MemoryWriter.ingest_candidates
|
|
302
|
+
# ------------------------------------------------------------------
|
|
303
|
+
def _reextract_chunk(self, chunk: dict, user_id: str, now_iso: str
|
|
304
|
+
) -> int:
|
|
305
|
+
"""Run dissim + pattern extractor on chunk text; commit through
|
|
306
|
+
writer.ingest_candidates so the standard pipeline runs.
|
|
307
|
+
|
|
308
|
+
Returns the number of candidates sent to the writer (NOT the
|
|
309
|
+
number of facts inserted — the writer's lifecycle may merge,
|
|
310
|
+
skip, or supersede duplicates, which is exactly the parity we
|
|
311
|
+
want with the ingest path).
|
|
312
|
+
"""
|
|
313
|
+
text = chunk["text"]
|
|
314
|
+
# normalize via idiolect if available (helps pronoun resolution
|
|
315
|
+
# on text that was stored raw without idiolect normalization)
|
|
316
|
+
if self.idiolect:
|
|
317
|
+
text = self.idiolect.normalize(user_id, text)
|
|
318
|
+
# split into clauses via dissim (recursive syntactic splitting)
|
|
319
|
+
clauses = [text]
|
|
320
|
+
if self.dissim:
|
|
321
|
+
simplified = self.dissim.simplify_text(text)
|
|
322
|
+
if simplified:
|
|
323
|
+
clauses = [c.text for c in simplified]
|
|
324
|
+
# extract candidates from each clause
|
|
325
|
+
from cortexm.bridge.patterns import ExtractionContext
|
|
326
|
+
from datetime import datetime, timezone
|
|
327
|
+
all_candidates: list = []
|
|
328
|
+
for clause in clauses:
|
|
329
|
+
ctx = ExtractionContext(
|
|
330
|
+
user_id=user_id,
|
|
331
|
+
ts=datetime.now(timezone.utc),
|
|
332
|
+
# the writer's name-of / lexicon helpers are private
|
|
333
|
+
# by convention only — call them to give the pattern
|
|
334
|
+
# extractor the same pronoun / entity hints that the
|
|
335
|
+
# ingest path gets
|
|
336
|
+
subject_name=self._safe_name_of(user_id),
|
|
337
|
+
lexicon=self._safe_lexicon(user_id))
|
|
338
|
+
try:
|
|
339
|
+
cs = self.pattern_extractor.extract(clause, ctx)
|
|
340
|
+
all_candidates.extend(cs)
|
|
341
|
+
except Exception:
|
|
342
|
+
continue
|
|
343
|
+
if not all_candidates:
|
|
344
|
+
return 0
|
|
345
|
+
# commit through the standard pipeline — `source` ends up in
|
|
346
|
+
# the provenance dict as `enriched_by`, so audits can always
|
|
347
|
+
# tell these facts came from the query-time path
|
|
348
|
+
try:
|
|
349
|
+
inserted = self.writer.ingest_candidates(
|
|
350
|
+
all_candidates, user_id=user_id,
|
|
351
|
+
chunk_id=chunk["id"],
|
|
352
|
+
ts=datetime.now(timezone.utc),
|
|
353
|
+
source="query_time_reextract")
|
|
354
|
+
return inserted
|
|
355
|
+
except Exception:
|
|
356
|
+
return 0
|
|
357
|
+
|
|
358
|
+
# ------------------------------------------------------------------
|
|
359
|
+
def _chunk_has_facts(self, chunk_id: str) -> bool:
|
|
360
|
+
"""A chunk counts as 'extracted' if any fact in the trace has
|
|
361
|
+
an EXTRACTED_FROM edge pointing at it. No schema migration
|
|
362
|
+
needed — uses the existing edges table."""
|
|
363
|
+
try:
|
|
364
|
+
row = self.store.conn.execute(
|
|
365
|
+
"SELECT 1 FROM edges WHERE dst=? AND kind='EXTRACTED_FROM' "
|
|
366
|
+
"LIMIT 1", (chunk_id,)).fetchone()
|
|
367
|
+
return row is not None
|
|
368
|
+
except Exception:
|
|
369
|
+
return False
|
|
370
|
+
|
|
371
|
+
def _facts_for_chunk(self, chunk_id: str) -> list[dict]:
|
|
372
|
+
"""Active facts whose source_id is this chunk."""
|
|
373
|
+
try:
|
|
374
|
+
rows = self.store.conn.execute(
|
|
375
|
+
"SELECT * FROM facts WHERE source_id=? AND is_active=1 "
|
|
376
|
+
"ORDER BY confidence DESC", (chunk_id,)).fetchall()
|
|
377
|
+
out = []
|
|
378
|
+
for r in rows:
|
|
379
|
+
d = dict(r)
|
|
380
|
+
# provenance is stored as JSON text — parse to dict so
|
|
381
|
+
# callers can do prov.get(...) without surprise
|
|
382
|
+
if isinstance(d.get("provenance"), str):
|
|
383
|
+
try:
|
|
384
|
+
import json
|
|
385
|
+
d["provenance"] = json.loads(
|
|
386
|
+
d["provenance"] or "{}")
|
|
387
|
+
except Exception:
|
|
388
|
+
d["provenance"] = {}
|
|
389
|
+
out.append(d)
|
|
390
|
+
return out
|
|
391
|
+
except Exception:
|
|
392
|
+
return []
|
|
393
|
+
|
|
394
|
+
# ------------------------------------------------------------------
|
|
395
|
+
def _safe_name_of(self, user_id: str) -> str | None:
|
|
396
|
+
"""Pull the user's display name from the writer if available."""
|
|
397
|
+
try:
|
|
398
|
+
return self.writer._name_of(user_id)
|
|
399
|
+
except Exception:
|
|
400
|
+
return None
|
|
401
|
+
|
|
402
|
+
def _safe_lexicon(self, user_id: str) -> set:
|
|
403
|
+
"""Pull the user's learned lexicon from the writer if available."""
|
|
404
|
+
try:
|
|
405
|
+
return self.writer._lexicon(user_id)
|
|
406
|
+
except Exception:
|
|
407
|
+
return set()
|
|
408
|
+
|
|
409
|
+
@staticmethod
|
|
410
|
+
def _parse_ts(valid_at: str | None):
|
|
411
|
+
if not valid_at:
|
|
412
|
+
return None
|
|
413
|
+
try:
|
|
414
|
+
from cortexm.trace.store import parse_ts
|
|
415
|
+
return parse_ts(valid_at)
|
|
416
|
+
except Exception:
|
|
417
|
+
return None
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
__all__ = ["QueryTimeExtractor", "RawChunk"]
|