contextos-memory-runtime 1.0.0rc2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- contextos/__init__.py +3 -0
- contextos/__main__.py +6 -0
- contextos/api/__init__.py +1 -0
- contextos/api/routes/__init__.py +1 -0
- contextos/api/routes/desktop.py +322 -0
- contextos/api/routes/ingest.py +17 -0
- contextos/api/routes/memories.py +84 -0
- contextos/api/routes/models.py +81 -0
- contextos/api/routes/retrieval.py +89 -0
- contextos/api/routes/system.py +216 -0
- contextos/api/server.py +195 -0
- contextos/benchmarks/__init__.py +1 -0
- contextos/benchmarks/compilation.py +245 -0
- contextos/benchmarks/connectors.py +423 -0
- contextos/benchmarks/explainability.py +103 -0
- contextos/benchmarks/final.py +406 -0
- contextos/benchmarks/graph.py +310 -0
- contextos/benchmarks/graph_adversarial.py +525 -0
- contextos/benchmarks/mcp.py +324 -0
- contextos/benchmarks/model_routing.py +203 -0
- contextos/benchmarks/optimization.py +305 -0
- contextos/benchmarks/rescue_integration.py +127 -0
- contextos/benchmarks/retrieval.py +266 -0
- contextos/benchmarks/temporal.py +377 -0
- contextos/benchmarks/temporal_hotpath.py +76 -0
- contextos/benchmarks/terminal.py +62 -0
- contextos/cli/__init__.py +1 -0
- contextos/cli/app.py +932 -0
- contextos/cli/dashboard.py +174 -0
- contextos/cli/formatters.py +299 -0
- contextos/config/__init__.py +1 -0
- contextos/config/settings.py +160 -0
- contextos/connectors/__init__.py +6 -0
- contextos/connectors/fake.py +11 -0
- contextos/connectors/json_import.py +125 -0
- contextos/connectors/local_files.py +102 -0
- contextos/connectors/manager.py +293 -0
- contextos/connectors/models.py +62 -0
- contextos/connectors/protocols.py +11 -0
- contextos/core/__init__.py +103 -0
- contextos/core/enums.py +489 -0
- contextos/core/exceptions.py +293 -0
- contextos/core/models.py +1147 -0
- contextos/core/protocols.py +549 -0
- contextos/daemon/__init__.py +1 -0
- contextos/daemon/manager.py +510 -0
- contextos/daemon/state.py +127 -0
- contextos/daemon/wiring.py +296 -0
- contextos/demo.py +217 -0
- contextos/embedding/__init__.py +1 -0
- contextos/embedding/deterministic.py +76 -0
- contextos/embedding/sentence_transformers.py +80 -0
- contextos/mcp/__init__.py +5 -0
- contextos/mcp/server.py +269 -0
- contextos/providers/__init__.py +13 -0
- contextos/providers/fake.py +217 -0
- contextos/providers/ollama.py +297 -0
- contextos/providers/openai_compatible.py +337 -0
- contextos/services/__init__.py +1 -0
- contextos/services/compilation.py +535 -0
- contextos/services/explainability.py +553 -0
- contextos/services/extraction.py +311 -0
- contextos/services/graph.py +524 -0
- contextos/services/graph_retrieval.py +143 -0
- contextos/services/ingestion.py +143 -0
- contextos/services/inspection.py +174 -0
- contextos/services/memory.py +291 -0
- contextos/services/model_service.py +409 -0
- contextos/services/optimization.py +426 -0
- contextos/services/privacy.py +331 -0
- contextos/services/retrieval.py +302 -0
- contextos/services/retrieval_index.py +88 -0
- contextos/services/router.py +302 -0
- contextos/services/secret_scanner.py +207 -0
- contextos/services/telemetry_query.py +102 -0
- contextos/services/temporal.py +500 -0
- contextos/services/token_counter.py +222 -0
- contextos/storage/__init__.py +1 -0
- contextos/storage/connector_repo.py +67 -0
- contextos/storage/database.py +497 -0
- contextos/storage/event_repo.py +137 -0
- contextos/storage/graph_repo.py +228 -0
- contextos/storage/lexical/__init__.py +1 -0
- contextos/storage/lexical/bm25.py +134 -0
- contextos/storage/memory_repo.py +589 -0
- contextos/storage/relation_repo.py +80 -0
- contextos/storage/telemetry_repo.py +481 -0
- contextos/storage/vector/__init__.py +1 -0
- contextos/storage/vector/in_memory.py +162 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/METADATA +143 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/RECORD +93 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/WHEEL +4 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,535 @@
|
|
|
1
|
+
"""Deterministic query-aware compilation of selected memory facts."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
import re
|
|
8
|
+
import time
|
|
9
|
+
from collections.abc import Iterable
|
|
10
|
+
from dataclasses import dataclass
|
|
11
|
+
from uuid import UUID
|
|
12
|
+
|
|
13
|
+
from contextos.core.enums import (
|
|
14
|
+
CandidateTemporalStatus,
|
|
15
|
+
CompilationStrategy,
|
|
16
|
+
CompilerInputKind,
|
|
17
|
+
CompressionLevel,
|
|
18
|
+
FactExclusionReason,
|
|
19
|
+
MemoryStatus,
|
|
20
|
+
PrivacyLevel,
|
|
21
|
+
)
|
|
22
|
+
from contextos.core.models import (
|
|
23
|
+
CompilationConfig,
|
|
24
|
+
CompilationTrace,
|
|
25
|
+
CompiledContext,
|
|
26
|
+
ContextFact,
|
|
27
|
+
ExcludedContextFact,
|
|
28
|
+
ScoredMemory,
|
|
29
|
+
SelectionResult,
|
|
30
|
+
StageTrace,
|
|
31
|
+
)
|
|
32
|
+
from contextos.core.protocols import TokenCounter
|
|
33
|
+
from contextos.services.optimization import (
|
|
34
|
+
information_tokens,
|
|
35
|
+
redundancy_similarity,
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
_NEGATION = re.compile(
|
|
40
|
+
r"\b(?:don't|doesn't|do not|does not|did not|never|no longer|not anymore|stopped)\b",
|
|
41
|
+
re.IGNORECASE,
|
|
42
|
+
)
|
|
43
|
+
_UNCERTAINTY = re.compile(
|
|
44
|
+
r"\b(?:might|may|maybe|perhaps|possibly|could|uncertain|not sure)\b",
|
|
45
|
+
re.IGNORECASE,
|
|
46
|
+
)
|
|
47
|
+
_HISTORICAL = re.compile(
|
|
48
|
+
r"\b(?:previously|before|formerly|historically|used to|during a previous|stopped)\b",
|
|
49
|
+
re.IGNORECASE,
|
|
50
|
+
)
|
|
51
|
+
_CURRENT = re.compile(r"\b(?:currently|now|today|still)\b", re.IGNORECASE)
|
|
52
|
+
_FUTURE = re.compile(r"\b(?:will|plan to|intends? to|going to|might learn)\b", re.IGNORECASE)
|
|
53
|
+
_CAUSAL = re.compile(
|
|
54
|
+
r"\b(?:because|due to|caused by|as a result|therefore|so that)\b",
|
|
55
|
+
re.IGNORECASE,
|
|
56
|
+
)
|
|
57
|
+
_WHY_QUERY = re.compile(r"\b(?:why|reason|cause|because)\b", re.IGNORECASE)
|
|
58
|
+
_CLAUSE_BOUNDARY = re.compile(r"(?<=[.!?])\s+|;\s+|,\s+(?=and\b)")
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
@dataclass(frozen=True)
|
|
62
|
+
class _CompilerInput:
|
|
63
|
+
scored: ScoredMemory
|
|
64
|
+
kind: CompilerInputKind
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _compiler_tokens(text: str) -> frozenset[str]:
|
|
68
|
+
"""Compiler-local lexical normalization without changing Phase 5."""
|
|
69
|
+
return frozenset(
|
|
70
|
+
"local" if token == "locally" else token
|
|
71
|
+
for token in information_tokens(text)
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def fact_is_supported(fact_text: str, source_text: str) -> bool:
|
|
76
|
+
"""Return whether fact tokens occur in source order without additions."""
|
|
77
|
+
pattern = r"[a-z0-9]+(?:[+#._-][a-z0-9]+)*"
|
|
78
|
+
fact_tokens = re.findall(pattern, fact_text.casefold())
|
|
79
|
+
source_tokens = iter(re.findall(pattern, source_text.casefold()))
|
|
80
|
+
return all(any(source == token for source in source_tokens) for token in fact_tokens)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
class QueryAwareContextCompiler:
|
|
84
|
+
"""Extract, merge, order, and serialize source-supported facts."""
|
|
85
|
+
|
|
86
|
+
def __init__(self, *, token_counter: TokenCounter) -> None:
|
|
87
|
+
self._token_counter = token_counter
|
|
88
|
+
|
|
89
|
+
async def compile(
|
|
90
|
+
self,
|
|
91
|
+
query: str,
|
|
92
|
+
memories: list[ScoredMemory] | SelectionResult,
|
|
93
|
+
config: CompilationConfig | None = None,
|
|
94
|
+
) -> CompiledContext:
|
|
95
|
+
cfg = config or CompilationConfig()
|
|
96
|
+
started = time.perf_counter()
|
|
97
|
+
stages: list[StageTrace] = []
|
|
98
|
+
inputs = self._compiler_inputs(memories)
|
|
99
|
+
input_tokens = sum(
|
|
100
|
+
self._token_counter.count(item.scored.memory.content) for item in inputs
|
|
101
|
+
)
|
|
102
|
+
normal_input_count = sum(
|
|
103
|
+
item.kind == CompilerInputKind.NORMAL_SELECTED for item in inputs
|
|
104
|
+
)
|
|
105
|
+
rescue_input_count = len(inputs) - normal_input_count
|
|
106
|
+
|
|
107
|
+
ir_started = time.perf_counter()
|
|
108
|
+
facts, excluded = self._build_ir(query, inputs, cfg)
|
|
109
|
+
ir_fact_count = len(facts)
|
|
110
|
+
stages.append(StageTrace(
|
|
111
|
+
stage_name="fact_ir",
|
|
112
|
+
input_count=len(inputs),
|
|
113
|
+
output_count=len(facts),
|
|
114
|
+
latency_ms=(time.perf_counter() - ir_started) * 1000,
|
|
115
|
+
input_tokens=input_tokens,
|
|
116
|
+
metadata={
|
|
117
|
+
"normal_selected": normal_input_count,
|
|
118
|
+
"oversized_rescue": rescue_input_count,
|
|
119
|
+
},
|
|
120
|
+
))
|
|
121
|
+
|
|
122
|
+
dedup_started = time.perf_counter()
|
|
123
|
+
if cfg.strategy != CompilationStrategy.RAW_CONCAT:
|
|
124
|
+
facts, duplicate_exclusions = self._deduplicate(facts)
|
|
125
|
+
excluded.extend(duplicate_exclusions)
|
|
126
|
+
stages.append(StageTrace(
|
|
127
|
+
stage_name="fact_deduplication",
|
|
128
|
+
input_count=len(facts) + len(duplicate_exclusions)
|
|
129
|
+
if cfg.strategy != CompilationStrategy.RAW_CONCAT
|
|
130
|
+
else len(facts),
|
|
131
|
+
output_count=len(facts),
|
|
132
|
+
latency_ms=(time.perf_counter() - dedup_started) * 1000,
|
|
133
|
+
))
|
|
134
|
+
|
|
135
|
+
serialization_started = time.perf_counter()
|
|
136
|
+
included: list[ContextFact] = []
|
|
137
|
+
for fact in facts:
|
|
138
|
+
proposed = self._serialize([*included, fact], cfg.format)
|
|
139
|
+
proposed_tokens = self._token_counter.count(proposed)
|
|
140
|
+
if proposed_tokens <= cfg.budget:
|
|
141
|
+
included.append(fact)
|
|
142
|
+
else:
|
|
143
|
+
excluded.append(ExcludedContextFact(
|
|
144
|
+
fact_id=fact.fact_id,
|
|
145
|
+
source_memory_ids=fact.source_memory_ids,
|
|
146
|
+
input_kind=fact.input_kind,
|
|
147
|
+
reason=FactExclusionReason.BUDGET,
|
|
148
|
+
token_cost=fact.token_cost,
|
|
149
|
+
))
|
|
150
|
+
context_text = self._serialize(included, cfg.format)
|
|
151
|
+
output_tokens = self._token_counter.count(context_text)
|
|
152
|
+
stages.append(StageTrace(
|
|
153
|
+
stage_name="serialization",
|
|
154
|
+
input_count=len(facts),
|
|
155
|
+
output_count=len(included),
|
|
156
|
+
latency_ms=(time.perf_counter() - serialization_started) * 1000,
|
|
157
|
+
input_tokens=sum(fact.token_cost for fact in facts),
|
|
158
|
+
output_tokens=output_tokens,
|
|
159
|
+
metadata={"format": cfg.format},
|
|
160
|
+
))
|
|
161
|
+
|
|
162
|
+
memory_ids = self._ordered_memory_ids(included)
|
|
163
|
+
provenance_map = {
|
|
164
|
+
fact.fact_id: list(fact.source_memory_ids) for fact in included
|
|
165
|
+
}
|
|
166
|
+
provenance_coverage = (
|
|
167
|
+
sum(bool(fact.source_memory_ids) for fact in included) / len(included)
|
|
168
|
+
if included
|
|
169
|
+
else 1.0
|
|
170
|
+
)
|
|
171
|
+
source_texts = {
|
|
172
|
+
item.scored.memory.id: item.scored.memory.content for item in inputs
|
|
173
|
+
}
|
|
174
|
+
unsupported_count = sum(
|
|
175
|
+
not any(
|
|
176
|
+
fact_is_supported(fact.text, source_texts.get(source_id, ""))
|
|
177
|
+
for source_id in fact.source_memory_ids
|
|
178
|
+
)
|
|
179
|
+
for fact in included
|
|
180
|
+
)
|
|
181
|
+
unsupported_rate = unsupported_count / len(included) if included else 0.0
|
|
182
|
+
total_latency = (time.perf_counter() - started) * 1000
|
|
183
|
+
compression_ratio = output_tokens / input_tokens if input_tokens else 0.0
|
|
184
|
+
utilization = output_tokens / cfg.budget if cfg.budget else 0.0
|
|
185
|
+
|
|
186
|
+
return CompiledContext(
|
|
187
|
+
query=query,
|
|
188
|
+
context_text=context_text,
|
|
189
|
+
total_tokens=output_tokens,
|
|
190
|
+
budget=cfg.budget,
|
|
191
|
+
memories_considered=len(inputs),
|
|
192
|
+
memories_included=len(memory_ids),
|
|
193
|
+
memories_excluded=len(inputs) - len(memory_ids),
|
|
194
|
+
compression_ratio=compression_ratio,
|
|
195
|
+
included_memory_ids=memory_ids,
|
|
196
|
+
included_fact_ids=[fact.fact_id for fact in included],
|
|
197
|
+
facts=included,
|
|
198
|
+
excluded_facts=excluded,
|
|
199
|
+
provenance_map=provenance_map,
|
|
200
|
+
input_tokens=input_tokens,
|
|
201
|
+
utilization=utilization,
|
|
202
|
+
unsupported_fact_rate=unsupported_rate,
|
|
203
|
+
provenance_coverage=provenance_coverage,
|
|
204
|
+
strategy=cfg.strategy,
|
|
205
|
+
compression_level=cfg.compression_level,
|
|
206
|
+
trace=CompilationTrace(
|
|
207
|
+
stages=stages,
|
|
208
|
+
memories_considered=len(inputs),
|
|
209
|
+
memories_included=len(memory_ids),
|
|
210
|
+
memories_excluded=len(inputs) - len(memory_ids),
|
|
211
|
+
normal_selected_inputs=normal_input_count,
|
|
212
|
+
oversized_rescue_inputs=rescue_input_count,
|
|
213
|
+
rescued_facts_included=sum(
|
|
214
|
+
fact.input_kind == CompilerInputKind.OVERSIZED_RESCUE
|
|
215
|
+
for fact in included
|
|
216
|
+
),
|
|
217
|
+
facts_created=ir_fact_count,
|
|
218
|
+
facts_included=len(included),
|
|
219
|
+
facts_excluded=len(excluded),
|
|
220
|
+
input_tokens=input_tokens,
|
|
221
|
+
output_tokens=output_tokens,
|
|
222
|
+
provenance_coverage=provenance_coverage,
|
|
223
|
+
total_latency_ms=total_latency,
|
|
224
|
+
),
|
|
225
|
+
)
|
|
226
|
+
|
|
227
|
+
def _build_ir(
|
|
228
|
+
self,
|
|
229
|
+
query: str,
|
|
230
|
+
memories: list[_CompilerInput],
|
|
231
|
+
config: CompilationConfig,
|
|
232
|
+
) -> tuple[list[ContextFact], list[ExcludedContextFact]]:
|
|
233
|
+
facts: list[ContextFact] = []
|
|
234
|
+
excluded: list[ExcludedContextFact] = []
|
|
235
|
+
query_terms = _compiler_tokens(query)
|
|
236
|
+
for compiler_input in memories:
|
|
237
|
+
scored = compiler_input.scored
|
|
238
|
+
memory = scored.memory
|
|
239
|
+
if memory.privacy_level == PrivacyLevel.RESTRICTED:
|
|
240
|
+
fact_id = self._fact_id(memory.content, [memory.id])
|
|
241
|
+
excluded.append(ExcludedContextFact(
|
|
242
|
+
fact_id=fact_id,
|
|
243
|
+
source_memory_ids=[memory.id],
|
|
244
|
+
input_kind=compiler_input.kind,
|
|
245
|
+
reason=FactExclusionReason.PRIVACY_RESTRICTED,
|
|
246
|
+
token_cost=self._token_counter.count(memory.content),
|
|
247
|
+
))
|
|
248
|
+
continue
|
|
249
|
+
|
|
250
|
+
if (
|
|
251
|
+
compiler_input.kind == CompilerInputKind.NORMAL_SELECTED
|
|
252
|
+
and (
|
|
253
|
+
config.strategy in {
|
|
254
|
+
CompilationStrategy.RAW_CONCAT,
|
|
255
|
+
CompilationStrategy.DEDUP_ONLY,
|
|
256
|
+
}
|
|
257
|
+
or config.compression_level == CompressionLevel.NONE
|
|
258
|
+
)
|
|
259
|
+
):
|
|
260
|
+
texts = [memory.content.strip()]
|
|
261
|
+
else:
|
|
262
|
+
clauses = self._split_clauses(memory.content)
|
|
263
|
+
scored_clauses = [
|
|
264
|
+
(clause, self._query_relevance(clause, query_terms))
|
|
265
|
+
for clause in clauses
|
|
266
|
+
]
|
|
267
|
+
texts = [
|
|
268
|
+
self._compress_clause(
|
|
269
|
+
clause,
|
|
270
|
+
query,
|
|
271
|
+
query_terms,
|
|
272
|
+
config.compression_level
|
|
273
|
+
if config.compression_level != CompressionLevel.NONE
|
|
274
|
+
else CompressionLevel.LIGHT,
|
|
275
|
+
)
|
|
276
|
+
for clause, relevance in scored_clauses
|
|
277
|
+
if relevance > 0.0
|
|
278
|
+
]
|
|
279
|
+
texts = [text for text in texts if text]
|
|
280
|
+
if not texts:
|
|
281
|
+
fact_id = self._fact_id(memory.content, [memory.id])
|
|
282
|
+
excluded.append(ExcludedContextFact(
|
|
283
|
+
fact_id=fact_id,
|
|
284
|
+
source_memory_ids=[memory.id],
|
|
285
|
+
input_kind=compiler_input.kind,
|
|
286
|
+
reason=FactExclusionReason.QUERY_IRRELEVANT,
|
|
287
|
+
token_cost=self._token_counter.count(memory.content),
|
|
288
|
+
))
|
|
289
|
+
continue
|
|
290
|
+
|
|
291
|
+
for text in texts:
|
|
292
|
+
fact = self._make_fact(
|
|
293
|
+
text, scored, query_terms, compiler_input.kind
|
|
294
|
+
)
|
|
295
|
+
facts.append(fact)
|
|
296
|
+
return facts, excluded
|
|
297
|
+
|
|
298
|
+
@staticmethod
|
|
299
|
+
def _split_clauses(text: str) -> list[str]:
|
|
300
|
+
return [
|
|
301
|
+
clause.strip()
|
|
302
|
+
for clause in _CLAUSE_BOUNDARY.split(" ".join(text.split()))
|
|
303
|
+
if clause.strip()
|
|
304
|
+
]
|
|
305
|
+
|
|
306
|
+
def _compress_clause(
|
|
307
|
+
self,
|
|
308
|
+
clause: str,
|
|
309
|
+
query: str,
|
|
310
|
+
query_terms: frozenset[str],
|
|
311
|
+
level: CompressionLevel,
|
|
312
|
+
) -> str:
|
|
313
|
+
text = clause.strip()
|
|
314
|
+
if not _WHY_QUERY.search(query) and not _NEGATION.search(text):
|
|
315
|
+
match = re.search(r"\s+(?:mainly\s+)?for\s+(.+?)([.!?]?)$", text, re.IGNORECASE)
|
|
316
|
+
if match:
|
|
317
|
+
reason_terms = _compiler_tokens(match.group(1))
|
|
318
|
+
if not (reason_terms & query_terms):
|
|
319
|
+
text = text[:match.start()].rstrip() + match.group(2)
|
|
320
|
+
if (
|
|
321
|
+
level == CompressionLevel.AGGRESSIVE
|
|
322
|
+
and not (_NEGATION.search(text) or _UNCERTAINTY.search(text))
|
|
323
|
+
):
|
|
324
|
+
text = re.sub(r"^(?:the\s+)?user\s+", "", text, flags=re.IGNORECASE)
|
|
325
|
+
if text:
|
|
326
|
+
text = text[0].upper() + text[1:]
|
|
327
|
+
return text
|
|
328
|
+
|
|
329
|
+
def _make_fact(
|
|
330
|
+
self,
|
|
331
|
+
text: str,
|
|
332
|
+
scored: ScoredMemory,
|
|
333
|
+
query_terms: frozenset[str],
|
|
334
|
+
input_kind: CompilerInputKind,
|
|
335
|
+
) -> ContextFact:
|
|
336
|
+
memory = scored.memory
|
|
337
|
+
temporal = (
|
|
338
|
+
memory.temporal_status
|
|
339
|
+
if memory.temporal_status != CandidateTemporalStatus.UNSPECIFIED
|
|
340
|
+
else self._temporal_status(text, memory.status)
|
|
341
|
+
)
|
|
342
|
+
event_ids = (
|
|
343
|
+
[memory.provenance_event_id] if memory.provenance_event_id is not None else []
|
|
344
|
+
)
|
|
345
|
+
return ContextFact(
|
|
346
|
+
fact_id=self._fact_id(text, [memory.id]),
|
|
347
|
+
text=text,
|
|
348
|
+
source_memory_ids=[memory.id],
|
|
349
|
+
input_kind=input_kind,
|
|
350
|
+
provenance_event_ids=event_ids,
|
|
351
|
+
memory_type=memory.type,
|
|
352
|
+
temporal_status=temporal,
|
|
353
|
+
confidence=memory.confidence,
|
|
354
|
+
importance=memory.importance,
|
|
355
|
+
negated=bool(_NEGATION.search(text)),
|
|
356
|
+
uncertain=bool(_UNCERTAINTY.search(text)),
|
|
357
|
+
causal=bool(_CAUSAL.search(text)),
|
|
358
|
+
query_relevance=self._query_relevance(text, query_terms),
|
|
359
|
+
token_cost=self._token_counter.count(text),
|
|
360
|
+
)
|
|
361
|
+
|
|
362
|
+
@staticmethod
|
|
363
|
+
def _query_relevance(text: str, query_terms: frozenset[str]) -> float:
|
|
364
|
+
if not query_terms:
|
|
365
|
+
return 1.0
|
|
366
|
+
fact_terms = _compiler_tokens(text)
|
|
367
|
+
return min(1.0, len(fact_terms & query_terms) / len(query_terms))
|
|
368
|
+
|
|
369
|
+
@staticmethod
|
|
370
|
+
def _temporal_status(
|
|
371
|
+
text: str, memory_status: MemoryStatus
|
|
372
|
+
) -> CandidateTemporalStatus:
|
|
373
|
+
if _HISTORICAL.search(text) or memory_status in {
|
|
374
|
+
MemoryStatus.HISTORICAL,
|
|
375
|
+
MemoryStatus.SUPERSEDED,
|
|
376
|
+
}:
|
|
377
|
+
return CandidateTemporalStatus.HISTORICAL
|
|
378
|
+
if _FUTURE.search(text):
|
|
379
|
+
return CandidateTemporalStatus.FUTURE
|
|
380
|
+
if _CURRENT.search(text) or memory_status == MemoryStatus.ACTIVE:
|
|
381
|
+
return CandidateTemporalStatus.CURRENT
|
|
382
|
+
return CandidateTemporalStatus.UNSPECIFIED
|
|
383
|
+
|
|
384
|
+
def _deduplicate(
|
|
385
|
+
self, facts: list[ContextFact]
|
|
386
|
+
) -> tuple[list[ContextFact], list[ExcludedContextFact]]:
|
|
387
|
+
kept: list[ContextFact] = []
|
|
388
|
+
excluded: list[ExcludedContextFact] = []
|
|
389
|
+
for fact in facts:
|
|
390
|
+
match_index = next(
|
|
391
|
+
(
|
|
392
|
+
index
|
|
393
|
+
for index, existing in enumerate(kept)
|
|
394
|
+
if self._merge_compatible(existing, fact)
|
|
395
|
+
),
|
|
396
|
+
None,
|
|
397
|
+
)
|
|
398
|
+
if match_index is None:
|
|
399
|
+
kept.append(fact)
|
|
400
|
+
continue
|
|
401
|
+
existing = kept[match_index]
|
|
402
|
+
merged = self._merge_facts(existing, fact)
|
|
403
|
+
kept[match_index] = merged
|
|
404
|
+
excluded.append(ExcludedContextFact(
|
|
405
|
+
fact_id=fact.fact_id,
|
|
406
|
+
source_memory_ids=fact.source_memory_ids,
|
|
407
|
+
input_kind=fact.input_kind,
|
|
408
|
+
reason=FactExclusionReason.DUPLICATE,
|
|
409
|
+
token_cost=fact.token_cost,
|
|
410
|
+
))
|
|
411
|
+
return kept, excluded
|
|
412
|
+
|
|
413
|
+
@staticmethod
|
|
414
|
+
def _merge_compatible(left: ContextFact, right: ContextFact) -> bool:
|
|
415
|
+
if (
|
|
416
|
+
left.memory_type != right.memory_type
|
|
417
|
+
or left.negated != right.negated
|
|
418
|
+
or left.uncertain != right.uncertain
|
|
419
|
+
or left.temporal_status != right.temporal_status
|
|
420
|
+
):
|
|
421
|
+
return False
|
|
422
|
+
similarity = redundancy_similarity(
|
|
423
|
+
information_tokens(left.text),
|
|
424
|
+
information_tokens(right.text),
|
|
425
|
+
)
|
|
426
|
+
return similarity >= 0.75
|
|
427
|
+
|
|
428
|
+
def _merge_facts(self, left: ContextFact, right: ContextFact) -> ContextFact:
|
|
429
|
+
representative = min(
|
|
430
|
+
(left, right),
|
|
431
|
+
key=lambda fact: (
|
|
432
|
+
-self._modifier_evidence(fact),
|
|
433
|
+
fact.token_cost,
|
|
434
|
+
fact.text.casefold(),
|
|
435
|
+
fact.fact_id,
|
|
436
|
+
),
|
|
437
|
+
)
|
|
438
|
+
source_ids = self._ordered_unique([*left.source_memory_ids, *right.source_memory_ids])
|
|
439
|
+
event_ids = self._ordered_unique(
|
|
440
|
+
[*left.provenance_event_ids, *right.provenance_event_ids]
|
|
441
|
+
)
|
|
442
|
+
return representative.model_copy(update={
|
|
443
|
+
"fact_id": self._fact_id(representative.text, source_ids),
|
|
444
|
+
"source_memory_ids": source_ids,
|
|
445
|
+
"provenance_event_ids": event_ids,
|
|
446
|
+
"confidence": max(left.confidence, right.confidence),
|
|
447
|
+
"importance": max(left.importance, right.importance),
|
|
448
|
+
"query_relevance": max(left.query_relevance, right.query_relevance),
|
|
449
|
+
"input_kind": (
|
|
450
|
+
CompilerInputKind.NORMAL_SELECTED
|
|
451
|
+
if CompilerInputKind.NORMAL_SELECTED
|
|
452
|
+
in {left.input_kind, right.input_kind}
|
|
453
|
+
else CompilerInputKind.OVERSIZED_RESCUE
|
|
454
|
+
),
|
|
455
|
+
})
|
|
456
|
+
|
|
457
|
+
@staticmethod
|
|
458
|
+
def _compiler_inputs(
|
|
459
|
+
memories: list[ScoredMemory] | SelectionResult,
|
|
460
|
+
) -> list[_CompilerInput]:
|
|
461
|
+
if isinstance(memories, SelectionResult):
|
|
462
|
+
selected = [
|
|
463
|
+
_CompilerInput(item, CompilerInputKind.NORMAL_SELECTED)
|
|
464
|
+
for item in memories.selected_memories
|
|
465
|
+
]
|
|
466
|
+
selected_ids = {item.scored.memory.id for item in selected}
|
|
467
|
+
rescued = [
|
|
468
|
+
_CompilerInput(item, CompilerInputKind.OVERSIZED_RESCUE)
|
|
469
|
+
for item in memories.compiler_rescue_candidates
|
|
470
|
+
if item.memory.id not in selected_ids
|
|
471
|
+
]
|
|
472
|
+
return [*selected, *rescued]
|
|
473
|
+
return [
|
|
474
|
+
_CompilerInput(item, CompilerInputKind.NORMAL_SELECTED)
|
|
475
|
+
for item in memories
|
|
476
|
+
]
|
|
477
|
+
|
|
478
|
+
@staticmethod
|
|
479
|
+
def _modifier_evidence(fact: ContextFact) -> int:
|
|
480
|
+
"""Prefer merged wording that explicitly carries critical modifiers."""
|
|
481
|
+
score = 0
|
|
482
|
+
if fact.negated and _NEGATION.search(fact.text):
|
|
483
|
+
score += 1
|
|
484
|
+
if fact.uncertain and _UNCERTAINTY.search(fact.text):
|
|
485
|
+
score += 1
|
|
486
|
+
if fact.causal and _CAUSAL.search(fact.text):
|
|
487
|
+
score += 1
|
|
488
|
+
if (
|
|
489
|
+
fact.temporal_status == CandidateTemporalStatus.HISTORICAL
|
|
490
|
+
and _HISTORICAL.search(fact.text)
|
|
491
|
+
):
|
|
492
|
+
score += 1
|
|
493
|
+
if (
|
|
494
|
+
fact.temporal_status == CandidateTemporalStatus.CURRENT
|
|
495
|
+
and _CURRENT.search(fact.text)
|
|
496
|
+
):
|
|
497
|
+
score += 1
|
|
498
|
+
if (
|
|
499
|
+
fact.temporal_status == CandidateTemporalStatus.FUTURE
|
|
500
|
+
and _FUTURE.search(fact.text)
|
|
501
|
+
):
|
|
502
|
+
score += 1
|
|
503
|
+
return score
|
|
504
|
+
|
|
505
|
+
@staticmethod
|
|
506
|
+
def _ordered_unique(values: Iterable[UUID]) -> list[UUID]:
|
|
507
|
+
return list(dict.fromkeys(values))
|
|
508
|
+
|
|
509
|
+
@staticmethod
|
|
510
|
+
def _ordered_memory_ids(facts: list[ContextFact]) -> list[UUID]:
|
|
511
|
+
return list(dict.fromkeys(
|
|
512
|
+
source_id for fact in facts for source_id in fact.source_memory_ids
|
|
513
|
+
))
|
|
514
|
+
|
|
515
|
+
@staticmethod
|
|
516
|
+
def _fact_id(text: str, source_ids: list[UUID]) -> str:
|
|
517
|
+
normalized = " ".join(text.casefold().split())
|
|
518
|
+
payload = normalized + "|" + "|".join(str(value) for value in source_ids)
|
|
519
|
+
return "cf_" + hashlib.sha256(payload.encode("utf-8")).hexdigest()[:16]
|
|
520
|
+
|
|
521
|
+
@staticmethod
|
|
522
|
+
def _serialize(facts: list[ContextFact], output_format: str) -> str:
|
|
523
|
+
if not facts:
|
|
524
|
+
return ""
|
|
525
|
+
if output_format == "json":
|
|
526
|
+
return json.dumps(
|
|
527
|
+
{"facts": [fact.text for fact in facts]},
|
|
528
|
+
ensure_ascii=False,
|
|
529
|
+
separators=(",", ":"),
|
|
530
|
+
)
|
|
531
|
+
return "USER CONTEXT\n" + "\n".join(f"- {fact.text}" for fact in facts)
|
|
532
|
+
|
|
533
|
+
|
|
534
|
+
# Compatibility name retained for API and external imports.
|
|
535
|
+
GreedyContextCompiler = QueryAwareContextCompiler
|