contextos-memory-runtime 1.0.0rc2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- contextos/__init__.py +3 -0
- contextos/__main__.py +6 -0
- contextos/api/__init__.py +1 -0
- contextos/api/routes/__init__.py +1 -0
- contextos/api/routes/desktop.py +322 -0
- contextos/api/routes/ingest.py +17 -0
- contextos/api/routes/memories.py +84 -0
- contextos/api/routes/models.py +81 -0
- contextos/api/routes/retrieval.py +89 -0
- contextos/api/routes/system.py +216 -0
- contextos/api/server.py +195 -0
- contextos/benchmarks/__init__.py +1 -0
- contextos/benchmarks/compilation.py +245 -0
- contextos/benchmarks/connectors.py +423 -0
- contextos/benchmarks/explainability.py +103 -0
- contextos/benchmarks/final.py +406 -0
- contextos/benchmarks/graph.py +310 -0
- contextos/benchmarks/graph_adversarial.py +525 -0
- contextos/benchmarks/mcp.py +324 -0
- contextos/benchmarks/model_routing.py +203 -0
- contextos/benchmarks/optimization.py +305 -0
- contextos/benchmarks/rescue_integration.py +127 -0
- contextos/benchmarks/retrieval.py +266 -0
- contextos/benchmarks/temporal.py +377 -0
- contextos/benchmarks/temporal_hotpath.py +76 -0
- contextos/benchmarks/terminal.py +62 -0
- contextos/cli/__init__.py +1 -0
- contextos/cli/app.py +932 -0
- contextos/cli/dashboard.py +174 -0
- contextos/cli/formatters.py +299 -0
- contextos/config/__init__.py +1 -0
- contextos/config/settings.py +160 -0
- contextos/connectors/__init__.py +6 -0
- contextos/connectors/fake.py +11 -0
- contextos/connectors/json_import.py +125 -0
- contextos/connectors/local_files.py +102 -0
- contextos/connectors/manager.py +293 -0
- contextos/connectors/models.py +62 -0
- contextos/connectors/protocols.py +11 -0
- contextos/core/__init__.py +103 -0
- contextos/core/enums.py +489 -0
- contextos/core/exceptions.py +293 -0
- contextos/core/models.py +1147 -0
- contextos/core/protocols.py +549 -0
- contextos/daemon/__init__.py +1 -0
- contextos/daemon/manager.py +510 -0
- contextos/daemon/state.py +127 -0
- contextos/daemon/wiring.py +296 -0
- contextos/demo.py +217 -0
- contextos/embedding/__init__.py +1 -0
- contextos/embedding/deterministic.py +76 -0
- contextos/embedding/sentence_transformers.py +80 -0
- contextos/mcp/__init__.py +5 -0
- contextos/mcp/server.py +269 -0
- contextos/providers/__init__.py +13 -0
- contextos/providers/fake.py +217 -0
- contextos/providers/ollama.py +297 -0
- contextos/providers/openai_compatible.py +337 -0
- contextos/services/__init__.py +1 -0
- contextos/services/compilation.py +535 -0
- contextos/services/explainability.py +553 -0
- contextos/services/extraction.py +311 -0
- contextos/services/graph.py +524 -0
- contextos/services/graph_retrieval.py +143 -0
- contextos/services/ingestion.py +143 -0
- contextos/services/inspection.py +174 -0
- contextos/services/memory.py +291 -0
- contextos/services/model_service.py +409 -0
- contextos/services/optimization.py +426 -0
- contextos/services/privacy.py +331 -0
- contextos/services/retrieval.py +302 -0
- contextos/services/retrieval_index.py +88 -0
- contextos/services/router.py +302 -0
- contextos/services/secret_scanner.py +207 -0
- contextos/services/telemetry_query.py +102 -0
- contextos/services/temporal.py +500 -0
- contextos/services/token_counter.py +222 -0
- contextos/storage/__init__.py +1 -0
- contextos/storage/connector_repo.py +67 -0
- contextos/storage/database.py +497 -0
- contextos/storage/event_repo.py +137 -0
- contextos/storage/graph_repo.py +228 -0
- contextos/storage/lexical/__init__.py +1 -0
- contextos/storage/lexical/bm25.py +134 -0
- contextos/storage/memory_repo.py +589 -0
- contextos/storage/relation_repo.py +80 -0
- contextos/storage/telemetry_repo.py +481 -0
- contextos/storage/vector/__init__.py +1 -0
- contextos/storage/vector/in_memory.py +162 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/METADATA +143 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/RECORD +93 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/WHEEL +4 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
"""Telemetry Query Service for metrics aggregation and dashboard reporting."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from datetime import UTC, datetime
|
|
6
|
+
from typing import TYPE_CHECKING, Any
|
|
7
|
+
|
|
8
|
+
if TYPE_CHECKING:
|
|
9
|
+
from uuid import UUID
|
|
10
|
+
|
|
11
|
+
from contextos.core.models import ModelInvocationTelemetry, TelemetrySummary
|
|
12
|
+
from contextos.core.protocols import TelemetryRepository
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class TelemetryQueryService:
|
|
16
|
+
"""Provides high-level analytical queries over recorded model invocations."""
|
|
17
|
+
|
|
18
|
+
def __init__(self, telemetry_repo: TelemetryRepository) -> None:
|
|
19
|
+
self._repo = telemetry_repo
|
|
20
|
+
|
|
21
|
+
async def record(self, telemetry: ModelInvocationTelemetry) -> None:
|
|
22
|
+
"""Record an invocation directly."""
|
|
23
|
+
await self._repo.record(telemetry)
|
|
24
|
+
|
|
25
|
+
async def get(self, invocation_id: UUID) -> ModelInvocationTelemetry | None:
|
|
26
|
+
return await self._repo.get(invocation_id)
|
|
27
|
+
|
|
28
|
+
async def list_recent(
|
|
29
|
+
self, limit: int = 50, model_id: str | None = None,
|
|
30
|
+
provider_id: str | None = None, start: datetime | None = None,
|
|
31
|
+
) -> list[ModelInvocationTelemetry]:
|
|
32
|
+
return await self._repo.list_recent(limit=limit, model_id=model_id,
|
|
33
|
+
provider_id=provider_id, start=start)
|
|
34
|
+
|
|
35
|
+
async def provider_model_breakdown(
|
|
36
|
+
self, start: datetime | None = None, provider_id: str | None = None,
|
|
37
|
+
model_id: str | None = None,
|
|
38
|
+
) -> list[dict[str, Any]]:
|
|
39
|
+
return await self._repo.provider_model_breakdown(start, provider_id, model_id)
|
|
40
|
+
|
|
41
|
+
async def summary_today(self) -> TelemetrySummary:
|
|
42
|
+
"""Aggregate telemetry for today (UTC start of day to now)."""
|
|
43
|
+
now = datetime.now(UTC)
|
|
44
|
+
start_of_day = datetime(now.year, now.month, now.day, tzinfo=UTC)
|
|
45
|
+
return await self._repo.summary(start=start_of_day, end=now)
|
|
46
|
+
|
|
47
|
+
async def summary_range(
|
|
48
|
+
self,
|
|
49
|
+
start: datetime | None = None,
|
|
50
|
+
end: datetime | None = None,
|
|
51
|
+
provider_id: str | None = None,
|
|
52
|
+
model_id: str | None = None,
|
|
53
|
+
success_only: bool = False,
|
|
54
|
+
) -> TelemetrySummary:
|
|
55
|
+
"""Aggregate telemetry over an arbitrary time window and filters."""
|
|
56
|
+
return await self._repo.summary(
|
|
57
|
+
start=start, end=end, provider_id=provider_id, model_id=model_id,
|
|
58
|
+
success_only=success_only,
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
async def by_provider(self, provider_id: str) -> TelemetrySummary:
|
|
62
|
+
"""Aggregate telemetry filtered by provider."""
|
|
63
|
+
return await self._repo.summary(provider_id=provider_id)
|
|
64
|
+
|
|
65
|
+
async def by_model(self, model_id: str) -> TelemetrySummary:
|
|
66
|
+
"""Aggregate telemetry filtered by model."""
|
|
67
|
+
return await self._repo.summary(model_id=model_id)
|
|
68
|
+
|
|
69
|
+
async def context_measurement_bases(
|
|
70
|
+
self, model_id: str | None = None, provider_id: str | None = None,
|
|
71
|
+
start: datetime | None = None,
|
|
72
|
+
) -> list[dict[str, str]]:
|
|
73
|
+
return await self._repo.context_measurement_bases(model_id, provider_id, start)
|
|
74
|
+
|
|
75
|
+
@staticmethod
|
|
76
|
+
def format_terminal_mock(telemetry: ModelInvocationTelemetry) -> dict[str, Any]:
|
|
77
|
+
"""Produce a CLI-friendly data structure suitable for future Phase 12 terminal UI."""
|
|
78
|
+
return {
|
|
79
|
+
"provider": telemetry.provider_id,
|
|
80
|
+
"model": telemetry.model_id,
|
|
81
|
+
"is_local": telemetry.is_local,
|
|
82
|
+
"candidate_context_tokens": telemetry.candidate_context_tokens,
|
|
83
|
+
"compiled_context_tokens": telemetry.compiled_context_tokens,
|
|
84
|
+
"context_tokens_avoided": telemetry.context_tokens_avoided,
|
|
85
|
+
"reduction_ratio_percent": f"{telemetry.reduction_ratio * 100.0:.1f}%",
|
|
86
|
+
"graph_expanded_memories": telemetry.graph_expanded_count,
|
|
87
|
+
"temporal_filtered_memories": telemetry.temporal_filtered_count,
|
|
88
|
+
"selected_memories": telemetry.selected_memory_count,
|
|
89
|
+
"compiled_facts": telemetry.compiled_fact_count,
|
|
90
|
+
"provider_input_tokens": telemetry.provider_input_tokens,
|
|
91
|
+
"provider_output_tokens": telemetry.provider_output_tokens,
|
|
92
|
+
"retrieval_ms": round(telemetry.retrieval_ms, 2),
|
|
93
|
+
"optimization_ms": round(telemetry.optimization_ms, 2),
|
|
94
|
+
"compilation_ms": round(telemetry.compilation_ms, 2),
|
|
95
|
+
"routing_ms": round(telemetry.routing_ms, 2),
|
|
96
|
+
"provider_ms": round(telemetry.provider_latency_ms, 2),
|
|
97
|
+
"end_to_end_ms": round(telemetry.end_to_end_ms, 2),
|
|
98
|
+
"token_measurement_source": telemetry.token_measurement_source.value,
|
|
99
|
+
"routing_policy": telemetry.routing_policy.value,
|
|
100
|
+
"routing_reason": telemetry.routing_reason,
|
|
101
|
+
"fallback_used": telemetry.fallback_used,
|
|
102
|
+
}
|
|
@@ -0,0 +1,500 @@
|
|
|
1
|
+
"""Deterministic temporal memory identity, resolution, and timeline lookup."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from datetime import datetime, timezone
|
|
7
|
+
from uuid import UUID
|
|
8
|
+
|
|
9
|
+
from contextos.core.enums import (
|
|
10
|
+
CandidateTemporalStatus,
|
|
11
|
+
MemoryStatus,
|
|
12
|
+
MemoryType,
|
|
13
|
+
TemporalOutcome,
|
|
14
|
+
TemporalPrecision,
|
|
15
|
+
)
|
|
16
|
+
from contextos.core.exceptions import InvalidTransitionError
|
|
17
|
+
from contextos.core.models import (
|
|
18
|
+
CandidateMemory,
|
|
19
|
+
Memory,
|
|
20
|
+
MemorySlot,
|
|
21
|
+
TemporalChange,
|
|
22
|
+
TemporalDecision,
|
|
23
|
+
TemporalResolutionResult,
|
|
24
|
+
)
|
|
25
|
+
from contextos.core.protocols import MemoryRepository
|
|
26
|
+
from contextos.services.optimization import information_tokens
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
_UNCERTAIN = re.compile(
|
|
30
|
+
r"\b(?:might|may|maybe|perhaps|possibly|could|i think|trying|tried|experimenting)\b",
|
|
31
|
+
re.I,
|
|
32
|
+
)
|
|
33
|
+
_FUTURE = re.compile(
|
|
34
|
+
r"\b(?:might learn|may learn|plan(?:s)? to|next year|later|will learn|intend(?:s)? to)\b",
|
|
35
|
+
re.I,
|
|
36
|
+
)
|
|
37
|
+
_HISTORICAL = re.compile(
|
|
38
|
+
r"\b(?:used to|previously|formerly|before|in the past)\b", re.I
|
|
39
|
+
)
|
|
40
|
+
_CORRECTION = re.compile(r"\b(?:correction|actually|to correct|rather than)\b", re.I)
|
|
41
|
+
_CHANGE = re.compile(
|
|
42
|
+
r"\b(?:now|currently|no longer|stopped|switched from|changed to|instead|anymore|used to prefer)\b",
|
|
43
|
+
re.I,
|
|
44
|
+
)
|
|
45
|
+
_NEGATION = re.compile(
|
|
46
|
+
r"\b(?:do not|does not|did not|don't|doesn't|didn't|never|no longer|stopped)\b",
|
|
47
|
+
re.I,
|
|
48
|
+
)
|
|
49
|
+
_LANGUAGES: tuple[tuple[str, str], ...] = (
|
|
50
|
+
("c++17", "c++17"), ("c++", "c++"), ("python", "python"),
|
|
51
|
+
("rust", "rust"), ("javascript", "javascript"), ("typescript", "typescript"),
|
|
52
|
+
("java", "java"), ("golang", "go"), ("go", "go"),
|
|
53
|
+
)
|
|
54
|
+
_MODELS: tuple[str, ...] = ("ollama", "localai", "qwen30b", "qwen14b", "qwen9b")
|
|
55
|
+
_MONTHS = (
|
|
56
|
+
"january|february|march|april|may|june|july|august|september|october|"
|
|
57
|
+
"november|december"
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class TemporalSlotAnalyzer:
|
|
62
|
+
"""Derive inspectable slots and temporal metadata from bounded cue rules."""
|
|
63
|
+
|
|
64
|
+
def prepare(self, memory: Memory) -> Memory:
|
|
65
|
+
text = memory.content
|
|
66
|
+
status = memory.temporal_status
|
|
67
|
+
if status == CandidateTemporalStatus.UNSPECIFIED:
|
|
68
|
+
if _FUTURE.search(text):
|
|
69
|
+
status = CandidateTemporalStatus.FUTURE
|
|
70
|
+
elif _HISTORICAL.search(text):
|
|
71
|
+
status = CandidateTemporalStatus.HISTORICAL
|
|
72
|
+
elif _CHANGE.search(text):
|
|
73
|
+
status = CandidateTemporalStatus.CURRENT
|
|
74
|
+
|
|
75
|
+
precision, expression, inferred_from = self._temporal_expression(text)
|
|
76
|
+
if memory.temporal_precision != TemporalPrecision.UNKNOWN:
|
|
77
|
+
precision = memory.temporal_precision
|
|
78
|
+
valid_from = memory.valid_from or inferred_from
|
|
79
|
+
return memory.model_copy(update={
|
|
80
|
+
"slot": memory.slot or self.slot_for(text, memory.type),
|
|
81
|
+
"temporal_status": status,
|
|
82
|
+
"temporal_precision": precision,
|
|
83
|
+
"temporal_expression": memory.temporal_expression or expression,
|
|
84
|
+
"valid_from": valid_from,
|
|
85
|
+
"uncertain": memory.uncertain or bool(_UNCERTAIN.search(text)),
|
|
86
|
+
"negated": memory.negated or bool(_NEGATION.search(text)),
|
|
87
|
+
})
|
|
88
|
+
|
|
89
|
+
def slot_for(self, text: str, memory_type: MemoryType) -> MemorySlot:
|
|
90
|
+
normalized = text.casefold()
|
|
91
|
+
language = self._first_value(normalized, _LANGUAGES)
|
|
92
|
+
if language:
|
|
93
|
+
if re.search(r"\b(?:beginner|proficient|advanced|expert)\b", normalized):
|
|
94
|
+
return MemorySlot(
|
|
95
|
+
subject="user", property="programming_skill",
|
|
96
|
+
scope=language, entity=language,
|
|
97
|
+
)
|
|
98
|
+
return MemorySlot(
|
|
99
|
+
subject="user",
|
|
100
|
+
property="programming_language",
|
|
101
|
+
scope=self._language_scope(normalized),
|
|
102
|
+
)
|
|
103
|
+
if "ram" in normalized and re.search(r"\b\d+\s*gb\b", normalized):
|
|
104
|
+
device = self._device_entity(normalized)
|
|
105
|
+
return MemorySlot(
|
|
106
|
+
subject="machine", property="ram_capacity",
|
|
107
|
+
scope=device or "global", entity=device,
|
|
108
|
+
)
|
|
109
|
+
if "docker" in normalized:
|
|
110
|
+
scope, entity = self._project_scope(normalized)
|
|
111
|
+
return MemorySlot(
|
|
112
|
+
subject="user", property="tool_usage", scope=scope, entity=entity,
|
|
113
|
+
qualifiers=("docker",),
|
|
114
|
+
)
|
|
115
|
+
if any(model in normalized for model in _MODELS):
|
|
116
|
+
return MemorySlot(subject="user", property="local_model", scope="local_machine")
|
|
117
|
+
if re.search(r"\b(?:attended|attend|did not attend|didn't attend)\b", normalized):
|
|
118
|
+
match = re.search(r"\bevent\s+([a-z0-9_-]+)", normalized)
|
|
119
|
+
entity = f"event_{match.group(1)}" if match else "event"
|
|
120
|
+
return MemorySlot(
|
|
121
|
+
subject="user", property="attendance", scope=entity, entity=entity
|
|
122
|
+
)
|
|
123
|
+
if any(word in normalized for word in ("concise", "short answers", "detailed answers")):
|
|
124
|
+
context = self._response_context(normalized)
|
|
125
|
+
return MemorySlot(
|
|
126
|
+
subject="user", property="response_style",
|
|
127
|
+
scope=context or "global", entity=context,
|
|
128
|
+
)
|
|
129
|
+
if "project" in normalized and any(
|
|
130
|
+
word in normalized for word in ("active", "paused", "complete", "cancelled")
|
|
131
|
+
):
|
|
132
|
+
scope, entity = self._project_scope(normalized)
|
|
133
|
+
return MemorySlot(
|
|
134
|
+
subject="user", property="project_status", scope=scope, entity=entity
|
|
135
|
+
)
|
|
136
|
+
concepts = sorted(information_tokens(text))
|
|
137
|
+
property_name = memory_type.value
|
|
138
|
+
scope = "_".join(concepts[:3]) if concepts else "global"
|
|
139
|
+
return MemorySlot(subject="user", property=property_name, scope=scope)
|
|
140
|
+
|
|
141
|
+
def value_for(self, memory: Memory) -> str:
|
|
142
|
+
normalized = memory.content.casefold()
|
|
143
|
+
if memory.slot and memory.slot.property == "programming_skill":
|
|
144
|
+
for level in ("beginner", "learning", "proficient", "advanced", "expert"):
|
|
145
|
+
if level in normalized:
|
|
146
|
+
return level
|
|
147
|
+
language = self._first_value(normalized, _LANGUAGES)
|
|
148
|
+
if language:
|
|
149
|
+
return language
|
|
150
|
+
ram = re.search(r"\b(\d+)\s*gb\s*ram\b", normalized)
|
|
151
|
+
if ram:
|
|
152
|
+
return f"{ram.group(1)}gb"
|
|
153
|
+
if memory.slot and memory.slot.property == "tool_usage":
|
|
154
|
+
return memory.slot.qualifiers[0] if memory.slot.qualifiers else "tool"
|
|
155
|
+
if memory.slot and memory.slot.property == "attendance":
|
|
156
|
+
return "attended"
|
|
157
|
+
for model in _MODELS:
|
|
158
|
+
if model in normalized:
|
|
159
|
+
return model
|
|
160
|
+
if "concise" in normalized or "short answers" in normalized:
|
|
161
|
+
return "concise"
|
|
162
|
+
if "detailed answers" in normalized:
|
|
163
|
+
return "detailed"
|
|
164
|
+
status = next(
|
|
165
|
+
(word for word in ("active", "paused", "complete", "cancelled")
|
|
166
|
+
if word in normalized),
|
|
167
|
+
None,
|
|
168
|
+
)
|
|
169
|
+
if status:
|
|
170
|
+
return status
|
|
171
|
+
ignored = {
|
|
172
|
+
"user", "currently", "now", "previously", "actually", "correction",
|
|
173
|
+
"mainly", "primarily", "use", "uses", "using", "for", "the", "a",
|
|
174
|
+
}
|
|
175
|
+
return " ".join(sorted(information_tokens(normalized) - ignored))
|
|
176
|
+
|
|
177
|
+
@staticmethod
|
|
178
|
+
def _first_value(text: str, values: tuple[tuple[str, str], ...]) -> str | None:
|
|
179
|
+
for needle, canonical in values:
|
|
180
|
+
if re.search(rf"(?<![a-z0-9]){re.escape(needle)}(?![a-z0-9])", text):
|
|
181
|
+
return canonical
|
|
182
|
+
return None
|
|
183
|
+
|
|
184
|
+
@staticmethod
|
|
185
|
+
def _language_scope(text: str) -> str:
|
|
186
|
+
if "machine learning" in text or re.search(r"\bml\b", text):
|
|
187
|
+
return "machine_learning"
|
|
188
|
+
if "data science" in text or re.search(r"\bdata\s+sci", text):
|
|
189
|
+
return "data_science"
|
|
190
|
+
if "interview" in text:
|
|
191
|
+
return "systems_interviews"
|
|
192
|
+
if "systems programming" in text or "for systems" in text:
|
|
193
|
+
return "systems_programming"
|
|
194
|
+
if "embedded" in text:
|
|
195
|
+
return "embedded"
|
|
196
|
+
if "automation" in text:
|
|
197
|
+
return "automation"
|
|
198
|
+
if "web development" in text or "for web" in text:
|
|
199
|
+
return "web_development"
|
|
200
|
+
return "global"
|
|
201
|
+
|
|
202
|
+
@staticmethod
|
|
203
|
+
def _device_entity(text: str) -> str | None:
|
|
204
|
+
"""Extract a machine/device entity from RAM-related text."""
|
|
205
|
+
for device in ("laptop", "desktop", "server", "workstation", "pc", "computer"):
|
|
206
|
+
if device in text:
|
|
207
|
+
return device
|
|
208
|
+
return None
|
|
209
|
+
|
|
210
|
+
@staticmethod
|
|
211
|
+
def _response_context(text: str) -> str | None:
|
|
212
|
+
"""Extract the context qualifier from 'concise/detailed answers for X'."""
|
|
213
|
+
match = re.search(
|
|
214
|
+
r"\bfor\s+([a-z]+(?:\s+[a-z]+)??)\s*(?:questions?|tasks?|topics?|work)?\s*[.,!?]?\s*$",
|
|
215
|
+
text,
|
|
216
|
+
)
|
|
217
|
+
if match:
|
|
218
|
+
raw = match.group(1).strip().replace(" ", "_")
|
|
219
|
+
return raw if raw else None
|
|
220
|
+
return None
|
|
221
|
+
|
|
222
|
+
@staticmethod
|
|
223
|
+
def _project_scope(text: str) -> tuple[str, str | None]:
|
|
224
|
+
match = re.search(r"\bproject\s+([a-z0-9_-]+)", text)
|
|
225
|
+
if not match:
|
|
226
|
+
return "global", None
|
|
227
|
+
entity = f"project_{match.group(1)}"
|
|
228
|
+
return entity, entity
|
|
229
|
+
|
|
230
|
+
@staticmethod
|
|
231
|
+
def _temporal_expression(
|
|
232
|
+
text: str,
|
|
233
|
+
) -> tuple[TemporalPrecision, str | None, datetime | None]:
|
|
234
|
+
timestamp = re.search(r"\b(20\d{2}-\d{2}-\d{2}T\d{2}:\d{2}(?::\d{2})?(?:Z|[+-]\d{2}:?\d{2}))\b", text)
|
|
235
|
+
if timestamp:
|
|
236
|
+
value = timestamp.group(1).replace("Z", "+00:00")
|
|
237
|
+
return TemporalPrecision.EXACT, timestamp.group(1), datetime.fromisoformat(value)
|
|
238
|
+
date = re.search(r"\b(20\d{2}-\d{2}-\d{2})\b", text)
|
|
239
|
+
if date:
|
|
240
|
+
return (
|
|
241
|
+
TemporalPrecision.DATE,
|
|
242
|
+
date.group(1),
|
|
243
|
+
datetime.strptime(date.group(1), "%Y-%m-%d").replace(tzinfo=timezone.utc),
|
|
244
|
+
)
|
|
245
|
+
month = re.search(rf"\b({_MONTHS})\s+(20\d{{2}})\b", text, re.I)
|
|
246
|
+
if month:
|
|
247
|
+
return TemporalPrecision.MONTH, month.group(0), None
|
|
248
|
+
year = re.search(r"\b(?:in\s+)?(20\d{2})\b", text)
|
|
249
|
+
if year:
|
|
250
|
+
return TemporalPrecision.YEAR, year.group(1), None
|
|
251
|
+
relative = re.search(
|
|
252
|
+
r"\b(now|currently|recently|last (?:week|month|year)|next year|before|previously|later)\b",
|
|
253
|
+
text,
|
|
254
|
+
re.I,
|
|
255
|
+
)
|
|
256
|
+
if relative:
|
|
257
|
+
return TemporalPrecision.RELATIVE, relative.group(0).casefold(), None
|
|
258
|
+
return TemporalPrecision.UNKNOWN, None, None
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
class TemporalMemoryService:
|
|
262
|
+
"""Resolve temporal candidates conservatively and apply one atomic plan."""
|
|
263
|
+
|
|
264
|
+
supersession_threshold = 0.70
|
|
265
|
+
|
|
266
|
+
def __init__(self, repository: MemoryRepository) -> None:
|
|
267
|
+
self._repository = repository
|
|
268
|
+
self._analyzer = TemporalSlotAnalyzer()
|
|
269
|
+
|
|
270
|
+
async def accept(
|
|
271
|
+
self,
|
|
272
|
+
candidate: CandidateMemory,
|
|
273
|
+
*,
|
|
274
|
+
provenance_event_id: UUID | None = None,
|
|
275
|
+
) -> TemporalResolutionResult:
|
|
276
|
+
memory = Memory(
|
|
277
|
+
content=candidate.content,
|
|
278
|
+
type=candidate.memory_type,
|
|
279
|
+
source_type=candidate.source_type,
|
|
280
|
+
source_uri=candidate.source_uri,
|
|
281
|
+
provenance_event_id=provenance_event_id,
|
|
282
|
+
status=MemoryStatus.CANDIDATE,
|
|
283
|
+
confidence=candidate.confidence,
|
|
284
|
+
importance=candidate.importance,
|
|
285
|
+
tags=candidate.tags,
|
|
286
|
+
observed_at=candidate.observed_at or datetime.now(timezone.utc),
|
|
287
|
+
valid_from=candidate.valid_from,
|
|
288
|
+
valid_to=candidate.valid_to,
|
|
289
|
+
temporal_precision=candidate.temporal_precision,
|
|
290
|
+
temporal_status=candidate.temporal_status,
|
|
291
|
+
temporal_expression=candidate.temporal_hint,
|
|
292
|
+
slot=candidate.slot,
|
|
293
|
+
uncertain=candidate.uncertain or bool(candidate.metadata.get("uncertain")),
|
|
294
|
+
negated=candidate.negated or bool(candidate.metadata.get("negated")),
|
|
295
|
+
)
|
|
296
|
+
return await self.resolve(memory)
|
|
297
|
+
|
|
298
|
+
async def resolve(self, candidate: Memory) -> TemporalResolutionResult:
|
|
299
|
+
if candidate.status in {MemoryStatus.DELETED, MemoryStatus.PURGED}:
|
|
300
|
+
raise InvalidTransitionError(
|
|
301
|
+
str(candidate.id), candidate.status.value, MemoryStatus.ACTIVE.value
|
|
302
|
+
)
|
|
303
|
+
prepared = self._analyzer.prepare(candidate)
|
|
304
|
+
assert prepared.slot is not None
|
|
305
|
+
decision = await self.decide(prepared)
|
|
306
|
+
if "late_ingestion" in decision.evidence:
|
|
307
|
+
prepared = prepared.model_copy(update={
|
|
308
|
+
"temporal_status": CandidateTemporalStatus.HISTORICAL
|
|
309
|
+
})
|
|
310
|
+
return await self._repository.apply_temporal_decision(prepared, decision)
|
|
311
|
+
|
|
312
|
+
async def decide(self, candidate: Memory) -> TemporalDecision:
|
|
313
|
+
assert candidate.slot is not None
|
|
314
|
+
exact = await self._repository.get_by_hash(candidate.content_hash)
|
|
315
|
+
if exact and exact.status not in {MemoryStatus.DELETED, MemoryStatus.PURGED}:
|
|
316
|
+
return self._decision(
|
|
317
|
+
candidate, TemporalOutcome.DUPLICATE, exact, ["exact_content_hash"], 1.0
|
|
318
|
+
)
|
|
319
|
+
|
|
320
|
+
timeline = await self._repository.list_by_slot(candidate.slot.key)
|
|
321
|
+
current = [
|
|
322
|
+
memory for memory in timeline
|
|
323
|
+
if memory.status == MemoryStatus.ACTIVE
|
|
324
|
+
and memory.temporal_status != CandidateTemporalStatus.FUTURE
|
|
325
|
+
]
|
|
326
|
+
related = self._latest(current)
|
|
327
|
+
if hasattr(self._repository, "latest_active_peer"):
|
|
328
|
+
scoped_peer = await self._repository.latest_active_peer(candidate.slot)
|
|
329
|
+
else:
|
|
330
|
+
# Compatibility for non-SQLite repository implementations.
|
|
331
|
+
all_temporal = await self._repository.list_temporal(limit=500)
|
|
332
|
+
scoped_peer = self._latest([
|
|
333
|
+
memory for memory in all_temporal
|
|
334
|
+
if memory.slot is not None
|
|
335
|
+
and memory.slot.subject == candidate.slot.subject
|
|
336
|
+
and memory.slot.property == candidate.slot.property
|
|
337
|
+
and memory.slot.key != candidate.slot.key
|
|
338
|
+
and memory.status == MemoryStatus.ACTIVE
|
|
339
|
+
])
|
|
340
|
+
|
|
341
|
+
if candidate.temporal_status == CandidateTemporalStatus.HISTORICAL:
|
|
342
|
+
return self._decision(
|
|
343
|
+
candidate, TemporalOutcome.ADD_NEW, related,
|
|
344
|
+
["historical_claim", "does_not_replace_current"], 0.98,
|
|
345
|
+
)
|
|
346
|
+
if candidate.temporal_status == CandidateTemporalStatus.FUTURE:
|
|
347
|
+
return self._decision(
|
|
348
|
+
candidate,
|
|
349
|
+
TemporalOutcome.COEXIST if related or scoped_peer else TemporalOutcome.ADD_NEW,
|
|
350
|
+
related or scoped_peer,
|
|
351
|
+
["future_intention", "current_state_preserved"], 0.98,
|
|
352
|
+
)
|
|
353
|
+
if candidate.uncertain:
|
|
354
|
+
return self._decision(
|
|
355
|
+
candidate,
|
|
356
|
+
TemporalOutcome.COEXIST if related or scoped_peer else TemporalOutcome.ADD_NEW,
|
|
357
|
+
related or scoped_peer,
|
|
358
|
+
["uncertain_claim", "conservative_resolution"], 0.80,
|
|
359
|
+
)
|
|
360
|
+
if related is None:
|
|
361
|
+
if scoped_peer is not None:
|
|
362
|
+
return self._decision(
|
|
363
|
+
candidate, TemporalOutcome.COEXIST, scoped_peer,
|
|
364
|
+
["same_property", "different_scope"], 0.99,
|
|
365
|
+
)
|
|
366
|
+
return self._decision(
|
|
367
|
+
candidate, TemporalOutcome.ADD_NEW, None, ["empty_slot"], 1.0
|
|
368
|
+
)
|
|
369
|
+
|
|
370
|
+
candidate_value = self._analyzer.value_for(candidate)
|
|
371
|
+
related_value = self._analyzer.value_for(related)
|
|
372
|
+
if candidate_value == related_value and candidate.negated == related.negated:
|
|
373
|
+
return self._decision(
|
|
374
|
+
candidate, TemporalOutcome.NO_CHANGE, related,
|
|
375
|
+
["same_slot", "same_normalized_value"], 0.98,
|
|
376
|
+
)
|
|
377
|
+
if self._is_out_of_order(candidate, related):
|
|
378
|
+
historical = candidate.model_copy(update={
|
|
379
|
+
"temporal_status": CandidateTemporalStatus.HISTORICAL
|
|
380
|
+
})
|
|
381
|
+
return self._decision(
|
|
382
|
+
historical, TemporalOutcome.ADD_NEW, related,
|
|
383
|
+
["effective_time_precedes_current", "late_ingestion"], 0.99,
|
|
384
|
+
)
|
|
385
|
+
if _CORRECTION.search(candidate.content):
|
|
386
|
+
return self._decision(
|
|
387
|
+
candidate, TemporalOutcome.CORRECT, related,
|
|
388
|
+
["same_slot", "explicit_correction"], 0.99,
|
|
389
|
+
)
|
|
390
|
+
if _CHANGE.search(candidate.content):
|
|
391
|
+
confidence = min(0.99, 0.65 + 0.35 * candidate.confidence)
|
|
392
|
+
if confidence >= self.supersession_threshold:
|
|
393
|
+
return self._decision(
|
|
394
|
+
candidate, TemporalOutcome.SUPERSEDE, related,
|
|
395
|
+
["same_slot", "explicit_change_cue"], confidence,
|
|
396
|
+
)
|
|
397
|
+
return self._decision(
|
|
398
|
+
candidate, TemporalOutcome.CONTRADICT, related,
|
|
399
|
+
["same_slot", "incompatible_value", "no_transition_evidence"], 0.90,
|
|
400
|
+
)
|
|
401
|
+
|
|
402
|
+
async def get_current_state(self, slot: MemorySlot | str) -> list[Memory]:
|
|
403
|
+
key = slot.key if isinstance(slot, MemorySlot) else slot
|
|
404
|
+
timeline = await self._repository.list_by_slot(key)
|
|
405
|
+
return [
|
|
406
|
+
memory for memory in timeline
|
|
407
|
+
if memory.status == MemoryStatus.ACTIVE
|
|
408
|
+
and memory.temporal_status != CandidateTemporalStatus.FUTURE
|
|
409
|
+
and not memory.uncertain
|
|
410
|
+
]
|
|
411
|
+
|
|
412
|
+
async def get_history(self, slot: MemorySlot | str) -> list[Memory]:
|
|
413
|
+
key = slot.key if isinstance(slot, MemorySlot) else slot
|
|
414
|
+
timeline = await self._repository.list_by_slot(key)
|
|
415
|
+
return sorted(timeline, key=self._timeline_order)
|
|
416
|
+
|
|
417
|
+
async def get_future(self, slot: MemorySlot | str | None = None) -> list[Memory]:
|
|
418
|
+
if slot is not None:
|
|
419
|
+
values = await self._repository.list_by_slot(
|
|
420
|
+
slot.key if isinstance(slot, MemorySlot) else slot
|
|
421
|
+
)
|
|
422
|
+
else:
|
|
423
|
+
values = await self._repository.list_temporal(limit=500)
|
|
424
|
+
return [
|
|
425
|
+
memory for memory in values
|
|
426
|
+
if memory.temporal_status == CandidateTemporalStatus.FUTURE
|
|
427
|
+
]
|
|
428
|
+
|
|
429
|
+
async def get_previous(self, memory_id: UUID) -> Memory | None:
|
|
430
|
+
memory = await self._repository.get(memory_id)
|
|
431
|
+
if memory is None or memory.supersedes is None:
|
|
432
|
+
return None
|
|
433
|
+
return await self._repository.get(memory.supersedes)
|
|
434
|
+
|
|
435
|
+
def _decision(
|
|
436
|
+
self,
|
|
437
|
+
candidate: Memory,
|
|
438
|
+
outcome: TemporalOutcome,
|
|
439
|
+
related: Memory | None,
|
|
440
|
+
evidence: list[str],
|
|
441
|
+
confidence: float,
|
|
442
|
+
) -> TemporalDecision:
|
|
443
|
+
assert candidate.slot is not None
|
|
444
|
+
changes: list[TemporalChange] = []
|
|
445
|
+
if outcome not in {TemporalOutcome.DUPLICATE, TemporalOutcome.NO_CHANGE}:
|
|
446
|
+
candidate_status = (
|
|
447
|
+
MemoryStatus.CONTRADICTED
|
|
448
|
+
if outcome == TemporalOutcome.CONTRADICT
|
|
449
|
+
else MemoryStatus.HISTORICAL
|
|
450
|
+
if candidate.temporal_status == CandidateTemporalStatus.HISTORICAL
|
|
451
|
+
else MemoryStatus.ACTIVE
|
|
452
|
+
)
|
|
453
|
+
changes.append(TemporalChange(
|
|
454
|
+
memory_id=candidate.id,
|
|
455
|
+
from_status=MemoryStatus.CANDIDATE,
|
|
456
|
+
to_status=candidate_status,
|
|
457
|
+
))
|
|
458
|
+
if related and outcome in {TemporalOutcome.SUPERSEDE, TemporalOutcome.CORRECT}:
|
|
459
|
+
changes.insert(0, TemporalChange(
|
|
460
|
+
memory_id=related.id,
|
|
461
|
+
from_status=related.status,
|
|
462
|
+
to_status=MemoryStatus.SUPERSEDED,
|
|
463
|
+
))
|
|
464
|
+
if related and outcome == TemporalOutcome.CONTRADICT:
|
|
465
|
+
changes.insert(0, TemporalChange(
|
|
466
|
+
memory_id=related.id,
|
|
467
|
+
from_status=related.status,
|
|
468
|
+
to_status=MemoryStatus.CONTRADICTED,
|
|
469
|
+
))
|
|
470
|
+
return TemporalDecision(
|
|
471
|
+
candidate_id=candidate.id,
|
|
472
|
+
slot=candidate.slot,
|
|
473
|
+
outcome=outcome,
|
|
474
|
+
compared_memory_ids=[related.id] if related else [],
|
|
475
|
+
related_memory_id=related.id if related else None,
|
|
476
|
+
evidence=evidence,
|
|
477
|
+
confidence=confidence,
|
|
478
|
+
changes=changes,
|
|
479
|
+
)
|
|
480
|
+
|
|
481
|
+
@staticmethod
|
|
482
|
+
def _latest(memories: list[Memory]) -> Memory | None:
|
|
483
|
+
if not memories:
|
|
484
|
+
return None
|
|
485
|
+
return max(memories, key=TemporalMemoryService._timeline_order)
|
|
486
|
+
|
|
487
|
+
@staticmethod
|
|
488
|
+
def _timeline_order(memory: Memory) -> tuple[datetime, datetime, str]:
|
|
489
|
+
return (
|
|
490
|
+
memory.valid_from or memory.observed_at,
|
|
491
|
+
memory.observed_at,
|
|
492
|
+
str(memory.id),
|
|
493
|
+
)
|
|
494
|
+
|
|
495
|
+
@staticmethod
|
|
496
|
+
def _is_out_of_order(candidate: Memory, current: Memory) -> bool:
|
|
497
|
+
if candidate.valid_from is None:
|
|
498
|
+
return False
|
|
499
|
+
current_effective = current.valid_from or current.observed_at
|
|
500
|
+
return candidate.valid_from < current_effective
|