contextos-memory-runtime 1.0.0rc2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- contextos/__init__.py +3 -0
- contextos/__main__.py +6 -0
- contextos/api/__init__.py +1 -0
- contextos/api/routes/__init__.py +1 -0
- contextos/api/routes/desktop.py +322 -0
- contextos/api/routes/ingest.py +17 -0
- contextos/api/routes/memories.py +84 -0
- contextos/api/routes/models.py +81 -0
- contextos/api/routes/retrieval.py +89 -0
- contextos/api/routes/system.py +216 -0
- contextos/api/server.py +195 -0
- contextos/benchmarks/__init__.py +1 -0
- contextos/benchmarks/compilation.py +245 -0
- contextos/benchmarks/connectors.py +423 -0
- contextos/benchmarks/explainability.py +103 -0
- contextos/benchmarks/final.py +406 -0
- contextos/benchmarks/graph.py +310 -0
- contextos/benchmarks/graph_adversarial.py +525 -0
- contextos/benchmarks/mcp.py +324 -0
- contextos/benchmarks/model_routing.py +203 -0
- contextos/benchmarks/optimization.py +305 -0
- contextos/benchmarks/rescue_integration.py +127 -0
- contextos/benchmarks/retrieval.py +266 -0
- contextos/benchmarks/temporal.py +377 -0
- contextos/benchmarks/temporal_hotpath.py +76 -0
- contextos/benchmarks/terminal.py +62 -0
- contextos/cli/__init__.py +1 -0
- contextos/cli/app.py +932 -0
- contextos/cli/dashboard.py +174 -0
- contextos/cli/formatters.py +299 -0
- contextos/config/__init__.py +1 -0
- contextos/config/settings.py +160 -0
- contextos/connectors/__init__.py +6 -0
- contextos/connectors/fake.py +11 -0
- contextos/connectors/json_import.py +125 -0
- contextos/connectors/local_files.py +102 -0
- contextos/connectors/manager.py +293 -0
- contextos/connectors/models.py +62 -0
- contextos/connectors/protocols.py +11 -0
- contextos/core/__init__.py +103 -0
- contextos/core/enums.py +489 -0
- contextos/core/exceptions.py +293 -0
- contextos/core/models.py +1147 -0
- contextos/core/protocols.py +549 -0
- contextos/daemon/__init__.py +1 -0
- contextos/daemon/manager.py +510 -0
- contextos/daemon/state.py +127 -0
- contextos/daemon/wiring.py +296 -0
- contextos/demo.py +217 -0
- contextos/embedding/__init__.py +1 -0
- contextos/embedding/deterministic.py +76 -0
- contextos/embedding/sentence_transformers.py +80 -0
- contextos/mcp/__init__.py +5 -0
- contextos/mcp/server.py +269 -0
- contextos/providers/__init__.py +13 -0
- contextos/providers/fake.py +217 -0
- contextos/providers/ollama.py +297 -0
- contextos/providers/openai_compatible.py +337 -0
- contextos/services/__init__.py +1 -0
- contextos/services/compilation.py +535 -0
- contextos/services/explainability.py +553 -0
- contextos/services/extraction.py +311 -0
- contextos/services/graph.py +524 -0
- contextos/services/graph_retrieval.py +143 -0
- contextos/services/ingestion.py +143 -0
- contextos/services/inspection.py +174 -0
- contextos/services/memory.py +291 -0
- contextos/services/model_service.py +409 -0
- contextos/services/optimization.py +426 -0
- contextos/services/privacy.py +331 -0
- contextos/services/retrieval.py +302 -0
- contextos/services/retrieval_index.py +88 -0
- contextos/services/router.py +302 -0
- contextos/services/secret_scanner.py +207 -0
- contextos/services/telemetry_query.py +102 -0
- contextos/services/temporal.py +500 -0
- contextos/services/token_counter.py +222 -0
- contextos/storage/__init__.py +1 -0
- contextos/storage/connector_repo.py +67 -0
- contextos/storage/database.py +497 -0
- contextos/storage/event_repo.py +137 -0
- contextos/storage/graph_repo.py +228 -0
- contextos/storage/lexical/__init__.py +1 -0
- contextos/storage/lexical/bm25.py +134 -0
- contextos/storage/memory_repo.py +589 -0
- contextos/storage/relation_repo.py +80 -0
- contextos/storage/telemetry_repo.py +481 -0
- contextos/storage/vector/__init__.py +1 -0
- contextos/storage/vector/in_memory.py +162 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/METADATA +143 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/RECORD +93 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/WHEEL +4 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,222 @@
|
|
|
1
|
+
"""Token counting service for ContextOS.
|
|
2
|
+
|
|
3
|
+
Supports exact provider usage, exact local tokenizers (tiktoken), model-family
|
|
4
|
+
profiles (Claude-like, Qwen-like), and deterministic approximation fallbacks.
|
|
5
|
+
Every token count is explicitly labeled with its measurement source:
|
|
6
|
+
- PROVIDER_REPORTED
|
|
7
|
+
- TOKENIZER_COUNTED
|
|
8
|
+
- APPROXIMATED
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import re
|
|
14
|
+
from typing import Protocol, runtime_checkable
|
|
15
|
+
|
|
16
|
+
import tiktoken
|
|
17
|
+
|
|
18
|
+
from contextos.core.enums import TokenMeasurementSource
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@runtime_checkable
|
|
22
|
+
class TokenCounter(Protocol):
|
|
23
|
+
"""Protocol for token counting."""
|
|
24
|
+
|
|
25
|
+
def count(self, text: str) -> int: ...
|
|
26
|
+
|
|
27
|
+
def count_batch(self, texts: list[str]) -> list[int]: ...
|
|
28
|
+
|
|
29
|
+
@property
|
|
30
|
+
def encoding_name(self) -> str: ...
|
|
31
|
+
|
|
32
|
+
@property
|
|
33
|
+
def measurement_source(self) -> TokenMeasurementSource: ...
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class TiktokenCounter:
|
|
37
|
+
"""Token counter using tiktoken (OpenAI models).
|
|
38
|
+
|
|
39
|
+
Implements TokenCounter with TOKENIZER_COUNTED source.
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
def __init__(self, encoding_name: str = "cl100k_base") -> None:
|
|
43
|
+
self._encoding_name = encoding_name
|
|
44
|
+
self._encoder = tiktoken.get_encoding(encoding_name)
|
|
45
|
+
|
|
46
|
+
def count(self, text: str) -> int:
|
|
47
|
+
"""Count tokens in text."""
|
|
48
|
+
if not text:
|
|
49
|
+
return 0
|
|
50
|
+
return len(self._encoder.encode(text))
|
|
51
|
+
|
|
52
|
+
def count_batch(self, texts: list[str]) -> list[int]:
|
|
53
|
+
"""Count tokens for a batch of texts."""
|
|
54
|
+
return [self.count(t) for t in texts]
|
|
55
|
+
|
|
56
|
+
@property
|
|
57
|
+
def encoding_name(self) -> str:
|
|
58
|
+
return self._encoding_name
|
|
59
|
+
|
|
60
|
+
@property
|
|
61
|
+
def measurement_source(self) -> TokenMeasurementSource:
|
|
62
|
+
return TokenMeasurementSource.TOKENIZER_COUNTED
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class ClaudeProfileTokenCounter:
|
|
66
|
+
"""Deterministic token counter profile modeled after Claude/Anthropic tokenization.
|
|
67
|
+
|
|
68
|
+
Uses Anthropic-like subword heuristics and whitespace/punctuation splitting.
|
|
69
|
+
Implements TokenCounter with APPROXIMATED source (calibrated synthetic profile).
|
|
70
|
+
"""
|
|
71
|
+
|
|
72
|
+
_STD_PATTERN = re.compile(
|
|
73
|
+
r"[A-Za-z0-9]+(?:'[A-Za-z]+)?|[^\w\s]|\s+",
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
def __init__(self, model_name: str = "claude-profile") -> None:
|
|
77
|
+
self._model_name = model_name
|
|
78
|
+
|
|
79
|
+
def count(self, text: str) -> int:
|
|
80
|
+
if not text:
|
|
81
|
+
return 0
|
|
82
|
+
# Deterministic Claude-like subword accounting:
|
|
83
|
+
# Standard words, punctuation, and subword chunks for long technical terms
|
|
84
|
+
matches = self._STD_PATTERN.findall(text)
|
|
85
|
+
total = 0
|
|
86
|
+
for token in matches:
|
|
87
|
+
if not token.strip():
|
|
88
|
+
# Whitespace tokens
|
|
89
|
+
total += max(1, len(token) // 4)
|
|
90
|
+
elif len(token) > 6:
|
|
91
|
+
# Subword split for long tokens
|
|
92
|
+
total += (len(token) + 3) // 4
|
|
93
|
+
else:
|
|
94
|
+
total += 1
|
|
95
|
+
return total
|
|
96
|
+
|
|
97
|
+
def count_batch(self, texts: list[str]) -> list[int]:
|
|
98
|
+
return [self.count(t) for t in texts]
|
|
99
|
+
|
|
100
|
+
@property
|
|
101
|
+
def encoding_name(self) -> str:
|
|
102
|
+
return f"profile-{self._model_name}"
|
|
103
|
+
|
|
104
|
+
@property
|
|
105
|
+
def measurement_source(self) -> TokenMeasurementSource:
|
|
106
|
+
return TokenMeasurementSource.APPROXIMATED
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
class QwenProfileTokenCounter:
|
|
110
|
+
"""Deterministic token counter profile modeled after Qwen 152k vocabulary.
|
|
111
|
+
|
|
112
|
+
Qwen's vocabulary compresses technical terms and code more densely while
|
|
113
|
+
treating punctuation and camelCase identifiers with byte-level BPE splits.
|
|
114
|
+
Implements TokenCounter with APPROXIMATED source (calibrated synthetic profile).
|
|
115
|
+
"""
|
|
116
|
+
|
|
117
|
+
_PATTERN = re.compile(
|
|
118
|
+
r"[A-Z]?[a-z]+|[A-Z]+(?=[A-Z][a-z]|\b)|[0-9]+|[^\w\s]|\s+",
|
|
119
|
+
)
|
|
120
|
+
|
|
121
|
+
def __init__(self, model_name: str = "qwen-profile") -> None:
|
|
122
|
+
self._model_name = model_name
|
|
123
|
+
|
|
124
|
+
def count(self, text: str) -> int:
|
|
125
|
+
if not text:
|
|
126
|
+
return 0
|
|
127
|
+
matches = self._PATTERN.findall(text)
|
|
128
|
+
total = 0
|
|
129
|
+
for token in matches:
|
|
130
|
+
if not token.strip():
|
|
131
|
+
total += max(1, len(token) // 5)
|
|
132
|
+
elif len(token) > 8:
|
|
133
|
+
total += (len(token) + 4) // 5
|
|
134
|
+
else:
|
|
135
|
+
total += 1
|
|
136
|
+
return total
|
|
137
|
+
|
|
138
|
+
def count_batch(self, texts: list[str]) -> list[int]:
|
|
139
|
+
return [self.count(t) for t in texts]
|
|
140
|
+
|
|
141
|
+
@property
|
|
142
|
+
def encoding_name(self) -> str:
|
|
143
|
+
return f"profile-{self._model_name}"
|
|
144
|
+
|
|
145
|
+
@property
|
|
146
|
+
def measurement_source(self) -> TokenMeasurementSource:
|
|
147
|
+
return TokenMeasurementSource.APPROXIMATED
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
class DeterministicWordTokenCounter:
|
|
151
|
+
"""Offline word-token approximation fallback.
|
|
152
|
+
|
|
153
|
+
Explicitly labeled with APPROXIMATED source.
|
|
154
|
+
"""
|
|
155
|
+
|
|
156
|
+
_TOKEN_PATTERN = re.compile(
|
|
157
|
+
r"[A-Za-z0-9]+(?:(?:[+#._-]+)[A-Za-z0-9]+)*|[^\w\s]"
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
def count(self, text: str) -> int:
|
|
161
|
+
if not text:
|
|
162
|
+
return 0
|
|
163
|
+
return len(self._TOKEN_PATTERN.findall(text))
|
|
164
|
+
|
|
165
|
+
def count_batch(self, texts: list[str]) -> list[int]:
|
|
166
|
+
return [self.count(text) for text in texts]
|
|
167
|
+
|
|
168
|
+
@property
|
|
169
|
+
def encoding_name(self) -> str:
|
|
170
|
+
return "deterministic-word-approximation"
|
|
171
|
+
|
|
172
|
+
@property
|
|
173
|
+
def measurement_source(self) -> TokenMeasurementSource:
|
|
174
|
+
return TokenMeasurementSource.APPROXIMATED
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def get_token_counter_for_model(
|
|
178
|
+
model_id: str,
|
|
179
|
+
tokenizer_family: str | None = None,
|
|
180
|
+
) -> TokenCounter:
|
|
181
|
+
"""Select the appropriate token counter for a model ID or family."""
|
|
182
|
+
fam = (tokenizer_family or "").lower()
|
|
183
|
+
mid = model_id.lower()
|
|
184
|
+
|
|
185
|
+
# FakeProvider's default is an offline simulation, not a real BPE tokenizer.
|
|
186
|
+
if fam in {"deterministic", "approximation"} or mid == "fake-default":
|
|
187
|
+
return DeterministicWordTokenCounter()
|
|
188
|
+
|
|
189
|
+
if "claude" in fam or "claude" in mid or "anthropic" in fam:
|
|
190
|
+
return ClaudeProfileTokenCounter(model_name=model_id)
|
|
191
|
+
|
|
192
|
+
if "qwen" in fam or "qwen" in mid:
|
|
193
|
+
return QwenProfileTokenCounter(model_name=model_id)
|
|
194
|
+
|
|
195
|
+
if "cl100k" in fam or "o200k" in fam or "gpt" in mid or "openai" in fam:
|
|
196
|
+
encoding = "o200k_base" if "o200k" in fam or "omni" in mid or "gpt-4o" in mid else "cl100k_base"
|
|
197
|
+
try:
|
|
198
|
+
return TiktokenCounter(encoding_name=encoding)
|
|
199
|
+
except Exception:
|
|
200
|
+
return TiktokenCounter("cl100k_base")
|
|
201
|
+
|
|
202
|
+
# Default fallback: try tiktoken cl100k_base
|
|
203
|
+
try:
|
|
204
|
+
return TiktokenCounter("cl100k_base")
|
|
205
|
+
except Exception:
|
|
206
|
+
return DeterministicWordTokenCounter()
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def recount_cross_model(
|
|
210
|
+
text: str,
|
|
211
|
+
target_models_or_families: list[str],
|
|
212
|
+
) -> dict[str, tuple[int, TokenMeasurementSource]]:
|
|
213
|
+
"""Recount the SAME compiled context text across distinct model profiles.
|
|
214
|
+
|
|
215
|
+
Returns a dict mapping model_or_family to (token_count, measurement_source).
|
|
216
|
+
"""
|
|
217
|
+
results: dict[str, tuple[int, TokenMeasurementSource]] = {}
|
|
218
|
+
for target in target_models_or_families:
|
|
219
|
+
counter = get_token_counter_for_model(target, target)
|
|
220
|
+
count = counter.count(text)
|
|
221
|
+
results[target] = (count, counter.measurement_source)
|
|
222
|
+
return results
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Storage package for ContextOS."""
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
"""Persistence for connector cursors and source identity, never source bodies."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
import json
|
|
4
|
+
from datetime import datetime, timezone
|
|
5
|
+
from uuid import UUID
|
|
6
|
+
import aiosqlite
|
|
7
|
+
from contextos.connectors.models import ConnectorItem, ConnectorSyncState
|
|
8
|
+
|
|
9
|
+
def _now() -> str: return datetime.now(timezone.utc).isoformat()
|
|
10
|
+
|
|
11
|
+
class SqliteConnectorRepository:
|
|
12
|
+
def __init__(self, connection: aiosqlite.Connection) -> None: self._db = connection
|
|
13
|
+
async def state(self, connector_id: str) -> ConnectorSyncState | None:
|
|
14
|
+
row = await (await self._db.execute("SELECT * FROM connector_state WHERE connector_id=?", (connector_id,))).fetchone()
|
|
15
|
+
if row is None: return None
|
|
16
|
+
return ConnectorSyncState(connector_id=row["connector_id"], connector_type=row["connector_type"], cursor=row["cursor"], enabled=bool(row["enabled"]), status=row["status"], error_code=row["error_code"], last_success_at=row["last_success_at"], last_attempt_at=row["last_attempt_at"])
|
|
17
|
+
async def save_state(self, state: ConnectorSyncState) -> None:
|
|
18
|
+
await self._db.execute("INSERT INTO connector_state(connector_id,connector_type,cursor,last_success_at,last_attempt_at,enabled,status,error_code) VALUES(?,?,?,?,?,?,?,?) ON CONFLICT(connector_id) DO UPDATE SET cursor=excluded.cursor,last_success_at=excluded.last_success_at,last_attempt_at=excluded.last_attempt_at,enabled=excluded.enabled,status=excluded.status,error_code=excluded.error_code", (state.connector_id,state.connector_type,state.cursor,state.last_success_at.isoformat() if state.last_success_at else None,state.last_attempt_at.isoformat() if state.last_attempt_at else None,int(state.enabled),state.status,state.error_code)); await self._db.commit()
|
|
19
|
+
async def item_is_current(self, connector_id: str, item: ConnectorItem, content_hash: str) -> bool:
|
|
20
|
+
row = await (await self._db.execute("SELECT revision,content_hash,deleted FROM connector_items WHERE connector_id=? AND external_id=?", (connector_id,item.external_id))).fetchone()
|
|
21
|
+
return row is not None and not row["deleted"] and row["revision"] == item.revision and row["content_hash"] == content_hash
|
|
22
|
+
async def save_item(self, connector_id: str, item: ConnectorItem, content_hash: str, memory_ids: list[UUID], deleted: bool = False) -> None:
|
|
23
|
+
is_deleted = int(deleted or item.deleted)
|
|
24
|
+
await self._db.execute("INSERT INTO connector_items(connector_id,external_id,revision,content_hash,source_uri,memory_ids,last_seen_at,deleted) VALUES(?,?,?,?,?,?,?,?) ON CONFLICT(connector_id,external_id) DO UPDATE SET revision=excluded.revision,content_hash=excluded.content_hash,source_uri=excluded.source_uri,memory_ids=excluded.memory_ids,last_seen_at=excluded.last_seen_at,deleted=excluded.deleted", (connector_id,item.external_id,item.revision,content_hash,item.source_uri,json.dumps([str(value) for value in memory_ids]),_now(),is_deleted)); await self._db.commit()
|
|
25
|
+
|
|
26
|
+
async def get_item_memory_ids(self, connector_id: str, external_id: str) -> list[UUID]:
|
|
27
|
+
row = await (await self._db.execute("SELECT memory_ids FROM connector_items WHERE connector_id=? AND external_id=?", (connector_id, external_id))).fetchone()
|
|
28
|
+
if row is None or not row["memory_ids"]:
|
|
29
|
+
return []
|
|
30
|
+
try:
|
|
31
|
+
raw = json.loads(row["memory_ids"])
|
|
32
|
+
return [UUID(val) for val in raw]
|
|
33
|
+
except Exception:
|
|
34
|
+
return []
|
|
35
|
+
|
|
36
|
+
async def count_active_references(self, memory_id: UUID) -> int:
|
|
37
|
+
target = str(memory_id)
|
|
38
|
+
cursor = await self._db.execute("SELECT memory_ids FROM connector_items WHERE deleted = 0")
|
|
39
|
+
rows = await cursor.fetchall()
|
|
40
|
+
count = 0
|
|
41
|
+
for row in rows:
|
|
42
|
+
if not row["memory_ids"]:
|
|
43
|
+
continue
|
|
44
|
+
try:
|
|
45
|
+
raw = json.loads(row["memory_ids"])
|
|
46
|
+
if target in raw:
|
|
47
|
+
count += 1
|
|
48
|
+
except Exception:
|
|
49
|
+
pass
|
|
50
|
+
return count
|
|
51
|
+
|
|
52
|
+
async def list_states(self) -> list[ConnectorSyncState]:
|
|
53
|
+
cursor = await self._db.execute("SELECT * FROM connector_state ORDER BY connector_id")
|
|
54
|
+
rows = await cursor.fetchall()
|
|
55
|
+
return [
|
|
56
|
+
ConnectorSyncState(
|
|
57
|
+
connector_id=row["connector_id"],
|
|
58
|
+
connector_type=row["connector_type"],
|
|
59
|
+
cursor=row["cursor"],
|
|
60
|
+
enabled=bool(row["enabled"]),
|
|
61
|
+
status=row["status"],
|
|
62
|
+
error_code=row["error_code"],
|
|
63
|
+
last_success_at=row["last_success_at"],
|
|
64
|
+
last_attempt_at=row["last_attempt_at"],
|
|
65
|
+
)
|
|
66
|
+
for row in rows
|
|
67
|
+
]
|