aether-context 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- aether_context/__init__.py +29 -0
- aether_context/_log.py +33 -0
- aether_context/cli.py +1191 -0
- aether_context/config.py +206 -0
- aether_context/context_pool.py +819 -0
- aether_context/encoder.py +213 -0
- aether_context/errors.py +93 -0
- aether_context/local_llm.py +846 -0
- aether_context/mpo.py +151 -0
- aether_context/py.typed +0 -0
- aether_context/quantize.py +86 -0
- aether_context/session.py +829 -0
- aether_context/slice_loader.py +501 -0
- aether_context/tokenizer.py +64 -0
- aether_context/ui.py +253 -0
- aether_context/witness.py +356 -0
- aether_context-0.3.0.dist-info/METADATA +429 -0
- aether_context-0.3.0.dist-info/RECORD +23 -0
- aether_context-0.3.0.dist-info/WHEEL +5 -0
- aether_context-0.3.0.dist-info/entry_points.txt +2 -0
- aether_context-0.3.0.dist-info/licenses/LICENSE +201 -0
- aether_context-0.3.0.dist-info/licenses/NOTICE.md +19 -0
- aether_context-0.3.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,829 @@
|
|
|
1
|
+
# aether-context (Unlimited Context)
|
|
2
|
+
# Copyright (c) 2026 Aether AI
|
|
3
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
4
|
+
"""B5 lifecycle controller — :class:`Session`, the process lifecycle of the engine.
|
|
5
|
+
|
|
6
|
+
This is the part a user drives:
|
|
7
|
+
|
|
8
|
+
from aether_context import Session
|
|
9
|
+
s = Session(model="ollama/qwen2.5", pool_gb=5)
|
|
10
|
+
print(s.run("Build me a full-stack weightlifting tracker app.").text)
|
|
11
|
+
|
|
12
|
+
It ties the four other parts together into the virtual-memory-for-attention lifecycle:
|
|
13
|
+
|
|
14
|
+
* **open** a fresh working window for the task;
|
|
15
|
+
* **stream + encode + fade** — as the model emits (and as input arrives), encode the spill
|
|
16
|
+
into the :class:`~aether_context.context_pool.ContextPool` while the
|
|
17
|
+
:class:`~aether_context.witness.Witness` fades the cold (the pool's governor holds the
|
|
18
|
+
byte budget after every write);
|
|
19
|
+
* **paged reason** — the :class:`~aether_context.slice_loader.Pager` keeps the right slices
|
|
20
|
+
resident, prefetched on a **background thread while the model generates** (the backend's
|
|
21
|
+
HTTP/subprocess call releases the GIL, so the prefetch genuinely overlaps generation);
|
|
22
|
+
* **close** — flush the pool and retain the run's abstracted **artifacts** (text + vector +
|
|
23
|
+
tags) locally for inspection. The session is fully local — nothing ever leaves your machine.
|
|
24
|
+
|
|
25
|
+
Lifecycle cadence
|
|
26
|
+
-----------------
|
|
27
|
+
The loop is a simple **CONTINUOUS** (perceive + encode every step) / **EVENT** (act on a real
|
|
28
|
+
event) / **VERIFY** (occasional bookkeeping) cadence. The only vector the engine ever stores or
|
|
29
|
+
compares is the 256-dim retrieval embedding.
|
|
30
|
+
|
|
31
|
+
Fail-soft (design law 3)
|
|
32
|
+
------------------------
|
|
33
|
+
The pager, encoder, and pool are *optimizations*, never correctness dependencies. Any error
|
|
34
|
+
inside encode/prefetch/recall is logged and the run continues on the model's native window — a
|
|
35
|
+
retrieval hiccup never crashes a long build.
|
|
36
|
+
"""
|
|
37
|
+
from __future__ import annotations
|
|
38
|
+
|
|
39
|
+
import threading
|
|
40
|
+
import uuid
|
|
41
|
+
import warnings
|
|
42
|
+
from dataclasses import dataclass, field
|
|
43
|
+
from pathlib import Path
|
|
44
|
+
from typing import Any, Iterator
|
|
45
|
+
|
|
46
|
+
import numpy as np
|
|
47
|
+
|
|
48
|
+
from aether_context._log import get_logger
|
|
49
|
+
from aether_context.config import PoolConfig, SessionConfig, reach_tokens
|
|
50
|
+
from aether_context.context_pool import ContextPool, Slice, slice_cost_bytes
|
|
51
|
+
from aether_context.encoder import StaticEncoder
|
|
52
|
+
from aether_context.errors import (
|
|
53
|
+
AetherContextError,
|
|
54
|
+
BackendUnavailable,
|
|
55
|
+
ModelNotPulled,
|
|
56
|
+
OllamaNotRunning,
|
|
57
|
+
)
|
|
58
|
+
from aether_context.local_llm import LocalLLM, load_model
|
|
59
|
+
from aether_context.mpo import ChainItem, MpoChain
|
|
60
|
+
from aether_context.slice_loader import Pager, SliceKey
|
|
61
|
+
from aether_context.tokenizer import from_backend
|
|
62
|
+
from aether_context.witness import Witness, retention_score, uniqueness_from_neighbors
|
|
63
|
+
|
|
64
|
+
logger = get_logger(__name__)
|
|
65
|
+
|
|
66
|
+
#: Default number of slices the pager keeps resident as the working set this turn.
|
|
67
|
+
DEFAULT_RESIDENT_K: int = 8
|
|
68
|
+
#: Salience floor for an encoded spill slice (so cold-but-real context is never zero-weighted).
|
|
69
|
+
_SPILL_SALIENCE_FLOOR: float = 0.30
|
|
70
|
+
#: Salience for an explicitly remembered fact — high, so the witness hardens it strongly.
|
|
71
|
+
_REMEMBER_SALIENCE: float = 0.95
|
|
72
|
+
#: Provenance tags written to a slice's ``meta["source"]`` (see SAFETY.md). They let a caller
|
|
73
|
+
#: tell user-planted memory from the model's own spilled notes and retrieve only trusted
|
|
74
|
+
#: sources — the primary guard against an agent's self-authored text becoming standing policy.
|
|
75
|
+
MEMORY_SOURCE_USER: str = "user" # planted by a human / the host (high authority)
|
|
76
|
+
MEMORY_SOURCE_MODEL: str = "model" # the model's own generated spill (NOT authoritative)
|
|
77
|
+
MEMORY_SOURCE_TOOL: str = "tool" # a tool/observation result (grounding, not policy)
|
|
78
|
+
#: Topic label assigned to the running reasoning region (the pager's working-set key).
|
|
79
|
+
_REASONING_TOPIC: str = "reasoning"
|
|
80
|
+
#: Resident in-RAM index estimate, MB per GB of reach (mirrors the README table / CLI's
|
|
81
|
+
#: _INDEX_MB_PER_GB: ~145 MB at 5 GB). Used only for the honest status RAM estimate.
|
|
82
|
+
_RESIDENT_RAM_MB_PER_GB: int = 29
|
|
83
|
+
#: Default transcript export filename (timestamp-free, written under cwd).
|
|
84
|
+
_DEFAULT_TRANSCRIPT_NAME: str = "aether-transcript.txt"
|
|
85
|
+
#: How much the Extended-Thinking toggle widens the resident working set (mock-honest).
|
|
86
|
+
_EXTENDED_RESIDENT_BONUS: int = 8
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
# ---------------------------------------------------------------------------
|
|
90
|
+
# Results & harvest candidates
|
|
91
|
+
# ---------------------------------------------------------------------------
|
|
92
|
+
@dataclass(frozen=True)
|
|
93
|
+
class HarvestCandidate:
|
|
94
|
+
"""An abstracted, durable artifact of a run, retained locally.
|
|
95
|
+
|
|
96
|
+
Deliberately tiny and self-contained: ``text`` + its 256-dim retrieval ``vector`` + plain
|
|
97
|
+
``tags`` — nothing else, no external state.
|
|
98
|
+
"""
|
|
99
|
+
|
|
100
|
+
text: str
|
|
101
|
+
vector: np.ndarray
|
|
102
|
+
tags: dict[str, Any] = field(default_factory=dict)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
@dataclass(frozen=True)
|
|
106
|
+
class RunResult:
|
|
107
|
+
"""The outcome of :meth:`Session.run`.
|
|
108
|
+
|
|
109
|
+
Fields:
|
|
110
|
+
text the model's full generated text for the task
|
|
111
|
+
stages ordered list of lifecycle stage records (open / stream / page / complete)
|
|
112
|
+
hit_rate the pager's measured retrieval hit rate over the run
|
|
113
|
+
spilled number of slices encoded into the pool during the run (encode-on-spill)
|
|
114
|
+
resident number of slices resident in the pager window at the end
|
|
115
|
+
overflowed whether the run exceeded the model's native ``context_window``
|
|
116
|
+
"""
|
|
117
|
+
|
|
118
|
+
text: str
|
|
119
|
+
stages: list[dict[str, Any]]
|
|
120
|
+
hit_rate: float
|
|
121
|
+
spilled: int = 0
|
|
122
|
+
resident: int = 0
|
|
123
|
+
overflowed: bool = False
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
# ---------------------------------------------------------------------------
|
|
127
|
+
# Session
|
|
128
|
+
# ---------------------------------------------------------------------------
|
|
129
|
+
class Session:
|
|
130
|
+
"""The engine lifecycle controller: open -> stream+encode+fade -> page -> close.
|
|
131
|
+
|
|
132
|
+
Construct with a model spec (or a :class:`~aether_context.local_llm.LocalLLM` object) and
|
|
133
|
+
a pool size; the session builds the local LLM (via
|
|
134
|
+
:func:`~aether_context.local_llm.load_model`), a :class:`ContextPool`, a :class:`Witness`,
|
|
135
|
+
and a :class:`Pager`. The session is fully local and offline — nothing leaves your machine.
|
|
136
|
+
|
|
137
|
+
Public surface:
|
|
138
|
+
* :meth:`run` ``(task) -> RunResult`` — the full lifecycle for one task.
|
|
139
|
+
* :meth:`ask` ``(msg) -> str`` — a convenience wrapper returning just the text.
|
|
140
|
+
* :meth:`stream` ``(task) -> Iterator[str]`` — stream the model's chunks, encoding spill.
|
|
141
|
+
* :meth:`remember` / :meth:`recall` — plant and retrieve a load-bearing fact.
|
|
142
|
+
* :meth:`harvest_candidates` — the run's durable artifacts.
|
|
143
|
+
* :meth:`close` + context-manager (``with Session(...) as s:``).
|
|
144
|
+
"""
|
|
145
|
+
|
|
146
|
+
def __init__(
|
|
147
|
+
self,
|
|
148
|
+
model: "str | LocalLLM",
|
|
149
|
+
pool_gb: int = 5,
|
|
150
|
+
*,
|
|
151
|
+
system: str | None = None,
|
|
152
|
+
max_tokens: int | None = None,
|
|
153
|
+
pull: bool = False,
|
|
154
|
+
fallback_to_mock: bool = True,
|
|
155
|
+
pool_dir: "str | Path | None" = None,
|
|
156
|
+
pool_index: str = "flat",
|
|
157
|
+
pool_mode: str = "separate",
|
|
158
|
+
pool_quantize: int = 0, # TurboVec: 0=float32 (default), 8=recall-safe (~4x), 4=lossy/flagged
|
|
159
|
+
context_window: int | None = None,
|
|
160
|
+
output_tokens: int | None = None,
|
|
161
|
+
resident_k: int = DEFAULT_RESIDENT_K,
|
|
162
|
+
pool_ceiling_bytes: int | None = None,
|
|
163
|
+
session_id: str | None = None,
|
|
164
|
+
mpo_chain: bool = True,
|
|
165
|
+
chain_width: int = 8,
|
|
166
|
+
chain_hops: int = 1,
|
|
167
|
+
chain_fanout: int = 4,
|
|
168
|
+
**cfg: Any,
|
|
169
|
+
) -> None:
|
|
170
|
+
self.id: str = session_id or f"sess-{uuid.uuid4().hex[:12]}"
|
|
171
|
+
self._closed: bool = False
|
|
172
|
+
self._resident_k: int = max(1, int(resident_k))
|
|
173
|
+
# Pool sharing discipline. "separate" (default) keeps every search/encode scoped to
|
|
174
|
+
# THIS session's namespace, so two sessions over one dir never see each other's
|
|
175
|
+
# slices. "shared" makes reach global (search/encode with session=None) so a named or
|
|
176
|
+
# persistent pool can be read across sessions. The PoolConfig records the same mode.
|
|
177
|
+
self.pool_mode: str = pool_mode
|
|
178
|
+
# Extended-Thinking toggle (honest): in the mock it only widens the resident set and
|
|
179
|
+
# surfaces in status; it is never a silent capability claim.
|
|
180
|
+
self.extended: bool = False
|
|
181
|
+
# Conversation transcript: one (role, text) tuple per ask()/run() turn, used by export().
|
|
182
|
+
self._transcript: list[tuple[str, str]] = []
|
|
183
|
+
# -- the model (THE WRAPPER) ------------------------------------------
|
|
184
|
+
self.local_llm: LocalLLM = self._build_model(
|
|
185
|
+
model,
|
|
186
|
+
pull=pull,
|
|
187
|
+
context_window=context_window,
|
|
188
|
+
output_tokens=output_tokens,
|
|
189
|
+
fallback_to_mock=fallback_to_mock,
|
|
190
|
+
**cfg,
|
|
191
|
+
)
|
|
192
|
+
self._count_tokens = from_backend(self.local_llm)
|
|
193
|
+
|
|
194
|
+
# -- session config (window fractions etc.) ---------------------------
|
|
195
|
+
self.config = SessionConfig(
|
|
196
|
+
model=model, system=system, max_tokens=max_tokens
|
|
197
|
+
)
|
|
198
|
+
|
|
199
|
+
# -- the pool ("disk") ------------------------------------------------
|
|
200
|
+
pool_config = PoolConfig(
|
|
201
|
+
pool_gb=pool_gb,
|
|
202
|
+
mode=pool_mode,
|
|
203
|
+
index=pool_index,
|
|
204
|
+
dir=Path(pool_dir) if pool_dir is not None else PoolConfig().dir,
|
|
205
|
+
quantize_bits=pool_quantize,
|
|
206
|
+
)
|
|
207
|
+
self.pool: ContextPool = ContextPool(
|
|
208
|
+
pool_config, ceiling_bytes=pool_ceiling_bytes
|
|
209
|
+
)
|
|
210
|
+
# Retain the reach (GB) for honest status reporting without reaching into pool internals.
|
|
211
|
+
self.pool_gb: int = int(pool_config.pool_gb)
|
|
212
|
+
|
|
213
|
+
# -- encoder ("encode-on-spill") --------------------------------------
|
|
214
|
+
self.encoder: StaticEncoder = StaticEncoder(dim=pool_config.dim)
|
|
215
|
+
|
|
216
|
+
# -- witness (page-replacement) + pager (the pager) -------------------
|
|
217
|
+
# In "shared" mode the cold path searches globally (session=None) so the resident
|
|
218
|
+
# window can draw on slices from any session; "separate" keeps the default
|
|
219
|
+
# session-scoped cold path (key.session) so namespaces never bleed.
|
|
220
|
+
self.witness: Witness = Witness()
|
|
221
|
+
self._rehydrate_pins()
|
|
222
|
+
self.pager: Pager = Pager(
|
|
223
|
+
self.pool,
|
|
224
|
+
self.encoder,
|
|
225
|
+
default_k=self._resident_k,
|
|
226
|
+
retrieve_fn=self._cold_retrieve,
|
|
227
|
+
)
|
|
228
|
+
|
|
229
|
+
# -- optional MPO context chain (assists retrieval; off by default) ----
|
|
230
|
+
# Cosine stays the retrieval mechanism; when enabled, the chain widens each cosine
|
|
231
|
+
# result with the slices most coupled to it on (cost, time) — connected context.
|
|
232
|
+
self.mpo_chain_enabled: bool = bool(mpo_chain)
|
|
233
|
+
self._chain_fanout: int = max(1, int(chain_fanout))
|
|
234
|
+
self.chain: MpoChain | None = (
|
|
235
|
+
MpoChain(width=max(1, int(chain_width)), hops=max(1, int(chain_hops)))
|
|
236
|
+
if self.mpo_chain_enabled else None
|
|
237
|
+
)
|
|
238
|
+
|
|
239
|
+
# -- run state --------------------------------------------------------
|
|
240
|
+
self._harvest: list[HarvestCandidate] = []
|
|
241
|
+
self._spill_seq: int = 0
|
|
242
|
+
self._clock: float = 0.0
|
|
243
|
+
|
|
244
|
+
# -- construction helpers -------------------------------------------------
|
|
245
|
+
@staticmethod
|
|
246
|
+
def _build_model(
|
|
247
|
+
model: "str | LocalLLM",
|
|
248
|
+
*,
|
|
249
|
+
pull: bool,
|
|
250
|
+
context_window: int | None,
|
|
251
|
+
output_tokens: int | None,
|
|
252
|
+
fallback_to_mock: bool = True,
|
|
253
|
+
**cfg: Any,
|
|
254
|
+
) -> LocalLLM:
|
|
255
|
+
"""Resolve ``model`` to a :class:`LocalLLM`, forwarding the relevant kwargs.
|
|
256
|
+
|
|
257
|
+
A bare :class:`LocalLLM` object is returned unchanged. A spec string is dispatched to
|
|
258
|
+
:func:`~aether_context.local_llm.load_model`; ``context_window`` / ``output_tokens`` /
|
|
259
|
+
``pull`` are forwarded only when meaningful (so the mock honors a tiny window and a
|
|
260
|
+
long output, and Ollama honors ``pull``).
|
|
261
|
+
|
|
262
|
+
When ``fallback_to_mock`` is true (the default) and the requested backend cannot be
|
|
263
|
+
loaded (daemon down, model not pulled, optional backend not installed), the engine
|
|
264
|
+
degrades to the deterministic mock model so a clean-clone / offline run never crashes.
|
|
265
|
+
The fallback is announced via :mod:`warnings` (visible on stderr regardless of logging
|
|
266
|
+
config) so it is never silent. Pass ``fallback_to_mock=False`` to fail loudly instead.
|
|
267
|
+
"""
|
|
268
|
+
if not isinstance(model, str):
|
|
269
|
+
return model # bring-your-own backend
|
|
270
|
+
kw: dict[str, Any] = dict(cfg)
|
|
271
|
+
if pull:
|
|
272
|
+
kw["pull"] = True
|
|
273
|
+
if context_window is not None:
|
|
274
|
+
kw["context_window"] = context_window
|
|
275
|
+
if output_tokens is not None:
|
|
276
|
+
# only the mock backend accepts output_tokens; forward it conditionally.
|
|
277
|
+
if model.strip() == "mock":
|
|
278
|
+
kw["output_tokens"] = output_tokens
|
|
279
|
+
try:
|
|
280
|
+
return load_model(model, **kw)
|
|
281
|
+
except (OllamaNotRunning, ModelNotPulled, BackendUnavailable) as exc:
|
|
282
|
+
if not fallback_to_mock or model.strip() == "mock":
|
|
283
|
+
raise
|
|
284
|
+
warnings.warn(
|
|
285
|
+
f"Could not load model {model!r} ({exc}); falling back to the deterministic "
|
|
286
|
+
"mock model. Output will be SYNTHETIC. Pass fallback_to_mock=False to fail "
|
|
287
|
+
"instead, or start the backend (e.g. `ollama serve` + `ollama pull <model>`).",
|
|
288
|
+
RuntimeWarning,
|
|
289
|
+
stacklevel=2,
|
|
290
|
+
)
|
|
291
|
+
mock_kw: dict[str, Any] = {}
|
|
292
|
+
if context_window is not None:
|
|
293
|
+
mock_kw["context_window"] = context_window
|
|
294
|
+
if output_tokens is not None:
|
|
295
|
+
mock_kw["output_tokens"] = output_tokens
|
|
296
|
+
return load_model("mock", **mock_kw)
|
|
297
|
+
|
|
298
|
+
# -- properties -----------------------------------------------------------
|
|
299
|
+
@property
|
|
300
|
+
def closed(self) -> bool:
|
|
301
|
+
"""Whether the session has been closed (pool flushed, artifacts retained)."""
|
|
302
|
+
return self._closed
|
|
303
|
+
|
|
304
|
+
@property
|
|
305
|
+
def context_window(self) -> int:
|
|
306
|
+
"""The model's native token window (the size the engine pages *around*)."""
|
|
307
|
+
try:
|
|
308
|
+
return int(self.local_llm.context_window)
|
|
309
|
+
except (AttributeError, ValueError, TypeError):
|
|
310
|
+
return 8192
|
|
311
|
+
|
|
312
|
+
def _key(self, topic: str = _REASONING_TOPIC) -> SliceKey:
|
|
313
|
+
"""The pager key for this session's region under ``topic``."""
|
|
314
|
+
return SliceKey(session=self.id, topic=topic)
|
|
315
|
+
|
|
316
|
+
def _scope(self) -> str | None:
|
|
317
|
+
"""The session id this session's searches are scoped to, or ``None`` when shared.
|
|
318
|
+
|
|
319
|
+
``"separate"`` (default) scopes every search/encode to :attr:`id` so two sessions
|
|
320
|
+
over one pool dir stay isolated. ``"shared"`` returns ``None`` so the search spans
|
|
321
|
+
every session's slices (global reach across sessions).
|
|
322
|
+
"""
|
|
323
|
+
return None if self.pool_mode == "shared" else self.id
|
|
324
|
+
|
|
325
|
+
def _cold_retrieve(self, key: SliceKey, query_vec: np.ndarray, k: int) -> list[Slice]:
|
|
326
|
+
"""Pager cold path honoring the session's pool mode (shared -> global search).
|
|
327
|
+
|
|
328
|
+
With ``mpo_chain`` enabled, cosine still selects the entry hits; the MPO chain then
|
|
329
|
+
widens the window with the slices most coupled to them — connected context. Fail-soft:
|
|
330
|
+
any chain error returns the plain cosine ``slices[:k]``.
|
|
331
|
+
"""
|
|
332
|
+
scope = self._scope()
|
|
333
|
+
if self.chain is None:
|
|
334
|
+
return self.pool.search(query_vec, k, session=scope)
|
|
335
|
+
candidates = self.pool.search(query_vec, max(k, k * self._chain_fanout), session=scope)
|
|
336
|
+
if len(candidates) <= k:
|
|
337
|
+
return candidates[:k]
|
|
338
|
+
try:
|
|
339
|
+
warm = {sl.id for sl in self.pager.window()}
|
|
340
|
+
items = [
|
|
341
|
+
ChainItem(id=sl.id, vector=sl.vector, c_t=self._c_t(sl, warm))
|
|
342
|
+
for sl in candidates
|
|
343
|
+
]
|
|
344
|
+
seed = [sl.id for sl in candidates[: min(2, k)]]
|
|
345
|
+
order = self.chain.expand(seed, items, width=k)
|
|
346
|
+
by_id = {sl.id: sl for sl in candidates}
|
|
347
|
+
widened = [by_id[i] for i in order if i in by_id]
|
|
348
|
+
return (widened or candidates)[:k]
|
|
349
|
+
except Exception as exc: # noqa: BLE001 - fail-soft: chain only ever augments
|
|
350
|
+
logger.warning("MPO chain expand failed (%s); serving cosine order", exc)
|
|
351
|
+
return candidates[:k]
|
|
352
|
+
|
|
353
|
+
@staticmethod
|
|
354
|
+
def _c_t(sl: Slice, warm: set[str]) -> tuple[float, float]:
|
|
355
|
+
"""The chain coordinate for a slice."""
|
|
356
|
+
return (float(sl.meta.get("_ct", 0.0)), float(sl.tokens) / (2.0 if sl.id in warm else 1.0))
|
|
357
|
+
|
|
358
|
+
# -- plant / recover a fact -----------------------------------------------
|
|
359
|
+
def pin(self, text: str, *, tags: dict[str, Any] | None = None,
|
|
360
|
+
source: str = MEMORY_SOURCE_USER) -> Slice | None:
|
|
361
|
+
"""Encode ``text`` as a **permanently retained** slice.
|
|
362
|
+
|
|
363
|
+
Equivalent to :meth:`remember` with ``pinned=True``: the slice never
|
|
364
|
+
fades and is never evicted, at any pool pressure, for the life of the
|
|
365
|
+
session. Use it for what the session must not lose -- an operating
|
|
366
|
+
contract, a hard constraint, an identity -- not for merely important
|
|
367
|
+
content, which ordinary salience already handles.
|
|
368
|
+
"""
|
|
369
|
+
return self.remember(text, tags={**(tags or {}), "pinned": True}, source=source)
|
|
370
|
+
|
|
371
|
+
def remember(self, text: str, *, tags: dict[str, Any] | None = None,
|
|
372
|
+
source: str = MEMORY_SOURCE_USER) -> Slice | None:
|
|
373
|
+
"""Encode ``text`` as a high-salience slice into the pool (a load-bearing fact).
|
|
374
|
+
|
|
375
|
+
This is how a durable constraint is established before a long run so it survives the
|
|
376
|
+
overflow: it is encoded into the pool immediately and hardened in the witness. The slice
|
|
377
|
+
is tagged ``meta["source"]`` (default :data:`MEMORY_SOURCE_USER` — high authority) so a
|
|
378
|
+
caller can later recall trusted memory only. Returns the stored :class:`Slice` (or
|
|
379
|
+
``None`` if encoding fails — fail-soft).
|
|
380
|
+
"""
|
|
381
|
+
meta = {"source": source, **(tags or {})}
|
|
382
|
+
return self._encode_slice(text, salience=_REMEMBER_SALIENCE, tags=meta)
|
|
383
|
+
|
|
384
|
+
def recall(self, query: str, k: int = DEFAULT_RESIDENT_K, *,
|
|
385
|
+
sources: set[str] | None = None) -> list[Slice]:
|
|
386
|
+
"""Retrieve the slices nearest to ``query`` from this session's pool region.
|
|
387
|
+
|
|
388
|
+
The hot path of recovery: embed ``query``, search the pool scoped to this session, and
|
|
389
|
+
return the nearest slices. Pass ``sources`` (e.g. ``{"user", "tool"}``) to retrieve only
|
|
390
|
+
trusted provenance and exclude the model's own spilled notes (see SAFETY.md). Fail-soft —
|
|
391
|
+
an encoder/search error yields ``[]`` (the run continues on the model's native window).
|
|
392
|
+
"""
|
|
393
|
+
return self._recall_local(query, k, sources)
|
|
394
|
+
|
|
395
|
+
def _recall_local(self, query: str, k: int,
|
|
396
|
+
sources: set[str] | None = None) -> list[Slice]:
|
|
397
|
+
try:
|
|
398
|
+
qvec = self.encoder.encode(query)
|
|
399
|
+
except AetherContextError as exc:
|
|
400
|
+
logger.warning("recall encode failed (%s); returning no local hits", exc)
|
|
401
|
+
return []
|
|
402
|
+
try:
|
|
403
|
+
scoped = self.pool.search(qvec, k, session=self._scope(), sources=sources)
|
|
404
|
+
if scoped:
|
|
405
|
+
return scoped
|
|
406
|
+
# Reopen case: a fresh Session over an existing pool dir has a new session id, so
|
|
407
|
+
# the prior run's slices live under a different namespace. Fall back to a global
|
|
408
|
+
# search so a disk-resident fact is still recoverable after a close + reopen.
|
|
409
|
+
# (In shared mode the scoped search is already global, so this is a harmless re-run.)
|
|
410
|
+
return self.pool.search(qvec, k, session=None, sources=sources)
|
|
411
|
+
except AetherContextError as exc:
|
|
412
|
+
logger.warning("recall pool search failed (%s); returning no local hits", exc)
|
|
413
|
+
return []
|
|
414
|
+
|
|
415
|
+
# -- encode-on-spill ------------------------------------------------------
|
|
416
|
+
def _encode_slice(
|
|
417
|
+
self, text: str, *, salience: float, tags: dict[str, Any]
|
|
418
|
+
) -> Slice | None:
|
|
419
|
+
"""Encode ``text`` into a pool slice and add it (fail-soft).
|
|
420
|
+
|
|
421
|
+
Returns the stored slice, or ``None`` if the encoder or the pool add fails — in which
|
|
422
|
+
case the run simply proceeds without that slice (the pager is an optimization). The
|
|
423
|
+
witness is touched so the slice participates in retention ranking.
|
|
424
|
+
"""
|
|
425
|
+
text = text.strip()
|
|
426
|
+
if not text:
|
|
427
|
+
return None
|
|
428
|
+
try:
|
|
429
|
+
vec = self.encoder.encode(text)
|
|
430
|
+
except Exception as exc: # noqa: BLE001 - fail-soft: encoder is an optimization
|
|
431
|
+
logger.warning("encode-on-spill failed (%s); slice dropped, run continues", exc)
|
|
432
|
+
return None
|
|
433
|
+
self._spill_seq += 1
|
|
434
|
+
sid = f"{self.id}:slice:{self._spill_seq}"
|
|
435
|
+
tokens = self._count_tokens(text)
|
|
436
|
+
score = self._salience(text, salience)
|
|
437
|
+
# Chain coordinate component for this slice (monotonic per encode).
|
|
438
|
+
meta = dict(tags)
|
|
439
|
+
meta.setdefault("_ct", float(self._spill_seq))
|
|
440
|
+
sl = Slice(
|
|
441
|
+
id=sid, session=self.id, vector=np.asarray(vec, dtype=np.float32),
|
|
442
|
+
text=text, tokens=int(tokens), meta=meta, score=score,
|
|
443
|
+
)
|
|
444
|
+
try:
|
|
445
|
+
self.pool.add(sl)
|
|
446
|
+
except AetherContextError as exc:
|
|
447
|
+
logger.warning("pool add failed (%s); slice dropped, run continues", exc)
|
|
448
|
+
return None
|
|
449
|
+
self._clock += 1.0
|
|
450
|
+
self.witness.touch(sid, salience=score, now=self._clock)
|
|
451
|
+
# A caller marks load-bearing content with ``pinned``: an agent's
|
|
452
|
+
# doctrine, its skills, its grounding. Salience cannot express that --
|
|
453
|
+
# a slice nothing queries decays exactly like one nothing needs -- so
|
|
454
|
+
# the tag is honoured here as permanence rather than as a hint.
|
|
455
|
+
if meta.get("pinned") is True:
|
|
456
|
+
self.witness.pin(sid)
|
|
457
|
+
# Every encoded durable artifact (planted fact or spill) is a run artifact:
|
|
458
|
+
# text + its 256-dim retrieval vector + tags, retained locally.
|
|
459
|
+
self._harvest.append(
|
|
460
|
+
HarvestCandidate(text=text, vector=sl.vector.copy(), tags=dict(meta))
|
|
461
|
+
)
|
|
462
|
+
return sl
|
|
463
|
+
|
|
464
|
+
def _salience(self, text: str, base: float) -> float:
|
|
465
|
+
"""Retention salience for a slice: SURPRISE x IMPACT x UNIQUENESS, floored.
|
|
466
|
+
|
|
467
|
+
Retention math: surprise ~ content density (token variety),
|
|
468
|
+
impact ~ the caller's base weight, uniqueness ~ 1/(1+near-neighbors already in pool).
|
|
469
|
+
Floored at :data:`_SPILL_SALIENCE_FLOOR` so real-but-cold context is never zeroed.
|
|
470
|
+
"""
|
|
471
|
+
words = text.split()
|
|
472
|
+
density = min(1.0, len(set(words)) / 64.0) if words else 0.0
|
|
473
|
+
neighbors = self._approx_neighbor_count(text)
|
|
474
|
+
uniqueness = uniqueness_from_neighbors(neighbors)
|
|
475
|
+
score = retention_score(surprise=density, impact=base, uniqueness=uniqueness)
|
|
476
|
+
return max(_SPILL_SALIENCE_FLOOR, float(score))
|
|
477
|
+
|
|
478
|
+
def _approx_neighbor_count(self, text: str) -> int:
|
|
479
|
+
"""Cheap near-neighbor count for uniqueness scoring (fail-soft, best-effort)."""
|
|
480
|
+
try:
|
|
481
|
+
qvec = self.encoder.encode(text)
|
|
482
|
+
hits = self.pool.search(qvec, k=4, session=self._scope())
|
|
483
|
+
except AetherContextError:
|
|
484
|
+
return 0
|
|
485
|
+
# count strongly-similar resident slices (cosine > 0.9)
|
|
486
|
+
return sum(1 for h in hits if float(np.dot(qvec, h.vector)) > 0.9)
|
|
487
|
+
|
|
488
|
+
# -- the lifecycle: run ---------------------------------------------------
|
|
489
|
+
def run(self, task: str) -> RunResult:
|
|
490
|
+
"""Run the full lifecycle for ``task`` and return a :class:`RunResult`.
|
|
491
|
+
|
|
492
|
+
Stages: **open** a fresh window, **stream** the model while encoding spill and paging
|
|
493
|
+
on a side thread, **page** the resident working set, then return. The pool budget is
|
|
494
|
+
held after every encoded write. Any pager/encoder/pool error is logged and the run
|
|
495
|
+
continues on the model's native window (fail-soft).
|
|
496
|
+
"""
|
|
497
|
+
if self._closed:
|
|
498
|
+
raise AetherContextError(
|
|
499
|
+
"run() called on a closed Session.",
|
|
500
|
+
hint="Open a new Session; a closed one has flushed its pool.",
|
|
501
|
+
)
|
|
502
|
+
stages: list[dict[str, Any]] = [
|
|
503
|
+
{"stage": "open", "task_tokens": self._count_tokens(task)}
|
|
504
|
+
]
|
|
505
|
+
|
|
506
|
+
text, spilled, overflowed = self._stream_and_encode(task, stages)
|
|
507
|
+
|
|
508
|
+
# paged reason: warm the working set from the final reasoning text (fail-soft).
|
|
509
|
+
self._page_working_set(task + "\n" + text, stages)
|
|
510
|
+
|
|
511
|
+
# Record the turn for export(): the user task then the model's reply (ask() funnels
|
|
512
|
+
# through run(), so logging here covers both without double-counting).
|
|
513
|
+
self._transcript.append(("user", task))
|
|
514
|
+
self._transcript.append(("assistant", text))
|
|
515
|
+
|
|
516
|
+
stages.append({"stage": "complete", "out_tokens": self._count_tokens(text)})
|
|
517
|
+
return RunResult(
|
|
518
|
+
text=text,
|
|
519
|
+
stages=stages,
|
|
520
|
+
hit_rate=self.pager.hit_rate(),
|
|
521
|
+
spilled=spilled,
|
|
522
|
+
resident=self.pager.warm_count,
|
|
523
|
+
overflowed=overflowed,
|
|
524
|
+
)
|
|
525
|
+
|
|
526
|
+
def _stream_and_encode(
|
|
527
|
+
self, task: str, stages: list[dict[str, Any]]
|
|
528
|
+
) -> tuple[str, int, bool]:
|
|
529
|
+
"""Stream the model for ``task`` while encoding spill and prefetching concurrently.
|
|
530
|
+
|
|
531
|
+
Returns ``(full_text, slices_spilled, overflowed)``. The prefetch runs on a background
|
|
532
|
+
thread so it overlaps generation (the backend call releases the GIL). Spill is encoded
|
|
533
|
+
whenever the running transcript crosses the window's trigger fraction.
|
|
534
|
+
"""
|
|
535
|
+
window = self.context_window
|
|
536
|
+
trigger_chars = int(window * self.config.trigger_fraction) * 4 # chars/4 budget
|
|
537
|
+
chunks: list[str] = []
|
|
538
|
+
spilled = 0
|
|
539
|
+
overflowed = False
|
|
540
|
+
pending = "" # text accumulated since the last spill
|
|
541
|
+
prefetch_thread: threading.Thread | None = None
|
|
542
|
+
|
|
543
|
+
try:
|
|
544
|
+
stream = self.local_llm.generate(
|
|
545
|
+
task, system=self.config.system, max_tokens=self.config.max_tokens
|
|
546
|
+
)
|
|
547
|
+
except AetherContextError as exc:
|
|
548
|
+
# a backend that fails to even start streaming: surface the typed error.
|
|
549
|
+
logger.error("model generate failed to start: %s", exc)
|
|
550
|
+
raise
|
|
551
|
+
|
|
552
|
+
for chunk in stream:
|
|
553
|
+
chunks.append(chunk)
|
|
554
|
+
pending += chunk
|
|
555
|
+
# when the running transcript crosses the trigger, encode the spill and launch a
|
|
556
|
+
# background prefetch from what we are reasoning about now (overlaps generation).
|
|
557
|
+
if trigger_chars > 0 and len(pending) >= trigger_chars:
|
|
558
|
+
overflowed = True
|
|
559
|
+
if self._encode_slice(
|
|
560
|
+
pending, salience=_SPILL_SALIENCE_FLOOR, tags={"kind": "spill", "source": MEMORY_SOURCE_MODEL}
|
|
561
|
+
):
|
|
562
|
+
spilled += 1
|
|
563
|
+
prefetch_thread = self._launch_prefetch(pending, prefetch_thread)
|
|
564
|
+
pending = ""
|
|
565
|
+
|
|
566
|
+
self._join(prefetch_thread)
|
|
567
|
+
|
|
568
|
+
# encode the final remainder as a slice too (so the tail is recoverable).
|
|
569
|
+
if pending.strip():
|
|
570
|
+
if self._encode_slice(
|
|
571
|
+
pending, salience=_SPILL_SALIENCE_FLOOR, tags={"kind": "spill", "source": MEMORY_SOURCE_MODEL}
|
|
572
|
+
):
|
|
573
|
+
spilled += 1
|
|
574
|
+
|
|
575
|
+
# fade the cold after the run (witness page-replacement step).
|
|
576
|
+
self._fade()
|
|
577
|
+
stages.append({"stage": "stream", "spilled": spilled, "overflowed": overflowed})
|
|
578
|
+
return "".join(chunks), spilled, overflowed
|
|
579
|
+
|
|
580
|
+
def _launch_prefetch(
|
|
581
|
+
self, reasoning_text: str, prior: threading.Thread | None
|
|
582
|
+
) -> threading.Thread | None:
|
|
583
|
+
"""Start a background prefetch from ``reasoning_text`` (fail-soft); join any prior.
|
|
584
|
+
|
|
585
|
+
The pager core is single-threaded; concurrency is the caller's job. We join the prior
|
|
586
|
+
thread first (so prefetches never pile up) then launch a fresh daemon thread. Any error
|
|
587
|
+
inside the thread is swallowed there — the pager is an optimization.
|
|
588
|
+
"""
|
|
589
|
+
self._join(prior)
|
|
590
|
+
|
|
591
|
+
def _work() -> None:
|
|
592
|
+
try:
|
|
593
|
+
self.pager.prefetch_from(self._key(), reasoning_text)
|
|
594
|
+
except Exception as exc: # noqa: BLE001 - fail-soft: never crash a run
|
|
595
|
+
logger.warning("background prefetch failed (%s); run continues", exc)
|
|
596
|
+
|
|
597
|
+
thread = threading.Thread(target=_work, daemon=True, name=f"prefetch-{self.id}")
|
|
598
|
+
thread.start()
|
|
599
|
+
return thread
|
|
600
|
+
|
|
601
|
+
@staticmethod
|
|
602
|
+
def _join(thread: threading.Thread | None) -> None:
|
|
603
|
+
"""Join a prefetch thread if present (bounded; it is pure CPU/IO over the pool)."""
|
|
604
|
+
if thread is not None and thread.is_alive():
|
|
605
|
+
thread.join(timeout=10.0)
|
|
606
|
+
|
|
607
|
+
def _page_working_set(self, reasoning_text: str, stages: list[dict[str, Any]]) -> None:
|
|
608
|
+
"""Warm the resident working set from the final reasoning (fail-soft, synchronous)."""
|
|
609
|
+
try:
|
|
610
|
+
self.pager.prefetch_from(self._key(), reasoning_text)
|
|
611
|
+
except Exception as exc: # noqa: BLE001 - fail-soft
|
|
612
|
+
logger.warning("final page-in failed (%s); run continues on native window", exc)
|
|
613
|
+
stages.append(
|
|
614
|
+
{
|
|
615
|
+
"stage": "page",
|
|
616
|
+
"resident": self.pager.warm_count,
|
|
617
|
+
"hit_rate": self.pager.hit_rate(),
|
|
618
|
+
}
|
|
619
|
+
)
|
|
620
|
+
|
|
621
|
+
def _rehydrate_pins(self) -> None:
|
|
622
|
+
"""Re-pin slices the pool already holds as permanent. Fail-soft.
|
|
623
|
+
|
|
624
|
+
The witness is in-memory; the pool is on disk. Without this a session
|
|
625
|
+
reopened against an existing pool starts with an empty witness, so
|
|
626
|
+
every previously pinned slice comes back as ordinary content and fades
|
|
627
|
+
on the first long run -- the pins would only hold until the next
|
|
628
|
+
restart, which is precisely when a durable memory is supposed to prove
|
|
629
|
+
itself.
|
|
630
|
+
|
|
631
|
+
The tag is the record. Permanence is re-derived from the slices
|
|
632
|
+
themselves rather than tracked in a second file that could disagree
|
|
633
|
+
with them.
|
|
634
|
+
"""
|
|
635
|
+
try:
|
|
636
|
+
stored = getattr(self.pool, "_slices", None)
|
|
637
|
+
if not stored:
|
|
638
|
+
return
|
|
639
|
+
pinned = 0
|
|
640
|
+
for sid, sl in stored.items():
|
|
641
|
+
meta = getattr(sl, "meta", None) or {}
|
|
642
|
+
if meta.get("pinned") is not True:
|
|
643
|
+
continue
|
|
644
|
+
if self.pool_mode == "separate":
|
|
645
|
+
# Never adopt another session's pins in a scoped pool.
|
|
646
|
+
if getattr(sl, "session", None) not in (None, self.id):
|
|
647
|
+
continue
|
|
648
|
+
# Anchored at 0.0 deliberately: a pinned slice never decays,
|
|
649
|
+
# so its anchor time carries no meaning, and the session
|
|
650
|
+
# clock does not exist yet at rehydration time.
|
|
651
|
+
self.witness.touch(
|
|
652
|
+
sid, salience=getattr(sl, "score", _REMEMBER_SALIENCE), now=0.0
|
|
653
|
+
)
|
|
654
|
+
self.witness.pin(sid)
|
|
655
|
+
pinned += 1
|
|
656
|
+
if pinned:
|
|
657
|
+
logger.debug("rehydrated %d pinned slice(s) from the pool", pinned)
|
|
658
|
+
except Exception as exc: # noqa: BLE001 - never block session open
|
|
659
|
+
logger.warning("pin rehydration skipped (%s)", exc)
|
|
660
|
+
|
|
661
|
+
def _fade(self) -> None:
|
|
662
|
+
"""Advance the witness clock and fade cold slices (page-replacement). Fail-soft."""
|
|
663
|
+
try:
|
|
664
|
+
self._clock += 1.0
|
|
665
|
+
self.witness.decay(self._clock)
|
|
666
|
+
except (ValueError, ArithmeticError) as exc: # defensive; decay is pure math
|
|
667
|
+
logger.debug("witness decay skipped (%s)", exc)
|
|
668
|
+
|
|
669
|
+
# -- convenience surface --------------------------------------------------
|
|
670
|
+
def ask(self, msg: str) -> str:
|
|
671
|
+
"""Run ``msg`` through the full lifecycle and return just the model's text."""
|
|
672
|
+
return self.run(msg).text
|
|
673
|
+
|
|
674
|
+
def stream(self, task: str) -> Iterator[str]:
|
|
675
|
+
"""Stream the model's chunks for ``task``, encoding spill as it flows.
|
|
676
|
+
|
|
677
|
+
Yields each text chunk as it arrives (so a UI can render live), while the same
|
|
678
|
+
encode-on-spill discipline runs underneath. The pool budget is held throughout. After
|
|
679
|
+
the stream is exhausted the witness fades the cold.
|
|
680
|
+
"""
|
|
681
|
+
if self._closed:
|
|
682
|
+
raise AetherContextError(
|
|
683
|
+
"stream() called on a closed Session.",
|
|
684
|
+
hint="Open a new Session; a closed one has flushed its pool.",
|
|
685
|
+
)
|
|
686
|
+
window = self.context_window
|
|
687
|
+
trigger_chars = int(window * self.config.trigger_fraction) * 4
|
|
688
|
+
pending = ""
|
|
689
|
+
prefetch_thread: threading.Thread | None = None
|
|
690
|
+
try:
|
|
691
|
+
stream = self.local_llm.generate(
|
|
692
|
+
task, system=self.config.system, max_tokens=self.config.max_tokens
|
|
693
|
+
)
|
|
694
|
+
except AetherContextError as exc:
|
|
695
|
+
logger.error("model generate failed to start: %s", exc)
|
|
696
|
+
raise
|
|
697
|
+
for chunk in stream:
|
|
698
|
+
pending += chunk
|
|
699
|
+
if trigger_chars > 0 and len(pending) >= trigger_chars:
|
|
700
|
+
self._encode_slice(
|
|
701
|
+
pending, salience=_SPILL_SALIENCE_FLOOR, tags={"kind": "spill", "source": MEMORY_SOURCE_MODEL}
|
|
702
|
+
)
|
|
703
|
+
prefetch_thread = self._launch_prefetch(pending, prefetch_thread)
|
|
704
|
+
pending = ""
|
|
705
|
+
yield chunk
|
|
706
|
+
self._join(prefetch_thread)
|
|
707
|
+
if pending.strip():
|
|
708
|
+
self._encode_slice(
|
|
709
|
+
pending, salience=_SPILL_SALIENCE_FLOOR, tags={"kind": "spill", "source": MEMORY_SOURCE_MODEL}
|
|
710
|
+
)
|
|
711
|
+
self._fade()
|
|
712
|
+
|
|
713
|
+
def harvest_candidates(self) -> list[HarvestCandidate]:
|
|
714
|
+
"""The run's abstracted durable artifacts (text + 256-dim vector + tags).
|
|
715
|
+
|
|
716
|
+
Retained locally; returned as a shallow copy so the caller cannot mutate session state.
|
|
717
|
+
"""
|
|
718
|
+
return list(self._harvest)
|
|
719
|
+
|
|
720
|
+
# -- clear / export / extended-thinking / status --------------------------
|
|
721
|
+
def clear(self, scope: str = "session") -> int:
|
|
722
|
+
"""Clear resident and (optionally) externalized state. Returns slices removed.
|
|
723
|
+
|
|
724
|
+
Two honest scopes:
|
|
725
|
+
|
|
726
|
+
* ``"resident"`` — empty only the in-memory resident window (the pager's warm set).
|
|
727
|
+
The reachable pool on disk is untouched, so this is :meth:`/new`: a fresh window
|
|
728
|
+
over the same reach. Always returns ``0`` (no pool slices were dropped).
|
|
729
|
+
* ``"session"`` (default) — also drop the slices THIS session externalized into the
|
|
730
|
+
pool. In ``"shared"`` mode the pool is global, so this empties the whole pool
|
|
731
|
+
(every session's slices). Returns the count of pool slices removed.
|
|
732
|
+
|
|
733
|
+
Resetting the witness keeps retention ranking consistent with the now-smaller pool.
|
|
734
|
+
Fail-soft: an empty/absent pool simply removes nothing.
|
|
735
|
+
"""
|
|
736
|
+
if scope not in ("resident", "session"):
|
|
737
|
+
raise AetherContextError(
|
|
738
|
+
f"clear scope must be 'resident' or 'session', got {scope!r}.",
|
|
739
|
+
hint="Use scope='resident' (window only) or scope='session' (window + slices).",
|
|
740
|
+
)
|
|
741
|
+
self.pager.reset()
|
|
742
|
+
if scope == "resident":
|
|
743
|
+
return 0
|
|
744
|
+
# session scope: drop this session's pool slices (or all of them when shared).
|
|
745
|
+
return self.pool.clear_session(self._scope())
|
|
746
|
+
|
|
747
|
+
def export(self, path: str | None = None) -> str:
|
|
748
|
+
"""Write the conversation transcript to ``path`` and return the path written.
|
|
749
|
+
|
|
750
|
+
With no ``path`` the transcript is written to ``aether-transcript.txt`` under the
|
|
751
|
+
current working directory (timestamp-free, so re-export overwrites in place). Each
|
|
752
|
+
turn is rendered as ``role: text`` on its own block. Returns the resolved path as a
|
|
753
|
+
string so a caller (the REPL ``/export``) can report exactly where it landed.
|
|
754
|
+
"""
|
|
755
|
+
target = Path(path) if path else Path.cwd() / _DEFAULT_TRANSCRIPT_NAME
|
|
756
|
+
lines = [f"{role}: {text}" for role, text in self._transcript]
|
|
757
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
758
|
+
target.write_text("\n\n".join(lines) + ("\n" if lines else ""), encoding="utf-8")
|
|
759
|
+
return str(target)
|
|
760
|
+
|
|
761
|
+
def toggle_extended(self) -> bool:
|
|
762
|
+
"""Flip the Extended-Thinking toggle and return its new value (honest, mock-safe).
|
|
763
|
+
|
|
764
|
+
In the mock backend this only widens the resident working set (more slices paged in
|
|
765
|
+
per turn) and surfaces in :meth:`status_dict`; it never silently changes the model's
|
|
766
|
+
real capability. Returns the toggle's new state so a REPL can report ``on``/``off``.
|
|
767
|
+
"""
|
|
768
|
+
self.extended = not self.extended
|
|
769
|
+
bonus = _EXTENDED_RESIDENT_BONUS if self.extended else 0
|
|
770
|
+
self.pager.default_k = self._resident_k + bonus
|
|
771
|
+
return self.extended
|
|
772
|
+
|
|
773
|
+
def status_dict(self) -> dict[str, Any]:
|
|
774
|
+
"""A snapshot of the session's state for the ``status`` command (all honest).
|
|
775
|
+
|
|
776
|
+
Fields: ``pool_gb`` (reach in GB), ``slices_used`` / ``capacity`` (governor budget),
|
|
777
|
+
``reach_tokens``, ``hit_rate`` (the pager's measured rate), ``resident_ram_mb``
|
|
778
|
+
(estimate ~29 MB/GB of reach), ``pool_mode``, ``index`` (the resolved kind actually
|
|
779
|
+
in use), ``model`` (the backend name), and ``extended`` (the toggle state).
|
|
780
|
+
"""
|
|
781
|
+
stats = self.pool.stats()
|
|
782
|
+
cost = slice_cost_bytes(int(stats["dim"]))
|
|
783
|
+
capacity = self.pool.ceiling_bytes // cost if cost else 0
|
|
784
|
+
return {
|
|
785
|
+
"pool_gb": self.pool_gb,
|
|
786
|
+
"slices_used": int(stats["count"]),
|
|
787
|
+
"capacity": int(capacity),
|
|
788
|
+
"reach_tokens": reach_tokens(self.pool_gb),
|
|
789
|
+
"hit_rate": self.pager.hit_rate(),
|
|
790
|
+
"resident_ram_mb": self.pool_gb * _RESIDENT_RAM_MB_PER_GB,
|
|
791
|
+
"pool_mode": self.pool_mode,
|
|
792
|
+
"index": stats["index"],
|
|
793
|
+
"model": getattr(self.local_llm, "name", str(self.config.model)),
|
|
794
|
+
"extended": self.extended,
|
|
795
|
+
"mpo_chain": self.mpo_chain_enabled,
|
|
796
|
+
}
|
|
797
|
+
|
|
798
|
+
# -- close + context manager ----------------------------------------------
|
|
799
|
+
def close(self) -> None:
|
|
800
|
+
"""Flush the pool and release its mmap. Idempotent.
|
|
801
|
+
|
|
802
|
+
Closing twice is a no-op. After close the session cannot ``run``/``stream`` again; the
|
|
803
|
+
run's artifacts remain available via :meth:`harvest_candidates` until then.
|
|
804
|
+
"""
|
|
805
|
+
if self._closed:
|
|
806
|
+
return
|
|
807
|
+
self._closed = True
|
|
808
|
+
self._flush_pool()
|
|
809
|
+
|
|
810
|
+
def _flush_pool(self) -> None:
|
|
811
|
+
"""Flush + release the pool's mmap (fail-soft; close must never raise)."""
|
|
812
|
+
try:
|
|
813
|
+
self.pool.close()
|
|
814
|
+
except AetherContextError as exc:
|
|
815
|
+
logger.warning("pool close failed (%s); state may be partially flushed", exc)
|
|
816
|
+
|
|
817
|
+
def __enter__(self) -> "Session":
|
|
818
|
+
return self
|
|
819
|
+
|
|
820
|
+
def __exit__(self, exc_type: Any, exc: Any, tb: Any) -> None:
|
|
821
|
+
self.close()
|
|
822
|
+
|
|
823
|
+
|
|
824
|
+
__all__ = [
|
|
825
|
+
"Session",
|
|
826
|
+
"RunResult",
|
|
827
|
+
"HarvestCandidate",
|
|
828
|
+
"DEFAULT_RESIDENT_K",
|
|
829
|
+
]
|