aether-context 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- aether_context/__init__.py +29 -0
- aether_context/_log.py +33 -0
- aether_context/cli.py +1191 -0
- aether_context/config.py +206 -0
- aether_context/context_pool.py +819 -0
- aether_context/encoder.py +213 -0
- aether_context/errors.py +93 -0
- aether_context/local_llm.py +846 -0
- aether_context/mpo.py +151 -0
- aether_context/py.typed +0 -0
- aether_context/quantize.py +86 -0
- aether_context/session.py +829 -0
- aether_context/slice_loader.py +501 -0
- aether_context/tokenizer.py +64 -0
- aether_context/ui.py +253 -0
- aether_context/witness.py +356 -0
- aether_context-0.3.0.dist-info/METADATA +429 -0
- aether_context-0.3.0.dist-info/RECORD +23 -0
- aether_context-0.3.0.dist-info/WHEEL +5 -0
- aether_context-0.3.0.dist-info/entry_points.txt +2 -0
- aether_context-0.3.0.dist-info/licenses/LICENSE +201 -0
- aether_context-0.3.0.dist-info/licenses/NOTICE.md +19 -0
- aether_context-0.3.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,846 @@
|
|
|
1
|
+
# aether-context (Unlimited Context)
|
|
2
|
+
# Copyright (c) 2026 Aether AI
|
|
3
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
4
|
+
"""THE WRAPPER — backend-agnostic local-LLM adapters.
|
|
5
|
+
|
|
6
|
+
This is the surface a user actually touches. The whole engine talks to *one* protocol
|
|
7
|
+
(:class:`LocalLLM`); every backend (Ollama, llama.cpp, HF, Mock) satisfies it so the
|
|
8
|
+
engine never special-cases a backend.
|
|
9
|
+
|
|
10
|
+
Design laws honored here:
|
|
11
|
+
* **Simple** — one spec string picks a backend (``"ollama/qwen2.5"``, ``"mock"``…).
|
|
12
|
+
* **Light core** — the primary Ollama path uses the Python standard library only
|
|
13
|
+
(``urllib``); no extra dependency for ``pip install aether-context``.
|
|
14
|
+
* **Forgiving** — daemon down / model missing raise *typed* errors from
|
|
15
|
+
:mod:`aether_context.errors`, each with an actionable ``.hint``. Metadata probes are
|
|
16
|
+
fail-soft (a failed ``/api/show`` degrades ``context_window`` to a fallback, never
|
|
17
|
+
raises into a run).
|
|
18
|
+
* **Streaming** — ``generate`` yields text chunks so the pager can prefetch the next
|
|
19
|
+
slices *while* the model is still talking.
|
|
20
|
+
|
|
21
|
+
No ``print()`` (use the logging seam), no bare ``except`` (catch specific, re-wrap as a
|
|
22
|
+
typed error).
|
|
23
|
+
"""
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import hashlib
|
|
27
|
+
import json
|
|
28
|
+
import os
|
|
29
|
+
import urllib.error
|
|
30
|
+
import urllib.request
|
|
31
|
+
from dataclasses import dataclass, field
|
|
32
|
+
from typing import Iterator, Optional, Protocol, runtime_checkable
|
|
33
|
+
|
|
34
|
+
from aether_context._log import get_logger
|
|
35
|
+
from aether_context.errors import (
|
|
36
|
+
AetherContextError,
|
|
37
|
+
BackendUnavailable,
|
|
38
|
+
ModelNotPulled,
|
|
39
|
+
OllamaNotRunning,
|
|
40
|
+
)
|
|
41
|
+
from aether_context.tokenizer import estimate
|
|
42
|
+
|
|
43
|
+
_log = get_logger(__name__)
|
|
44
|
+
|
|
45
|
+
#: Fallback context window (tokens) when a backend cannot report its own.
|
|
46
|
+
DEFAULT_CONTEXT_WINDOW = 8192
|
|
47
|
+
#: Default Ollama daemon host.
|
|
48
|
+
DEFAULT_OLLAMA_HOST = "http://localhost:11434"
|
|
49
|
+
#: Recognised backends in a spec string.
|
|
50
|
+
_BACKENDS = ("ollama", "llamacpp", "hf", "mock", "openai")
|
|
51
|
+
#: HTTP timeout (seconds) for Ollama requests. Generation can be slow → generous.
|
|
52
|
+
_OLLAMA_TIMEOUT = 600
|
|
53
|
+
#: Best-effort metadata probe timeout (seconds) — short; failure is non-fatal.
|
|
54
|
+
_OLLAMA_PROBE_TIMEOUT = 5
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
# ---------------------------------------------------------------------------
|
|
58
|
+
# The protocol every adapter satisfies.
|
|
59
|
+
# ---------------------------------------------------------------------------
|
|
60
|
+
@runtime_checkable
|
|
61
|
+
class LocalLLM(Protocol):
|
|
62
|
+
"""Backend-agnostic local-model contract.
|
|
63
|
+
|
|
64
|
+
``generate`` **streams** text chunks (yield once with the full text if a backend
|
|
65
|
+
cannot stream — it still works, you just lose the free prefetch concurrency).
|
|
66
|
+
"""
|
|
67
|
+
|
|
68
|
+
name: str
|
|
69
|
+
|
|
70
|
+
@property
|
|
71
|
+
def context_window(self) -> int:
|
|
72
|
+
"""Token window the model exposes this turn (read-only; may be lazily probed)."""
|
|
73
|
+
...
|
|
74
|
+
|
|
75
|
+
def generate(
|
|
76
|
+
self,
|
|
77
|
+
prompt: str,
|
|
78
|
+
*,
|
|
79
|
+
system: Optional[str] = None,
|
|
80
|
+
stop: Optional[list[str]] = None,
|
|
81
|
+
max_tokens: Optional[int] = None,
|
|
82
|
+
) -> Iterator[str]:
|
|
83
|
+
"""Stream model output for ``prompt`` as a sequence of text chunks."""
|
|
84
|
+
...
|
|
85
|
+
|
|
86
|
+
def count_tokens(self, text: str) -> int:
|
|
87
|
+
"""Return the token count of ``text`` for budget math."""
|
|
88
|
+
...
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
# ---------------------------------------------------------------------------
|
|
92
|
+
# Spec parsing.
|
|
93
|
+
# ---------------------------------------------------------------------------
|
|
94
|
+
@dataclass(frozen=True)
|
|
95
|
+
class ModelSpec:
|
|
96
|
+
"""Parsed model spec: a backend, a model ref, and pass-through options."""
|
|
97
|
+
|
|
98
|
+
backend: str
|
|
99
|
+
ref: str
|
|
100
|
+
options: dict = field(default_factory=dict)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
_SPEC_HINT = (
|
|
104
|
+
"Use one of: 'ollama/qwen2.5' (or a bare name -> ollama), "
|
|
105
|
+
"'llamacpp:/path/to/model.gguf', 'hf/org/model', or 'mock'."
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def parse_spec(spec: str) -> ModelSpec:
|
|
110
|
+
"""Parse a spec string into a :class:`ModelSpec`.
|
|
111
|
+
|
|
112
|
+
Grammar (one obvious format)::
|
|
113
|
+
|
|
114
|
+
ollama/<name> -> Ollama (tags ok: ollama/llama3.1:8b)
|
|
115
|
+
<name> -> bare name assumed Ollama
|
|
116
|
+
llamacpp:<path> -> llama.cpp over a .gguf (first ':' splits; drive ':' kept)
|
|
117
|
+
hf/<org>/<model> -> HF transformers
|
|
118
|
+
mock -> built-in deterministic model
|
|
119
|
+
|
|
120
|
+
Raises :class:`~aether_context.errors.BackendUnavailable` (carrying a ``.hint`` with
|
|
121
|
+
the format) for anything unparseable or an unknown backend.
|
|
122
|
+
"""
|
|
123
|
+
if not isinstance(spec, str):
|
|
124
|
+
raise BackendUnavailable(
|
|
125
|
+
f"Model spec must be a string (backend/ref). Got: {type(spec).__name__}",
|
|
126
|
+
hint=_SPEC_HINT,
|
|
127
|
+
)
|
|
128
|
+
text = spec.strip()
|
|
129
|
+
if not text:
|
|
130
|
+
raise BackendUnavailable(
|
|
131
|
+
"Model spec is empty. Must be backend/ref "
|
|
132
|
+
"(e.g. ollama/qwen2.5, llamacpp:/path/model.gguf, hf/org/model, mock).",
|
|
133
|
+
hint=_SPEC_HINT,
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
# mock — exact bare token.
|
|
137
|
+
if text == "mock":
|
|
138
|
+
return ModelSpec(backend="mock", ref="mock")
|
|
139
|
+
|
|
140
|
+
# llamacpp:<path> — split on the FIRST colon only so a Windows drive ':' survives.
|
|
141
|
+
if text.startswith("llamacpp:"):
|
|
142
|
+
ref = text[len("llamacpp:"):]
|
|
143
|
+
if not ref:
|
|
144
|
+
raise BackendUnavailable(
|
|
145
|
+
"llamacpp spec needs a path: 'llamacpp:/path/to/model.gguf'",
|
|
146
|
+
hint=_SPEC_HINT,
|
|
147
|
+
)
|
|
148
|
+
return ModelSpec(backend="llamacpp", ref=ref)
|
|
149
|
+
|
|
150
|
+
# slash-prefixed backends: ollama/<ref>, hf/<org/model>.
|
|
151
|
+
if "/" in text:
|
|
152
|
+
head, ref = text.split("/", 1)
|
|
153
|
+
if head == "ollama":
|
|
154
|
+
return ModelSpec(backend="ollama", ref=ref)
|
|
155
|
+
if head == "hf":
|
|
156
|
+
return ModelSpec(backend="hf", ref=ref)
|
|
157
|
+
if head == "openai":
|
|
158
|
+
return ModelSpec(backend="openai", ref=ref)
|
|
159
|
+
if head in _BACKENDS:
|
|
160
|
+
return ModelSpec(backend=head, ref=ref)
|
|
161
|
+
raise BackendUnavailable(
|
|
162
|
+
f"Unknown backend '{head}' in spec '{spec}'. "
|
|
163
|
+
"Must be backend/ref (e.g. ollama/qwen2.5, llamacpp:/path/model.gguf, "
|
|
164
|
+
"hf/org/model, mock).",
|
|
165
|
+
hint=_SPEC_HINT,
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
# Bare name -> assumed Ollama (documented convenience; see docs/local-models.md).
|
|
169
|
+
return ModelSpec(backend="ollama", ref=text)
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
# ---------------------------------------------------------------------------
|
|
173
|
+
# load_model — the one public entry point.
|
|
174
|
+
# ---------------------------------------------------------------------------
|
|
175
|
+
def load_model(spec: "str | LocalLLM", **kw: object) -> LocalLLM:
|
|
176
|
+
"""Resolve ``spec`` to a :class:`LocalLLM`.
|
|
177
|
+
|
|
178
|
+
* A :class:`LocalLLM` object is returned unchanged (bring-your-own backend).
|
|
179
|
+
* A spec string is parsed and dispatched to the matching adapter; extra ``**kw`` are
|
|
180
|
+
forwarded to the adapter constructor.
|
|
181
|
+
|
|
182
|
+
Raises :class:`~aether_context.errors.BackendUnavailable` for an unknown backend or a
|
|
183
|
+
backend whose optional dependency is not installed.
|
|
184
|
+
"""
|
|
185
|
+
# Already a backend object? Pass it straight through (duck-typed protocol).
|
|
186
|
+
if not isinstance(spec, str) and isinstance(spec, LocalLLM):
|
|
187
|
+
return spec
|
|
188
|
+
|
|
189
|
+
if not isinstance(spec, str):
|
|
190
|
+
raise BackendUnavailable(
|
|
191
|
+
f"model must be a spec string or a LocalLLM object. Got: {type(spec).__name__}",
|
|
192
|
+
hint=_SPEC_HINT,
|
|
193
|
+
)
|
|
194
|
+
|
|
195
|
+
parsed = parse_spec(spec)
|
|
196
|
+
backend = parsed.backend
|
|
197
|
+
if backend == "mock":
|
|
198
|
+
return MockLLM(name=parsed.ref, **kw) # type: ignore[arg-type]
|
|
199
|
+
if backend == "ollama":
|
|
200
|
+
return OllamaLLM(parsed.ref, **kw) # type: ignore[arg-type]
|
|
201
|
+
if backend == "llamacpp":
|
|
202
|
+
return LlamaCppLLM(parsed.ref, **kw) # type: ignore[arg-type]
|
|
203
|
+
if backend == "hf":
|
|
204
|
+
return HFLLM(parsed.ref, **kw) # type: ignore[arg-type]
|
|
205
|
+
if backend == "openai":
|
|
206
|
+
return OpenAICompatLLM(parsed.ref, **kw) # type: ignore[arg-type]
|
|
207
|
+
raise BackendUnavailable(
|
|
208
|
+
f"Unknown backend '{backend}'.", hint=_SPEC_HINT
|
|
209
|
+
)
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
# ---------------------------------------------------------------------------
|
|
213
|
+
# MockLLM — deterministic, dependency-free.
|
|
214
|
+
# ---------------------------------------------------------------------------
|
|
215
|
+
#: Vocabulary the mock draws from (deterministically) to build pseudo-output.
|
|
216
|
+
_MOCK_WORDS = (
|
|
217
|
+
"plan step build module function class test verify refactor encode slice pool "
|
|
218
|
+
"window context retrieve prefetch witness fade harden session token vector index "
|
|
219
|
+
"the and then so we now next done note check ok yes done finally indeed thus"
|
|
220
|
+
).split()
|
|
221
|
+
#: Mock chunk width in characters (>1 chunk for any non-trivial output → streaming).
|
|
222
|
+
_MOCK_CHUNK_CHARS = 32
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
@dataclass
|
|
226
|
+
class MockLLM:
|
|
227
|
+
"""Deterministic, offline, zero-dependency model.
|
|
228
|
+
|
|
229
|
+
Output is derived from a hash of ``(prompt, system)`` so the same input always
|
|
230
|
+
produces the same text — letting tests assert properties and the bench run a hermetic
|
|
231
|
+
baseline. ``output_tokens`` controls *length* independently of ``context_window`` so a
|
|
232
|
+
test/bench can force overflow with a tiny window and a long generation.
|
|
233
|
+
"""
|
|
234
|
+
|
|
235
|
+
name: str = "mock"
|
|
236
|
+
context_window: int = DEFAULT_CONTEXT_WINDOW
|
|
237
|
+
output_tokens: int = 64
|
|
238
|
+
|
|
239
|
+
def _build_text(self, prompt: str, system: Optional[str]) -> str:
|
|
240
|
+
seed_src = f"{system or ''}\x00{prompt}".encode("utf-8")
|
|
241
|
+
digest = hashlib.sha256(seed_src).digest()
|
|
242
|
+
# Expand the 32-byte digest deterministically to as many words as we need.
|
|
243
|
+
words: list[str] = []
|
|
244
|
+
i = 0
|
|
245
|
+
# ~6 chars per word incl. space; target enough words for output_tokens (chars/4).
|
|
246
|
+
target_words = max(1, (self.output_tokens * 4) // 6)
|
|
247
|
+
while len(words) < target_words:
|
|
248
|
+
# rehash with a counter so we never run out of entropy
|
|
249
|
+
block = hashlib.sha256(digest + i.to_bytes(4, "big")).digest()
|
|
250
|
+
for b in block:
|
|
251
|
+
words.append(_MOCK_WORDS[b % len(_MOCK_WORDS)])
|
|
252
|
+
if len(words) >= target_words:
|
|
253
|
+
break
|
|
254
|
+
i += 1
|
|
255
|
+
return " ".join(words)
|
|
256
|
+
|
|
257
|
+
@property
|
|
258
|
+
def is_streaming(self) -> bool:
|
|
259
|
+
"""True iff a representative ``generate`` yields more than one chunk."""
|
|
260
|
+
return len(self._build_text("probe", None)) > _MOCK_CHUNK_CHARS
|
|
261
|
+
|
|
262
|
+
def generate(
|
|
263
|
+
self,
|
|
264
|
+
prompt: str,
|
|
265
|
+
*,
|
|
266
|
+
system: Optional[str] = None,
|
|
267
|
+
stop: Optional[list[str]] = None,
|
|
268
|
+
max_tokens: Optional[int] = None,
|
|
269
|
+
) -> Iterator[str]:
|
|
270
|
+
"""Yield deterministic pseudo-output in fixed-width chunks (streaming)."""
|
|
271
|
+
if not isinstance(prompt, str):
|
|
272
|
+
raise TypeError(f"prompt must be str, got {type(prompt).__name__}")
|
|
273
|
+
text = self._build_text(prompt, system)
|
|
274
|
+
|
|
275
|
+
# Apply max_tokens cap (chars/4 estimate → char budget).
|
|
276
|
+
if max_tokens is not None and max_tokens >= 0:
|
|
277
|
+
char_budget = max_tokens * 4
|
|
278
|
+
text = text[:char_budget]
|
|
279
|
+
|
|
280
|
+
# Apply stop sequences: truncate at the earliest occurrence of any stop string.
|
|
281
|
+
if stop:
|
|
282
|
+
cut = len(text)
|
|
283
|
+
for s in stop:
|
|
284
|
+
if s:
|
|
285
|
+
idx = text.find(s)
|
|
286
|
+
if idx != -1:
|
|
287
|
+
cut = min(cut, idx)
|
|
288
|
+
text = text[:cut]
|
|
289
|
+
|
|
290
|
+
for start in range(0, len(text), _MOCK_CHUNK_CHARS):
|
|
291
|
+
yield text[start:start + _MOCK_CHUNK_CHARS]
|
|
292
|
+
|
|
293
|
+
def count_tokens(self, text: str) -> int:
|
|
294
|
+
"""Estimate token count (chars/4) — the backend-agnostic budget rule."""
|
|
295
|
+
return estimate(text)
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
# ---------------------------------------------------------------------------
|
|
299
|
+
# OllamaLLM — primary path, stdlib urllib only.
|
|
300
|
+
# ---------------------------------------------------------------------------
|
|
301
|
+
class OllamaLLM:
|
|
302
|
+
"""Ollama adapter over HTTP using only the standard library (``urllib``).
|
|
303
|
+
|
|
304
|
+
Talks to ``<host>/api/chat`` with ``stream=true``. ``context_window`` is auto-detected
|
|
305
|
+
from ``<host>/api/show`` metadata (``model_info.*context_length`` / ``num_ctx``),
|
|
306
|
+
falling back to :data:`DEFAULT_CONTEXT_WINDOW` if the probe fails (fail-soft). Token
|
|
307
|
+
counts use the chars/4 estimate (Ollama exposes no count endpoint).
|
|
308
|
+
"""
|
|
309
|
+
|
|
310
|
+
def __init__(
|
|
311
|
+
self,
|
|
312
|
+
ref: str,
|
|
313
|
+
*,
|
|
314
|
+
host: str = DEFAULT_OLLAMA_HOST,
|
|
315
|
+
context_window: Optional[int] = None,
|
|
316
|
+
pull: bool = False,
|
|
317
|
+
model_options: Optional[dict] = None,
|
|
318
|
+
) -> None:
|
|
319
|
+
self.name: str = ref
|
|
320
|
+
self.host: str = host.rstrip("/")
|
|
321
|
+
self._pull: bool = pull
|
|
322
|
+
self._model_options: dict = dict(model_options or {})
|
|
323
|
+
# Lazily resolved; an explicit value wins, else probe (fail-soft), else fallback.
|
|
324
|
+
self._context_window: Optional[int] = context_window
|
|
325
|
+
|
|
326
|
+
# -- metadata (fail-soft) ------------------------------------------------
|
|
327
|
+
@property
|
|
328
|
+
def context_window(self) -> int:
|
|
329
|
+
"""Token window, auto-detected via ``/api/show`` (fallback on any failure)."""
|
|
330
|
+
if self._context_window is None:
|
|
331
|
+
self._context_window = self._probe_context_window()
|
|
332
|
+
return self._context_window
|
|
333
|
+
|
|
334
|
+
def _probe_context_window(self) -> int:
|
|
335
|
+
"""Best-effort ``/api/show`` probe. Never raises — degrades to the fallback."""
|
|
336
|
+
try:
|
|
337
|
+
payload = json.dumps({"model": self.name}).encode("utf-8")
|
|
338
|
+
req = urllib.request.Request(
|
|
339
|
+
f"{self.host}/api/show",
|
|
340
|
+
data=payload,
|
|
341
|
+
headers={"Content-Type": "application/json"},
|
|
342
|
+
method="POST",
|
|
343
|
+
)
|
|
344
|
+
with urllib.request.urlopen(req, timeout=_OLLAMA_PROBE_TIMEOUT) as resp:
|
|
345
|
+
body = json.loads(resp.read().decode("utf-8"))
|
|
346
|
+
return self._extract_context_length(body)
|
|
347
|
+
except (urllib.error.URLError, OSError, ValueError, KeyError, TypeError) as exc:
|
|
348
|
+
_log.debug("ollama /api/show probe failed, using fallback window: %s", exc)
|
|
349
|
+
return DEFAULT_CONTEXT_WINDOW
|
|
350
|
+
|
|
351
|
+
@staticmethod
|
|
352
|
+
def _extract_context_length(body: dict) -> int:
|
|
353
|
+
"""Pull a context length out of /api/show metadata; fallback if absent."""
|
|
354
|
+
info = body.get("model_info") or {}
|
|
355
|
+
for key, value in info.items():
|
|
356
|
+
if key.endswith("context_length") and isinstance(value, int) and value > 0:
|
|
357
|
+
return value
|
|
358
|
+
params = body.get("parameters")
|
|
359
|
+
if isinstance(params, str):
|
|
360
|
+
for line in params.splitlines():
|
|
361
|
+
parts = line.split()
|
|
362
|
+
if len(parts) == 2 and parts[0] == "num_ctx" and parts[1].isdigit():
|
|
363
|
+
return int(parts[1])
|
|
364
|
+
return DEFAULT_CONTEXT_WINDOW
|
|
365
|
+
|
|
366
|
+
# -- generation ----------------------------------------------------------
|
|
367
|
+
def generate(
|
|
368
|
+
self,
|
|
369
|
+
prompt: str,
|
|
370
|
+
*,
|
|
371
|
+
system: Optional[str] = None,
|
|
372
|
+
stop: Optional[list[str]] = None,
|
|
373
|
+
max_tokens: Optional[int] = None,
|
|
374
|
+
) -> Iterator[str]:
|
|
375
|
+
"""Stream chunks from ``/api/chat``. Typed, forgiving errors on failure."""
|
|
376
|
+
if self._pull:
|
|
377
|
+
self._ensure_pulled()
|
|
378
|
+
messages: list[dict] = []
|
|
379
|
+
if system:
|
|
380
|
+
messages.append({"role": "system", "content": system})
|
|
381
|
+
messages.append({"role": "user", "content": prompt})
|
|
382
|
+
|
|
383
|
+
options = dict(self._model_options)
|
|
384
|
+
if stop:
|
|
385
|
+
options["stop"] = list(stop)
|
|
386
|
+
if max_tokens is not None:
|
|
387
|
+
options["num_predict"] = int(max_tokens)
|
|
388
|
+
|
|
389
|
+
body: dict = {"model": self.name, "messages": messages, "stream": True}
|
|
390
|
+
if options:
|
|
391
|
+
body["options"] = options
|
|
392
|
+
|
|
393
|
+
yield from self._stream_chat(body)
|
|
394
|
+
|
|
395
|
+
def _stream_chat(self, body: dict) -> Iterator[str]:
|
|
396
|
+
payload = json.dumps(body).encode("utf-8")
|
|
397
|
+
req = urllib.request.Request(
|
|
398
|
+
f"{self.host}/api/chat",
|
|
399
|
+
data=payload,
|
|
400
|
+
headers={"Content-Type": "application/json"},
|
|
401
|
+
method="POST",
|
|
402
|
+
)
|
|
403
|
+
try:
|
|
404
|
+
resp = urllib.request.urlopen(req, timeout=_OLLAMA_TIMEOUT)
|
|
405
|
+
except urllib.error.HTTPError as exc:
|
|
406
|
+
raise self._http_error_to_typed(exc) from exc
|
|
407
|
+
except (urllib.error.URLError, OSError) as exc:
|
|
408
|
+
raise OllamaNotRunning(
|
|
409
|
+
f"Could not reach the Ollama daemon at {self.host}: {exc}"
|
|
410
|
+
) from exc
|
|
411
|
+
|
|
412
|
+
with resp:
|
|
413
|
+
for raw in resp:
|
|
414
|
+
line = raw.decode("utf-8").strip()
|
|
415
|
+
if not line:
|
|
416
|
+
continue
|
|
417
|
+
try:
|
|
418
|
+
obj = json.loads(line)
|
|
419
|
+
except json.JSONDecodeError as exc:
|
|
420
|
+
_log.debug("skipping non-JSON ollama stream line: %s", exc)
|
|
421
|
+
continue
|
|
422
|
+
if obj.get("error"):
|
|
423
|
+
raise ModelNotPulled(
|
|
424
|
+
f"Ollama error for model '{self.name}': {obj['error']}"
|
|
425
|
+
)
|
|
426
|
+
chunk = (obj.get("message") or {}).get("content")
|
|
427
|
+
if chunk:
|
|
428
|
+
yield chunk
|
|
429
|
+
if obj.get("done"):
|
|
430
|
+
break
|
|
431
|
+
|
|
432
|
+
def _http_error_to_typed(self, exc: urllib.error.HTTPError) -> AetherContextError:
|
|
433
|
+
"""Map an Ollama HTTP error to a typed, hinted aether-context error."""
|
|
434
|
+
detail = ""
|
|
435
|
+
try:
|
|
436
|
+
detail = exc.read().decode("utf-8", errors="replace")
|
|
437
|
+
except OSError:
|
|
438
|
+
detail = ""
|
|
439
|
+
if exc.code == 404 or "not found" in detail.lower():
|
|
440
|
+
return ModelNotPulled(
|
|
441
|
+
f"Ollama model '{self.name}' is not pulled: {detail or exc}",
|
|
442
|
+
hint=f"Pull it first: `ollama pull {self.name}` (or pass pull=True).",
|
|
443
|
+
)
|
|
444
|
+
return OllamaNotRunning(
|
|
445
|
+
f"Ollama returned HTTP {exc.code} from {self.host}: {detail or exc}"
|
|
446
|
+
)
|
|
447
|
+
|
|
448
|
+
def _ensure_pulled(self) -> None:
|
|
449
|
+
"""Best-effort ``/api/pull`` when ``pull=True``. Typed error on failure."""
|
|
450
|
+
payload = json.dumps({"model": self.name, "stream": False}).encode("utf-8")
|
|
451
|
+
req = urllib.request.Request(
|
|
452
|
+
f"{self.host}/api/pull",
|
|
453
|
+
data=payload,
|
|
454
|
+
headers={"Content-Type": "application/json"},
|
|
455
|
+
method="POST",
|
|
456
|
+
)
|
|
457
|
+
try:
|
|
458
|
+
with urllib.request.urlopen(req, timeout=_OLLAMA_TIMEOUT) as resp:
|
|
459
|
+
resp.read()
|
|
460
|
+
except urllib.error.HTTPError as exc:
|
|
461
|
+
raise self._http_error_to_typed(exc) from exc
|
|
462
|
+
except (urllib.error.URLError, OSError) as exc:
|
|
463
|
+
raise OllamaNotRunning(
|
|
464
|
+
f"Could not reach the Ollama daemon at {self.host} to pull "
|
|
465
|
+
f"'{self.name}': {exc}"
|
|
466
|
+
) from exc
|
|
467
|
+
|
|
468
|
+
# -- token counting ------------------------------------------------------
|
|
469
|
+
def count_tokens(self, text: str) -> int:
|
|
470
|
+
"""Estimate token count (chars/4) — Ollama has no count endpoint."""
|
|
471
|
+
return estimate(text)
|
|
472
|
+
|
|
473
|
+
|
|
474
|
+
# ---------------------------------------------------------------------------
|
|
475
|
+
# OpenAICompatLLM — vendor-neutral OpenAI-compatible API (OpenRouter / OpenAI / vLLM).
|
|
476
|
+
# ---------------------------------------------------------------------------
|
|
477
|
+
#: OpenRouter default base URL when only OPENROUTER_API_KEY is set.
|
|
478
|
+
_OPENROUTER_BASE = "https://openrouter.ai/api/v1"
|
|
479
|
+
#: HTTP timeout (s) for the OpenAI-compatible adapter. Kept tight: a coding turn that
|
|
480
|
+
#: hasn't responded in 3 min is a hung connection — fail fast so the batch can continue
|
|
481
|
+
#: (a 600s hang once stalled a whole sequential run ~30 min). Override via OPENAI_HTTP_TIMEOUT.
|
|
482
|
+
_OPENAI_TIMEOUT = int(os.environ.get("OPENAI_HTTP_TIMEOUT", "180"))
|
|
483
|
+
|
|
484
|
+
|
|
485
|
+
class OpenAICompatLLM:
|
|
486
|
+
"""Vendor-neutral OpenAI-compatible adapter (OpenRouter / OpenAI / vLLM / …).
|
|
487
|
+
|
|
488
|
+
Implements the :class:`LocalLLM` protocol (streaming ``generate``) and adds
|
|
489
|
+
``chat(messages, tools)`` (non-streaming, function-calling) for the eval harness's agent
|
|
490
|
+
loop. stdlib ``urllib`` only — no extra dependency. ``context_window`` is caller-provided
|
|
491
|
+
(these APIs do not report it); ``count_tokens`` is the chars/4 estimate.
|
|
492
|
+
|
|
493
|
+
Config resolution: explicit ``base_url``/``api_key`` win; else ``OPENROUTER_API_KEY``
|
|
494
|
+
(with the OpenRouter base URL); else ``OPENAI_BASE_URL``/``OPENAI_API_KEY``. The key is
|
|
495
|
+
never logged or placed in an error message.
|
|
496
|
+
"""
|
|
497
|
+
|
|
498
|
+
def __init__(
|
|
499
|
+
self,
|
|
500
|
+
ref: str,
|
|
501
|
+
*,
|
|
502
|
+
base_url: Optional[str] = None,
|
|
503
|
+
api_key: Optional[str] = None,
|
|
504
|
+
context_window: Optional[int] = None,
|
|
505
|
+
model_options: Optional[dict] = None,
|
|
506
|
+
) -> None:
|
|
507
|
+
self.name: str = ref
|
|
508
|
+
key = api_key or os.environ.get("OPENROUTER_API_KEY") or os.environ.get("OPENAI_API_KEY")
|
|
509
|
+
if base_url is None:
|
|
510
|
+
if os.environ.get("OPENROUTER_API_KEY") and not api_key:
|
|
511
|
+
base_url = _OPENROUTER_BASE
|
|
512
|
+
else:
|
|
513
|
+
base_url = os.environ.get("OPENAI_BASE_URL")
|
|
514
|
+
if not key:
|
|
515
|
+
raise BackendUnavailable(
|
|
516
|
+
"No API key for the OpenAI-compatible backend.",
|
|
517
|
+
hint="Set OPENROUTER_API_KEY (or OPENAI_API_KEY), or pass api_key=...",
|
|
518
|
+
)
|
|
519
|
+
if not base_url:
|
|
520
|
+
raise BackendUnavailable(
|
|
521
|
+
"No base_url for the OpenAI-compatible backend.",
|
|
522
|
+
hint="Set OPENROUTER_API_KEY (uses openrouter.ai), OPENAI_BASE_URL, or pass base_url=...",
|
|
523
|
+
)
|
|
524
|
+
self.base_url: str = base_url.rstrip("/")
|
|
525
|
+
self.api_key: str = key
|
|
526
|
+
self._context_window: int = context_window or DEFAULT_CONTEXT_WINDOW
|
|
527
|
+
self._model_options: dict = dict(model_options or {})
|
|
528
|
+
|
|
529
|
+
@property
|
|
530
|
+
def context_window(self) -> int:
|
|
531
|
+
return self._context_window
|
|
532
|
+
|
|
533
|
+
# -- low-level HTTP (one place; mockable) --------------------------------
|
|
534
|
+
def _open(self, req: "urllib.request.Request"):
|
|
535
|
+
return urllib.request.urlopen(req, timeout=_OPENAI_TIMEOUT)
|
|
536
|
+
|
|
537
|
+
def _request(self, payload: dict, *, stream: bool):
|
|
538
|
+
body = json.dumps(payload).encode("utf-8")
|
|
539
|
+
req = urllib.request.Request(
|
|
540
|
+
f"{self.base_url}/chat/completions",
|
|
541
|
+
data=body,
|
|
542
|
+
headers={
|
|
543
|
+
"Content-Type": "application/json",
|
|
544
|
+
"Authorization": f"Bearer {self.api_key}",
|
|
545
|
+
"Accept": "text/event-stream" if stream else "application/json",
|
|
546
|
+
},
|
|
547
|
+
method="POST",
|
|
548
|
+
)
|
|
549
|
+
try:
|
|
550
|
+
return self._open(req)
|
|
551
|
+
except urllib.error.HTTPError as exc:
|
|
552
|
+
detail = ""
|
|
553
|
+
try:
|
|
554
|
+
detail = exc.read().decode("utf-8", errors="replace")[:300]
|
|
555
|
+
except OSError:
|
|
556
|
+
detail = ""
|
|
557
|
+
raise BackendUnavailable(
|
|
558
|
+
f"OpenAI-compatible API HTTP {exc.code} from {self.base_url}.",
|
|
559
|
+
hint=f"Check model/key/credits. Detail: {detail}",
|
|
560
|
+
) from exc
|
|
561
|
+
except (urllib.error.URLError, OSError) as exc:
|
|
562
|
+
raise BackendUnavailable(
|
|
563
|
+
f"Could not reach the API at {self.base_url}: {exc}",
|
|
564
|
+
hint="Check the base_url and your network.",
|
|
565
|
+
) from exc
|
|
566
|
+
|
|
567
|
+
# -- streaming generate (LocalLLM protocol) ------------------------------
|
|
568
|
+
def generate(
|
|
569
|
+
self,
|
|
570
|
+
prompt: str,
|
|
571
|
+
*,
|
|
572
|
+
system: Optional[str] = None,
|
|
573
|
+
stop: Optional[list[str]] = None,
|
|
574
|
+
max_tokens: Optional[int] = None,
|
|
575
|
+
) -> Iterator[str]:
|
|
576
|
+
"""Stream content chunks from ``/chat/completions``; reasoning deltas are ignored."""
|
|
577
|
+
messages: list[dict] = []
|
|
578
|
+
if system:
|
|
579
|
+
messages.append({"role": "system", "content": system})
|
|
580
|
+
messages.append({"role": "user", "content": prompt})
|
|
581
|
+
payload: dict = {"model": self.name, "messages": messages, "stream": True,
|
|
582
|
+
**self._model_options}
|
|
583
|
+
if stop:
|
|
584
|
+
payload["stop"] = list(stop)
|
|
585
|
+
if max_tokens is not None:
|
|
586
|
+
payload["max_tokens"] = int(max_tokens)
|
|
587
|
+
resp = self._request(payload, stream=True)
|
|
588
|
+
with resp:
|
|
589
|
+
for raw in resp:
|
|
590
|
+
line = raw.decode("utf-8").strip()
|
|
591
|
+
if not line or not line.startswith("data:"):
|
|
592
|
+
continue
|
|
593
|
+
data = line[len("data:"):].strip()
|
|
594
|
+
if data == "[DONE]":
|
|
595
|
+
break
|
|
596
|
+
try:
|
|
597
|
+
obj = json.loads(data)
|
|
598
|
+
except json.JSONDecodeError:
|
|
599
|
+
continue
|
|
600
|
+
delta = (obj.get("choices") or [{}])[0].get("delta") or {}
|
|
601
|
+
chunk = delta.get("content") # delta.reasoning intentionally ignored
|
|
602
|
+
if chunk:
|
|
603
|
+
yield chunk
|
|
604
|
+
|
|
605
|
+
# -- non-streaming tool-calling chat (eval harness) ----------------------
|
|
606
|
+
def chat(self, messages: list[dict], tools: Optional[list[dict]] = None,
|
|
607
|
+
*, max_tokens: Optional[int] = None) -> dict:
|
|
608
|
+
"""Return ``{content, tool_calls, usage}`` for an OpenAI-style chat call."""
|
|
609
|
+
payload: dict = {"model": self.name, "messages": messages, "stream": False,
|
|
610
|
+
# Ask OpenRouter to report the real charged cost + cached-token detail
|
|
611
|
+
# in ``usage`` (ignored by APIs that don't support the field).
|
|
612
|
+
"usage": {"include": True},
|
|
613
|
+
**self._model_options}
|
|
614
|
+
if tools:
|
|
615
|
+
payload["tools"] = tools
|
|
616
|
+
if max_tokens is not None:
|
|
617
|
+
payload["max_tokens"] = int(max_tokens)
|
|
618
|
+
resp = self._request(payload, stream=False)
|
|
619
|
+
with resp:
|
|
620
|
+
obj = json.loads(resp.read().decode("utf-8"))
|
|
621
|
+
msg = (obj.get("choices") or [{}])[0].get("message") or {}
|
|
622
|
+
return {
|
|
623
|
+
"content": msg.get("content"),
|
|
624
|
+
"tool_calls": msg.get("tool_calls") or [],
|
|
625
|
+
"usage": obj.get("usage") or {},
|
|
626
|
+
}
|
|
627
|
+
|
|
628
|
+
def count_tokens(self, text: str) -> int:
|
|
629
|
+
"""Estimate token count (chars/4) — these APIs expose no count endpoint."""
|
|
630
|
+
return estimate(text)
|
|
631
|
+
|
|
632
|
+
|
|
633
|
+
# ---------------------------------------------------------------------------
|
|
634
|
+
# LlamaCppLLM — guarded import of llama_cpp.
|
|
635
|
+
# ---------------------------------------------------------------------------
|
|
636
|
+
class LlamaCppLLM:
|
|
637
|
+
"""llama.cpp adapter (``llama-cpp-python``). Import-guarded.
|
|
638
|
+
|
|
639
|
+
Requires the ``[llamacpp]`` extra. Constructing it without the dependency raises a
|
|
640
|
+
typed :class:`~aether_context.errors.BackendUnavailable` with the install hint.
|
|
641
|
+
"""
|
|
642
|
+
|
|
643
|
+
def __init__(
|
|
644
|
+
self,
|
|
645
|
+
model_path: str,
|
|
646
|
+
*,
|
|
647
|
+
context_window: Optional[int] = None,
|
|
648
|
+
model_options: Optional[dict] = None,
|
|
649
|
+
) -> None:
|
|
650
|
+
try:
|
|
651
|
+
from llama_cpp import Llama # type: ignore[import-not-found]
|
|
652
|
+
except ImportError as exc:
|
|
653
|
+
raise BackendUnavailable(
|
|
654
|
+
f"The llama.cpp backend needs 'llama-cpp-python': {exc}",
|
|
655
|
+
hint="Install it: pip install \"aether-context[llamacpp]\"",
|
|
656
|
+
) from exc
|
|
657
|
+
|
|
658
|
+
opts = dict(model_options or {})
|
|
659
|
+
n_ctx = opts.pop("n_ctx", context_window or 0) # 0 -> llama.cpp auto-detects
|
|
660
|
+
try:
|
|
661
|
+
self._llama = Llama(model_path=model_path, n_ctx=n_ctx, **opts)
|
|
662
|
+
except (OSError, ValueError) as exc:
|
|
663
|
+
raise BackendUnavailable(
|
|
664
|
+
f"Could not load gguf at '{model_path}': {exc}",
|
|
665
|
+
hint="Check the .gguf path and that model_options are valid for llama.cpp.",
|
|
666
|
+
) from exc
|
|
667
|
+
|
|
668
|
+
self.name: str = model_path
|
|
669
|
+
detected = getattr(self._llama, "n_ctx", None)
|
|
670
|
+
try:
|
|
671
|
+
ctx = int(detected()) if callable(detected) else int(detected or 0)
|
|
672
|
+
except (TypeError, ValueError):
|
|
673
|
+
ctx = 0
|
|
674
|
+
self.context_window: int = context_window or ctx or DEFAULT_CONTEXT_WINDOW
|
|
675
|
+
|
|
676
|
+
def generate(
|
|
677
|
+
self,
|
|
678
|
+
prompt: str,
|
|
679
|
+
*,
|
|
680
|
+
system: Optional[str] = None,
|
|
681
|
+
stop: Optional[list[str]] = None,
|
|
682
|
+
max_tokens: Optional[int] = None,
|
|
683
|
+
) -> Iterator[str]:
|
|
684
|
+
"""Stream chunks via ``create_chat_completion(stream=True)``."""
|
|
685
|
+
messages: list[dict] = []
|
|
686
|
+
if system:
|
|
687
|
+
messages.append({"role": "system", "content": system})
|
|
688
|
+
messages.append({"role": "user", "content": prompt})
|
|
689
|
+
kwargs: dict = {"messages": messages, "stream": True}
|
|
690
|
+
if stop:
|
|
691
|
+
kwargs["stop"] = list(stop)
|
|
692
|
+
if max_tokens is not None:
|
|
693
|
+
kwargs["max_tokens"] = int(max_tokens)
|
|
694
|
+
try:
|
|
695
|
+
for part in self._llama.create_chat_completion(**kwargs):
|
|
696
|
+
delta = (part.get("choices") or [{}])[0].get("delta") or {}
|
|
697
|
+
chunk = delta.get("content")
|
|
698
|
+
if chunk:
|
|
699
|
+
yield chunk
|
|
700
|
+
except (RuntimeError, ValueError, KeyError) as exc:
|
|
701
|
+
raise BackendUnavailable(
|
|
702
|
+
f"llama.cpp generation failed: {exc}",
|
|
703
|
+
hint="Verify the model and n_ctx; reduce max_tokens if out of memory.",
|
|
704
|
+
) from exc
|
|
705
|
+
|
|
706
|
+
def count_tokens(self, text: str) -> int:
|
|
707
|
+
"""Token count via llama.cpp's real tokenizer; estimate on failure."""
|
|
708
|
+
try:
|
|
709
|
+
return len(self._llama.tokenize(text.encode("utf-8")))
|
|
710
|
+
except (RuntimeError, ValueError, AttributeError) as exc:
|
|
711
|
+
_log.debug("llama.cpp tokenize failed, using estimate: %s", exc)
|
|
712
|
+
return estimate(text)
|
|
713
|
+
|
|
714
|
+
|
|
715
|
+
# ---------------------------------------------------------------------------
|
|
716
|
+
# HFLLM — guarded import of transformers.
|
|
717
|
+
# ---------------------------------------------------------------------------
|
|
718
|
+
class HFLLM:
|
|
719
|
+
"""Hugging Face transformers adapter. Import-guarded.
|
|
720
|
+
|
|
721
|
+
Requires the ``[hf]`` extra (transformers + torch). Uses a ``TextIteratorStreamer`` so
|
|
722
|
+
generation still streams. ``context_window`` comes from the model config; token counts
|
|
723
|
+
use the model's own tokenizer.
|
|
724
|
+
"""
|
|
725
|
+
|
|
726
|
+
def __init__(
|
|
727
|
+
self,
|
|
728
|
+
model_ref: str,
|
|
729
|
+
*,
|
|
730
|
+
context_window: Optional[int] = None,
|
|
731
|
+
model_options: Optional[dict] = None,
|
|
732
|
+
) -> None:
|
|
733
|
+
try:
|
|
734
|
+
from transformers import ( # type: ignore[import-not-found]
|
|
735
|
+
AutoModelForCausalLM,
|
|
736
|
+
AutoTokenizer,
|
|
737
|
+
)
|
|
738
|
+
except ImportError as exc:
|
|
739
|
+
raise BackendUnavailable(
|
|
740
|
+
f"The HF backend needs 'transformers' (and torch): {exc}",
|
|
741
|
+
hint="Install it: pip install \"aether-context[hf]\"",
|
|
742
|
+
) from exc
|
|
743
|
+
|
|
744
|
+
opts = dict(model_options or {})
|
|
745
|
+
opts.setdefault("device_map", "auto")
|
|
746
|
+
try:
|
|
747
|
+
self._tokenizer = AutoTokenizer.from_pretrained(model_ref)
|
|
748
|
+
self._model = AutoModelForCausalLM.from_pretrained(model_ref, **opts)
|
|
749
|
+
except (OSError, ValueError) as exc:
|
|
750
|
+
raise BackendUnavailable(
|
|
751
|
+
f"Could not load HF model '{model_ref}': {exc}",
|
|
752
|
+
hint="Check the org/model id and your network/cache for the download.",
|
|
753
|
+
) from exc
|
|
754
|
+
|
|
755
|
+
self.name: str = model_ref
|
|
756
|
+
cfg = getattr(self._model, "config", None)
|
|
757
|
+
cfg_ctx = getattr(cfg, "max_position_embeddings", None)
|
|
758
|
+
self.context_window: int = (
|
|
759
|
+
context_window
|
|
760
|
+
or (cfg_ctx if isinstance(cfg_ctx, int) and cfg_ctx > 0 else DEFAULT_CONTEXT_WINDOW)
|
|
761
|
+
)
|
|
762
|
+
|
|
763
|
+
def generate(
|
|
764
|
+
self,
|
|
765
|
+
prompt: str,
|
|
766
|
+
*,
|
|
767
|
+
system: Optional[str] = None,
|
|
768
|
+
stop: Optional[list[str]] = None,
|
|
769
|
+
max_tokens: Optional[int] = None,
|
|
770
|
+
) -> Iterator[str]:
|
|
771
|
+
"""Stream chunks via a background generate + ``TextIteratorStreamer``."""
|
|
772
|
+
import threading
|
|
773
|
+
|
|
774
|
+
from transformers import TextIteratorStreamer # type: ignore[import-not-found]
|
|
775
|
+
|
|
776
|
+
messages: list[dict] = []
|
|
777
|
+
if system:
|
|
778
|
+
messages.append({"role": "system", "content": system})
|
|
779
|
+
messages.append({"role": "user", "content": prompt})
|
|
780
|
+
try:
|
|
781
|
+
inputs = self._tokenizer.apply_chat_template(
|
|
782
|
+
messages, add_generation_prompt=True, return_tensors="pt"
|
|
783
|
+
).to(self._model.device)
|
|
784
|
+
except (ValueError, AttributeError) as exc:
|
|
785
|
+
raise BackendUnavailable(
|
|
786
|
+
f"HF chat templating failed for '{self.name}': {exc}",
|
|
787
|
+
hint="The model may lack a chat template; try a chat/instruct variant.",
|
|
788
|
+
) from exc
|
|
789
|
+
|
|
790
|
+
streamer = TextIteratorStreamer(
|
|
791
|
+
self._tokenizer, skip_prompt=True, skip_special_tokens=True
|
|
792
|
+
)
|
|
793
|
+
gen_kwargs: dict = {
|
|
794
|
+
"input_ids": inputs,
|
|
795
|
+
"streamer": streamer,
|
|
796
|
+
"max_new_tokens": int(max_tokens) if max_tokens is not None else 512,
|
|
797
|
+
}
|
|
798
|
+
thread = threading.Thread(target=self._model.generate, kwargs=gen_kwargs)
|
|
799
|
+
thread.start()
|
|
800
|
+
emitted = ""
|
|
801
|
+
for chunk in streamer:
|
|
802
|
+
if not chunk:
|
|
803
|
+
continue
|
|
804
|
+
if stop:
|
|
805
|
+
emitted += chunk
|
|
806
|
+
cut = self._first_stop(emitted, stop)
|
|
807
|
+
if cut is not None:
|
|
808
|
+
remainder = emitted[:cut][len(emitted) - len(chunk):]
|
|
809
|
+
if remainder:
|
|
810
|
+
yield remainder
|
|
811
|
+
break
|
|
812
|
+
yield chunk
|
|
813
|
+
thread.join()
|
|
814
|
+
|
|
815
|
+
@staticmethod
|
|
816
|
+
def _first_stop(text: str, stop: list[str]) -> Optional[int]:
|
|
817
|
+
cut: Optional[int] = None
|
|
818
|
+
for s in stop:
|
|
819
|
+
if s:
|
|
820
|
+
idx = text.find(s)
|
|
821
|
+
if idx != -1:
|
|
822
|
+
cut = idx if cut is None else min(cut, idx)
|
|
823
|
+
return cut
|
|
824
|
+
|
|
825
|
+
def count_tokens(self, text: str) -> int:
|
|
826
|
+
"""Token count via the model's tokenizer; estimate on failure."""
|
|
827
|
+
try:
|
|
828
|
+
return len(self._tokenizer.encode(text))
|
|
829
|
+
except (RuntimeError, ValueError, AttributeError) as exc:
|
|
830
|
+
_log.debug("HF tokenizer encode failed, using estimate: %s", exc)
|
|
831
|
+
return estimate(text)
|
|
832
|
+
|
|
833
|
+
|
|
834
|
+
__all__ = [
|
|
835
|
+
"LocalLLM",
|
|
836
|
+
"ModelSpec",
|
|
837
|
+
"parse_spec",
|
|
838
|
+
"load_model",
|
|
839
|
+
"MockLLM",
|
|
840
|
+
"OllamaLLM",
|
|
841
|
+
"OpenAICompatLLM",
|
|
842
|
+
"LlamaCppLLM",
|
|
843
|
+
"HFLLM",
|
|
844
|
+
"DEFAULT_CONTEXT_WINDOW",
|
|
845
|
+
"DEFAULT_OLLAMA_HOST",
|
|
846
|
+
]
|