aether-context 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,846 @@
1
+ # aether-context (Unlimited Context)
2
+ # Copyright (c) 2026 Aether AI
3
+ # SPDX-License-Identifier: Apache-2.0
4
+ """THE WRAPPER — backend-agnostic local-LLM adapters.
5
+
6
+ This is the surface a user actually touches. The whole engine talks to *one* protocol
7
+ (:class:`LocalLLM`); every backend (Ollama, llama.cpp, HF, Mock) satisfies it so the
8
+ engine never special-cases a backend.
9
+
10
+ Design laws honored here:
11
+ * **Simple** — one spec string picks a backend (``"ollama/qwen2.5"``, ``"mock"``…).
12
+ * **Light core** — the primary Ollama path uses the Python standard library only
13
+ (``urllib``); no extra dependency for ``pip install aether-context``.
14
+ * **Forgiving** — daemon down / model missing raise *typed* errors from
15
+ :mod:`aether_context.errors`, each with an actionable ``.hint``. Metadata probes are
16
+ fail-soft (a failed ``/api/show`` degrades ``context_window`` to a fallback, never
17
+ raises into a run).
18
+ * **Streaming** — ``generate`` yields text chunks so the pager can prefetch the next
19
+ slices *while* the model is still talking.
20
+
21
+ No ``print()`` (use the logging seam), no bare ``except`` (catch specific, re-wrap as a
22
+ typed error).
23
+ """
24
+ from __future__ import annotations
25
+
26
+ import hashlib
27
+ import json
28
+ import os
29
+ import urllib.error
30
+ import urllib.request
31
+ from dataclasses import dataclass, field
32
+ from typing import Iterator, Optional, Protocol, runtime_checkable
33
+
34
+ from aether_context._log import get_logger
35
+ from aether_context.errors import (
36
+ AetherContextError,
37
+ BackendUnavailable,
38
+ ModelNotPulled,
39
+ OllamaNotRunning,
40
+ )
41
+ from aether_context.tokenizer import estimate
42
+
43
+ _log = get_logger(__name__)
44
+
45
+ #: Fallback context window (tokens) when a backend cannot report its own.
46
+ DEFAULT_CONTEXT_WINDOW = 8192
47
+ #: Default Ollama daemon host.
48
+ DEFAULT_OLLAMA_HOST = "http://localhost:11434"
49
+ #: Recognised backends in a spec string.
50
+ _BACKENDS = ("ollama", "llamacpp", "hf", "mock", "openai")
51
+ #: HTTP timeout (seconds) for Ollama requests. Generation can be slow → generous.
52
+ _OLLAMA_TIMEOUT = 600
53
+ #: Best-effort metadata probe timeout (seconds) — short; failure is non-fatal.
54
+ _OLLAMA_PROBE_TIMEOUT = 5
55
+
56
+
57
+ # ---------------------------------------------------------------------------
58
+ # The protocol every adapter satisfies.
59
+ # ---------------------------------------------------------------------------
60
+ @runtime_checkable
61
+ class LocalLLM(Protocol):
62
+ """Backend-agnostic local-model contract.
63
+
64
+ ``generate`` **streams** text chunks (yield once with the full text if a backend
65
+ cannot stream — it still works, you just lose the free prefetch concurrency).
66
+ """
67
+
68
+ name: str
69
+
70
+ @property
71
+ def context_window(self) -> int:
72
+ """Token window the model exposes this turn (read-only; may be lazily probed)."""
73
+ ...
74
+
75
+ def generate(
76
+ self,
77
+ prompt: str,
78
+ *,
79
+ system: Optional[str] = None,
80
+ stop: Optional[list[str]] = None,
81
+ max_tokens: Optional[int] = None,
82
+ ) -> Iterator[str]:
83
+ """Stream model output for ``prompt`` as a sequence of text chunks."""
84
+ ...
85
+
86
+ def count_tokens(self, text: str) -> int:
87
+ """Return the token count of ``text`` for budget math."""
88
+ ...
89
+
90
+
91
+ # ---------------------------------------------------------------------------
92
+ # Spec parsing.
93
+ # ---------------------------------------------------------------------------
94
+ @dataclass(frozen=True)
95
+ class ModelSpec:
96
+ """Parsed model spec: a backend, a model ref, and pass-through options."""
97
+
98
+ backend: str
99
+ ref: str
100
+ options: dict = field(default_factory=dict)
101
+
102
+
103
+ _SPEC_HINT = (
104
+ "Use one of: 'ollama/qwen2.5' (or a bare name -> ollama), "
105
+ "'llamacpp:/path/to/model.gguf', 'hf/org/model', or 'mock'."
106
+ )
107
+
108
+
109
+ def parse_spec(spec: str) -> ModelSpec:
110
+ """Parse a spec string into a :class:`ModelSpec`.
111
+
112
+ Grammar (one obvious format)::
113
+
114
+ ollama/<name> -> Ollama (tags ok: ollama/llama3.1:8b)
115
+ <name> -> bare name assumed Ollama
116
+ llamacpp:<path> -> llama.cpp over a .gguf (first ':' splits; drive ':' kept)
117
+ hf/<org>/<model> -> HF transformers
118
+ mock -> built-in deterministic model
119
+
120
+ Raises :class:`~aether_context.errors.BackendUnavailable` (carrying a ``.hint`` with
121
+ the format) for anything unparseable or an unknown backend.
122
+ """
123
+ if not isinstance(spec, str):
124
+ raise BackendUnavailable(
125
+ f"Model spec must be a string (backend/ref). Got: {type(spec).__name__}",
126
+ hint=_SPEC_HINT,
127
+ )
128
+ text = spec.strip()
129
+ if not text:
130
+ raise BackendUnavailable(
131
+ "Model spec is empty. Must be backend/ref "
132
+ "(e.g. ollama/qwen2.5, llamacpp:/path/model.gguf, hf/org/model, mock).",
133
+ hint=_SPEC_HINT,
134
+ )
135
+
136
+ # mock — exact bare token.
137
+ if text == "mock":
138
+ return ModelSpec(backend="mock", ref="mock")
139
+
140
+ # llamacpp:<path> — split on the FIRST colon only so a Windows drive ':' survives.
141
+ if text.startswith("llamacpp:"):
142
+ ref = text[len("llamacpp:"):]
143
+ if not ref:
144
+ raise BackendUnavailable(
145
+ "llamacpp spec needs a path: 'llamacpp:/path/to/model.gguf'",
146
+ hint=_SPEC_HINT,
147
+ )
148
+ return ModelSpec(backend="llamacpp", ref=ref)
149
+
150
+ # slash-prefixed backends: ollama/<ref>, hf/<org/model>.
151
+ if "/" in text:
152
+ head, ref = text.split("/", 1)
153
+ if head == "ollama":
154
+ return ModelSpec(backend="ollama", ref=ref)
155
+ if head == "hf":
156
+ return ModelSpec(backend="hf", ref=ref)
157
+ if head == "openai":
158
+ return ModelSpec(backend="openai", ref=ref)
159
+ if head in _BACKENDS:
160
+ return ModelSpec(backend=head, ref=ref)
161
+ raise BackendUnavailable(
162
+ f"Unknown backend '{head}' in spec '{spec}'. "
163
+ "Must be backend/ref (e.g. ollama/qwen2.5, llamacpp:/path/model.gguf, "
164
+ "hf/org/model, mock).",
165
+ hint=_SPEC_HINT,
166
+ )
167
+
168
+ # Bare name -> assumed Ollama (documented convenience; see docs/local-models.md).
169
+ return ModelSpec(backend="ollama", ref=text)
170
+
171
+
172
+ # ---------------------------------------------------------------------------
173
+ # load_model — the one public entry point.
174
+ # ---------------------------------------------------------------------------
175
+ def load_model(spec: "str | LocalLLM", **kw: object) -> LocalLLM:
176
+ """Resolve ``spec`` to a :class:`LocalLLM`.
177
+
178
+ * A :class:`LocalLLM` object is returned unchanged (bring-your-own backend).
179
+ * A spec string is parsed and dispatched to the matching adapter; extra ``**kw`` are
180
+ forwarded to the adapter constructor.
181
+
182
+ Raises :class:`~aether_context.errors.BackendUnavailable` for an unknown backend or a
183
+ backend whose optional dependency is not installed.
184
+ """
185
+ # Already a backend object? Pass it straight through (duck-typed protocol).
186
+ if not isinstance(spec, str) and isinstance(spec, LocalLLM):
187
+ return spec
188
+
189
+ if not isinstance(spec, str):
190
+ raise BackendUnavailable(
191
+ f"model must be a spec string or a LocalLLM object. Got: {type(spec).__name__}",
192
+ hint=_SPEC_HINT,
193
+ )
194
+
195
+ parsed = parse_spec(spec)
196
+ backend = parsed.backend
197
+ if backend == "mock":
198
+ return MockLLM(name=parsed.ref, **kw) # type: ignore[arg-type]
199
+ if backend == "ollama":
200
+ return OllamaLLM(parsed.ref, **kw) # type: ignore[arg-type]
201
+ if backend == "llamacpp":
202
+ return LlamaCppLLM(parsed.ref, **kw) # type: ignore[arg-type]
203
+ if backend == "hf":
204
+ return HFLLM(parsed.ref, **kw) # type: ignore[arg-type]
205
+ if backend == "openai":
206
+ return OpenAICompatLLM(parsed.ref, **kw) # type: ignore[arg-type]
207
+ raise BackendUnavailable(
208
+ f"Unknown backend '{backend}'.", hint=_SPEC_HINT
209
+ )
210
+
211
+
212
+ # ---------------------------------------------------------------------------
213
+ # MockLLM — deterministic, dependency-free.
214
+ # ---------------------------------------------------------------------------
215
+ #: Vocabulary the mock draws from (deterministically) to build pseudo-output.
216
+ _MOCK_WORDS = (
217
+ "plan step build module function class test verify refactor encode slice pool "
218
+ "window context retrieve prefetch witness fade harden session token vector index "
219
+ "the and then so we now next done note check ok yes done finally indeed thus"
220
+ ).split()
221
+ #: Mock chunk width in characters (>1 chunk for any non-trivial output → streaming).
222
+ _MOCK_CHUNK_CHARS = 32
223
+
224
+
225
+ @dataclass
226
+ class MockLLM:
227
+ """Deterministic, offline, zero-dependency model.
228
+
229
+ Output is derived from a hash of ``(prompt, system)`` so the same input always
230
+ produces the same text — letting tests assert properties and the bench run a hermetic
231
+ baseline. ``output_tokens`` controls *length* independently of ``context_window`` so a
232
+ test/bench can force overflow with a tiny window and a long generation.
233
+ """
234
+
235
+ name: str = "mock"
236
+ context_window: int = DEFAULT_CONTEXT_WINDOW
237
+ output_tokens: int = 64
238
+
239
+ def _build_text(self, prompt: str, system: Optional[str]) -> str:
240
+ seed_src = f"{system or ''}\x00{prompt}".encode("utf-8")
241
+ digest = hashlib.sha256(seed_src).digest()
242
+ # Expand the 32-byte digest deterministically to as many words as we need.
243
+ words: list[str] = []
244
+ i = 0
245
+ # ~6 chars per word incl. space; target enough words for output_tokens (chars/4).
246
+ target_words = max(1, (self.output_tokens * 4) // 6)
247
+ while len(words) < target_words:
248
+ # rehash with a counter so we never run out of entropy
249
+ block = hashlib.sha256(digest + i.to_bytes(4, "big")).digest()
250
+ for b in block:
251
+ words.append(_MOCK_WORDS[b % len(_MOCK_WORDS)])
252
+ if len(words) >= target_words:
253
+ break
254
+ i += 1
255
+ return " ".join(words)
256
+
257
+ @property
258
+ def is_streaming(self) -> bool:
259
+ """True iff a representative ``generate`` yields more than one chunk."""
260
+ return len(self._build_text("probe", None)) > _MOCK_CHUNK_CHARS
261
+
262
+ def generate(
263
+ self,
264
+ prompt: str,
265
+ *,
266
+ system: Optional[str] = None,
267
+ stop: Optional[list[str]] = None,
268
+ max_tokens: Optional[int] = None,
269
+ ) -> Iterator[str]:
270
+ """Yield deterministic pseudo-output in fixed-width chunks (streaming)."""
271
+ if not isinstance(prompt, str):
272
+ raise TypeError(f"prompt must be str, got {type(prompt).__name__}")
273
+ text = self._build_text(prompt, system)
274
+
275
+ # Apply max_tokens cap (chars/4 estimate → char budget).
276
+ if max_tokens is not None and max_tokens >= 0:
277
+ char_budget = max_tokens * 4
278
+ text = text[:char_budget]
279
+
280
+ # Apply stop sequences: truncate at the earliest occurrence of any stop string.
281
+ if stop:
282
+ cut = len(text)
283
+ for s in stop:
284
+ if s:
285
+ idx = text.find(s)
286
+ if idx != -1:
287
+ cut = min(cut, idx)
288
+ text = text[:cut]
289
+
290
+ for start in range(0, len(text), _MOCK_CHUNK_CHARS):
291
+ yield text[start:start + _MOCK_CHUNK_CHARS]
292
+
293
+ def count_tokens(self, text: str) -> int:
294
+ """Estimate token count (chars/4) — the backend-agnostic budget rule."""
295
+ return estimate(text)
296
+
297
+
298
+ # ---------------------------------------------------------------------------
299
+ # OllamaLLM — primary path, stdlib urllib only.
300
+ # ---------------------------------------------------------------------------
301
+ class OllamaLLM:
302
+ """Ollama adapter over HTTP using only the standard library (``urllib``).
303
+
304
+ Talks to ``<host>/api/chat`` with ``stream=true``. ``context_window`` is auto-detected
305
+ from ``<host>/api/show`` metadata (``model_info.*context_length`` / ``num_ctx``),
306
+ falling back to :data:`DEFAULT_CONTEXT_WINDOW` if the probe fails (fail-soft). Token
307
+ counts use the chars/4 estimate (Ollama exposes no count endpoint).
308
+ """
309
+
310
+ def __init__(
311
+ self,
312
+ ref: str,
313
+ *,
314
+ host: str = DEFAULT_OLLAMA_HOST,
315
+ context_window: Optional[int] = None,
316
+ pull: bool = False,
317
+ model_options: Optional[dict] = None,
318
+ ) -> None:
319
+ self.name: str = ref
320
+ self.host: str = host.rstrip("/")
321
+ self._pull: bool = pull
322
+ self._model_options: dict = dict(model_options or {})
323
+ # Lazily resolved; an explicit value wins, else probe (fail-soft), else fallback.
324
+ self._context_window: Optional[int] = context_window
325
+
326
+ # -- metadata (fail-soft) ------------------------------------------------
327
+ @property
328
+ def context_window(self) -> int:
329
+ """Token window, auto-detected via ``/api/show`` (fallback on any failure)."""
330
+ if self._context_window is None:
331
+ self._context_window = self._probe_context_window()
332
+ return self._context_window
333
+
334
+ def _probe_context_window(self) -> int:
335
+ """Best-effort ``/api/show`` probe. Never raises — degrades to the fallback."""
336
+ try:
337
+ payload = json.dumps({"model": self.name}).encode("utf-8")
338
+ req = urllib.request.Request(
339
+ f"{self.host}/api/show",
340
+ data=payload,
341
+ headers={"Content-Type": "application/json"},
342
+ method="POST",
343
+ )
344
+ with urllib.request.urlopen(req, timeout=_OLLAMA_PROBE_TIMEOUT) as resp:
345
+ body = json.loads(resp.read().decode("utf-8"))
346
+ return self._extract_context_length(body)
347
+ except (urllib.error.URLError, OSError, ValueError, KeyError, TypeError) as exc:
348
+ _log.debug("ollama /api/show probe failed, using fallback window: %s", exc)
349
+ return DEFAULT_CONTEXT_WINDOW
350
+
351
+ @staticmethod
352
+ def _extract_context_length(body: dict) -> int:
353
+ """Pull a context length out of /api/show metadata; fallback if absent."""
354
+ info = body.get("model_info") or {}
355
+ for key, value in info.items():
356
+ if key.endswith("context_length") and isinstance(value, int) and value > 0:
357
+ return value
358
+ params = body.get("parameters")
359
+ if isinstance(params, str):
360
+ for line in params.splitlines():
361
+ parts = line.split()
362
+ if len(parts) == 2 and parts[0] == "num_ctx" and parts[1].isdigit():
363
+ return int(parts[1])
364
+ return DEFAULT_CONTEXT_WINDOW
365
+
366
+ # -- generation ----------------------------------------------------------
367
+ def generate(
368
+ self,
369
+ prompt: str,
370
+ *,
371
+ system: Optional[str] = None,
372
+ stop: Optional[list[str]] = None,
373
+ max_tokens: Optional[int] = None,
374
+ ) -> Iterator[str]:
375
+ """Stream chunks from ``/api/chat``. Typed, forgiving errors on failure."""
376
+ if self._pull:
377
+ self._ensure_pulled()
378
+ messages: list[dict] = []
379
+ if system:
380
+ messages.append({"role": "system", "content": system})
381
+ messages.append({"role": "user", "content": prompt})
382
+
383
+ options = dict(self._model_options)
384
+ if stop:
385
+ options["stop"] = list(stop)
386
+ if max_tokens is not None:
387
+ options["num_predict"] = int(max_tokens)
388
+
389
+ body: dict = {"model": self.name, "messages": messages, "stream": True}
390
+ if options:
391
+ body["options"] = options
392
+
393
+ yield from self._stream_chat(body)
394
+
395
+ def _stream_chat(self, body: dict) -> Iterator[str]:
396
+ payload = json.dumps(body).encode("utf-8")
397
+ req = urllib.request.Request(
398
+ f"{self.host}/api/chat",
399
+ data=payload,
400
+ headers={"Content-Type": "application/json"},
401
+ method="POST",
402
+ )
403
+ try:
404
+ resp = urllib.request.urlopen(req, timeout=_OLLAMA_TIMEOUT)
405
+ except urllib.error.HTTPError as exc:
406
+ raise self._http_error_to_typed(exc) from exc
407
+ except (urllib.error.URLError, OSError) as exc:
408
+ raise OllamaNotRunning(
409
+ f"Could not reach the Ollama daemon at {self.host}: {exc}"
410
+ ) from exc
411
+
412
+ with resp:
413
+ for raw in resp:
414
+ line = raw.decode("utf-8").strip()
415
+ if not line:
416
+ continue
417
+ try:
418
+ obj = json.loads(line)
419
+ except json.JSONDecodeError as exc:
420
+ _log.debug("skipping non-JSON ollama stream line: %s", exc)
421
+ continue
422
+ if obj.get("error"):
423
+ raise ModelNotPulled(
424
+ f"Ollama error for model '{self.name}': {obj['error']}"
425
+ )
426
+ chunk = (obj.get("message") or {}).get("content")
427
+ if chunk:
428
+ yield chunk
429
+ if obj.get("done"):
430
+ break
431
+
432
+ def _http_error_to_typed(self, exc: urllib.error.HTTPError) -> AetherContextError:
433
+ """Map an Ollama HTTP error to a typed, hinted aether-context error."""
434
+ detail = ""
435
+ try:
436
+ detail = exc.read().decode("utf-8", errors="replace")
437
+ except OSError:
438
+ detail = ""
439
+ if exc.code == 404 or "not found" in detail.lower():
440
+ return ModelNotPulled(
441
+ f"Ollama model '{self.name}' is not pulled: {detail or exc}",
442
+ hint=f"Pull it first: `ollama pull {self.name}` (or pass pull=True).",
443
+ )
444
+ return OllamaNotRunning(
445
+ f"Ollama returned HTTP {exc.code} from {self.host}: {detail or exc}"
446
+ )
447
+
448
+ def _ensure_pulled(self) -> None:
449
+ """Best-effort ``/api/pull`` when ``pull=True``. Typed error on failure."""
450
+ payload = json.dumps({"model": self.name, "stream": False}).encode("utf-8")
451
+ req = urllib.request.Request(
452
+ f"{self.host}/api/pull",
453
+ data=payload,
454
+ headers={"Content-Type": "application/json"},
455
+ method="POST",
456
+ )
457
+ try:
458
+ with urllib.request.urlopen(req, timeout=_OLLAMA_TIMEOUT) as resp:
459
+ resp.read()
460
+ except urllib.error.HTTPError as exc:
461
+ raise self._http_error_to_typed(exc) from exc
462
+ except (urllib.error.URLError, OSError) as exc:
463
+ raise OllamaNotRunning(
464
+ f"Could not reach the Ollama daemon at {self.host} to pull "
465
+ f"'{self.name}': {exc}"
466
+ ) from exc
467
+
468
+ # -- token counting ------------------------------------------------------
469
+ def count_tokens(self, text: str) -> int:
470
+ """Estimate token count (chars/4) — Ollama has no count endpoint."""
471
+ return estimate(text)
472
+
473
+
474
+ # ---------------------------------------------------------------------------
475
+ # OpenAICompatLLM — vendor-neutral OpenAI-compatible API (OpenRouter / OpenAI / vLLM).
476
+ # ---------------------------------------------------------------------------
477
+ #: OpenRouter default base URL when only OPENROUTER_API_KEY is set.
478
+ _OPENROUTER_BASE = "https://openrouter.ai/api/v1"
479
+ #: HTTP timeout (s) for the OpenAI-compatible adapter. Kept tight: a coding turn that
480
+ #: hasn't responded in 3 min is a hung connection — fail fast so the batch can continue
481
+ #: (a 600s hang once stalled a whole sequential run ~30 min). Override via OPENAI_HTTP_TIMEOUT.
482
+ _OPENAI_TIMEOUT = int(os.environ.get("OPENAI_HTTP_TIMEOUT", "180"))
483
+
484
+
485
+ class OpenAICompatLLM:
486
+ """Vendor-neutral OpenAI-compatible adapter (OpenRouter / OpenAI / vLLM / …).
487
+
488
+ Implements the :class:`LocalLLM` protocol (streaming ``generate``) and adds
489
+ ``chat(messages, tools)`` (non-streaming, function-calling) for the eval harness's agent
490
+ loop. stdlib ``urllib`` only — no extra dependency. ``context_window`` is caller-provided
491
+ (these APIs do not report it); ``count_tokens`` is the chars/4 estimate.
492
+
493
+ Config resolution: explicit ``base_url``/``api_key`` win; else ``OPENROUTER_API_KEY``
494
+ (with the OpenRouter base URL); else ``OPENAI_BASE_URL``/``OPENAI_API_KEY``. The key is
495
+ never logged or placed in an error message.
496
+ """
497
+
498
+ def __init__(
499
+ self,
500
+ ref: str,
501
+ *,
502
+ base_url: Optional[str] = None,
503
+ api_key: Optional[str] = None,
504
+ context_window: Optional[int] = None,
505
+ model_options: Optional[dict] = None,
506
+ ) -> None:
507
+ self.name: str = ref
508
+ key = api_key or os.environ.get("OPENROUTER_API_KEY") or os.environ.get("OPENAI_API_KEY")
509
+ if base_url is None:
510
+ if os.environ.get("OPENROUTER_API_KEY") and not api_key:
511
+ base_url = _OPENROUTER_BASE
512
+ else:
513
+ base_url = os.environ.get("OPENAI_BASE_URL")
514
+ if not key:
515
+ raise BackendUnavailable(
516
+ "No API key for the OpenAI-compatible backend.",
517
+ hint="Set OPENROUTER_API_KEY (or OPENAI_API_KEY), or pass api_key=...",
518
+ )
519
+ if not base_url:
520
+ raise BackendUnavailable(
521
+ "No base_url for the OpenAI-compatible backend.",
522
+ hint="Set OPENROUTER_API_KEY (uses openrouter.ai), OPENAI_BASE_URL, or pass base_url=...",
523
+ )
524
+ self.base_url: str = base_url.rstrip("/")
525
+ self.api_key: str = key
526
+ self._context_window: int = context_window or DEFAULT_CONTEXT_WINDOW
527
+ self._model_options: dict = dict(model_options or {})
528
+
529
+ @property
530
+ def context_window(self) -> int:
531
+ return self._context_window
532
+
533
+ # -- low-level HTTP (one place; mockable) --------------------------------
534
+ def _open(self, req: "urllib.request.Request"):
535
+ return urllib.request.urlopen(req, timeout=_OPENAI_TIMEOUT)
536
+
537
+ def _request(self, payload: dict, *, stream: bool):
538
+ body = json.dumps(payload).encode("utf-8")
539
+ req = urllib.request.Request(
540
+ f"{self.base_url}/chat/completions",
541
+ data=body,
542
+ headers={
543
+ "Content-Type": "application/json",
544
+ "Authorization": f"Bearer {self.api_key}",
545
+ "Accept": "text/event-stream" if stream else "application/json",
546
+ },
547
+ method="POST",
548
+ )
549
+ try:
550
+ return self._open(req)
551
+ except urllib.error.HTTPError as exc:
552
+ detail = ""
553
+ try:
554
+ detail = exc.read().decode("utf-8", errors="replace")[:300]
555
+ except OSError:
556
+ detail = ""
557
+ raise BackendUnavailable(
558
+ f"OpenAI-compatible API HTTP {exc.code} from {self.base_url}.",
559
+ hint=f"Check model/key/credits. Detail: {detail}",
560
+ ) from exc
561
+ except (urllib.error.URLError, OSError) as exc:
562
+ raise BackendUnavailable(
563
+ f"Could not reach the API at {self.base_url}: {exc}",
564
+ hint="Check the base_url and your network.",
565
+ ) from exc
566
+
567
+ # -- streaming generate (LocalLLM protocol) ------------------------------
568
+ def generate(
569
+ self,
570
+ prompt: str,
571
+ *,
572
+ system: Optional[str] = None,
573
+ stop: Optional[list[str]] = None,
574
+ max_tokens: Optional[int] = None,
575
+ ) -> Iterator[str]:
576
+ """Stream content chunks from ``/chat/completions``; reasoning deltas are ignored."""
577
+ messages: list[dict] = []
578
+ if system:
579
+ messages.append({"role": "system", "content": system})
580
+ messages.append({"role": "user", "content": prompt})
581
+ payload: dict = {"model": self.name, "messages": messages, "stream": True,
582
+ **self._model_options}
583
+ if stop:
584
+ payload["stop"] = list(stop)
585
+ if max_tokens is not None:
586
+ payload["max_tokens"] = int(max_tokens)
587
+ resp = self._request(payload, stream=True)
588
+ with resp:
589
+ for raw in resp:
590
+ line = raw.decode("utf-8").strip()
591
+ if not line or not line.startswith("data:"):
592
+ continue
593
+ data = line[len("data:"):].strip()
594
+ if data == "[DONE]":
595
+ break
596
+ try:
597
+ obj = json.loads(data)
598
+ except json.JSONDecodeError:
599
+ continue
600
+ delta = (obj.get("choices") or [{}])[0].get("delta") or {}
601
+ chunk = delta.get("content") # delta.reasoning intentionally ignored
602
+ if chunk:
603
+ yield chunk
604
+
605
+ # -- non-streaming tool-calling chat (eval harness) ----------------------
606
+ def chat(self, messages: list[dict], tools: Optional[list[dict]] = None,
607
+ *, max_tokens: Optional[int] = None) -> dict:
608
+ """Return ``{content, tool_calls, usage}`` for an OpenAI-style chat call."""
609
+ payload: dict = {"model": self.name, "messages": messages, "stream": False,
610
+ # Ask OpenRouter to report the real charged cost + cached-token detail
611
+ # in ``usage`` (ignored by APIs that don't support the field).
612
+ "usage": {"include": True},
613
+ **self._model_options}
614
+ if tools:
615
+ payload["tools"] = tools
616
+ if max_tokens is not None:
617
+ payload["max_tokens"] = int(max_tokens)
618
+ resp = self._request(payload, stream=False)
619
+ with resp:
620
+ obj = json.loads(resp.read().decode("utf-8"))
621
+ msg = (obj.get("choices") or [{}])[0].get("message") or {}
622
+ return {
623
+ "content": msg.get("content"),
624
+ "tool_calls": msg.get("tool_calls") or [],
625
+ "usage": obj.get("usage") or {},
626
+ }
627
+
628
+ def count_tokens(self, text: str) -> int:
629
+ """Estimate token count (chars/4) — these APIs expose no count endpoint."""
630
+ return estimate(text)
631
+
632
+
633
+ # ---------------------------------------------------------------------------
634
+ # LlamaCppLLM — guarded import of llama_cpp.
635
+ # ---------------------------------------------------------------------------
636
+ class LlamaCppLLM:
637
+ """llama.cpp adapter (``llama-cpp-python``). Import-guarded.
638
+
639
+ Requires the ``[llamacpp]`` extra. Constructing it without the dependency raises a
640
+ typed :class:`~aether_context.errors.BackendUnavailable` with the install hint.
641
+ """
642
+
643
+ def __init__(
644
+ self,
645
+ model_path: str,
646
+ *,
647
+ context_window: Optional[int] = None,
648
+ model_options: Optional[dict] = None,
649
+ ) -> None:
650
+ try:
651
+ from llama_cpp import Llama # type: ignore[import-not-found]
652
+ except ImportError as exc:
653
+ raise BackendUnavailable(
654
+ f"The llama.cpp backend needs 'llama-cpp-python': {exc}",
655
+ hint="Install it: pip install \"aether-context[llamacpp]\"",
656
+ ) from exc
657
+
658
+ opts = dict(model_options or {})
659
+ n_ctx = opts.pop("n_ctx", context_window or 0) # 0 -> llama.cpp auto-detects
660
+ try:
661
+ self._llama = Llama(model_path=model_path, n_ctx=n_ctx, **opts)
662
+ except (OSError, ValueError) as exc:
663
+ raise BackendUnavailable(
664
+ f"Could not load gguf at '{model_path}': {exc}",
665
+ hint="Check the .gguf path and that model_options are valid for llama.cpp.",
666
+ ) from exc
667
+
668
+ self.name: str = model_path
669
+ detected = getattr(self._llama, "n_ctx", None)
670
+ try:
671
+ ctx = int(detected()) if callable(detected) else int(detected or 0)
672
+ except (TypeError, ValueError):
673
+ ctx = 0
674
+ self.context_window: int = context_window or ctx or DEFAULT_CONTEXT_WINDOW
675
+
676
+ def generate(
677
+ self,
678
+ prompt: str,
679
+ *,
680
+ system: Optional[str] = None,
681
+ stop: Optional[list[str]] = None,
682
+ max_tokens: Optional[int] = None,
683
+ ) -> Iterator[str]:
684
+ """Stream chunks via ``create_chat_completion(stream=True)``."""
685
+ messages: list[dict] = []
686
+ if system:
687
+ messages.append({"role": "system", "content": system})
688
+ messages.append({"role": "user", "content": prompt})
689
+ kwargs: dict = {"messages": messages, "stream": True}
690
+ if stop:
691
+ kwargs["stop"] = list(stop)
692
+ if max_tokens is not None:
693
+ kwargs["max_tokens"] = int(max_tokens)
694
+ try:
695
+ for part in self._llama.create_chat_completion(**kwargs):
696
+ delta = (part.get("choices") or [{}])[0].get("delta") or {}
697
+ chunk = delta.get("content")
698
+ if chunk:
699
+ yield chunk
700
+ except (RuntimeError, ValueError, KeyError) as exc:
701
+ raise BackendUnavailable(
702
+ f"llama.cpp generation failed: {exc}",
703
+ hint="Verify the model and n_ctx; reduce max_tokens if out of memory.",
704
+ ) from exc
705
+
706
+ def count_tokens(self, text: str) -> int:
707
+ """Token count via llama.cpp's real tokenizer; estimate on failure."""
708
+ try:
709
+ return len(self._llama.tokenize(text.encode("utf-8")))
710
+ except (RuntimeError, ValueError, AttributeError) as exc:
711
+ _log.debug("llama.cpp tokenize failed, using estimate: %s", exc)
712
+ return estimate(text)
713
+
714
+
715
+ # ---------------------------------------------------------------------------
716
+ # HFLLM — guarded import of transformers.
717
+ # ---------------------------------------------------------------------------
718
+ class HFLLM:
719
+ """Hugging Face transformers adapter. Import-guarded.
720
+
721
+ Requires the ``[hf]`` extra (transformers + torch). Uses a ``TextIteratorStreamer`` so
722
+ generation still streams. ``context_window`` comes from the model config; token counts
723
+ use the model's own tokenizer.
724
+ """
725
+
726
+ def __init__(
727
+ self,
728
+ model_ref: str,
729
+ *,
730
+ context_window: Optional[int] = None,
731
+ model_options: Optional[dict] = None,
732
+ ) -> None:
733
+ try:
734
+ from transformers import ( # type: ignore[import-not-found]
735
+ AutoModelForCausalLM,
736
+ AutoTokenizer,
737
+ )
738
+ except ImportError as exc:
739
+ raise BackendUnavailable(
740
+ f"The HF backend needs 'transformers' (and torch): {exc}",
741
+ hint="Install it: pip install \"aether-context[hf]\"",
742
+ ) from exc
743
+
744
+ opts = dict(model_options or {})
745
+ opts.setdefault("device_map", "auto")
746
+ try:
747
+ self._tokenizer = AutoTokenizer.from_pretrained(model_ref)
748
+ self._model = AutoModelForCausalLM.from_pretrained(model_ref, **opts)
749
+ except (OSError, ValueError) as exc:
750
+ raise BackendUnavailable(
751
+ f"Could not load HF model '{model_ref}': {exc}",
752
+ hint="Check the org/model id and your network/cache for the download.",
753
+ ) from exc
754
+
755
+ self.name: str = model_ref
756
+ cfg = getattr(self._model, "config", None)
757
+ cfg_ctx = getattr(cfg, "max_position_embeddings", None)
758
+ self.context_window: int = (
759
+ context_window
760
+ or (cfg_ctx if isinstance(cfg_ctx, int) and cfg_ctx > 0 else DEFAULT_CONTEXT_WINDOW)
761
+ )
762
+
763
+ def generate(
764
+ self,
765
+ prompt: str,
766
+ *,
767
+ system: Optional[str] = None,
768
+ stop: Optional[list[str]] = None,
769
+ max_tokens: Optional[int] = None,
770
+ ) -> Iterator[str]:
771
+ """Stream chunks via a background generate + ``TextIteratorStreamer``."""
772
+ import threading
773
+
774
+ from transformers import TextIteratorStreamer # type: ignore[import-not-found]
775
+
776
+ messages: list[dict] = []
777
+ if system:
778
+ messages.append({"role": "system", "content": system})
779
+ messages.append({"role": "user", "content": prompt})
780
+ try:
781
+ inputs = self._tokenizer.apply_chat_template(
782
+ messages, add_generation_prompt=True, return_tensors="pt"
783
+ ).to(self._model.device)
784
+ except (ValueError, AttributeError) as exc:
785
+ raise BackendUnavailable(
786
+ f"HF chat templating failed for '{self.name}': {exc}",
787
+ hint="The model may lack a chat template; try a chat/instruct variant.",
788
+ ) from exc
789
+
790
+ streamer = TextIteratorStreamer(
791
+ self._tokenizer, skip_prompt=True, skip_special_tokens=True
792
+ )
793
+ gen_kwargs: dict = {
794
+ "input_ids": inputs,
795
+ "streamer": streamer,
796
+ "max_new_tokens": int(max_tokens) if max_tokens is not None else 512,
797
+ }
798
+ thread = threading.Thread(target=self._model.generate, kwargs=gen_kwargs)
799
+ thread.start()
800
+ emitted = ""
801
+ for chunk in streamer:
802
+ if not chunk:
803
+ continue
804
+ if stop:
805
+ emitted += chunk
806
+ cut = self._first_stop(emitted, stop)
807
+ if cut is not None:
808
+ remainder = emitted[:cut][len(emitted) - len(chunk):]
809
+ if remainder:
810
+ yield remainder
811
+ break
812
+ yield chunk
813
+ thread.join()
814
+
815
+ @staticmethod
816
+ def _first_stop(text: str, stop: list[str]) -> Optional[int]:
817
+ cut: Optional[int] = None
818
+ for s in stop:
819
+ if s:
820
+ idx = text.find(s)
821
+ if idx != -1:
822
+ cut = idx if cut is None else min(cut, idx)
823
+ return cut
824
+
825
+ def count_tokens(self, text: str) -> int:
826
+ """Token count via the model's tokenizer; estimate on failure."""
827
+ try:
828
+ return len(self._tokenizer.encode(text))
829
+ except (RuntimeError, ValueError, AttributeError) as exc:
830
+ _log.debug("HF tokenizer encode failed, using estimate: %s", exc)
831
+ return estimate(text)
832
+
833
+
834
+ __all__ = [
835
+ "LocalLLM",
836
+ "ModelSpec",
837
+ "parse_spec",
838
+ "load_model",
839
+ "MockLLM",
840
+ "OllamaLLM",
841
+ "OpenAICompatLLM",
842
+ "LlamaCppLLM",
843
+ "HFLLM",
844
+ "DEFAULT_CONTEXT_WINDOW",
845
+ "DEFAULT_OLLAMA_HOST",
846
+ ]