aether-context 0.3.0__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- aether_context-0.4.0/PKG-INFO +398 -0
- aether_context-0.4.0/README.md +348 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/__init__.py +1 -1
- aether_context-0.4.0/aether_context/contracts/__init__.py +534 -0
- aether_context-0.4.0/aether_context/contracts/schema-v1.json +2201 -0
- aether_context-0.4.0/aether_context/crypto.py +105 -0
- aether_context-0.4.0/aether_context/engine.py +1637 -0
- aether_context-0.4.0/aether_context/policy/__init__.py +66 -0
- aether_context-0.4.0/aether_context/retention.py +647 -0
- aether_context-0.4.0/aether_context/scale.py +298 -0
- aether_context-0.4.0/aether_context/service/__init__.py +279 -0
- aether_context-0.4.0/aether_context/storage_v2/__init__.py +125 -0
- aether_context-0.4.0/aether_context.egg-info/PKG-INFO +398 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/aether_context.egg-info/SOURCES.txt +14 -1
- {aether_context-0.3.0 → aether_context-0.4.0}/aether_context.egg-info/entry_points.txt +1 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/aether_context.egg-info/requires.txt +6 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/pyproject.toml +6 -3
- aether_context-0.4.0/tests/test_context_ipc.py +65 -0
- aether_context-0.4.0/tests/test_hosted_context.py +423 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_release_parity.py +32 -7
- aether_context-0.4.0/tests/test_retention_scale.py +1601 -0
- aether_context-0.4.0/tests/test_workflow_yaml.py +85 -0
- aether_context-0.3.0/PKG-INFO +0 -429
- aether_context-0.3.0/README.md +0 -384
- aether_context-0.3.0/aether_context.egg-info/PKG-INFO +0 -429
- {aether_context-0.3.0 → aether_context-0.4.0}/LICENSE +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/NOTICE.md +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/_log.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/cli.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/config.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/context_pool.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/encoder.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/errors.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/local_llm.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/mpo.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/py.typed +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/quantize.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/session.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/slice_loader.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/tokenizer.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/ui.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/witness.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/aether_context.egg-info/dependency_links.txt +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/aether_context.egg-info/top_level.txt +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/setup.cfg +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_aether_agent.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agent_commands.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agent_config.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agent_library.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agent_profile.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agent_profile_commands.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agent_runner.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agent_runner_define_command.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agent_runner_lock.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agent_slash.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agent_slash_agents_columns.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agent_slash_cmd.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agent_store.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agents_view.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_api_eval.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_auth.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_brains.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_bridge.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_cli.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_cli_disk.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_cli_dispatch.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_cli_doctor.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_cli_setup.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_cli_surface.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_config.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_context_pool.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_encoder.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_end_to_end.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_errors.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_hardware.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_install_security.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_local_llm.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_local_llm_openai.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_moat_seal.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_mpo.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_multi_runner.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_ollama_ctl.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_ollama_integration.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_onboarding.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_pool_quantize.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_profiles.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_progress.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_protocol_lockstep.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_quantize.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_repl_agents.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_repl_multi.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_repl_preflight.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_run_agent_events_system.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_safety.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_session.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_session_fallback.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_slash.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_slash_agents.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_slash_custom_invoke.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_slash_ollama.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_slice_loader.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_smoke.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_tokenizer.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_toolparse.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_transport.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_ui.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_web_tools.py +0 -0
- {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_witness.py +0 -0
|
@@ -0,0 +1,398 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: aether-context
|
|
3
|
+
Version: 0.4.0
|
|
4
|
+
Summary: Unlimited Context — virtual memory for an LLM's attention. Local-first, numpy-only core.
|
|
5
|
+
Author: Aether AI
|
|
6
|
+
Maintainer: Aether AI
|
|
7
|
+
License: Apache-2.0
|
|
8
|
+
Project-URL: Homepage, https://github.com/AetherAI3/Unlimited-Context-LLM
|
|
9
|
+
Project-URL: Repository, https://github.com/AetherAI3/Unlimited-Context-LLM.git
|
|
10
|
+
Project-URL: Documentation, https://github.com/AetherAI3/Unlimited-Context-LLM#readme
|
|
11
|
+
Project-URL: Issues, https://github.com/AetherAI3/Unlimited-Context-LLM/issues
|
|
12
|
+
Project-URL: Changelog, https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/CHANGELOG.md
|
|
13
|
+
Keywords: llm,context,retrieval,ollama,local,rag,agents
|
|
14
|
+
Classifier: Development Status :: 4 - Beta
|
|
15
|
+
Classifier: Intended Audience :: Developers
|
|
16
|
+
Classifier: License :: OSI Approved :: Apache Software License
|
|
17
|
+
Classifier: Operating System :: OS Independent
|
|
18
|
+
Classifier: Programming Language :: Python :: 3
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
23
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
24
|
+
Requires-Python: <3.15,>=3.10
|
|
25
|
+
Description-Content-Type: text/markdown
|
|
26
|
+
License-File: LICENSE
|
|
27
|
+
License-File: NOTICE.md
|
|
28
|
+
Requires-Dist: numpy<3.0,>=1.24
|
|
29
|
+
Provides-Extra: ollama
|
|
30
|
+
Provides-Extra: llamacpp
|
|
31
|
+
Requires-Dist: llama-cpp-python>=0.2; extra == "llamacpp"
|
|
32
|
+
Provides-Extra: hf
|
|
33
|
+
Requires-Dist: transformers>=4.40; extra == "hf"
|
|
34
|
+
Requires-Dist: torch>=2.2; extra == "hf"
|
|
35
|
+
Provides-Extra: fast
|
|
36
|
+
Requires-Dist: hnswlib>=0.8; extra == "fast"
|
|
37
|
+
Provides-Extra: hosted
|
|
38
|
+
Requires-Dist: pydantic<3,>=2.10; extra == "hosted"
|
|
39
|
+
Requires-Dist: cryptography<51,>=44; extra == "hosted"
|
|
40
|
+
Provides-Extra: all
|
|
41
|
+
Requires-Dist: aether-context[fast,hf,llamacpp]; extra == "all"
|
|
42
|
+
Provides-Extra: dev
|
|
43
|
+
Requires-Dist: pytest>=8; extra == "dev"
|
|
44
|
+
Requires-Dist: pytest-cov; extra == "dev"
|
|
45
|
+
Requires-Dist: ruff>=0.5; extra == "dev"
|
|
46
|
+
Requires-Dist: mypy>=1.10; extra == "dev"
|
|
47
|
+
Requires-Dist: pyyaml>=6; extra == "dev"
|
|
48
|
+
Requires-Dist: aether-context[hosted]; extra == "dev"
|
|
49
|
+
Dynamic: license-file
|
|
50
|
+
|
|
51
|
+
<div align="center">
|
|
52
|
+
|
|
53
|
+
# ⚡ Unlimited Context
|
|
54
|
+
|
|
55
|
+
**Virtual memory for an LLM's attention.** Keep a billion-token pool on your own disk; the model
|
|
56
|
+
reaches it in slices, one small window at a time. Local-first, offline, free.
|
|
57
|
+
|
|
58
|
+
<img alt="npx aether-context: guided setup, a clean doctor check, and pool status - real terminal output" width="716" src="https://raw.githubusercontent.com/AetherAI3/Unlimited-Context-LLM/main/assets/demo.gif">
|
|
59
|
+
|
|
60
|
+
[](https://pypi.org/project/aether-context/)
|
|
61
|
+
[](https://www.npmjs.com/package/aether-context)
|
|
62
|
+
[](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/LICENSE)
|
|
63
|
+
[](https://www.python.org)
|
|
64
|
+
[](https://aethersystems.net)
|
|
65
|
+
[](https://github.com/AetherAI3/Unlimited-Context-LLM/stargazers)
|
|
66
|
+
|
|
67
|
+
[Site](https://aetherai3.github.io/Unlimited-Context-LLM/) ·
|
|
68
|
+
[Install](https://github.com/AetherAI3/Unlimited-Context-LLM#install) ·
|
|
69
|
+
[How it works](https://github.com/AetherAI3/Unlimited-Context-LLM#how-it-works) ·
|
|
70
|
+
[The proof](https://github.com/AetherAI3/Unlimited-Context-LLM#the-proof) ·
|
|
71
|
+
[Sizing](https://github.com/AetherAI3/Unlimited-Context-LLM#sizing-disk-reach-and-ram) ·
|
|
72
|
+
[Safety](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/SAFETY.md)
|
|
73
|
+
|
|
74
|
+
</div>
|
|
75
|
+
|
|
76
|
+
---
|
|
77
|
+
|
|
78
|
+
> **Your context window didn't get bigger. Its *reach* did.**
|
|
79
|
+
> The model keeps its small window. The engine keeps a vast store on your disk and pulls the
|
|
80
|
+
> *right slice* back in while the model reasons. A small local model stays coherent across runs
|
|
81
|
+
> that would blow past any context window.
|
|
82
|
+
|
|
83
|
+
## Install
|
|
84
|
+
|
|
85
|
+
```bash
|
|
86
|
+
pip install aether-context
|
|
87
|
+
aether-context setup
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
`setup` sizes the pool, checks for a local model, and verifies the engine end to end. It works
|
|
91
|
+
with no daemon, no network and no model pulled — the check runs against the built-in mock model.
|
|
92
|
+
|
|
93
|
+
```python
|
|
94
|
+
from aether_context import Session
|
|
95
|
+
|
|
96
|
+
s = Session(model="ollama/qwen2.5", pool_gb=5)
|
|
97
|
+
s.run("Build me a full-stack weightlifting tracker app.")
|
|
98
|
+
# runs long. stays coherent. walk away.
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
Prefer npm? Same software, same release. The npm package is a launcher that installs the Python
|
|
102
|
+
engine into a private virtualenv for you (it needs Python 3.10+ on your PATH):
|
|
103
|
+
|
|
104
|
+
```bash
|
|
105
|
+
npx aether-context setup
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
<details>
|
|
109
|
+
<summary>Other install routes</summary>
|
|
110
|
+
|
|
111
|
+
```bash
|
|
112
|
+
# Straight from source, always the latest main:
|
|
113
|
+
pip install git+https://github.com/AetherAI3/Unlimited-Context-LLM.git
|
|
114
|
+
|
|
115
|
+
# Isolated, if you only want the CLI:
|
|
116
|
+
pipx install aether-context
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
The distribution name is **`aether-context`** — `pip install unlimited-context` is not this
|
|
120
|
+
package.
|
|
121
|
+
</details>
|
|
122
|
+
|
|
123
|
+
That's the whole thing. One small model, one command, a billion tokens of reach behind it.
|
|
124
|
+
|
|
125
|
+
## The problem
|
|
126
|
+
|
|
127
|
+
Long agentic runs all die the same way. The model fills its window, starts **compressing** its own
|
|
128
|
+
history, silently drops the one detail that mattered three steps ago — and drifts. You've seen it:
|
|
129
|
+
the runaway PR, the agent that rewrites a function it already wrote, the build that falls apart at
|
|
130
|
+
hour two. Bigger windows just delay it, and a crammed 1M-token window **rots in the middle** anyway.
|
|
131
|
+
|
|
132
|
+
The fix isn't a bigger window. It's to stop throwing the overflow away. Instead of summarizing what
|
|
133
|
+
spills over, Unlimited Context **encodes** it to a local pool on your disk and **recovers** the
|
|
134
|
+
right slice exactly when it's needed. Nothing load-bearing is silently lost.
|
|
135
|
+
|
|
136
|
+
<p align="center"><strong>Compress & forget ✗ → Encode & recover ✓</strong></p>
|
|
137
|
+
|
|
138
|
+
## How it works
|
|
139
|
+
|
|
140
|
+
It's **virtual memory, for attention.** Map it to an OS and it clicks:
|
|
141
|
+
|
|
142
|
+
| OS | Unlimited Context |
|
|
143
|
+
|---|---|
|
|
144
|
+
| RAM | the **resident window** the model sees now (small, fast) |
|
|
145
|
+
| Disk | the **context pool** — your encoded memory (~5 GB ≈ ~1B tokens) |
|
|
146
|
+
| Pager | the **slice loader** — prefetches the next slice from what the model is reasoning about *right now* |
|
|
147
|
+
| Page-replacement | the **retention policy** — useful slices *stay*, stale ones *fade*, anything relevant again comes back |
|
|
148
|
+
|
|
149
|
+
The pager runs concurrently with generation, so most of the fetch hides behind the model's own
|
|
150
|
+
thinking. Full explainer: [`docs/how-it-works.md`](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/docs/how-it-works.md).
|
|
151
|
+
|
|
152
|
+
## What you get
|
|
153
|
+
|
|
154
|
+
- 🧠 **Unbounded reach** — ~1B tokens of encoded context in ~5 GB on disk; the model reaches it in slices.
|
|
155
|
+
- 🧩 **MPO context chain** — recall pulls the whole connected thread, not isolated nearest-neighbors.
|
|
156
|
+
- 🪟 **Curated beats crammed** — a small, relevant resident window outperforms a stuffed one (no lost-in-the-middle) — and costs less.
|
|
157
|
+
- 🔒 **Local-first** — your context never leaves your machine. Free storage, full privacy, works offline.
|
|
158
|
+
- 🤖 **Any model** — Llama, Qwen, Mistral, Phi — via Ollama, llama.cpp, or Hugging Face, or your own API-backed model.
|
|
159
|
+
- 📉 **Coherence you can measure** — the head-to-head is committed: same model, engine on vs off.
|
|
160
|
+
|
|
161
|
+
## The proof
|
|
162
|
+
|
|
163
|
+
Not a synthetic micro-benchmark — a **real, paid, end-to-end run.** A reasoning model
|
|
164
|
+
(`deepseek-v4-pro`, via OpenRouter) driven through a **40-turn agent session that overflows its
|
|
165
|
+
window** (2,000-token window, 60 real `microsoft/vscode` issues), measured **engine off vs on** —
|
|
166
|
+
one live run, **$0.19**, 2026-06-14.
|
|
167
|
+
|
|
168
|
+
- **The model stops forgetting.** Recall of early facts after they fall out of the window:
|
|
169
|
+
**0.15 → 1.00.** The baseline drifts and forgets; the engine holds every early fact — zero drift.
|
|
170
|
+
- **Failure turns into success on the real work.** Tasks completed correctly: **3 / 20 → 20 / 20.**
|
|
171
|
+
- **Cheaper, not just better.** **−24%** total cost, **−54%** in the back half — the engine sends a
|
|
172
|
+
compact recalled slice instead of dragging the whole transcript into every call.
|
|
173
|
+
|
|
174
|
+
| Metric | Off (baseline) | On (engine) | Change |
|
|
175
|
+
|---|:---:|:---:|:---:|
|
|
176
|
+
| **Recall coherence** (early facts still correct) | 0.15 | **1.00** | **6.7×** |
|
|
177
|
+
| **Work outcome** (tasks done right) | 3 / 20 | **20 / 20** | **3 → 20** |
|
|
178
|
+
| **Cost — full session** | $0.0711 | **$0.0542** | **−24%** |
|
|
179
|
+
| **Cost — back half (recall phase)** | $0.00117/turn | **$0.00053/turn** | **−54%** |
|
|
180
|
+
|
|
181
|
+
<p align="center">
|
|
182
|
+
<img alt="Cumulative cost and recall coherence vs turn — engine off vs on" width="780"
|
|
183
|
+
src="https://raw.githubusercontent.com/AetherAI3/Unlimited-Context-LLM/main/docs/benchmarks/artifacts/2026-06-14-deepseek-v4-pro/api_eval_plot.png">
|
|
184
|
+
</p>
|
|
185
|
+
|
|
186
|
+
**Committed data:** [full write-up](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/docs/benchmarks/2026-06-14-deepseek-v4-pro-session-eval.md) ·
|
|
187
|
+
[raw artifacts](https://github.com/AetherAI3/Unlimited-Context-LLM/tree/main/docs/benchmarks/artifacts/2026-06-14-deepseek-v4-pro)
|
|
188
|
+
(`api_eval_results.json`, `api_eval_series.csv`, `api_eval_plot.png`, `RESULTS.md`) · reproduce with
|
|
189
|
+
`python -m bench.api_eval --model deepseek/deepseek-v4-pro --repo microsoft/vscode --arms off,on,on_chain --plot`
|
|
190
|
+
|
|
191
|
+
<sub>**Scope, honestly:** this run used a hosted reasoning model, not a local one — the mechanism is
|
|
192
|
+
backend-agnostic, but the headline number is not a local number. It measures the **engine**
|
|
193
|
+
(retrieve-on-overflow memory), not the MPO chain: on this single-fact recall task the chain
|
|
194
|
+
**ties** plain recall (both 1.00), and its multi-slice edge is **synthetic-only so far**
|
|
195
|
+
(`bench/chain_recall.py`: connected-context recall 0.15 → 0.78), with the live `thread` run
|
|
196
|
+
**pending**, not yet claimed. The 2,000-token window is deliberately tiny to force overflow, so a
|
|
197
|
+
realistic window shows a smaller (still real) gain. N = 20 recall turns, single run.</sub>
|
|
198
|
+
|
|
199
|
+
## Sizing: disk, reach and RAM
|
|
200
|
+
|
|
201
|
+
First run drops you into a slider — pick how much your model gets to remember:
|
|
202
|
+
|
|
203
|
+
```text
|
|
204
|
+
$ aether-context init
|
|
205
|
+
──────────────────────────────────────────────────────────────────
|
|
206
|
+
⚡ choose your context pool encoded reach · not a window
|
|
207
|
+
──────────────────────────────────────────────────────────────────
|
|
208
|
+
▸ 5 GB ████░░░░░░░░░░░░ ~1.16B tokens a big project (floor)
|
|
209
|
+
10 GB ████████░░░░░░░░ ~2.33B tokens a large monorepo + docs
|
|
210
|
+
15 GB ████████████░░░░ ~3.49B tokens multiple repos / long runs
|
|
211
|
+
20 GB ████████████████ ~4.65B tokens massive corpus / power user
|
|
212
|
+
──────────────────────────────────────────────────────────────────
|
|
213
|
+
reach ≈ pool_GB × 233M tokens custom: --pool 12 (any size ≥ 5 GB)
|
|
214
|
+
↑/↓ slide ↵ confirm
|
|
215
|
+
|
|
216
|
+
pool [5]: 10
|
|
217
|
+
✓ 10 GB → your model can now reach ~2.33 billion tokens
|
|
218
|
+
```
|
|
219
|
+
|
|
220
|
+
One table for the whole trade-off — disk in, reach out, RAM cost, and how many isolated sessions
|
|
221
|
+
fit on a small machine:
|
|
222
|
+
|
|
223
|
+
| Pool | Slices | Encoded reach | Index RAM | Sessions on 8 GB (separate pools) |
|
|
224
|
+
|:----:|:------:|:-------------:|:---------:|:---------------------------------:|
|
|
225
|
+
| **5 GB** *(floor)* | 2.27M | **~1.16B tokens** | ~146 MB | ~13 |
|
|
226
|
+
| 10 GB | 4.55M | **~2.33B tokens** | ~291 MB | ~7 |
|
|
227
|
+
| 15 GB | 6.82M | **~3.49B tokens** | ~436 MB | ~4 |
|
|
228
|
+
| 20 GB | 9.09M | **~4.65B tokens** | ~582 MB | ~3 |
|
|
229
|
+
|
|
230
|
+
Roughly double the session counts on a 16 GB machine. Where the numbers come from: ~2.2 KB per
|
|
231
|
+
slice (a 256-dim vector + compressed text + metadata) ÷ 512 tokens per slice → **~455K slices/GB →
|
|
232
|
+
~233M tokens of reach per GB**. So `reach ≈ pool_GB × 233M`. At the 5 GB floor that's about
|
|
233
|
+
**9,000×** a 128K window. Bump the pool anytime with `aether-context --pool 20`.
|
|
234
|
+
|
|
235
|
+
**RAM is a formula, not a mystery.** Vectors live on disk (mmap'd) — only the small index graph and
|
|
236
|
+
a hot working set are ever resident:
|
|
237
|
+
|
|
238
|
+
```
|
|
239
|
+
RAM ≈ ~180 MB base (engine + shared static encoder)
|
|
240
|
+
+ ~29 MB per GB of pool (resident index)
|
|
241
|
+
+ ~30 MB per active session
|
|
242
|
+
```
|
|
243
|
+
|
|
244
|
+
**Sharing the pool is the biggest RAM lever.** `--pool-mode separate` *(default)* gives every
|
|
245
|
+
session its own pool and index — fully isolated and private, but you pay one index per session, so
|
|
246
|
+
RAM scales with `N × pool` (that's the last column above). `--pool-mode shared` pays for the index
|
|
247
|
+
**once**; each extra session adds only ~30 MB, so 50–70+ sessions fit and CPU becomes the limit
|
|
248
|
+
instead of memory. The trade-off is that sessions can see each other's context. Use shared for
|
|
249
|
+
related work on one project, separate for unrelated tasks.
|
|
250
|
+
|
|
251
|
+
**How much building is that?** A ~128K window fills after well under an hour of active agent work,
|
|
252
|
+
then starts compacting and forgetting. Assuming a busy coding agent encodes ~300K–1M keep-worthy
|
|
253
|
+
tokens an hour, a 5 GB pool covers on the order of **1,200–3,900 hours** before it even fills —
|
|
254
|
+
weeks of nonstop building. Because the retention policy fades stale slices, the pool never
|
|
255
|
+
hard-stops anyway; it just keeps what's relevant.
|
|
256
|
+
|
|
257
|
+
<div align="center">
|
|
258
|
+
<img width="880" alt="Coding time per pool size" src="https://raw.githubusercontent.com/AetherAI3/Unlimited-Context-LLM/main/assets/coding-time-per-pool.png">
|
|
259
|
+
</div>
|
|
260
|
+
|
|
261
|
+
> **Honest:** that's encoded **reach**, retrieved in slices — not a bigger attention window, and it
|
|
262
|
+
> rides on retrieval hit rate. A bigger pool buys more reachable codebase or corpus *per session*,
|
|
263
|
+
> never more concurrent sessions (those are RAM-bound). `--index tiered` is reserved for a future
|
|
264
|
+
> paged-graph index and currently runs the flat index — it does not yet reduce resident RAM.
|
|
265
|
+
|
|
266
|
+
## Commands
|
|
267
|
+
|
|
268
|
+
| Command | What it's for |
|
|
269
|
+
|---|---|
|
|
270
|
+
| `aether-context setup` | **Start here.** Guided first run: size the pool, check your model, verify the engine. |
|
|
271
|
+
| `aether-context init` | Pick your pool size — the on-disk storage slider — on first run. |
|
|
272
|
+
| `aether-context run "<task>"` | One-shot a task with full reach, then print the result. |
|
|
273
|
+
| `aether-context run "<task>" --no-mpo-chain` | Same, with the MPO context chain disabled (plain cosine). |
|
|
274
|
+
| `aether-context chat` | Open an interactive session; type `/status` anytime, `/clear` to reset. |
|
|
275
|
+
| `aether-context status` | See pool size, slices used, reach, and hit rate at a glance. |
|
|
276
|
+
| `aether-context doctor` | Check Ollama, your model, disk, and RAM before a long run. |
|
|
277
|
+
| `aether-context --pool 20` | Resize the pool anytime (non-destructive re-index). |
|
|
278
|
+
|
|
279
|
+
> **Tip:** run `aether-context doctor` first — it catches the three things that ever go wrong
|
|
280
|
+
> (Ollama down, model not pulled, not enough disk) and prints the exact fix.
|
|
281
|
+
|
|
282
|
+
## MPO: the context chain
|
|
283
|
+
|
|
284
|
+
Plain semantic search returns isolated nearest-neighbors — the single closest slices, ripped out of
|
|
285
|
+
the thread they belonged to. Recall a fact and you often miss the three slices around it that made
|
|
286
|
+
it make sense.
|
|
287
|
+
|
|
288
|
+
The **MPO context chain** links the session's slices into one connected structure, so when cosine
|
|
289
|
+
pulls an entry slice, the chain pulls in the slices most coupled to it — widening the working set
|
|
290
|
+
with the *connected thread*, not stray hits. Cosine is still the retrieval mechanism; the chain
|
|
291
|
+
assists it.
|
|
292
|
+
|
|
293
|
+
The chain is Aether-tuned, deterministic and fully local — no training, no network. It is purely
|
|
294
|
+
**additive**: it only ever *adds* connected context, never blocks or replaces a hit, and on any
|
|
295
|
+
hiccup it falls back cleanly to plain cosine. In a planted-thread benchmark it lifts
|
|
296
|
+
connected-context recall from **0.15 (cosine alone) to 0.78** — over 5× more of the right thread in
|
|
297
|
+
the window. That result is synthetic so far; see the caveat under [The proof](https://github.com/AetherAI3/Unlimited-Context-LLM#the-proof).
|
|
298
|
+
|
|
299
|
+
On by default:
|
|
300
|
+
|
|
301
|
+
```python
|
|
302
|
+
Session(model="ollama/qwen2.5", pool_gb=10) # chain on by default
|
|
303
|
+
Session(model="ollama/qwen2.5", pool_gb=10, mpo_chain=False) # plain cosine
|
|
304
|
+
```
|
|
305
|
+
```bash
|
|
306
|
+
aether-context run "..." --no-mpo-chain # disable for one run
|
|
307
|
+
```
|
|
308
|
+
|
|
309
|
+
## The `aether` coding terminal
|
|
310
|
+
|
|
311
|
+
**`aether`** is an open-source agentic coding terminal that runs on this engine. Turns run on your
|
|
312
|
+
local [Ollama](https://ollama.com) by default — no account, no network; sign in and they switch to
|
|
313
|
+
the Aether cloud API. It ships as its own package:
|
|
314
|
+
|
|
315
|
+
```bash
|
|
316
|
+
pip install aether-agent # or: npm install -g aether-agents
|
|
317
|
+
```
|
|
318
|
+
|
|
319
|
+
`aether` opens the REPL, `aether "<prompt>"` is a one-shot turn, and `aether code "<task>"` is an
|
|
320
|
+
autonomous coding run on the Unlimited Context brain (test-gated, git-checkpointed). Full command
|
|
321
|
+
list, slash commands and backend settings live at
|
|
322
|
+
[AetherAI3/aether-agent](https://github.com/AetherAI3/aether-agent).
|
|
323
|
+
|
|
324
|
+
<sub>The `aether_agent/` directory in *this* repo is the Python-native twin — same commands, same
|
|
325
|
+
backend, same tools — kept here for development and deliberately **not** published from this
|
|
326
|
+
package: PyPI's `aether-agent` already owns that import path and the `aether` command, so shipping
|
|
327
|
+
a second copy would silently overwrite it wherever both are installed. From a clone with Ollama up,
|
|
328
|
+
`python -m aether_agent.smoke` runs the SSRF guard, a real local turn, a web search and fetch, and
|
|
329
|
+
the cloud path when signed in.</sub>
|
|
330
|
+
|
|
331
|
+
## Safety and use policy
|
|
332
|
+
|
|
333
|
+
Giving a model durable memory is powerful, and the failure modes are real: runaway agents,
|
|
334
|
+
grounding drift, an agent's own notes hardening into its rules. What those are and what we do about
|
|
335
|
+
them is written up in
|
|
336
|
+
**[Ethical & Safety Measures](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/SAFETY.md)**.
|
|
337
|
+
Use of the project is governed by the
|
|
338
|
+
**[Acceptable Use Policy](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/USE_POLICY.md)**
|
|
339
|
+
— by using the project you agree to and are bound by its terms.
|
|
340
|
+
|
|
341
|
+
## Honest about the word "unlimited"
|
|
342
|
+
|
|
343
|
+
"Unlimited" means **reach, not attention.** Your model keeps its native window; the engine makes it
|
|
344
|
+
*reach* a billion-token pool in slices, via fast retrieval. The whole thing rides on retrieval hit
|
|
345
|
+
rate — when that's high, and the loader is built to keep it high, the pool feels like one seamless
|
|
346
|
+
context. When it isn't, you get a miss, and a miss looks like forgetting. The measured evidence for
|
|
347
|
+
all of this is in [The proof](https://github.com/AetherAI3/Unlimited-Context-LLM#the-proof), caveats included.
|
|
348
|
+
|
|
349
|
+
## Contributing
|
|
350
|
+
|
|
351
|
+
**PRs and issues are welcome** — start with
|
|
352
|
+
[CONTRIBUTING.md](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/CONTRIBUTING.md).
|
|
353
|
+
There are open issues tagged
|
|
354
|
+
[good first issue](https://github.com/AetherAI3/Unlimited-Context-LLM/issues?q=is%3Aissue+is%3Aopen+label%3A%22good+first+issue%22)
|
|
355
|
+
and [help wanted](https://github.com/AetherAI3/Unlimited-Context-LLM/issues?q=is%3Aissue+is%3Aopen+label%3A%22help+wanted%22)
|
|
356
|
+
right now, including an LM Studio backend, a Windows quickstart, and a recall-quality benchmark at
|
|
357
|
+
100K / 1M / 10M tokens.
|
|
358
|
+
|
|
359
|
+
Runnable examples live in
|
|
360
|
+
[`examples/`](https://github.com/AetherAI3/Unlimited-Context-LLM/tree/main/examples) — start with
|
|
361
|
+
[`quickstart.py`](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/examples/quickstart.py),
|
|
362
|
+
then [`coding_agent.py`](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/examples/coding_agent.py).
|
|
363
|
+
|
|
364
|
+
If the engine earns its place in your setup, **a star helps other people find it.**
|
|
365
|
+
|
|
366
|
+
## Citation
|
|
367
|
+
|
|
368
|
+
If Unlimited Context helps your work, please cite it. Built and maintained by **Aether AI**.
|
|
369
|
+
|
|
370
|
+
```bibtex
|
|
371
|
+
@software{unlimited_context_2026,
|
|
372
|
+
title = {Unlimited Context (aether-context): virtual memory for LLM attention},
|
|
373
|
+
author = {Barrante, Brandon},
|
|
374
|
+
organization = {Aether AI},
|
|
375
|
+
year = {2026},
|
|
376
|
+
url = {https://github.com/AetherAI3/Unlimited-Context-LLM},
|
|
377
|
+
license = {Apache-2.0}
|
|
378
|
+
}
|
|
379
|
+
```
|
|
380
|
+
|
|
381
|
+
GitHub's "Cite this repository" button reads
|
|
382
|
+
[`CITATION.cff`](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/CITATION.cff) directly.
|
|
383
|
+
|
|
384
|
+
## License
|
|
385
|
+
|
|
386
|
+
**Apache-2.0.** Use it, fork it, ship it in your product.
|
|
387
|
+
|
|
388
|
+
---
|
|
389
|
+
|
|
390
|
+
<div align="center">
|
|
391
|
+
|
|
392
|
+
Built by **Aether AI** · [aethersystems.net](https://aethersystems.net)
|
|
393
|
+
|
|
394
|
+
<img width="880" alt="Aether" src="https://raw.githubusercontent.com/AetherAI3/Unlimited-Context-LLM/main/assets/aether-footer.jpg">
|
|
395
|
+
|
|
396
|
+
*Unbounded reach for the model you already run.*
|
|
397
|
+
|
|
398
|
+
</div>
|