aether-context 0.3.0__tar.gz → 0.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (96) hide show
  1. aether_context-0.3.1/PKG-INFO +394 -0
  2. aether_context-0.3.1/README.md +348 -0
  3. {aether_context-0.3.0 → aether_context-0.3.1}/aether_context/__init__.py +1 -1
  4. aether_context-0.3.1/aether_context.egg-info/PKG-INFO +394 -0
  5. {aether_context-0.3.0 → aether_context-0.3.1}/aether_context.egg-info/SOURCES.txt +2 -1
  6. {aether_context-0.3.0 → aether_context-0.3.1}/aether_context.egg-info/requires.txt +1 -0
  7. {aether_context-0.3.0 → aether_context-0.3.1}/pyproject.toml +1 -1
  8. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_release_parity.py +16 -3
  9. aether_context-0.3.1/tests/test_workflow_yaml.py +85 -0
  10. aether_context-0.3.0/PKG-INFO +0 -429
  11. aether_context-0.3.0/README.md +0 -384
  12. aether_context-0.3.0/aether_context.egg-info/PKG-INFO +0 -429
  13. {aether_context-0.3.0 → aether_context-0.3.1}/LICENSE +0 -0
  14. {aether_context-0.3.0 → aether_context-0.3.1}/NOTICE.md +0 -0
  15. {aether_context-0.3.0 → aether_context-0.3.1}/aether_context/_log.py +0 -0
  16. {aether_context-0.3.0 → aether_context-0.3.1}/aether_context/cli.py +0 -0
  17. {aether_context-0.3.0 → aether_context-0.3.1}/aether_context/config.py +0 -0
  18. {aether_context-0.3.0 → aether_context-0.3.1}/aether_context/context_pool.py +0 -0
  19. {aether_context-0.3.0 → aether_context-0.3.1}/aether_context/encoder.py +0 -0
  20. {aether_context-0.3.0 → aether_context-0.3.1}/aether_context/errors.py +0 -0
  21. {aether_context-0.3.0 → aether_context-0.3.1}/aether_context/local_llm.py +0 -0
  22. {aether_context-0.3.0 → aether_context-0.3.1}/aether_context/mpo.py +0 -0
  23. {aether_context-0.3.0 → aether_context-0.3.1}/aether_context/py.typed +0 -0
  24. {aether_context-0.3.0 → aether_context-0.3.1}/aether_context/quantize.py +0 -0
  25. {aether_context-0.3.0 → aether_context-0.3.1}/aether_context/session.py +0 -0
  26. {aether_context-0.3.0 → aether_context-0.3.1}/aether_context/slice_loader.py +0 -0
  27. {aether_context-0.3.0 → aether_context-0.3.1}/aether_context/tokenizer.py +0 -0
  28. {aether_context-0.3.0 → aether_context-0.3.1}/aether_context/ui.py +0 -0
  29. {aether_context-0.3.0 → aether_context-0.3.1}/aether_context/witness.py +0 -0
  30. {aether_context-0.3.0 → aether_context-0.3.1}/aether_context.egg-info/dependency_links.txt +0 -0
  31. {aether_context-0.3.0 → aether_context-0.3.1}/aether_context.egg-info/entry_points.txt +0 -0
  32. {aether_context-0.3.0 → aether_context-0.3.1}/aether_context.egg-info/top_level.txt +0 -0
  33. {aether_context-0.3.0 → aether_context-0.3.1}/setup.cfg +0 -0
  34. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_aether_agent.py +0 -0
  35. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_agent_commands.py +0 -0
  36. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_agent_config.py +0 -0
  37. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_agent_library.py +0 -0
  38. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_agent_profile.py +0 -0
  39. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_agent_profile_commands.py +0 -0
  40. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_agent_runner.py +0 -0
  41. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_agent_runner_define_command.py +0 -0
  42. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_agent_runner_lock.py +0 -0
  43. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_agent_slash.py +0 -0
  44. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_agent_slash_agents_columns.py +0 -0
  45. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_agent_slash_cmd.py +0 -0
  46. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_agent_store.py +0 -0
  47. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_agents_view.py +0 -0
  48. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_api_eval.py +0 -0
  49. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_auth.py +0 -0
  50. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_brains.py +0 -0
  51. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_bridge.py +0 -0
  52. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_cli.py +0 -0
  53. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_cli_disk.py +0 -0
  54. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_cli_dispatch.py +0 -0
  55. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_cli_doctor.py +0 -0
  56. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_cli_setup.py +0 -0
  57. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_cli_surface.py +0 -0
  58. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_config.py +0 -0
  59. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_context_pool.py +0 -0
  60. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_encoder.py +0 -0
  61. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_end_to_end.py +0 -0
  62. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_errors.py +0 -0
  63. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_hardware.py +0 -0
  64. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_install_security.py +0 -0
  65. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_local_llm.py +0 -0
  66. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_local_llm_openai.py +0 -0
  67. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_moat_seal.py +0 -0
  68. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_mpo.py +0 -0
  69. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_multi_runner.py +0 -0
  70. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_ollama_ctl.py +0 -0
  71. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_ollama_integration.py +0 -0
  72. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_onboarding.py +0 -0
  73. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_pool_quantize.py +0 -0
  74. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_profiles.py +0 -0
  75. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_progress.py +0 -0
  76. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_protocol_lockstep.py +0 -0
  77. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_quantize.py +0 -0
  78. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_repl_agents.py +0 -0
  79. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_repl_multi.py +0 -0
  80. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_repl_preflight.py +0 -0
  81. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_run_agent_events_system.py +0 -0
  82. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_safety.py +0 -0
  83. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_session.py +0 -0
  84. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_session_fallback.py +0 -0
  85. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_slash.py +0 -0
  86. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_slash_agents.py +0 -0
  87. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_slash_custom_invoke.py +0 -0
  88. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_slash_ollama.py +0 -0
  89. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_slice_loader.py +0 -0
  90. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_smoke.py +0 -0
  91. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_tokenizer.py +0 -0
  92. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_toolparse.py +0 -0
  93. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_transport.py +0 -0
  94. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_ui.py +0 -0
  95. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_web_tools.py +0 -0
  96. {aether_context-0.3.0 → aether_context-0.3.1}/tests/test_witness.py +0 -0
@@ -0,0 +1,394 @@
1
+ Metadata-Version: 2.4
2
+ Name: aether-context
3
+ Version: 0.3.1
4
+ Summary: Unlimited Context — virtual memory for an LLM's attention. Local-first, numpy-only core.
5
+ Author: Aether AI
6
+ Maintainer: Aether AI
7
+ License: Apache-2.0
8
+ Project-URL: Homepage, https://github.com/AetherAI3/Unlimited-Context-LLM
9
+ Project-URL: Repository, https://github.com/AetherAI3/Unlimited-Context-LLM.git
10
+ Project-URL: Documentation, https://github.com/AetherAI3/Unlimited-Context-LLM#readme
11
+ Project-URL: Issues, https://github.com/AetherAI3/Unlimited-Context-LLM/issues
12
+ Project-URL: Changelog, https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/CHANGELOG.md
13
+ Keywords: llm,context,retrieval,ollama,local,rag,agents
14
+ Classifier: Development Status :: 4 - Beta
15
+ Classifier: Intended Audience :: Developers
16
+ Classifier: License :: OSI Approved :: Apache Software License
17
+ Classifier: Operating System :: OS Independent
18
+ Classifier: Programming Language :: Python :: 3
19
+ Classifier: Programming Language :: Python :: 3.10
20
+ Classifier: Programming Language :: Python :: 3.11
21
+ Classifier: Programming Language :: Python :: 3.12
22
+ Classifier: Programming Language :: Python :: 3.13
23
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
24
+ Requires-Python: <3.15,>=3.10
25
+ Description-Content-Type: text/markdown
26
+ License-File: LICENSE
27
+ License-File: NOTICE.md
28
+ Requires-Dist: numpy<3.0,>=1.24
29
+ Provides-Extra: ollama
30
+ Provides-Extra: llamacpp
31
+ Requires-Dist: llama-cpp-python>=0.2; extra == "llamacpp"
32
+ Provides-Extra: hf
33
+ Requires-Dist: transformers>=4.40; extra == "hf"
34
+ Requires-Dist: torch>=2.2; extra == "hf"
35
+ Provides-Extra: fast
36
+ Requires-Dist: hnswlib>=0.8; extra == "fast"
37
+ Provides-Extra: all
38
+ Requires-Dist: aether-context[fast,hf,llamacpp]; extra == "all"
39
+ Provides-Extra: dev
40
+ Requires-Dist: pytest>=8; extra == "dev"
41
+ Requires-Dist: pytest-cov; extra == "dev"
42
+ Requires-Dist: ruff>=0.5; extra == "dev"
43
+ Requires-Dist: mypy>=1.10; extra == "dev"
44
+ Requires-Dist: pyyaml>=6; extra == "dev"
45
+ Dynamic: license-file
46
+
47
+ <div align="center">
48
+
49
+ # ⚡ Unlimited Context
50
+
51
+ **Virtual memory for an LLM's attention.** Keep a billion-token pool on your own disk; the model
52
+ reaches it in slices, one small window at a time. Local-first, offline, free.
53
+
54
+ <img alt="npx aether-context: guided setup, a clean doctor check, and pool status - real terminal output" width="716" src="https://raw.githubusercontent.com/AetherAI3/Unlimited-Context-LLM/main/assets/demo.gif">
55
+
56
+ [![PyPI](https://img.shields.io/pypi/v/aether-context?style=flat-square&logo=pypi&logoColor=white&color=06b6d4)](https://pypi.org/project/aether-context/)
57
+ [![npm](https://img.shields.io/npm/v/aether-context?style=flat-square&logo=npm&logoColor=white&color=cb3837)](https://www.npmjs.com/package/aether-context)
58
+ [![License](https://img.shields.io/badge/License-Apache_2.0-06b6d4?style=flat-square)](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/LICENSE)
59
+ [![Python](https://img.shields.io/badge/Python-3.10%2B-14b8a6?style=flat-square&logo=python&logoColor=white)](https://www.python.org)
60
+ [![Built by Aether](https://img.shields.io/badge/Built_by-Aether-7c3aed?style=flat-square)](https://aethersystems.net)
61
+ [![Stars](https://img.shields.io/github/stars/AetherAI3/Unlimited-Context-LLM?style=flat-square&logo=github&color=eab308)](https://github.com/AetherAI3/Unlimited-Context-LLM/stargazers)
62
+
63
+ [Site](https://aetherai3.github.io/Unlimited-Context-LLM/) ·
64
+ [Install](https://github.com/AetherAI3/Unlimited-Context-LLM#install) ·
65
+ [How it works](https://github.com/AetherAI3/Unlimited-Context-LLM#how-it-works) ·
66
+ [The proof](https://github.com/AetherAI3/Unlimited-Context-LLM#the-proof) ·
67
+ [Sizing](https://github.com/AetherAI3/Unlimited-Context-LLM#sizing-disk-reach-and-ram) ·
68
+ [Safety](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/SAFETY.md)
69
+
70
+ </div>
71
+
72
+ ---
73
+
74
+ > **Your context window didn't get bigger. Its *reach* did.**
75
+ > The model keeps its small window. The engine keeps a vast store on your disk and pulls the
76
+ > *right slice* back in while the model reasons. A small local model stays coherent across runs
77
+ > that would blow past any context window.
78
+
79
+ ## Install
80
+
81
+ ```bash
82
+ pip install aether-context
83
+ aether-context setup
84
+ ```
85
+
86
+ `setup` sizes the pool, checks for a local model, and verifies the engine end to end. It works
87
+ with no daemon, no network and no model pulled — the check runs against the built-in mock model.
88
+
89
+ ```python
90
+ from aether_context import Session
91
+
92
+ s = Session(model="ollama/qwen2.5", pool_gb=5)
93
+ s.run("Build me a full-stack weightlifting tracker app.")
94
+ # runs long. stays coherent. walk away.
95
+ ```
96
+
97
+ Prefer npm? Same software, same release. The npm package is a launcher that installs the Python
98
+ engine into a private virtualenv for you (it needs Python 3.10+ on your PATH):
99
+
100
+ ```bash
101
+ npx aether-context setup
102
+ ```
103
+
104
+ <details>
105
+ <summary>Other install routes</summary>
106
+
107
+ ```bash
108
+ # Straight from source, always the latest main:
109
+ pip install git+https://github.com/AetherAI3/Unlimited-Context-LLM.git
110
+
111
+ # Isolated, if you only want the CLI:
112
+ pipx install aether-context
113
+ ```
114
+
115
+ The distribution name is **`aether-context`** — `pip install unlimited-context` is not this
116
+ package.
117
+ </details>
118
+
119
+ That's the whole thing. One small model, one command, a billion tokens of reach behind it.
120
+
121
+ ## The problem
122
+
123
+ Long agentic runs all die the same way. The model fills its window, starts **compressing** its own
124
+ history, silently drops the one detail that mattered three steps ago — and drifts. You've seen it:
125
+ the runaway PR, the agent that rewrites a function it already wrote, the build that falls apart at
126
+ hour two. Bigger windows just delay it, and a crammed 1M-token window **rots in the middle** anyway.
127
+
128
+ The fix isn't a bigger window. It's to stop throwing the overflow away. Instead of summarizing what
129
+ spills over, Unlimited Context **encodes** it to a local pool on your disk and **recovers** the
130
+ right slice exactly when it's needed. Nothing load-bearing is silently lost.
131
+
132
+ <p align="center"><strong>Compress &amp; forget ✗ &nbsp;→&nbsp; Encode &amp; recover ✓</strong></p>
133
+
134
+ ## How it works
135
+
136
+ It's **virtual memory, for attention.** Map it to an OS and it clicks:
137
+
138
+ | OS | Unlimited Context |
139
+ |---|---|
140
+ | RAM | the **resident window** the model sees now (small, fast) |
141
+ | Disk | the **context pool** — your encoded memory (~5 GB ≈ ~1B tokens) |
142
+ | Pager | the **slice loader** — prefetches the next slice from what the model is reasoning about *right now* |
143
+ | Page-replacement | the **retention policy** — useful slices *stay*, stale ones *fade*, anything relevant again comes back |
144
+
145
+ The pager runs concurrently with generation, so most of the fetch hides behind the model's own
146
+ thinking. Full explainer: [`docs/how-it-works.md`](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/docs/how-it-works.md).
147
+
148
+ ## What you get
149
+
150
+ - 🧠 **Unbounded reach** — ~1B tokens of encoded context in ~5 GB on disk; the model reaches it in slices.
151
+ - 🧩 **MPO context chain** — recall pulls the whole connected thread, not isolated nearest-neighbors.
152
+ - 🪟 **Curated beats crammed** — a small, relevant resident window outperforms a stuffed one (no lost-in-the-middle) — and costs less.
153
+ - 🔒 **Local-first** — your context never leaves your machine. Free storage, full privacy, works offline.
154
+ - 🤖 **Any model** — Llama, Qwen, Mistral, Phi — via Ollama, llama.cpp, or Hugging Face, or your own API-backed model.
155
+ - 📉 **Coherence you can measure** — the head-to-head is committed: same model, engine on vs off.
156
+
157
+ ## The proof
158
+
159
+ Not a synthetic micro-benchmark — a **real, paid, end-to-end run.** A reasoning model
160
+ (`deepseek-v4-pro`, via OpenRouter) driven through a **40-turn agent session that overflows its
161
+ window** (2,000-token window, 60 real `microsoft/vscode` issues), measured **engine off vs on** —
162
+ one live run, **$0.19**, 2026-06-14.
163
+
164
+ - **The model stops forgetting.** Recall of early facts after they fall out of the window:
165
+ **0.15 → 1.00.** The baseline drifts and forgets; the engine holds every early fact — zero drift.
166
+ - **Failure turns into success on the real work.** Tasks completed correctly: **3 / 20 → 20 / 20.**
167
+ - **Cheaper, not just better.** **−24%** total cost, **−54%** in the back half — the engine sends a
168
+ compact recalled slice instead of dragging the whole transcript into every call.
169
+
170
+ | Metric | Off (baseline) | On (engine) | Change |
171
+ |---|:---:|:---:|:---:|
172
+ | **Recall coherence** (early facts still correct) | 0.15 | **1.00** | **6.7×** |
173
+ | **Work outcome** (tasks done right) | 3 / 20 | **20 / 20** | **3 → 20** |
174
+ | **Cost — full session** | $0.0711 | **$0.0542** | **−24%** |
175
+ | **Cost — back half (recall phase)** | $0.00117/turn | **$0.00053/turn** | **−54%** |
176
+
177
+ <p align="center">
178
+ <img alt="Cumulative cost and recall coherence vs turn — engine off vs on" width="780"
179
+ src="https://raw.githubusercontent.com/AetherAI3/Unlimited-Context-LLM/main/docs/benchmarks/artifacts/2026-06-14-deepseek-v4-pro/api_eval_plot.png">
180
+ </p>
181
+
182
+ **Committed data:** [full write-up](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/docs/benchmarks/2026-06-14-deepseek-v4-pro-session-eval.md) ·
183
+ [raw artifacts](https://github.com/AetherAI3/Unlimited-Context-LLM/tree/main/docs/benchmarks/artifacts/2026-06-14-deepseek-v4-pro)
184
+ (`api_eval_results.json`, `api_eval_series.csv`, `api_eval_plot.png`, `RESULTS.md`) · reproduce with
185
+ `python -m bench.api_eval --model deepseek/deepseek-v4-pro --repo microsoft/vscode --arms off,on,on_chain --plot`
186
+
187
+ <sub>**Scope, honestly:** this run used a hosted reasoning model, not a local one — the mechanism is
188
+ backend-agnostic, but the headline number is not a local number. It measures the **engine**
189
+ (retrieve-on-overflow memory), not the MPO chain: on this single-fact recall task the chain
190
+ **ties** plain recall (both 1.00), and its multi-slice edge is **synthetic-only so far**
191
+ (`bench/chain_recall.py`: connected-context recall 0.15 → 0.78), with the live `thread` run
192
+ **pending**, not yet claimed. The 2,000-token window is deliberately tiny to force overflow, so a
193
+ realistic window shows a smaller (still real) gain. N = 20 recall turns, single run.</sub>
194
+
195
+ ## Sizing: disk, reach and RAM
196
+
197
+ First run drops you into a slider — pick how much your model gets to remember:
198
+
199
+ ```text
200
+ $ aether-context init
201
+ ──────────────────────────────────────────────────────────────────
202
+ ⚡ choose your context pool encoded reach · not a window
203
+ ──────────────────────────────────────────────────────────────────
204
+ ▸ 5 GB ████░░░░░░░░░░░░ ~1.16B tokens a big project (floor)
205
+ 10 GB ████████░░░░░░░░ ~2.33B tokens a large monorepo + docs
206
+ 15 GB ████████████░░░░ ~3.49B tokens multiple repos / long runs
207
+ 20 GB ████████████████ ~4.65B tokens massive corpus / power user
208
+ ──────────────────────────────────────────────────────────────────
209
+ reach ≈ pool_GB × 233M tokens custom: --pool 12 (any size ≥ 5 GB)
210
+ ↑/↓ slide ↵ confirm
211
+
212
+ pool [5]: 10
213
+ ✓ 10 GB → your model can now reach ~2.33 billion tokens
214
+ ```
215
+
216
+ One table for the whole trade-off — disk in, reach out, RAM cost, and how many isolated sessions
217
+ fit on a small machine:
218
+
219
+ | Pool | Slices | Encoded reach | Index RAM | Sessions on 8 GB (separate pools) |
220
+ |:----:|:------:|:-------------:|:---------:|:---------------------------------:|
221
+ | **5 GB** *(floor)* | 2.27M | **~1.16B tokens** | ~146 MB | ~13 |
222
+ | 10 GB | 4.55M | **~2.33B tokens** | ~291 MB | ~7 |
223
+ | 15 GB | 6.82M | **~3.49B tokens** | ~436 MB | ~4 |
224
+ | 20 GB | 9.09M | **~4.65B tokens** | ~582 MB | ~3 |
225
+
226
+ Roughly double the session counts on a 16 GB machine. Where the numbers come from: ~2.2 KB per
227
+ slice (a 256-dim vector + compressed text + metadata) ÷ 512 tokens per slice → **~455K slices/GB →
228
+ ~233M tokens of reach per GB**. So `reach ≈ pool_GB × 233M`. At the 5 GB floor that's about
229
+ **9,000×** a 128K window. Bump the pool anytime with `aether-context --pool 20`.
230
+
231
+ **RAM is a formula, not a mystery.** Vectors live on disk (mmap'd) — only the small index graph and
232
+ a hot working set are ever resident:
233
+
234
+ ```
235
+ RAM ≈ ~180 MB base (engine + shared static encoder)
236
+ + ~29 MB per GB of pool (resident index)
237
+ + ~30 MB per active session
238
+ ```
239
+
240
+ **Sharing the pool is the biggest RAM lever.** `--pool-mode separate` *(default)* gives every
241
+ session its own pool and index — fully isolated and private, but you pay one index per session, so
242
+ RAM scales with `N × pool` (that's the last column above). `--pool-mode shared` pays for the index
243
+ **once**; each extra session adds only ~30 MB, so 50–70+ sessions fit and CPU becomes the limit
244
+ instead of memory. The trade-off is that sessions can see each other's context. Use shared for
245
+ related work on one project, separate for unrelated tasks.
246
+
247
+ **How much building is that?** A ~128K window fills after well under an hour of active agent work,
248
+ then starts compacting and forgetting. Assuming a busy coding agent encodes ~300K–1M keep-worthy
249
+ tokens an hour, a 5 GB pool covers on the order of **1,200–3,900 hours** before it even fills —
250
+ weeks of nonstop building. Because the retention policy fades stale slices, the pool never
251
+ hard-stops anyway; it just keeps what's relevant.
252
+
253
+ <div align="center">
254
+ <img width="880" alt="Coding time per pool size" src="https://raw.githubusercontent.com/AetherAI3/Unlimited-Context-LLM/main/assets/coding-time-per-pool.png">
255
+ </div>
256
+
257
+ > **Honest:** that's encoded **reach**, retrieved in slices — not a bigger attention window, and it
258
+ > rides on retrieval hit rate. A bigger pool buys more reachable codebase or corpus *per session*,
259
+ > never more concurrent sessions (those are RAM-bound). `--index tiered` is reserved for a future
260
+ > paged-graph index and currently runs the flat index — it does not yet reduce resident RAM.
261
+
262
+ ## Commands
263
+
264
+ | Command | What it's for |
265
+ |---|---|
266
+ | `aether-context setup` | **Start here.** Guided first run: size the pool, check your model, verify the engine. |
267
+ | `aether-context init` | Pick your pool size — the on-disk storage slider — on first run. |
268
+ | `aether-context run "<task>"` | One-shot a task with full reach, then print the result. |
269
+ | `aether-context run "<task>" --no-mpo-chain` | Same, with the MPO context chain disabled (plain cosine). |
270
+ | `aether-context chat` | Open an interactive session; type `/status` anytime, `/clear` to reset. |
271
+ | `aether-context status` | See pool size, slices used, reach, and hit rate at a glance. |
272
+ | `aether-context doctor` | Check Ollama, your model, disk, and RAM before a long run. |
273
+ | `aether-context --pool 20` | Resize the pool anytime (non-destructive re-index). |
274
+
275
+ > **Tip:** run `aether-context doctor` first — it catches the three things that ever go wrong
276
+ > (Ollama down, model not pulled, not enough disk) and prints the exact fix.
277
+
278
+ ## MPO: the context chain
279
+
280
+ Plain semantic search returns isolated nearest-neighbors — the single closest slices, ripped out of
281
+ the thread they belonged to. Recall a fact and you often miss the three slices around it that made
282
+ it make sense.
283
+
284
+ The **MPO context chain** links the session's slices into one connected structure, so when cosine
285
+ pulls an entry slice, the chain pulls in the slices most coupled to it — widening the working set
286
+ with the *connected thread*, not stray hits. Cosine is still the retrieval mechanism; the chain
287
+ assists it.
288
+
289
+ The chain is Aether-tuned, deterministic and fully local — no training, no network. It is purely
290
+ **additive**: it only ever *adds* connected context, never blocks or replaces a hit, and on any
291
+ hiccup it falls back cleanly to plain cosine. In a planted-thread benchmark it lifts
292
+ connected-context recall from **0.15 (cosine alone) to 0.78** — over 5× more of the right thread in
293
+ the window. That result is synthetic so far; see the caveat under [The proof](https://github.com/AetherAI3/Unlimited-Context-LLM#the-proof).
294
+
295
+ On by default:
296
+
297
+ ```python
298
+ Session(model="ollama/qwen2.5", pool_gb=10) # chain on by default
299
+ Session(model="ollama/qwen2.5", pool_gb=10, mpo_chain=False) # plain cosine
300
+ ```
301
+ ```bash
302
+ aether-context run "..." --no-mpo-chain # disable for one run
303
+ ```
304
+
305
+ ## The `aether` coding terminal
306
+
307
+ **`aether`** is an open-source agentic coding terminal that runs on this engine. Turns run on your
308
+ local [Ollama](https://ollama.com) by default — no account, no network; sign in and they switch to
309
+ the Aether cloud API. It ships as its own package:
310
+
311
+ ```bash
312
+ pip install aether-agent # or: npm install -g aether-agents
313
+ ```
314
+
315
+ `aether` opens the REPL, `aether "<prompt>"` is a one-shot turn, and `aether code "<task>"` is an
316
+ autonomous coding run on the Unlimited Context brain (test-gated, git-checkpointed). Full command
317
+ list, slash commands and backend settings live at
318
+ [AetherAI3/aether-agent](https://github.com/AetherAI3/aether-agent).
319
+
320
+ <sub>The `aether_agent/` directory in *this* repo is the Python-native twin — same commands, same
321
+ backend, same tools — kept here for development and deliberately **not** published from this
322
+ package: PyPI's `aether-agent` already owns that import path and the `aether` command, so shipping
323
+ a second copy would silently overwrite it wherever both are installed. From a clone with Ollama up,
324
+ `python -m aether_agent.smoke` runs the SSRF guard, a real local turn, a web search and fetch, and
325
+ the cloud path when signed in.</sub>
326
+
327
+ ## Safety and use policy
328
+
329
+ Giving a model durable memory is powerful, and the failure modes are real: runaway agents,
330
+ grounding drift, an agent's own notes hardening into its rules. What those are and what we do about
331
+ them is written up in
332
+ **[Ethical & Safety Measures](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/SAFETY.md)**.
333
+ Use of the project is governed by the
334
+ **[Acceptable Use Policy](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/USE_POLICY.md)**
335
+ — by using the project you agree to and are bound by its terms.
336
+
337
+ ## Honest about the word "unlimited"
338
+
339
+ "Unlimited" means **reach, not attention.** Your model keeps its native window; the engine makes it
340
+ *reach* a billion-token pool in slices, via fast retrieval. The whole thing rides on retrieval hit
341
+ rate — when that's high, and the loader is built to keep it high, the pool feels like one seamless
342
+ context. When it isn't, you get a miss, and a miss looks like forgetting. The measured evidence for
343
+ all of this is in [The proof](https://github.com/AetherAI3/Unlimited-Context-LLM#the-proof), caveats included.
344
+
345
+ ## Contributing
346
+
347
+ **PRs and issues are welcome** — start with
348
+ [CONTRIBUTING.md](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/CONTRIBUTING.md).
349
+ There are open issues tagged
350
+ [good first issue](https://github.com/AetherAI3/Unlimited-Context-LLM/issues?q=is%3Aissue+is%3Aopen+label%3A%22good+first+issue%22)
351
+ and [help wanted](https://github.com/AetherAI3/Unlimited-Context-LLM/issues?q=is%3Aissue+is%3Aopen+label%3A%22help+wanted%22)
352
+ right now, including an LM Studio backend, a Windows quickstart, and a recall-quality benchmark at
353
+ 100K / 1M / 10M tokens.
354
+
355
+ Runnable examples live in
356
+ [`examples/`](https://github.com/AetherAI3/Unlimited-Context-LLM/tree/main/examples) — start with
357
+ [`quickstart.py`](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/examples/quickstart.py),
358
+ then [`coding_agent.py`](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/examples/coding_agent.py).
359
+
360
+ If the engine earns its place in your setup, **a star helps other people find it.**
361
+
362
+ ## Citation
363
+
364
+ If Unlimited Context helps your work, please cite it. Built and maintained by **Aether AI**.
365
+
366
+ ```bibtex
367
+ @software{unlimited_context_2026,
368
+ title = {Unlimited Context (aether-context): virtual memory for LLM attention},
369
+ author = {Barrante, Brandon},
370
+ organization = {Aether AI},
371
+ year = {2026},
372
+ url = {https://github.com/AetherAI3/Unlimited-Context-LLM},
373
+ license = {Apache-2.0}
374
+ }
375
+ ```
376
+
377
+ GitHub's "Cite this repository" button reads
378
+ [`CITATION.cff`](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/CITATION.cff) directly.
379
+
380
+ ## License
381
+
382
+ **Apache-2.0.** Use it, fork it, ship it in your product.
383
+
384
+ ---
385
+
386
+ <div align="center">
387
+
388
+ Built by **Aether AI** · [aethersystems.net](https://aethersystems.net)
389
+
390
+ <img width="880" alt="Aether" src="https://raw.githubusercontent.com/AetherAI3/Unlimited-Context-LLM/main/assets/aether-footer.jpg">
391
+
392
+ *Unbounded reach for the model you already run.*
393
+
394
+ </div>