aether-context 0.3.0__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (108) hide show
  1. aether_context-0.4.0/PKG-INFO +398 -0
  2. aether_context-0.4.0/README.md +348 -0
  3. {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/__init__.py +1 -1
  4. aether_context-0.4.0/aether_context/contracts/__init__.py +534 -0
  5. aether_context-0.4.0/aether_context/contracts/schema-v1.json +2201 -0
  6. aether_context-0.4.0/aether_context/crypto.py +105 -0
  7. aether_context-0.4.0/aether_context/engine.py +1637 -0
  8. aether_context-0.4.0/aether_context/policy/__init__.py +66 -0
  9. aether_context-0.4.0/aether_context/retention.py +647 -0
  10. aether_context-0.4.0/aether_context/scale.py +298 -0
  11. aether_context-0.4.0/aether_context/service/__init__.py +279 -0
  12. aether_context-0.4.0/aether_context/storage_v2/__init__.py +125 -0
  13. aether_context-0.4.0/aether_context.egg-info/PKG-INFO +398 -0
  14. {aether_context-0.3.0 → aether_context-0.4.0}/aether_context.egg-info/SOURCES.txt +14 -1
  15. {aether_context-0.3.0 → aether_context-0.4.0}/aether_context.egg-info/entry_points.txt +1 -0
  16. {aether_context-0.3.0 → aether_context-0.4.0}/aether_context.egg-info/requires.txt +6 -0
  17. {aether_context-0.3.0 → aether_context-0.4.0}/pyproject.toml +6 -3
  18. aether_context-0.4.0/tests/test_context_ipc.py +65 -0
  19. aether_context-0.4.0/tests/test_hosted_context.py +423 -0
  20. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_release_parity.py +32 -7
  21. aether_context-0.4.0/tests/test_retention_scale.py +1601 -0
  22. aether_context-0.4.0/tests/test_workflow_yaml.py +85 -0
  23. aether_context-0.3.0/PKG-INFO +0 -429
  24. aether_context-0.3.0/README.md +0 -384
  25. aether_context-0.3.0/aether_context.egg-info/PKG-INFO +0 -429
  26. {aether_context-0.3.0 → aether_context-0.4.0}/LICENSE +0 -0
  27. {aether_context-0.3.0 → aether_context-0.4.0}/NOTICE.md +0 -0
  28. {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/_log.py +0 -0
  29. {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/cli.py +0 -0
  30. {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/config.py +0 -0
  31. {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/context_pool.py +0 -0
  32. {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/encoder.py +0 -0
  33. {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/errors.py +0 -0
  34. {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/local_llm.py +0 -0
  35. {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/mpo.py +0 -0
  36. {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/py.typed +0 -0
  37. {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/quantize.py +0 -0
  38. {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/session.py +0 -0
  39. {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/slice_loader.py +0 -0
  40. {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/tokenizer.py +0 -0
  41. {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/ui.py +0 -0
  42. {aether_context-0.3.0 → aether_context-0.4.0}/aether_context/witness.py +0 -0
  43. {aether_context-0.3.0 → aether_context-0.4.0}/aether_context.egg-info/dependency_links.txt +0 -0
  44. {aether_context-0.3.0 → aether_context-0.4.0}/aether_context.egg-info/top_level.txt +0 -0
  45. {aether_context-0.3.0 → aether_context-0.4.0}/setup.cfg +0 -0
  46. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_aether_agent.py +0 -0
  47. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agent_commands.py +0 -0
  48. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agent_config.py +0 -0
  49. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agent_library.py +0 -0
  50. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agent_profile.py +0 -0
  51. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agent_profile_commands.py +0 -0
  52. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agent_runner.py +0 -0
  53. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agent_runner_define_command.py +0 -0
  54. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agent_runner_lock.py +0 -0
  55. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agent_slash.py +0 -0
  56. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agent_slash_agents_columns.py +0 -0
  57. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agent_slash_cmd.py +0 -0
  58. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agent_store.py +0 -0
  59. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_agents_view.py +0 -0
  60. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_api_eval.py +0 -0
  61. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_auth.py +0 -0
  62. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_brains.py +0 -0
  63. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_bridge.py +0 -0
  64. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_cli.py +0 -0
  65. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_cli_disk.py +0 -0
  66. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_cli_dispatch.py +0 -0
  67. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_cli_doctor.py +0 -0
  68. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_cli_setup.py +0 -0
  69. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_cli_surface.py +0 -0
  70. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_config.py +0 -0
  71. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_context_pool.py +0 -0
  72. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_encoder.py +0 -0
  73. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_end_to_end.py +0 -0
  74. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_errors.py +0 -0
  75. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_hardware.py +0 -0
  76. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_install_security.py +0 -0
  77. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_local_llm.py +0 -0
  78. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_local_llm_openai.py +0 -0
  79. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_moat_seal.py +0 -0
  80. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_mpo.py +0 -0
  81. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_multi_runner.py +0 -0
  82. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_ollama_ctl.py +0 -0
  83. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_ollama_integration.py +0 -0
  84. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_onboarding.py +0 -0
  85. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_pool_quantize.py +0 -0
  86. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_profiles.py +0 -0
  87. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_progress.py +0 -0
  88. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_protocol_lockstep.py +0 -0
  89. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_quantize.py +0 -0
  90. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_repl_agents.py +0 -0
  91. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_repl_multi.py +0 -0
  92. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_repl_preflight.py +0 -0
  93. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_run_agent_events_system.py +0 -0
  94. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_safety.py +0 -0
  95. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_session.py +0 -0
  96. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_session_fallback.py +0 -0
  97. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_slash.py +0 -0
  98. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_slash_agents.py +0 -0
  99. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_slash_custom_invoke.py +0 -0
  100. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_slash_ollama.py +0 -0
  101. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_slice_loader.py +0 -0
  102. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_smoke.py +0 -0
  103. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_tokenizer.py +0 -0
  104. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_toolparse.py +0 -0
  105. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_transport.py +0 -0
  106. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_ui.py +0 -0
  107. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_web_tools.py +0 -0
  108. {aether_context-0.3.0 → aether_context-0.4.0}/tests/test_witness.py +0 -0
@@ -0,0 +1,398 @@
1
+ Metadata-Version: 2.4
2
+ Name: aether-context
3
+ Version: 0.4.0
4
+ Summary: Unlimited Context — virtual memory for an LLM's attention. Local-first, numpy-only core.
5
+ Author: Aether AI
6
+ Maintainer: Aether AI
7
+ License: Apache-2.0
8
+ Project-URL: Homepage, https://github.com/AetherAI3/Unlimited-Context-LLM
9
+ Project-URL: Repository, https://github.com/AetherAI3/Unlimited-Context-LLM.git
10
+ Project-URL: Documentation, https://github.com/AetherAI3/Unlimited-Context-LLM#readme
11
+ Project-URL: Issues, https://github.com/AetherAI3/Unlimited-Context-LLM/issues
12
+ Project-URL: Changelog, https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/CHANGELOG.md
13
+ Keywords: llm,context,retrieval,ollama,local,rag,agents
14
+ Classifier: Development Status :: 4 - Beta
15
+ Classifier: Intended Audience :: Developers
16
+ Classifier: License :: OSI Approved :: Apache Software License
17
+ Classifier: Operating System :: OS Independent
18
+ Classifier: Programming Language :: Python :: 3
19
+ Classifier: Programming Language :: Python :: 3.10
20
+ Classifier: Programming Language :: Python :: 3.11
21
+ Classifier: Programming Language :: Python :: 3.12
22
+ Classifier: Programming Language :: Python :: 3.13
23
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
24
+ Requires-Python: <3.15,>=3.10
25
+ Description-Content-Type: text/markdown
26
+ License-File: LICENSE
27
+ License-File: NOTICE.md
28
+ Requires-Dist: numpy<3.0,>=1.24
29
+ Provides-Extra: ollama
30
+ Provides-Extra: llamacpp
31
+ Requires-Dist: llama-cpp-python>=0.2; extra == "llamacpp"
32
+ Provides-Extra: hf
33
+ Requires-Dist: transformers>=4.40; extra == "hf"
34
+ Requires-Dist: torch>=2.2; extra == "hf"
35
+ Provides-Extra: fast
36
+ Requires-Dist: hnswlib>=0.8; extra == "fast"
37
+ Provides-Extra: hosted
38
+ Requires-Dist: pydantic<3,>=2.10; extra == "hosted"
39
+ Requires-Dist: cryptography<51,>=44; extra == "hosted"
40
+ Provides-Extra: all
41
+ Requires-Dist: aether-context[fast,hf,llamacpp]; extra == "all"
42
+ Provides-Extra: dev
43
+ Requires-Dist: pytest>=8; extra == "dev"
44
+ Requires-Dist: pytest-cov; extra == "dev"
45
+ Requires-Dist: ruff>=0.5; extra == "dev"
46
+ Requires-Dist: mypy>=1.10; extra == "dev"
47
+ Requires-Dist: pyyaml>=6; extra == "dev"
48
+ Requires-Dist: aether-context[hosted]; extra == "dev"
49
+ Dynamic: license-file
50
+
51
+ <div align="center">
52
+
53
+ # ⚡ Unlimited Context
54
+
55
+ **Virtual memory for an LLM's attention.** Keep a billion-token pool on your own disk; the model
56
+ reaches it in slices, one small window at a time. Local-first, offline, free.
57
+
58
+ <img alt="npx aether-context: guided setup, a clean doctor check, and pool status - real terminal output" width="716" src="https://raw.githubusercontent.com/AetherAI3/Unlimited-Context-LLM/main/assets/demo.gif">
59
+
60
+ [![PyPI](https://img.shields.io/pypi/v/aether-context?style=flat-square&logo=pypi&logoColor=white&color=06b6d4)](https://pypi.org/project/aether-context/)
61
+ [![npm](https://img.shields.io/npm/v/aether-context?style=flat-square&logo=npm&logoColor=white&color=cb3837)](https://www.npmjs.com/package/aether-context)
62
+ [![License](https://img.shields.io/badge/License-Apache_2.0-06b6d4?style=flat-square)](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/LICENSE)
63
+ [![Python](https://img.shields.io/badge/Python-3.10%2B-14b8a6?style=flat-square&logo=python&logoColor=white)](https://www.python.org)
64
+ [![Built by Aether](https://img.shields.io/badge/Built_by-Aether-7c3aed?style=flat-square)](https://aethersystems.net)
65
+ [![Stars](https://img.shields.io/github/stars/AetherAI3/Unlimited-Context-LLM?style=flat-square&logo=github&color=eab308)](https://github.com/AetherAI3/Unlimited-Context-LLM/stargazers)
66
+
67
+ [Site](https://aetherai3.github.io/Unlimited-Context-LLM/) ·
68
+ [Install](https://github.com/AetherAI3/Unlimited-Context-LLM#install) ·
69
+ [How it works](https://github.com/AetherAI3/Unlimited-Context-LLM#how-it-works) ·
70
+ [The proof](https://github.com/AetherAI3/Unlimited-Context-LLM#the-proof) ·
71
+ [Sizing](https://github.com/AetherAI3/Unlimited-Context-LLM#sizing-disk-reach-and-ram) ·
72
+ [Safety](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/SAFETY.md)
73
+
74
+ </div>
75
+
76
+ ---
77
+
78
+ > **Your context window didn't get bigger. Its *reach* did.**
79
+ > The model keeps its small window. The engine keeps a vast store on your disk and pulls the
80
+ > *right slice* back in while the model reasons. A small local model stays coherent across runs
81
+ > that would blow past any context window.
82
+
83
+ ## Install
84
+
85
+ ```bash
86
+ pip install aether-context
87
+ aether-context setup
88
+ ```
89
+
90
+ `setup` sizes the pool, checks for a local model, and verifies the engine end to end. It works
91
+ with no daemon, no network and no model pulled — the check runs against the built-in mock model.
92
+
93
+ ```python
94
+ from aether_context import Session
95
+
96
+ s = Session(model="ollama/qwen2.5", pool_gb=5)
97
+ s.run("Build me a full-stack weightlifting tracker app.")
98
+ # runs long. stays coherent. walk away.
99
+ ```
100
+
101
+ Prefer npm? Same software, same release. The npm package is a launcher that installs the Python
102
+ engine into a private virtualenv for you (it needs Python 3.10+ on your PATH):
103
+
104
+ ```bash
105
+ npx aether-context setup
106
+ ```
107
+
108
+ <details>
109
+ <summary>Other install routes</summary>
110
+
111
+ ```bash
112
+ # Straight from source, always the latest main:
113
+ pip install git+https://github.com/AetherAI3/Unlimited-Context-LLM.git
114
+
115
+ # Isolated, if you only want the CLI:
116
+ pipx install aether-context
117
+ ```
118
+
119
+ The distribution name is **`aether-context`** — `pip install unlimited-context` is not this
120
+ package.
121
+ </details>
122
+
123
+ That's the whole thing. One small model, one command, a billion tokens of reach behind it.
124
+
125
+ ## The problem
126
+
127
+ Long agentic runs all die the same way. The model fills its window, starts **compressing** its own
128
+ history, silently drops the one detail that mattered three steps ago — and drifts. You've seen it:
129
+ the runaway PR, the agent that rewrites a function it already wrote, the build that falls apart at
130
+ hour two. Bigger windows just delay it, and a crammed 1M-token window **rots in the middle** anyway.
131
+
132
+ The fix isn't a bigger window. It's to stop throwing the overflow away. Instead of summarizing what
133
+ spills over, Unlimited Context **encodes** it to a local pool on your disk and **recovers** the
134
+ right slice exactly when it's needed. Nothing load-bearing is silently lost.
135
+
136
+ <p align="center"><strong>Compress &amp; forget ✗ &nbsp;→&nbsp; Encode &amp; recover ✓</strong></p>
137
+
138
+ ## How it works
139
+
140
+ It's **virtual memory, for attention.** Map it to an OS and it clicks:
141
+
142
+ | OS | Unlimited Context |
143
+ |---|---|
144
+ | RAM | the **resident window** the model sees now (small, fast) |
145
+ | Disk | the **context pool** — your encoded memory (~5 GB ≈ ~1B tokens) |
146
+ | Pager | the **slice loader** — prefetches the next slice from what the model is reasoning about *right now* |
147
+ | Page-replacement | the **retention policy** — useful slices *stay*, stale ones *fade*, anything relevant again comes back |
148
+
149
+ The pager runs concurrently with generation, so most of the fetch hides behind the model's own
150
+ thinking. Full explainer: [`docs/how-it-works.md`](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/docs/how-it-works.md).
151
+
152
+ ## What you get
153
+
154
+ - 🧠 **Unbounded reach** — ~1B tokens of encoded context in ~5 GB on disk; the model reaches it in slices.
155
+ - 🧩 **MPO context chain** — recall pulls the whole connected thread, not isolated nearest-neighbors.
156
+ - 🪟 **Curated beats crammed** — a small, relevant resident window outperforms a stuffed one (no lost-in-the-middle) — and costs less.
157
+ - 🔒 **Local-first** — your context never leaves your machine. Free storage, full privacy, works offline.
158
+ - 🤖 **Any model** — Llama, Qwen, Mistral, Phi — via Ollama, llama.cpp, or Hugging Face, or your own API-backed model.
159
+ - 📉 **Coherence you can measure** — the head-to-head is committed: same model, engine on vs off.
160
+
161
+ ## The proof
162
+
163
+ Not a synthetic micro-benchmark — a **real, paid, end-to-end run.** A reasoning model
164
+ (`deepseek-v4-pro`, via OpenRouter) driven through a **40-turn agent session that overflows its
165
+ window** (2,000-token window, 60 real `microsoft/vscode` issues), measured **engine off vs on** —
166
+ one live run, **$0.19**, 2026-06-14.
167
+
168
+ - **The model stops forgetting.** Recall of early facts after they fall out of the window:
169
+ **0.15 → 1.00.** The baseline drifts and forgets; the engine holds every early fact — zero drift.
170
+ - **Failure turns into success on the real work.** Tasks completed correctly: **3 / 20 → 20 / 20.**
171
+ - **Cheaper, not just better.** **−24%** total cost, **−54%** in the back half — the engine sends a
172
+ compact recalled slice instead of dragging the whole transcript into every call.
173
+
174
+ | Metric | Off (baseline) | On (engine) | Change |
175
+ |---|:---:|:---:|:---:|
176
+ | **Recall coherence** (early facts still correct) | 0.15 | **1.00** | **6.7×** |
177
+ | **Work outcome** (tasks done right) | 3 / 20 | **20 / 20** | **3 → 20** |
178
+ | **Cost — full session** | $0.0711 | **$0.0542** | **−24%** |
179
+ | **Cost — back half (recall phase)** | $0.00117/turn | **$0.00053/turn** | **−54%** |
180
+
181
+ <p align="center">
182
+ <img alt="Cumulative cost and recall coherence vs turn — engine off vs on" width="780"
183
+ src="https://raw.githubusercontent.com/AetherAI3/Unlimited-Context-LLM/main/docs/benchmarks/artifacts/2026-06-14-deepseek-v4-pro/api_eval_plot.png">
184
+ </p>
185
+
186
+ **Committed data:** [full write-up](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/docs/benchmarks/2026-06-14-deepseek-v4-pro-session-eval.md) ·
187
+ [raw artifacts](https://github.com/AetherAI3/Unlimited-Context-LLM/tree/main/docs/benchmarks/artifacts/2026-06-14-deepseek-v4-pro)
188
+ (`api_eval_results.json`, `api_eval_series.csv`, `api_eval_plot.png`, `RESULTS.md`) · reproduce with
189
+ `python -m bench.api_eval --model deepseek/deepseek-v4-pro --repo microsoft/vscode --arms off,on,on_chain --plot`
190
+
191
+ <sub>**Scope, honestly:** this run used a hosted reasoning model, not a local one — the mechanism is
192
+ backend-agnostic, but the headline number is not a local number. It measures the **engine**
193
+ (retrieve-on-overflow memory), not the MPO chain: on this single-fact recall task the chain
194
+ **ties** plain recall (both 1.00), and its multi-slice edge is **synthetic-only so far**
195
+ (`bench/chain_recall.py`: connected-context recall 0.15 → 0.78), with the live `thread` run
196
+ **pending**, not yet claimed. The 2,000-token window is deliberately tiny to force overflow, so a
197
+ realistic window shows a smaller (still real) gain. N = 20 recall turns, single run.</sub>
198
+
199
+ ## Sizing: disk, reach and RAM
200
+
201
+ First run drops you into a slider — pick how much your model gets to remember:
202
+
203
+ ```text
204
+ $ aether-context init
205
+ ──────────────────────────────────────────────────────────────────
206
+ ⚡ choose your context pool encoded reach · not a window
207
+ ──────────────────────────────────────────────────────────────────
208
+ ▸ 5 GB ████░░░░░░░░░░░░ ~1.16B tokens a big project (floor)
209
+ 10 GB ████████░░░░░░░░ ~2.33B tokens a large monorepo + docs
210
+ 15 GB ████████████░░░░ ~3.49B tokens multiple repos / long runs
211
+ 20 GB ████████████████ ~4.65B tokens massive corpus / power user
212
+ ──────────────────────────────────────────────────────────────────
213
+ reach ≈ pool_GB × 233M tokens custom: --pool 12 (any size ≥ 5 GB)
214
+ ↑/↓ slide ↵ confirm
215
+
216
+ pool [5]: 10
217
+ ✓ 10 GB → your model can now reach ~2.33 billion tokens
218
+ ```
219
+
220
+ One table for the whole trade-off — disk in, reach out, RAM cost, and how many isolated sessions
221
+ fit on a small machine:
222
+
223
+ | Pool | Slices | Encoded reach | Index RAM | Sessions on 8 GB (separate pools) |
224
+ |:----:|:------:|:-------------:|:---------:|:---------------------------------:|
225
+ | **5 GB** *(floor)* | 2.27M | **~1.16B tokens** | ~146 MB | ~13 |
226
+ | 10 GB | 4.55M | **~2.33B tokens** | ~291 MB | ~7 |
227
+ | 15 GB | 6.82M | **~3.49B tokens** | ~436 MB | ~4 |
228
+ | 20 GB | 9.09M | **~4.65B tokens** | ~582 MB | ~3 |
229
+
230
+ Roughly double the session counts on a 16 GB machine. Where the numbers come from: ~2.2 KB per
231
+ slice (a 256-dim vector + compressed text + metadata) ÷ 512 tokens per slice → **~455K slices/GB →
232
+ ~233M tokens of reach per GB**. So `reach ≈ pool_GB × 233M`. At the 5 GB floor that's about
233
+ **9,000×** a 128K window. Bump the pool anytime with `aether-context --pool 20`.
234
+
235
+ **RAM is a formula, not a mystery.** Vectors live on disk (mmap'd) — only the small index graph and
236
+ a hot working set are ever resident:
237
+
238
+ ```
239
+ RAM ≈ ~180 MB base (engine + shared static encoder)
240
+ + ~29 MB per GB of pool (resident index)
241
+ + ~30 MB per active session
242
+ ```
243
+
244
+ **Sharing the pool is the biggest RAM lever.** `--pool-mode separate` *(default)* gives every
245
+ session its own pool and index — fully isolated and private, but you pay one index per session, so
246
+ RAM scales with `N × pool` (that's the last column above). `--pool-mode shared` pays for the index
247
+ **once**; each extra session adds only ~30 MB, so 50–70+ sessions fit and CPU becomes the limit
248
+ instead of memory. The trade-off is that sessions can see each other's context. Use shared for
249
+ related work on one project, separate for unrelated tasks.
250
+
251
+ **How much building is that?** A ~128K window fills after well under an hour of active agent work,
252
+ then starts compacting and forgetting. Assuming a busy coding agent encodes ~300K–1M keep-worthy
253
+ tokens an hour, a 5 GB pool covers on the order of **1,200–3,900 hours** before it even fills —
254
+ weeks of nonstop building. Because the retention policy fades stale slices, the pool never
255
+ hard-stops anyway; it just keeps what's relevant.
256
+
257
+ <div align="center">
258
+ <img width="880" alt="Coding time per pool size" src="https://raw.githubusercontent.com/AetherAI3/Unlimited-Context-LLM/main/assets/coding-time-per-pool.png">
259
+ </div>
260
+
261
+ > **Honest:** that's encoded **reach**, retrieved in slices — not a bigger attention window, and it
262
+ > rides on retrieval hit rate. A bigger pool buys more reachable codebase or corpus *per session*,
263
+ > never more concurrent sessions (those are RAM-bound). `--index tiered` is reserved for a future
264
+ > paged-graph index and currently runs the flat index — it does not yet reduce resident RAM.
265
+
266
+ ## Commands
267
+
268
+ | Command | What it's for |
269
+ |---|---|
270
+ | `aether-context setup` | **Start here.** Guided first run: size the pool, check your model, verify the engine. |
271
+ | `aether-context init` | Pick your pool size — the on-disk storage slider — on first run. |
272
+ | `aether-context run "<task>"` | One-shot a task with full reach, then print the result. |
273
+ | `aether-context run "<task>" --no-mpo-chain` | Same, with the MPO context chain disabled (plain cosine). |
274
+ | `aether-context chat` | Open an interactive session; type `/status` anytime, `/clear` to reset. |
275
+ | `aether-context status` | See pool size, slices used, reach, and hit rate at a glance. |
276
+ | `aether-context doctor` | Check Ollama, your model, disk, and RAM before a long run. |
277
+ | `aether-context --pool 20` | Resize the pool anytime (non-destructive re-index). |
278
+
279
+ > **Tip:** run `aether-context doctor` first — it catches the three things that ever go wrong
280
+ > (Ollama down, model not pulled, not enough disk) and prints the exact fix.
281
+
282
+ ## MPO: the context chain
283
+
284
+ Plain semantic search returns isolated nearest-neighbors — the single closest slices, ripped out of
285
+ the thread they belonged to. Recall a fact and you often miss the three slices around it that made
286
+ it make sense.
287
+
288
+ The **MPO context chain** links the session's slices into one connected structure, so when cosine
289
+ pulls an entry slice, the chain pulls in the slices most coupled to it — widening the working set
290
+ with the *connected thread*, not stray hits. Cosine is still the retrieval mechanism; the chain
291
+ assists it.
292
+
293
+ The chain is Aether-tuned, deterministic and fully local — no training, no network. It is purely
294
+ **additive**: it only ever *adds* connected context, never blocks or replaces a hit, and on any
295
+ hiccup it falls back cleanly to plain cosine. In a planted-thread benchmark it lifts
296
+ connected-context recall from **0.15 (cosine alone) to 0.78** — over 5× more of the right thread in
297
+ the window. That result is synthetic so far; see the caveat under [The proof](https://github.com/AetherAI3/Unlimited-Context-LLM#the-proof).
298
+
299
+ On by default:
300
+
301
+ ```python
302
+ Session(model="ollama/qwen2.5", pool_gb=10) # chain on by default
303
+ Session(model="ollama/qwen2.5", pool_gb=10, mpo_chain=False) # plain cosine
304
+ ```
305
+ ```bash
306
+ aether-context run "..." --no-mpo-chain # disable for one run
307
+ ```
308
+
309
+ ## The `aether` coding terminal
310
+
311
+ **`aether`** is an open-source agentic coding terminal that runs on this engine. Turns run on your
312
+ local [Ollama](https://ollama.com) by default — no account, no network; sign in and they switch to
313
+ the Aether cloud API. It ships as its own package:
314
+
315
+ ```bash
316
+ pip install aether-agent # or: npm install -g aether-agents
317
+ ```
318
+
319
+ `aether` opens the REPL, `aether "<prompt>"` is a one-shot turn, and `aether code "<task>"` is an
320
+ autonomous coding run on the Unlimited Context brain (test-gated, git-checkpointed). Full command
321
+ list, slash commands and backend settings live at
322
+ [AetherAI3/aether-agent](https://github.com/AetherAI3/aether-agent).
323
+
324
+ <sub>The `aether_agent/` directory in *this* repo is the Python-native twin — same commands, same
325
+ backend, same tools — kept here for development and deliberately **not** published from this
326
+ package: PyPI's `aether-agent` already owns that import path and the `aether` command, so shipping
327
+ a second copy would silently overwrite it wherever both are installed. From a clone with Ollama up,
328
+ `python -m aether_agent.smoke` runs the SSRF guard, a real local turn, a web search and fetch, and
329
+ the cloud path when signed in.</sub>
330
+
331
+ ## Safety and use policy
332
+
333
+ Giving a model durable memory is powerful, and the failure modes are real: runaway agents,
334
+ grounding drift, an agent's own notes hardening into its rules. What those are and what we do about
335
+ them is written up in
336
+ **[Ethical & Safety Measures](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/SAFETY.md)**.
337
+ Use of the project is governed by the
338
+ **[Acceptable Use Policy](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/USE_POLICY.md)**
339
+ — by using the project you agree to and are bound by its terms.
340
+
341
+ ## Honest about the word "unlimited"
342
+
343
+ "Unlimited" means **reach, not attention.** Your model keeps its native window; the engine makes it
344
+ *reach* a billion-token pool in slices, via fast retrieval. The whole thing rides on retrieval hit
345
+ rate — when that's high, and the loader is built to keep it high, the pool feels like one seamless
346
+ context. When it isn't, you get a miss, and a miss looks like forgetting. The measured evidence for
347
+ all of this is in [The proof](https://github.com/AetherAI3/Unlimited-Context-LLM#the-proof), caveats included.
348
+
349
+ ## Contributing
350
+
351
+ **PRs and issues are welcome** — start with
352
+ [CONTRIBUTING.md](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/CONTRIBUTING.md).
353
+ There are open issues tagged
354
+ [good first issue](https://github.com/AetherAI3/Unlimited-Context-LLM/issues?q=is%3Aissue+is%3Aopen+label%3A%22good+first+issue%22)
355
+ and [help wanted](https://github.com/AetherAI3/Unlimited-Context-LLM/issues?q=is%3Aissue+is%3Aopen+label%3A%22help+wanted%22)
356
+ right now, including an LM Studio backend, a Windows quickstart, and a recall-quality benchmark at
357
+ 100K / 1M / 10M tokens.
358
+
359
+ Runnable examples live in
360
+ [`examples/`](https://github.com/AetherAI3/Unlimited-Context-LLM/tree/main/examples) — start with
361
+ [`quickstart.py`](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/examples/quickstart.py),
362
+ then [`coding_agent.py`](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/examples/coding_agent.py).
363
+
364
+ If the engine earns its place in your setup, **a star helps other people find it.**
365
+
366
+ ## Citation
367
+
368
+ If Unlimited Context helps your work, please cite it. Built and maintained by **Aether AI**.
369
+
370
+ ```bibtex
371
+ @software{unlimited_context_2026,
372
+ title = {Unlimited Context (aether-context): virtual memory for LLM attention},
373
+ author = {Barrante, Brandon},
374
+ organization = {Aether AI},
375
+ year = {2026},
376
+ url = {https://github.com/AetherAI3/Unlimited-Context-LLM},
377
+ license = {Apache-2.0}
378
+ }
379
+ ```
380
+
381
+ GitHub's "Cite this repository" button reads
382
+ [`CITATION.cff`](https://github.com/AetherAI3/Unlimited-Context-LLM/blob/main/CITATION.cff) directly.
383
+
384
+ ## License
385
+
386
+ **Apache-2.0.** Use it, fork it, ship it in your product.
387
+
388
+ ---
389
+
390
+ <div align="center">
391
+
392
+ Built by **Aether AI** · [aethersystems.net](https://aethersystems.net)
393
+
394
+ <img width="880" alt="Aether" src="https://raw.githubusercontent.com/AetherAI3/Unlimited-Context-LLM/main/assets/aether-footer.jpg">
395
+
396
+ *Unbounded reach for the model you already run.*
397
+
398
+ </div>