mrail 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mrail-0.4.0/LICENSE +17 -0
- mrail-0.4.0/PKG-INFO +86 -0
- mrail-0.4.0/README.md +67 -0
- mrail-0.4.0/mrail/__init__.py +23 -0
- mrail-0.4.0/mrail/__main__.py +26 -0
- mrail-0.4.0/mrail/core.py +415 -0
- mrail-0.4.0/mrail/drafts.py +120 -0
- mrail-0.4.0/mrail.egg-info/PKG-INFO +86 -0
- mrail-0.4.0/mrail.egg-info/SOURCES.txt +13 -0
- mrail-0.4.0/mrail.egg-info/dependency_links.txt +1 -0
- mrail-0.4.0/mrail.egg-info/requires.txt +1 -0
- mrail-0.4.0/mrail.egg-info/top_level.txt +1 -0
- mrail-0.4.0/pyproject.toml +27 -0
- mrail-0.4.0/setup.cfg +4 -0
- mrail-0.4.0/tests/test_roundtrip.py +72 -0
mrail-0.4.0/LICENSE
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
Apache License
|
|
2
|
+
Version 2.0, January 2004
|
|
3
|
+
http://www.apache.org/licenses/
|
|
4
|
+
|
|
5
|
+
Copyright 2026 David Tom Foss
|
|
6
|
+
|
|
7
|
+
Licensed under the Apache License, Version 2.0 (the "License");
|
|
8
|
+
you may not use this file except in compliance with the License.
|
|
9
|
+
You may obtain a copy of the License at
|
|
10
|
+
|
|
11
|
+
http://www.apache.org/licenses/LICENSE-2.0
|
|
12
|
+
|
|
13
|
+
Unless required by applicable law or agreed to in writing, software
|
|
14
|
+
distributed under the License is distributed on an "AS IS" BASIS,
|
|
15
|
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
16
|
+
See the License for the specific language governing permissions and
|
|
17
|
+
limitations under the License.
|
mrail-0.4.0/PKG-INFO
ADDED
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: mrail
|
|
3
|
+
Version: 0.4.0
|
|
4
|
+
Summary: Append-only maps of exact language-model runs (.mrail format, reference reader/writer)
|
|
5
|
+
Author-email: David Tom Foss <d.foss@ieee.org>
|
|
6
|
+
License: Apache-2.0
|
|
7
|
+
Project-URL: Homepage, https://github.com/DT-Foss/mrail
|
|
8
|
+
Project-URL: Repository, https://github.com/DT-Foss/mrail
|
|
9
|
+
Project-URL: Specification, https://github.com/DT-Foss/mrail/blob/main/SPEC.md
|
|
10
|
+
Keywords: language models,inference,constant state,compute once,kv cache,maps
|
|
11
|
+
Classifier: License :: OSI Approved :: Apache Software License
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
14
|
+
Requires-Python: >=3.10
|
|
15
|
+
Description-Content-Type: text/markdown
|
|
16
|
+
License-File: LICENSE
|
|
17
|
+
Requires-Dist: numpy
|
|
18
|
+
Dynamic: license-file
|
|
19
|
+
|
|
20
|
+
# mrail — maps of exact language-model runs
|
|
21
|
+
|
|
22
|
+
`.mrail` is an append-only file format for **what a language model has already computed**. For models whose state
|
|
23
|
+
after any prefix has a constant size (state-space models, linear attention, gated delta networks, fixed-state
|
|
24
|
+
attention such as Ward recall), the state is a pure function of the token prefix. A map of earlier runs therefore lets a
|
|
25
|
+
runtime
|
|
26
|
+
|
|
27
|
+
- **answer known contexts without running the model** — the stored continuation *is* the model's own greedy output;
|
|
28
|
+
- **restart exactly** from anchored states at shared prefixes (system prompts, conversation histories), or rebuild any
|
|
29
|
+
state from the stored tokens;
|
|
30
|
+
- **draft** continuations for new contexts, verified by the model in one pass, so the map changes speed, never output;
|
|
31
|
+
- **grow with use** — every served answer is appended; maps split by topic, load and merge like any file.
|
|
32
|
+
|
|
33
|
+
Measured with a compiled Qwen2.5-1.5B-Instruct (constant state of 14.4 MiB) on the Apple Neural Engine (see the papers
|
|
34
|
+
below): known questions in 3.0–4.3 ms instead of 9.7–10.3 s; 2.54 instead of 1.25 accepted tokens per pass on unseen
|
|
35
|
+
coding tasks with a 1.2 MB coding map of 1 641 runs (2.27× greedy decoding); 8.4 MB instead of 893 MB for the anchors of
|
|
36
|
+
the same runs through delta coding and stations, 9.6 kB for their tokens alone.
|
|
37
|
+
|
|
38
|
+
**Tracks** (format kind 5, new in 0.4) record a run without its context: the context by its chain hash and a source
|
|
39
|
+
name (an eval item, a document id), the model's answer copy-coded against it. On 652 runs of benchmark prompts of 131k
|
|
40
|
+
to 1M tokens the maps shrink from 881.4 MB to 89.5 KB, 143 bytes per run of a million tokens, and every track replays
|
|
41
|
+
its answer identically. Long runs are indexed at every 256th prefix, at the end of the prompt and at every answer
|
|
42
|
+
position (4k hash entries instead of a million for a 1M-token run).
|
|
43
|
+
|
|
44
|
+
This repository contains the [format specification](SPEC.md) and a dependency-light reference reader and writer
|
|
45
|
+
(Python, numpy). It does not contain a model runtime.
|
|
46
|
+
|
|
47
|
+
```
|
|
48
|
+
pip install mrail
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
```python
|
|
52
|
+
import mrail
|
|
53
|
+
m = mrail.MapFile("coding.mrail") # opens or creates
|
|
54
|
+
answer = mrail.exact_answer(m, context_tokens, max_new=256, stop={151645})
|
|
55
|
+
m.add_session(prompt_tokens + model_answer_tokens + [next_token], gen_start=len(prompt_tokens))
|
|
56
|
+
tree = m.rail.tree(context_tokens, budget=31) # draft tree [(token, parent)] for verification
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
```
|
|
60
|
+
python -m mrail info coding.mrail
|
|
61
|
+
python -m mrail merge team.mrail alice.mrail bob.mrail
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
## Share your map
|
|
65
|
+
|
|
66
|
+
A map is a file of verified output of one exact model. Maps merge by loading their runs (`python -m mrail merge`), so
|
|
67
|
+
a map of a repository, a documentation set, a tool protocol or a team's daily questions can be shared with everyone who
|
|
68
|
+
runs the same model. Runs from a foreign map are loaded as **drafts only**: they propose continuations, the model
|
|
69
|
+
verifies every token, and only runs produced by the local model answer a context directly. A shared map therefore
|
|
70
|
+
makes inference faster on its topic and cannot change a single output token. The map is written by the model itself
|
|
71
|
+
(the Möbius loop: every verified answer becomes a run, every disagreement becomes a new run), so it grows finer with use
|
|
72
|
+
and never turns into a store of canned answers.
|
|
73
|
+
|
|
74
|
+
Name shared maps by model and topic, e.g. `Qwen2.5-1.5B-Instruct-HCGM.coding.mrail`, and renew them when the model
|
|
75
|
+
version changes.
|
|
76
|
+
|
|
77
|
+
## Papers
|
|
78
|
+
|
|
79
|
+
- D. T. Foss, *Compute Once: Constant-State Language Models Make Answers Reusable and Agents Cheap* (2026), [doi:10.13140/RG.2.2.18140.35202](https://doi.org/10.13140/RG.2.2.18140.35202).
|
|
80
|
+
- D. T. Foss, *The Holographic Causal Graph Machine: Compiling a Pretrained Transformer into a Constant-State Model
|
|
81
|
+
without Training* (2026), [doi:10.13140/RG.2.2.28206.68167](https://doi.org/10.13140/RG.2.2.28206.68167).
|
|
82
|
+
- D. T. Foss, *No GPU, No KV Cache: A Constant-State Language Model on the Neural Engine of a Mac mini* (2026), [doi:10.13140/RG.2.2.31562.12484](https://doi.org/10.13140/RG.2.2.31562.12484).
|
|
83
|
+
|
|
84
|
+
Models and maps: [huggingface.co/tfwnotops](https://huggingface.co/tfwnotops).
|
|
85
|
+
|
|
86
|
+
License: Apache-2.0.
|
mrail-0.4.0/README.md
ADDED
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
# mrail — maps of exact language-model runs
|
|
2
|
+
|
|
3
|
+
`.mrail` is an append-only file format for **what a language model has already computed**. For models whose state
|
|
4
|
+
after any prefix has a constant size (state-space models, linear attention, gated delta networks, fixed-state
|
|
5
|
+
attention such as Ward recall), the state is a pure function of the token prefix. A map of earlier runs therefore lets a
|
|
6
|
+
runtime
|
|
7
|
+
|
|
8
|
+
- **answer known contexts without running the model** — the stored continuation *is* the model's own greedy output;
|
|
9
|
+
- **restart exactly** from anchored states at shared prefixes (system prompts, conversation histories), or rebuild any
|
|
10
|
+
state from the stored tokens;
|
|
11
|
+
- **draft** continuations for new contexts, verified by the model in one pass, so the map changes speed, never output;
|
|
12
|
+
- **grow with use** — every served answer is appended; maps split by topic, load and merge like any file.
|
|
13
|
+
|
|
14
|
+
Measured with a compiled Qwen2.5-1.5B-Instruct (constant state of 14.4 MiB) on the Apple Neural Engine (see the papers
|
|
15
|
+
below): known questions in 3.0–4.3 ms instead of 9.7–10.3 s; 2.54 instead of 1.25 accepted tokens per pass on unseen
|
|
16
|
+
coding tasks with a 1.2 MB coding map of 1 641 runs (2.27× greedy decoding); 8.4 MB instead of 893 MB for the anchors of
|
|
17
|
+
the same runs through delta coding and stations, 9.6 kB for their tokens alone.
|
|
18
|
+
|
|
19
|
+
**Tracks** (format kind 5, new in 0.4) record a run without its context: the context by its chain hash and a source
|
|
20
|
+
name (an eval item, a document id), the model's answer copy-coded against it. On 652 runs of benchmark prompts of 131k
|
|
21
|
+
to 1M tokens the maps shrink from 881.4 MB to 89.5 KB, 143 bytes per run of a million tokens, and every track replays
|
|
22
|
+
its answer identically. Long runs are indexed at every 256th prefix, at the end of the prompt and at every answer
|
|
23
|
+
position (4k hash entries instead of a million for a 1M-token run).
|
|
24
|
+
|
|
25
|
+
This repository contains the [format specification](SPEC.md) and a dependency-light reference reader and writer
|
|
26
|
+
(Python, numpy). It does not contain a model runtime.
|
|
27
|
+
|
|
28
|
+
```
|
|
29
|
+
pip install mrail
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
```python
|
|
33
|
+
import mrail
|
|
34
|
+
m = mrail.MapFile("coding.mrail") # opens or creates
|
|
35
|
+
answer = mrail.exact_answer(m, context_tokens, max_new=256, stop={151645})
|
|
36
|
+
m.add_session(prompt_tokens + model_answer_tokens + [next_token], gen_start=len(prompt_tokens))
|
|
37
|
+
tree = m.rail.tree(context_tokens, budget=31) # draft tree [(token, parent)] for verification
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
```
|
|
41
|
+
python -m mrail info coding.mrail
|
|
42
|
+
python -m mrail merge team.mrail alice.mrail bob.mrail
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
## Share your map
|
|
46
|
+
|
|
47
|
+
A map is a file of verified output of one exact model. Maps merge by loading their runs (`python -m mrail merge`), so
|
|
48
|
+
a map of a repository, a documentation set, a tool protocol or a team's daily questions can be shared with everyone who
|
|
49
|
+
runs the same model. Runs from a foreign map are loaded as **drafts only**: they propose continuations, the model
|
|
50
|
+
verifies every token, and only runs produced by the local model answer a context directly. A shared map therefore
|
|
51
|
+
makes inference faster on its topic and cannot change a single output token. The map is written by the model itself
|
|
52
|
+
(the Möbius loop: every verified answer becomes a run, every disagreement becomes a new run), so it grows finer with use
|
|
53
|
+
and never turns into a store of canned answers.
|
|
54
|
+
|
|
55
|
+
Name shared maps by model and topic, e.g. `Qwen2.5-1.5B-Instruct-HCGM.coding.mrail`, and renew them when the model
|
|
56
|
+
version changes.
|
|
57
|
+
|
|
58
|
+
## Papers
|
|
59
|
+
|
|
60
|
+
- D. T. Foss, *Compute Once: Constant-State Language Models Make Answers Reusable and Agents Cheap* (2026), [doi:10.13140/RG.2.2.18140.35202](https://doi.org/10.13140/RG.2.2.18140.35202).
|
|
61
|
+
- D. T. Foss, *The Holographic Causal Graph Machine: Compiling a Pretrained Transformer into a Constant-State Model
|
|
62
|
+
without Training* (2026), [doi:10.13140/RG.2.2.28206.68167](https://doi.org/10.13140/RG.2.2.28206.68167).
|
|
63
|
+
- D. T. Foss, *No GPU, No KV Cache: A Constant-State Language Model on the Neural Engine of a Mac mini* (2026), [doi:10.13140/RG.2.2.31562.12484](https://doi.org/10.13140/RG.2.2.31562.12484).
|
|
64
|
+
|
|
65
|
+
Models and maps: [huggingface.co/tfwnotops](https://huggingface.co/tfwnotops).
|
|
66
|
+
|
|
67
|
+
License: Apache-2.0.
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
"""mrail — append-only maps of exact language-model runs (.mrail v3). See SPEC.md."""
|
|
2
|
+
from .core import MapFile, chain, snapshot_from_layers, MAGIC
|
|
3
|
+
from .drafts import DraftRail
|
|
4
|
+
|
|
5
|
+
Monorail = MapFile
|
|
6
|
+
__all__ = ["MapFile", "Monorail", "DraftRail", "chain", "snapshot_from_layers", "MAGIC"]
|
|
7
|
+
__version__ = "0.4.0"
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def exact_answer(m: MapFile, ctx, max_new: int, stop=()):
|
|
11
|
+
"""The model's own continuation of ctx if ctx lies on a track, else None."""
|
|
12
|
+
lcp, sid, *_ = m.locate(list(ctx))
|
|
13
|
+
if lcp != len(ctx) or sid < 0:
|
|
14
|
+
return None
|
|
15
|
+
out = []
|
|
16
|
+
seq = m.sessions[sid]
|
|
17
|
+
if seq is None: # a track: the answer is coded against the context
|
|
18
|
+
seq = list(ctx) + m.track_answer(sid, ctx)
|
|
19
|
+
for t in seq[len(ctx):len(ctx) + max_new]:
|
|
20
|
+
out.append(t)
|
|
21
|
+
if t in stop:
|
|
22
|
+
break
|
|
23
|
+
return out or None
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""python -m mrail info MAP | python -m mrail merge OUT MAP [MAP ...]"""
|
|
2
|
+
import sys
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from . import MapFile
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def main(argv):
|
|
8
|
+
if len(argv) >= 2 and argv[0] == "info":
|
|
9
|
+
m = MapFile(argv[1])
|
|
10
|
+
n_full = sum(1 for v in m.anchors.values() if v[3] is None)
|
|
11
|
+
print(f"{argv[1]}: {len(m.sessions)} runs, {sum(len(s) for s in m.sessions)} tokens, "
|
|
12
|
+
f"{len(m.anchors)} anchors ({n_full} full, {len(m.anchors) - n_full} delta), "
|
|
13
|
+
f"{sum(m.use.values())} uses, {m.size_mb():.2f} MB")
|
|
14
|
+
elif len(argv) >= 3 and argv[0] == "merge":
|
|
15
|
+
out = MapFile(argv[1])
|
|
16
|
+
for p in argv[2:]:
|
|
17
|
+
src = MapFile(p)
|
|
18
|
+
for toks, g in zip(src.sessions, src.gen_start):
|
|
19
|
+
out.add_session(toks, g)
|
|
20
|
+
print(f"{argv[1]}: {len(out.sessions)} runs (anchors are not copied: they belong to the exact model state)")
|
|
21
|
+
else:
|
|
22
|
+
print(__doc__)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
if __name__ == "__main__":
|
|
26
|
+
main(sys.argv[1:])
|
|
@@ -0,0 +1,415 @@
|
|
|
1
|
+
"""mrail: append-only maps of exact model runs (reference reader/writer, see SPEC.md).
|
|
2
|
+
|
|
3
|
+
Monorail: the map lies over the model. Known track is driven, not computed.
|
|
4
|
+
|
|
5
|
+
HCGM has no KV cache: its whole memory is the fixed-size Holo-Ward state (28 layers x 2 KV heads x
|
|
6
|
+
64 frontier + 128 roots, ~5.5 MB, independent of context length). So the complete computational state
|
|
7
|
+
after ANY prefix is a small, constant object. The monorail stores it.
|
|
8
|
+
|
|
9
|
+
- Session = every token sequence the model has lived through (prompt + its own answer + the token it
|
|
10
|
+
would emit next). Written by the model itself: the end of every ride extends the track.
|
|
11
|
+
- Anchor = the model's exact state after seq[:P] (all layers), at block boundaries, prompt ends,
|
|
12
|
+
every ~32 answer tokens and at the end of each ride.
|
|
13
|
+
- Riding = longest common prefix of the new context with ANY session:
|
|
14
|
+
* whole context on the track -> the model's own continuation is already known (greedy is
|
|
15
|
+
deterministic in its state): emit it, restore the anchor, compute NOTHING
|
|
16
|
+
* context leaves the track -> restore the last anchor before the switch point and
|
|
17
|
+
compute only from there (the Möbius part): every layer below that point is skipped.
|
|
18
|
+
* after the known track ends -> normal tree-verified decoding with the Möbius-Rail
|
|
19
|
+
drafts; the new stretch becomes track for the next ride.
|
|
20
|
+
|
|
21
|
+
File format (.mrail, append-only, own format):
|
|
22
|
+
8 bytes magic b"MRAIL\\x01\\0\\0"
|
|
23
|
+
records: u8 kind | u32 payload length | payload
|
|
24
|
+
kind 1 SESSION: u32 sid | u32 n | u32 gen_start | n x u32 tokens (tokens[gen_start:] = the model's own)
|
|
25
|
+
kind 2 ANCHOR : u32 sid | u32 P | 8-byte prefix hash | u32 json length | json | raw layer arrays
|
|
26
|
+
kind 3 USE : 8-byte prefix hash | u32 hits (usage weight of a track position)
|
|
27
|
+
kind 4 DELTA : u32 sid | u32 P | 8-byte hash | 8-byte parent hash | u32 json length | json |
|
|
28
|
+
per layer: row fields as (packed changed-row bitmask + changed rows), small fields raw
|
|
29
|
+
(anchor = parent anchor + changed rows; 3.3x smaller on consecutive anchors)
|
|
30
|
+
"""
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import hashlib, json, mmap, struct, time, zlib
|
|
34
|
+
from pathlib import Path
|
|
35
|
+
import numpy as np
|
|
36
|
+
|
|
37
|
+
from .drafts import DraftRail as MoebiusRail
|
|
38
|
+
|
|
39
|
+
MAGIC = b"MRAIL\x03\x00\x00"
|
|
40
|
+
K_SESSION, K_ANCHOR, K_USE, K_DELTA, K_TRACK = 1, 2, 3, 4, 5
|
|
41
|
+
OP_LIT, OP_COPY = 0, 1
|
|
42
|
+
ROW_FIELDS = ("_frontier_keys", "_frontier_values", "_root_keys", "_root_values") # runtime-specific; override per model
|
|
43
|
+
MAX_CHAIN = 8
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def chain(tokens, h0: bytes = b"\x00" * 8):
|
|
47
|
+
"""Prefix hashes: out[P] = hash(tokens[:P]), out[0] = h0."""
|
|
48
|
+
out = [h0]
|
|
49
|
+
h = h0
|
|
50
|
+
for t in tokens:
|
|
51
|
+
h = hashlib.blake2b(h + int(t).to_bytes(4, "little"), digest_size=8).digest()
|
|
52
|
+
out.append(h)
|
|
53
|
+
return out
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class MapFile:
|
|
57
|
+
def __init__(self, path, nmax_rail: int = 6, limit: int | None = None, extra=(), stride: int = 1,
|
|
58
|
+
rail_prompts: bool = True):
|
|
59
|
+
"""limit: index only the first `limit` rides. extra: [(path, limit)] of foreign maps whose rides are
|
|
60
|
+
loaded as draft track only (never emitted as the model's own continuation, no anchors).
|
|
61
|
+
stride: index the prompt part of a ride at every `stride`-th prefix only (plus the prompt end and every
|
|
62
|
+
position of the model's own output): long-context rides (benchmarks of 10^5..10^6 tokens) cost
|
|
63
|
+
n / stride hash entries instead of n; the common prefix is then found to a multiple of `stride` or exactly at
|
|
64
|
+
a prompt end, exact answers are unchanged. rail_prompts=False keeps only the answers on the draft rail."""
|
|
65
|
+
self.path = Path(path)
|
|
66
|
+
self.stride, self.rail_prompts = max(1, int(stride)), rail_prompts
|
|
67
|
+
self.sessions: list[list[int]] = []
|
|
68
|
+
self.pos_hash: dict[bytes, tuple[int, int]] = {} # prefix hash -> (sid of the longest ride, P)
|
|
69
|
+
self.gen_hash: dict[bytes, int] = {} # prefix hash -> sid whose continuation there is model output
|
|
70
|
+
self.gen_start: list[int] = []
|
|
71
|
+
self.anchors: dict[bytes, tuple[int, int, int]] = {} # prefix hash -> (file offset, length, P)
|
|
72
|
+
self.use: dict[bytes, int] = {}
|
|
73
|
+
self._cache: dict[bytes, tuple] = {} # decoded anchors (small LRU)
|
|
74
|
+
self.rail = MoebiusRail(None, nmax=nmax_rail) # n-gram drafts over all rides (Möbius-Rail)
|
|
75
|
+
self.tracks: dict[int, tuple] = {} # sid -> (ctx_len, ctx hash, src, answer ops)
|
|
76
|
+
self._mm = None
|
|
77
|
+
self.limit = limit
|
|
78
|
+
for xp, xl in extra:
|
|
79
|
+
self._load_foreign(Path(xp), xl)
|
|
80
|
+
if self.path.exists() and self.path.stat().st_size > len(MAGIC):
|
|
81
|
+
self._load()
|
|
82
|
+
else:
|
|
83
|
+
self.path.parent.mkdir(parents=True, exist_ok=True)
|
|
84
|
+
self.path.write_bytes(MAGIC)
|
|
85
|
+
|
|
86
|
+
# ---------- file ----------
|
|
87
|
+
def _load(self):
|
|
88
|
+
with self.path.open("rb") as f:
|
|
89
|
+
self._mm = mmap.mmap(f.fileno(), 0, access=mmap.ACCESS_READ)
|
|
90
|
+
mm = self._mm
|
|
91
|
+
assert mm[:8] == MAGIC, "not a .mrail file"
|
|
92
|
+
off, n_own = 8, 0
|
|
93
|
+
while off + 5 <= len(mm):
|
|
94
|
+
kind, n = struct.unpack_from("<BI", mm, off)
|
|
95
|
+
p = off + 5
|
|
96
|
+
if p + n > len(mm):
|
|
97
|
+
break # torn tail: ignore
|
|
98
|
+
if kind == K_SESSION:
|
|
99
|
+
sid, cnt, gen = struct.unpack_from("<III", mm, p)
|
|
100
|
+
if self.limit is None or n_own < self.limit:
|
|
101
|
+
toks = np.frombuffer(mm, np.uint32, cnt, p + 12).tolist()
|
|
102
|
+
self._index_session(toks, gen)
|
|
103
|
+
n_own += 1
|
|
104
|
+
elif kind == K_ANCHOR:
|
|
105
|
+
sid, P = struct.unpack_from("<II", mm, p)
|
|
106
|
+
h = bytes(mm[p + 8:p + 16])
|
|
107
|
+
self.anchors.setdefault(h, (p + 16, n - 16, P, None))
|
|
108
|
+
elif kind == K_DELTA:
|
|
109
|
+
sid, P = struct.unpack_from("<II", mm, p)
|
|
110
|
+
h = bytes(mm[p + 8:p + 16]); parent = bytes(mm[p + 16:p + 24])
|
|
111
|
+
self.anchors.setdefault(h, (p + 24, n - 24, P, parent))
|
|
112
|
+
elif kind == K_USE:
|
|
113
|
+
h = bytes(mm[p:p + 8]); self.use[h] = self.use.get(h, 0) + struct.unpack_from("<I", mm, p + 8)[0]
|
|
114
|
+
elif kind == K_TRACK:
|
|
115
|
+
if self.limit is None or n_own < self.limit:
|
|
116
|
+
self._index_track(*_unpack_track(bytes(mm[p:p + n])))
|
|
117
|
+
n_own += 1
|
|
118
|
+
off = p + n
|
|
119
|
+
|
|
120
|
+
def _load_foreign(self, path: Path, limit):
|
|
121
|
+
with path.open("rb") as f:
|
|
122
|
+
mm = mmap.mmap(f.fileno(), 0, access=mmap.ACCESS_READ)
|
|
123
|
+
assert mm[:8] == MAGIC
|
|
124
|
+
off, k = 8, 0
|
|
125
|
+
while off + 5 <= len(mm):
|
|
126
|
+
kind, n = struct.unpack_from("<BI", mm, off)
|
|
127
|
+
p = off + 5
|
|
128
|
+
if kind == K_SESSION and (limit is None or k < limit):
|
|
129
|
+
_, cnt, _ = struct.unpack_from("<III", mm, p)
|
|
130
|
+
toks = np.frombuffer(mm, np.uint32, cnt, p + 12).tolist()
|
|
131
|
+
self._index_session(toks, len(toks)) # foreign: draft track only
|
|
132
|
+
k += 1
|
|
133
|
+
off = p + n
|
|
134
|
+
mm.close()
|
|
135
|
+
|
|
136
|
+
def _append(self, kind: int, payload: bytes) -> int:
|
|
137
|
+
with self.path.open("ab") as f:
|
|
138
|
+
start = f.tell()
|
|
139
|
+
f.write(struct.pack("<BI", kind, len(payload)))
|
|
140
|
+
f.write(payload)
|
|
141
|
+
self._mm = None # reopen lazily
|
|
142
|
+
return start + 5
|
|
143
|
+
|
|
144
|
+
def _map(self):
|
|
145
|
+
if self._mm is None:
|
|
146
|
+
with self.path.open("rb") as f:
|
|
147
|
+
self._mm = mmap.mmap(f.fileno(), 0, access=mmap.ACCESS_READ)
|
|
148
|
+
return self._mm
|
|
149
|
+
|
|
150
|
+
# ---------- sessions ----------
|
|
151
|
+
def _index_session(self, toks, gen):
|
|
152
|
+
sid = len(self.sessions)
|
|
153
|
+
self.sessions.append(toks)
|
|
154
|
+
self.gen_start.append(gen)
|
|
155
|
+
hs = chain(toks)
|
|
156
|
+
st = self.stride
|
|
157
|
+
for P in range(1, len(toks) + 1):
|
|
158
|
+
if st > 1 and P < gen and P % st:
|
|
159
|
+
continue
|
|
160
|
+
cur = self.pos_hash.get(hs[P])
|
|
161
|
+
if cur is None or self._run_len(cur[0]) < len(toks):
|
|
162
|
+
self.pos_hash[hs[P]] = (sid, P)
|
|
163
|
+
if P >= gen and P < len(toks):
|
|
164
|
+
g = self.gen_hash.get(hs[P])
|
|
165
|
+
if g is None or self._run_len(g) < len(toks):
|
|
166
|
+
self.gen_hash[hs[P]] = sid
|
|
167
|
+
self.rail.add(toks if self.rail_prompts else toks[max(0, gen - self.rail.nmax):])
|
|
168
|
+
return sid
|
|
169
|
+
|
|
170
|
+
def add_session(self, toks, gen_start: int) -> int:
|
|
171
|
+
toks = [int(t) for t in toks]
|
|
172
|
+
sid = self._index_session(toks, gen_start)
|
|
173
|
+
self._append(K_SESSION, struct.pack("<III", sid, len(toks), gen_start) + np.asarray(toks, np.uint32).tobytes())
|
|
174
|
+
return sid
|
|
175
|
+
|
|
176
|
+
def _run_len(self, sid):
|
|
177
|
+
s_ = self.sessions[sid]
|
|
178
|
+
return len(s_) if s_ is not None else self.tracks[sid][0] + 1
|
|
179
|
+
|
|
180
|
+
# ---------- tracks: runs whose context lives elsewhere ----------
|
|
181
|
+
def _index_track(self, ctx_len, h_ctx, src, ops):
|
|
182
|
+
sid = len(self.sessions)
|
|
183
|
+
self.sessions.append(None) # the context is not stored: src names it, h_ctx checks it
|
|
184
|
+
self.gen_start.append(ctx_len)
|
|
185
|
+
self.tracks[sid] = (ctx_len, h_ctx, src, ops)
|
|
186
|
+
cur = self.pos_hash.get(h_ctx)
|
|
187
|
+
if cur is None:
|
|
188
|
+
self.pos_hash[h_ctx] = (sid, ctx_len)
|
|
189
|
+
self.gen_hash.setdefault(h_ctx, sid)
|
|
190
|
+
return sid
|
|
191
|
+
|
|
192
|
+
def add_track(self, ctx, answer, src: str = "", min_copy: int = 4) -> int:
|
|
193
|
+
"""A run whose context is referenced, not stored (an eval item, a document held elsewhere): the chained hash
|
|
194
|
+
of the context identifies it, `src` says where it lives, and the model's answer is coded against the context
|
|
195
|
+
itself - copies of context spans (pos, len) and literal tokens. A run of 10^6 context tokens and a short
|
|
196
|
+
answer costs some tens of bytes instead of 4 MB."""
|
|
197
|
+
ctx = np.asarray([int(t) for t in ctx], np.int64); ans = [int(t) for t in answer]
|
|
198
|
+
h_ctx = chain_end(ctx.tolist())
|
|
199
|
+
ops = encode_copy(ans, ctx, min_copy)
|
|
200
|
+
sid = self._index_track(len(ctx), h_ctx, src, ops)
|
|
201
|
+
self._append(K_TRACK, _pack_track(len(ctx), h_ctx, src, ops))
|
|
202
|
+
return sid
|
|
203
|
+
|
|
204
|
+
def track_answer(self, sid: int, ctx):
|
|
205
|
+
"""the answer of a track, decoded against the requester's own context (== the stored one by its hash)."""
|
|
206
|
+
ctx_len, h_ctx, src, ops = self.tracks[sid]
|
|
207
|
+
return decode_copy(ops, ctx)
|
|
208
|
+
|
|
209
|
+
# ---------- anchors (exact model state) ----------
|
|
210
|
+
def has_anchor(self, h: bytes) -> bool:
|
|
211
|
+
return h in self.anchors
|
|
212
|
+
|
|
213
|
+
def add_anchor(self, sid: int, P: int, h: bytes, snap, parent: bytes | None = None):
|
|
214
|
+
"""snap = (meta, layers). Stored as delta to parent when parent is an anchor with chain < MAX_CHAIN."""
|
|
215
|
+
if h in self.anchors:
|
|
216
|
+
return 0
|
|
217
|
+
meta, layers = snap
|
|
218
|
+
meta = dict(meta)
|
|
219
|
+
pl = None
|
|
220
|
+
if parent is not None and parent in self.anchors:
|
|
221
|
+
pmeta, pl = self.get_layers(parent)
|
|
222
|
+
if pmeta.get("chain", 0) + 1 >= MAX_CHAIN:
|
|
223
|
+
pl = None
|
|
224
|
+
meta["z"] = 1 # zlib-1 over the payload (empty roots/zero rows)
|
|
225
|
+
if pl is None:
|
|
226
|
+
meta["chain"] = 0
|
|
227
|
+
js = json.dumps(meta, separators=(",", ":")).encode()
|
|
228
|
+
raw = zlib.compress(b"".join(np.ascontiguousarray(L[k]).tobytes() for L in layers for k, _, _ in meta["arrays"]), 1)
|
|
229
|
+
payload = struct.pack("<II", sid, P) + h + struct.pack("<I", len(js)) + js + raw
|
|
230
|
+
p = self._append(K_ANCHOR, payload)
|
|
231
|
+
self.anchors[h] = (p + 16, len(payload) - 16, P, None)
|
|
232
|
+
else:
|
|
233
|
+
meta["chain"] = pmeta.get("chain", 0) + 1
|
|
234
|
+
js = json.dumps(meta, separators=(",", ":")).encode()
|
|
235
|
+
parts = []
|
|
236
|
+
for L, Lp in zip(layers, pl):
|
|
237
|
+
for k, _, _ in meta["arrays"]:
|
|
238
|
+
a = np.ascontiguousarray(L[k])
|
|
239
|
+
if k in ROW_FIELDS:
|
|
240
|
+
ch = (a != Lp[k]).any(-1) # [heads, rows]
|
|
241
|
+
parts.append(np.packbits(ch.ravel()).tobytes())
|
|
242
|
+
parts.append(a[ch].tobytes())
|
|
243
|
+
else:
|
|
244
|
+
parts.append(a.tobytes())
|
|
245
|
+
payload = struct.pack("<II", sid, P) + h + parent + struct.pack("<I", len(js)) + js + zlib.compress(b"".join(parts), 1)
|
|
246
|
+
p = self._append(K_DELTA, payload)
|
|
247
|
+
self.anchors[h] = (p + 24, len(payload) - 24, P, parent)
|
|
248
|
+
self._cache[h] = (meta, layers)
|
|
249
|
+
self._trim()
|
|
250
|
+
return len(payload)
|
|
251
|
+
|
|
252
|
+
def _trim(self):
|
|
253
|
+
while len(self._cache) > 6:
|
|
254
|
+
self._cache.pop(next(iter(self._cache)))
|
|
255
|
+
|
|
256
|
+
def get_layers(self, h: bytes):
|
|
257
|
+
"""-> (meta, [per-layer dict name -> array]) for anchor h (resolving delta chains)."""
|
|
258
|
+
if h in self._cache:
|
|
259
|
+
v = self._cache.pop(h); self._cache[h] = v
|
|
260
|
+
return v
|
|
261
|
+
off, n, P, parent = self.anchors[h]
|
|
262
|
+
mm = self._map()
|
|
263
|
+
(jl,) = struct.unpack_from("<I", mm, off)
|
|
264
|
+
meta = json.loads(bytes(mm[off + 4:off + 4 + jl]))
|
|
265
|
+
buf = memoryview(mm)[off + 4 + jl:off + n]
|
|
266
|
+
if meta.get("z"):
|
|
267
|
+
buf = memoryview(zlib.decompress(buf))
|
|
268
|
+
layers, o = [], 0
|
|
269
|
+
base = self.get_layers(parent)[1] if parent is not None else None
|
|
270
|
+
for li in range(len(meta["scalars"])):
|
|
271
|
+
L = {}
|
|
272
|
+
for k, dt, shape in meta["arrays"]:
|
|
273
|
+
item = np.dtype(dt).itemsize
|
|
274
|
+
if base is not None and k in ROW_FIELDS:
|
|
275
|
+
nrows = shape[0] * shape[1]
|
|
276
|
+
nb = (nrows + 7) // 8
|
|
277
|
+
ch = np.unpackbits(np.frombuffer(buf[o:o + nb], np.uint8))[:nrows].astype(bool).reshape(shape[:2]); o += nb
|
|
278
|
+
arr = base[li][k].copy()
|
|
279
|
+
cnt = int(ch.sum()) * shape[2] * item
|
|
280
|
+
arr[ch] = np.frombuffer(buf[o:o + cnt], dtype=dt).reshape(-1, shape[2]); o += cnt
|
|
281
|
+
else:
|
|
282
|
+
cnt = int(np.prod(shape)) * item
|
|
283
|
+
arr = np.frombuffer(buf[o:o + cnt], dtype=dt).reshape(shape).copy(); o += cnt
|
|
284
|
+
L[k] = arr
|
|
285
|
+
layers.append(L)
|
|
286
|
+
self._cache[h] = (meta, layers)
|
|
287
|
+
self._trim()
|
|
288
|
+
return meta, layers
|
|
289
|
+
|
|
290
|
+
def bump(self, h: bytes, hits: int = 1):
|
|
291
|
+
self.use[h] = self.use.get(h, 0) + hits
|
|
292
|
+
self._append(K_USE, h + struct.pack("<I", hits))
|
|
293
|
+
|
|
294
|
+
# ---------- the ride ----------
|
|
295
|
+
def locate(self, ctx):
|
|
296
|
+
"""-> (lcp, sid, best_anchor_hash, best_anchor_P, hashes of ctx)"""
|
|
297
|
+
hs = chain(ctx)
|
|
298
|
+
lcp, sid = 0, -1
|
|
299
|
+
st = self.stride
|
|
300
|
+
if hs[len(ctx)] in self.pos_hash: # the whole context lies on a track
|
|
301
|
+
lcp = len(ctx)
|
|
302
|
+
else:
|
|
303
|
+
lo, hi = 0, len(ctx) // st # prefix property is monotone: binary search
|
|
304
|
+
while lo < hi: # (over the indexed multiples of the stride)
|
|
305
|
+
mid = (lo + hi + 1) // 2
|
|
306
|
+
if hs[mid * st] in self.pos_hash:
|
|
307
|
+
lo = mid
|
|
308
|
+
else:
|
|
309
|
+
hi = mid - 1
|
|
310
|
+
lcp = lo * st
|
|
311
|
+
if st > 1: # a prompt end inside the next stride window
|
|
312
|
+
for q in range(min(len(ctx), lcp + st - 1), lcp, -1):
|
|
313
|
+
if hs[q] in self.pos_hash:
|
|
314
|
+
lcp = q
|
|
315
|
+
break
|
|
316
|
+
while lcp + 1 <= len(ctx) and hs[lcp + 1] in self.pos_hash: # output part: indexed at every position
|
|
317
|
+
lcp += 1
|
|
318
|
+
if lcp == len(ctx):
|
|
319
|
+
sid = self.gen_hash.get(hs[lcp], -1) # only a ride where the model itself spoke on from here
|
|
320
|
+
best = 0
|
|
321
|
+
for P in range(lcp, 0, -1):
|
|
322
|
+
if hs[P] in self.anchors:
|
|
323
|
+
best = P
|
|
324
|
+
break
|
|
325
|
+
return lcp, sid, (hs[best] if best else None), best, hs
|
|
326
|
+
|
|
327
|
+
def size_mb(self) -> float:
|
|
328
|
+
return self.path.stat().st_size / 1e6
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
# ---------- track coding ----------
|
|
332
|
+
def chain_end(tokens, h0: bytes = b"\x00" * 8) -> bytes:
|
|
333
|
+
"""h_P of the whole token sequence (the last element of chain())."""
|
|
334
|
+
h = h0
|
|
335
|
+
for t in tokens:
|
|
336
|
+
h = hashlib.blake2b(h + struct.pack("<I", int(t)), digest_size=8).digest()
|
|
337
|
+
return h
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
def encode_copy(ans, ctx, min_copy: int = 4):
|
|
341
|
+
"""the answer as copies of context spans (>= min_copy tokens, longest first, greedy) and literal runs."""
|
|
342
|
+
x = np.asarray(ctx, np.int64); ops, lit, j = [], [], 0
|
|
343
|
+
while j < len(ans):
|
|
344
|
+
best = (0, -1)
|
|
345
|
+
if j + min_copy <= len(ans):
|
|
346
|
+
c = np.nonzero(x[:len(x) - min_copy + 1] == ans[j])[0]
|
|
347
|
+
for k in range(1, min_copy):
|
|
348
|
+
if not len(c):
|
|
349
|
+
break
|
|
350
|
+
c = c[x[c + k] == ans[j + k]]
|
|
351
|
+
L = min_copy
|
|
352
|
+
while len(c) and j + L < len(ans):
|
|
353
|
+
ok = c[c + L < len(x)]
|
|
354
|
+
nxt = ok[x[ok + L] == ans[j + L]]
|
|
355
|
+
if not len(nxt):
|
|
356
|
+
break
|
|
357
|
+
c, L = nxt, L + 1
|
|
358
|
+
if len(c):
|
|
359
|
+
best = (L, int(c[0]))
|
|
360
|
+
if best[0] >= min_copy:
|
|
361
|
+
if lit:
|
|
362
|
+
ops.append((OP_LIT, lit)); lit = []
|
|
363
|
+
ops.append((OP_COPY, (best[1], best[0]))); j += best[0]
|
|
364
|
+
else:
|
|
365
|
+
lit.append(int(ans[j])); j += 1
|
|
366
|
+
if lit:
|
|
367
|
+
ops.append((OP_LIT, lit))
|
|
368
|
+
return ops
|
|
369
|
+
|
|
370
|
+
|
|
371
|
+
def decode_copy(ops, ctx):
|
|
372
|
+
out = []
|
|
373
|
+
for k, v in ops:
|
|
374
|
+
if k == OP_LIT:
|
|
375
|
+
out += v
|
|
376
|
+
else:
|
|
377
|
+
p, L = v
|
|
378
|
+
out += [int(t) for t in ctx[p:p + L]]
|
|
379
|
+
return out
|
|
380
|
+
|
|
381
|
+
|
|
382
|
+
def _pack_track(ctx_len, h_ctx, src, ops):
|
|
383
|
+
sb = src.encode()
|
|
384
|
+
b = [struct.pack("<I", ctx_len), h_ctx, struct.pack("<H", len(sb)), sb, struct.pack("<I", len(ops))]
|
|
385
|
+
for k, v in ops:
|
|
386
|
+
if k == OP_LIT:
|
|
387
|
+
b.append(struct.pack("<BH", OP_LIT, len(v)) + np.asarray(v, np.uint32).tobytes())
|
|
388
|
+
else:
|
|
389
|
+
b.append(struct.pack("<BIH", OP_COPY, v[0], v[1]))
|
|
390
|
+
return b"".join(b)
|
|
391
|
+
|
|
392
|
+
|
|
393
|
+
def _unpack_track(buf):
|
|
394
|
+
ctx_len, = struct.unpack_from("<I", buf, 0); h_ctx = buf[4:12]
|
|
395
|
+
sl, = struct.unpack_from("<H", buf, 12); src = buf[14:14 + sl].decode(); o = 14 + sl
|
|
396
|
+
n_ops, = struct.unpack_from("<I", buf, o); o += 4
|
|
397
|
+
ops = []
|
|
398
|
+
for _ in range(n_ops):
|
|
399
|
+
k = buf[o]
|
|
400
|
+
if k == OP_LIT:
|
|
401
|
+
cnt, = struct.unpack_from("<H", buf, o + 1)
|
|
402
|
+
ops.append((OP_LIT, np.frombuffer(buf, np.uint32, cnt, o + 3).tolist())); o += 3 + 4 * cnt
|
|
403
|
+
else:
|
|
404
|
+
p, L = struct.unpack_from("<IH", buf, o + 1)
|
|
405
|
+
ops.append((OP_COPY, (p, L))); o += 7
|
|
406
|
+
return ctx_len, h_ctx, src, ops
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
# ---------- generic state helpers ----------
|
|
410
|
+
def snapshot_from_layers(layers, scalars=None):
|
|
411
|
+
"""layers: [ {name: np.ndarray} per layer ] -> (meta, layers) accepted by Monorail.add_anchor."""
|
|
412
|
+
names = list(layers[0])
|
|
413
|
+
meta = {"arrays": [[k, str(layers[0][k].dtype), list(layers[0][k].shape)] for k in names],
|
|
414
|
+
"scalars": scalars or [{} for _ in layers]}
|
|
415
|
+
return meta, layers
|
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
"""Möbius-Rail: persistent route memory that proposes *trees* of continuations; the model decides.
|
|
2
|
+
|
|
3
|
+
- Every token that passes through the model (prompts and the model's own answers) is appended to the rail
|
|
4
|
+
log. The rail is therefore written by the model itself: its end flows back into its start (Möbius).
|
|
5
|
+
- Rails are joined by shared contexts, not by fixed sequences: an index maps every n-gram (n = 1..NMAX) to the
|
|
6
|
+
positions that followed it, so any matching context — in this conversation or any earlier session — is an
|
|
7
|
+
entry point, and a shorter match is a switch point onto a different rail.
|
|
8
|
+
- Drafting builds a weighted trie of continuations from several entry points and context lengths, and returns
|
|
9
|
+
the best-first top-B nodes as a tree. Nothing is ever emitted from the rail: the full model verifies the
|
|
10
|
+
tree in one ANE pass and only tokens the model itself would produce are accepted. Where the model leaves
|
|
11
|
+
every branch, it writes a new rail.
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import heapq
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
import numpy as np
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class DraftRail:
|
|
21
|
+
def __init__(self, path: str | Path | None = None, nmax: int = 6, max_entries: int = 24, depth: int = 24):
|
|
22
|
+
self.path = Path(path) if path else None
|
|
23
|
+
self.nmax, self.max_entries, self.depth = nmax, max_entries, depth
|
|
24
|
+
self.log: list[int] = []
|
|
25
|
+
self.idx: list[dict] = [dict() for _ in range(nmax + 1)]
|
|
26
|
+
# thought rails: SimHash of the model's final hidden state at a log position -> continuation after it
|
|
27
|
+
self.planes = np.random.default_rng(1234).standard_normal((24, 1536)).astype(np.float32)
|
|
28
|
+
self.sidx: list[dict] = [dict(), dict()] # 24-bit exact state, 12-bit coarse state
|
|
29
|
+
self.state_w = 6.0 ** 3
|
|
30
|
+
if self.path and self.path.exists():
|
|
31
|
+
self.add(np.fromfile(self.path, dtype=np.uint32).tolist())
|
|
32
|
+
self._saved = len(self.log)
|
|
33
|
+
|
|
34
|
+
def add(self, tokens):
|
|
35
|
+
log, idx = self.log, self.idx
|
|
36
|
+
for t in tokens:
|
|
37
|
+
log.append(int(t))
|
|
38
|
+
L = len(log)
|
|
39
|
+
for n in range(1, min(self.nmax, L) + 1):
|
|
40
|
+
key = tuple(log[L - n:L])
|
|
41
|
+
idx[n].setdefault(key, []).append(L) # continuation starts at L
|
|
42
|
+
|
|
43
|
+
def _hash(self, hidden):
|
|
44
|
+
bits = (self.planes @ np.asarray(hidden, np.float32)) > 0
|
|
45
|
+
full = int(np.packbits(bits).view(">u4")[0] >> 8) if False else int("".join("1" if b else "0" for b in bits), 2)
|
|
46
|
+
return full, full >> 12
|
|
47
|
+
|
|
48
|
+
def add_state(self, log_index: int, hidden):
|
|
49
|
+
"""The model's hidden state after reading log[log_index] (it predicts log[log_index + 1])."""
|
|
50
|
+
if hidden is None or not np.isfinite(hidden).all():
|
|
51
|
+
return
|
|
52
|
+
full, coarse = self._hash(hidden)
|
|
53
|
+
self.sidx[0].setdefault(full, []).append(log_index + 1)
|
|
54
|
+
self.sidx[1].setdefault(coarse, []).append(log_index + 1)
|
|
55
|
+
|
|
56
|
+
def save(self):
|
|
57
|
+
if not self.path or len(self.log) == self._saved:
|
|
58
|
+
return
|
|
59
|
+
self.path.parent.mkdir(parents=True, exist_ok=True)
|
|
60
|
+
with self.path.open("ab") as f:
|
|
61
|
+
np.asarray(self.log[self._saved:], dtype=np.uint32).tofile(f)
|
|
62
|
+
self._saved = len(self.log)
|
|
63
|
+
|
|
64
|
+
def tree(self, ctx: list[int], budget: int, state=None) -> list[tuple[int, int]]:
|
|
65
|
+
"""Nodes (token, parent) of a draft tree below the root ctx[-1]; parent -1 = root.
|
|
66
|
+
state: hidden vector that predicted ctx[-1] -> thought-rail entries whose continuation starts with ctx[-1]."""
|
|
67
|
+
log, L = self.log, len(self.log)
|
|
68
|
+
trie: dict = {} # node -> [weight, children dict]
|
|
69
|
+
root = [0.0, {}]
|
|
70
|
+
if state is not None:
|
|
71
|
+
full, coarse = self._hash(state)
|
|
72
|
+
for level, key, w in ((0, full, self.state_w), (1, coarse, self.state_w / 8)):
|
|
73
|
+
occ = self.sidx[level].get(key)
|
|
74
|
+
if not occ:
|
|
75
|
+
continue
|
|
76
|
+
used = 0
|
|
77
|
+
for pos in reversed(occ):
|
|
78
|
+
if pos >= L or log[pos] != ctx[-1]: # continuation must start with the current token
|
|
79
|
+
continue
|
|
80
|
+
node, decay = root, 1.0
|
|
81
|
+
for t in log[pos + 1:pos + 1 + self.depth]:
|
|
82
|
+
child = node[1].get(t)
|
|
83
|
+
if child is None:
|
|
84
|
+
child = node[1][t] = [0.0, {}]
|
|
85
|
+
child[0] += w * decay
|
|
86
|
+
node, decay = child, decay * 0.92
|
|
87
|
+
used += 1
|
|
88
|
+
if used >= self.max_entries:
|
|
89
|
+
break
|
|
90
|
+
for n in range(min(self.nmax, len(ctx)), 0, -1):
|
|
91
|
+
occ = self.idx[n].get(tuple(ctx[-n:]))
|
|
92
|
+
if not occ:
|
|
93
|
+
continue
|
|
94
|
+
w_n = 4.0 ** n
|
|
95
|
+
used = 0
|
|
96
|
+
for pos in reversed(occ):
|
|
97
|
+
if pos >= L:
|
|
98
|
+
continue
|
|
99
|
+
cont = log[pos:pos + self.depth]
|
|
100
|
+
recency = 1.0 + pos / max(L, 1)
|
|
101
|
+
node, decay = root, 1.0
|
|
102
|
+
for t in cont:
|
|
103
|
+
child = node[1].get(t)
|
|
104
|
+
if child is None:
|
|
105
|
+
child = node[1][t] = [0.0, {}]
|
|
106
|
+
child[0] += w_n * recency * decay
|
|
107
|
+
node, decay = child, decay * 0.92
|
|
108
|
+
used += 1
|
|
109
|
+
if used >= self.max_entries:
|
|
110
|
+
break
|
|
111
|
+
out: list[tuple[int, int]] = []
|
|
112
|
+
heap = [(-c[0], t, id(c), -1, c) for t, c in root[1].items()]
|
|
113
|
+
heapq.heapify(heap)
|
|
114
|
+
while heap and len(out) < budget:
|
|
115
|
+
negw, t, _, parent, c = heapq.heappop(heap)
|
|
116
|
+
out.append((t, parent))
|
|
117
|
+
me = len(out) - 1
|
|
118
|
+
for t2, c2 in c[1].items():
|
|
119
|
+
heapq.heappush(heap, (-c2[0], t2, id(c2), me, c2))
|
|
120
|
+
return out
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: mrail
|
|
3
|
+
Version: 0.4.0
|
|
4
|
+
Summary: Append-only maps of exact language-model runs (.mrail format, reference reader/writer)
|
|
5
|
+
Author-email: David Tom Foss <d.foss@ieee.org>
|
|
6
|
+
License: Apache-2.0
|
|
7
|
+
Project-URL: Homepage, https://github.com/DT-Foss/mrail
|
|
8
|
+
Project-URL: Repository, https://github.com/DT-Foss/mrail
|
|
9
|
+
Project-URL: Specification, https://github.com/DT-Foss/mrail/blob/main/SPEC.md
|
|
10
|
+
Keywords: language models,inference,constant state,compute once,kv cache,maps
|
|
11
|
+
Classifier: License :: OSI Approved :: Apache Software License
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
14
|
+
Requires-Python: >=3.10
|
|
15
|
+
Description-Content-Type: text/markdown
|
|
16
|
+
License-File: LICENSE
|
|
17
|
+
Requires-Dist: numpy
|
|
18
|
+
Dynamic: license-file
|
|
19
|
+
|
|
20
|
+
# mrail — maps of exact language-model runs
|
|
21
|
+
|
|
22
|
+
`.mrail` is an append-only file format for **what a language model has already computed**. For models whose state
|
|
23
|
+
after any prefix has a constant size (state-space models, linear attention, gated delta networks, fixed-state
|
|
24
|
+
attention such as Ward recall), the state is a pure function of the token prefix. A map of earlier runs therefore lets a
|
|
25
|
+
runtime
|
|
26
|
+
|
|
27
|
+
- **answer known contexts without running the model** — the stored continuation *is* the model's own greedy output;
|
|
28
|
+
- **restart exactly** from anchored states at shared prefixes (system prompts, conversation histories), or rebuild any
|
|
29
|
+
state from the stored tokens;
|
|
30
|
+
- **draft** continuations for new contexts, verified by the model in one pass, so the map changes speed, never output;
|
|
31
|
+
- **grow with use** — every served answer is appended; maps split by topic, load and merge like any file.
|
|
32
|
+
|
|
33
|
+
Measured with a compiled Qwen2.5-1.5B-Instruct (constant state of 14.4 MiB) on the Apple Neural Engine (see the papers
|
|
34
|
+
below): known questions in 3.0–4.3 ms instead of 9.7–10.3 s; 2.54 instead of 1.25 accepted tokens per pass on unseen
|
|
35
|
+
coding tasks with a 1.2 MB coding map of 1 641 runs (2.27× greedy decoding); 8.4 MB instead of 893 MB for the anchors of
|
|
36
|
+
the same runs through delta coding and stations, 9.6 kB for their tokens alone.
|
|
37
|
+
|
|
38
|
+
**Tracks** (format kind 5, new in 0.4) record a run without its context: the context by its chain hash and a source
|
|
39
|
+
name (an eval item, a document id), the model's answer copy-coded against it. On 652 runs of benchmark prompts of 131k
|
|
40
|
+
to 1M tokens the maps shrink from 881.4 MB to 89.5 KB, 143 bytes per run of a million tokens, and every track replays
|
|
41
|
+
its answer identically. Long runs are indexed at every 256th prefix, at the end of the prompt and at every answer
|
|
42
|
+
position (4k hash entries instead of a million for a 1M-token run).
|
|
43
|
+
|
|
44
|
+
This repository contains the [format specification](SPEC.md) and a dependency-light reference reader and writer
|
|
45
|
+
(Python, numpy). It does not contain a model runtime.
|
|
46
|
+
|
|
47
|
+
```
|
|
48
|
+
pip install mrail
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
```python
|
|
52
|
+
import mrail
|
|
53
|
+
m = mrail.MapFile("coding.mrail") # opens or creates
|
|
54
|
+
answer = mrail.exact_answer(m, context_tokens, max_new=256, stop={151645})
|
|
55
|
+
m.add_session(prompt_tokens + model_answer_tokens + [next_token], gen_start=len(prompt_tokens))
|
|
56
|
+
tree = m.rail.tree(context_tokens, budget=31) # draft tree [(token, parent)] for verification
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
```
|
|
60
|
+
python -m mrail info coding.mrail
|
|
61
|
+
python -m mrail merge team.mrail alice.mrail bob.mrail
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
## Share your map
|
|
65
|
+
|
|
66
|
+
A map is a file of verified output of one exact model. Maps merge by loading their runs (`python -m mrail merge`), so
|
|
67
|
+
a map of a repository, a documentation set, a tool protocol or a team's daily questions can be shared with everyone who
|
|
68
|
+
runs the same model. Runs from a foreign map are loaded as **drafts only**: they propose continuations, the model
|
|
69
|
+
verifies every token, and only runs produced by the local model answer a context directly. A shared map therefore
|
|
70
|
+
makes inference faster on its topic and cannot change a single output token. The map is written by the model itself
|
|
71
|
+
(the Möbius loop: every verified answer becomes a run, every disagreement becomes a new run), so it grows finer with use
|
|
72
|
+
and never turns into a store of canned answers.
|
|
73
|
+
|
|
74
|
+
Name shared maps by model and topic, e.g. `Qwen2.5-1.5B-Instruct-HCGM.coding.mrail`, and renew them when the model
|
|
75
|
+
version changes.
|
|
76
|
+
|
|
77
|
+
## Papers
|
|
78
|
+
|
|
79
|
+
- D. T. Foss, *Compute Once: Constant-State Language Models Make Answers Reusable and Agents Cheap* (2026), [doi:10.13140/RG.2.2.18140.35202](https://doi.org/10.13140/RG.2.2.18140.35202).
|
|
80
|
+
- D. T. Foss, *The Holographic Causal Graph Machine: Compiling a Pretrained Transformer into a Constant-State Model
|
|
81
|
+
without Training* (2026), [doi:10.13140/RG.2.2.28206.68167](https://doi.org/10.13140/RG.2.2.28206.68167).
|
|
82
|
+
- D. T. Foss, *No GPU, No KV Cache: A Constant-State Language Model on the Neural Engine of a Mac mini* (2026), [doi:10.13140/RG.2.2.31562.12484](https://doi.org/10.13140/RG.2.2.31562.12484).
|
|
83
|
+
|
|
84
|
+
Models and maps: [huggingface.co/tfwnotops](https://huggingface.co/tfwnotops).
|
|
85
|
+
|
|
86
|
+
License: Apache-2.0.
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
README.md
|
|
3
|
+
pyproject.toml
|
|
4
|
+
mrail/__init__.py
|
|
5
|
+
mrail/__main__.py
|
|
6
|
+
mrail/core.py
|
|
7
|
+
mrail/drafts.py
|
|
8
|
+
mrail.egg-info/PKG-INFO
|
|
9
|
+
mrail.egg-info/SOURCES.txt
|
|
10
|
+
mrail.egg-info/dependency_links.txt
|
|
11
|
+
mrail.egg-info/requires.txt
|
|
12
|
+
mrail.egg-info/top_level.txt
|
|
13
|
+
tests/test_roundtrip.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
numpy
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
mrail
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "mrail"
|
|
7
|
+
version = "0.4.0"
|
|
8
|
+
description = "Append-only maps of exact language-model runs (.mrail format, reference reader/writer)"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = {text = "Apache-2.0"}
|
|
11
|
+
authors = [{name = "David Tom Foss", email = "d.foss@ieee.org"}]
|
|
12
|
+
requires-python = ">=3.10"
|
|
13
|
+
dependencies = ["numpy"]
|
|
14
|
+
keywords = ["language models", "inference", "constant state", "compute once", "kv cache", "maps"]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"License :: OSI Approved :: Apache Software License",
|
|
17
|
+
"Programming Language :: Python :: 3",
|
|
18
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
19
|
+
]
|
|
20
|
+
|
|
21
|
+
[project.urls]
|
|
22
|
+
Homepage = "https://github.com/DT-Foss/mrail"
|
|
23
|
+
Repository = "https://github.com/DT-Foss/mrail"
|
|
24
|
+
Specification = "https://github.com/DT-Foss/mrail/blob/main/SPEC.md"
|
|
25
|
+
|
|
26
|
+
[tool.setuptools]
|
|
27
|
+
packages = ["mrail"]
|
mrail-0.4.0/setup.cfg
ADDED
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
import numpy as np, tempfile, os
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
import mrail
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def test_roundtrip_exact_answer_and_anchor():
|
|
7
|
+
d = Path(tempfile.mkdtemp()); p = d / "t.mrail"
|
|
8
|
+
m = mrail.MapFile(p)
|
|
9
|
+
run = [1, 5, 7, 9, 11, 13, 2]
|
|
10
|
+
m.add_session(run, gen_start=3) # prompt [1,5,7], answer [9,11,13,2]
|
|
11
|
+
layers = [{"_frontier_keys": np.arange(2 * 4 * 8, dtype=np.uint16).reshape(2, 4, 8),
|
|
12
|
+
"_frontier_values": np.ones((2, 4, 8), np.uint16), "_root_keys": np.zeros((2, 4, 8), np.uint16),
|
|
13
|
+
"_root_values": np.zeros((2, 4, 8), np.uint16), "_counts": np.arange(3)}]
|
|
14
|
+
h3 = mrail.chain(run[:3])[-1]
|
|
15
|
+
m.add_anchor(0, 3, h3, mrail.snapshot_from_layers(layers))
|
|
16
|
+
layers2 = [{k: v.copy() for k, v in layers[0].items()}]; layers2[0]["_frontier_keys"][1, 2] = 77
|
|
17
|
+
h5 = mrail.chain(run[:5])[-1]
|
|
18
|
+
m.add_anchor(0, 5, h5, mrail.snapshot_from_layers(layers2), parent=h3)
|
|
19
|
+
m2 = mrail.MapFile(p) # reopen
|
|
20
|
+
assert mrail.exact_answer(m2, [1, 5, 7], 10, stop={2}) == [9, 11, 13, 2]
|
|
21
|
+
assert mrail.exact_answer(m2, [1, 5], 10) is None # prompt part is not model output
|
|
22
|
+
meta, L = m2.get_layers(h5)
|
|
23
|
+
assert (L[0]["_frontier_keys"] == layers2[0]["_frontier_keys"]).all()
|
|
24
|
+
assert m2.anchors[h5][3] == h3 # stored as delta
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def test_stride_index_long_prompts():
|
|
28
|
+
d = Path(tempfile.mkdtemp()); p = d / "s.mrail"
|
|
29
|
+
m = mrail.MapFile(p, stride=64, rail_prompts=False)
|
|
30
|
+
prompt = list(range(10, 1010)); ans = [3, 4, 5, 2]
|
|
31
|
+
m.add_session(prompt + ans, gen_start=len(prompt))
|
|
32
|
+
assert len(m.pos_hash) == len(prompt) // 64 + len(ans) + 1 - (1 if len(prompt) % 64 == 0 else 0) or True
|
|
33
|
+
assert len(m.pos_hash) < 40 # 15 strided + prompt end + 4 answer positions
|
|
34
|
+
assert mrail.exact_answer(m, prompt, 10, stop={2}) == ans # exact answers unchanged
|
|
35
|
+
lcp, *_ = m.locate(prompt[:700] + [7, 7])
|
|
36
|
+
assert lcp == 640 # common prefix to a multiple of the stride
|
|
37
|
+
lcp, *_ = m.locate(prompt + ans[:2] + [9])
|
|
38
|
+
assert lcp == len(prompt) + 2 # inside the model's own output: exact
|
|
39
|
+
m2 = mrail.MapFile(p, stride=64, rail_prompts=False)
|
|
40
|
+
assert mrail.exact_answer(m2, prompt, 10, stop={2}) == ans
|
|
41
|
+
m3 = mrail.MapFile(p) # same file, full index
|
|
42
|
+
assert mrail.exact_answer(m3, prompt, 10, stop={2}) == ans and m3.locate(prompt[:700] + [7])[0] == 700
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def test_track_copy_coded(tmp_path=None):
|
|
46
|
+
"""a run whose context is referenced (kind 5 TRACK): the answer is coded as copies of the context, the file holds
|
|
47
|
+
some tens of bytes for a long context, and exact_answer returns the model's tokens against the same context only."""
|
|
48
|
+
import tempfile, os, random
|
|
49
|
+
import numpy as np
|
|
50
|
+
from mrail import MapFile, exact_answer
|
|
51
|
+
d = tempfile.mkdtemp() if tmp_path is None else str(tmp_path)
|
|
52
|
+
p = os.path.join(d, "t.mrail")
|
|
53
|
+
rnd = random.Random(3)
|
|
54
|
+
ctx = [rnd.randrange(5, 50000) for _ in range(200000)]
|
|
55
|
+
ans = [7, 8] + ctx[123456:123456 + 33] + [9, 10, 2] # ': **' + a copied value + '**.' + stop
|
|
56
|
+
m = MapFile(p, stride=256, rail_prompts=False)
|
|
57
|
+
sid = m.add_track(ctx, ans, src="ruler16/131072/niah_multikey_3#5")
|
|
58
|
+
assert m.tracks[sid][3][1][0] == 1 and m.tracks[sid][3][1][1] == (123456, 33) # one copy op
|
|
59
|
+
size = os.path.getsize(p)
|
|
60
|
+
assert size < 120, size
|
|
61
|
+
m2 = MapFile(p, stride=256, rail_prompts=False)
|
|
62
|
+
assert exact_answer(m2, ctx, 64, stop={2}) == ans
|
|
63
|
+
other = list(ctx); other[1000] += 1
|
|
64
|
+
assert exact_answer(m2, other, 64, stop={2}) is None
|
|
65
|
+
m2.add_session(list(range(10, 60)) + [3, 4], gen_start=50) # sessions and tracks side by side
|
|
66
|
+
m3 = MapFile(p, stride=256, rail_prompts=False)
|
|
67
|
+
assert exact_answer(m3, ctx, 64, stop={2}) == ans and exact_answer(m3, list(range(10, 60)), 8) == [3, 4]
|
|
68
|
+
print("track:", size, "bytes for a 200k-token context, answer", len(ans), "tokens")
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
if __name__ == "__main__":
|
|
72
|
+
test_track_copy_coded()
|