@hviana/sema 0.9.0 → 0.9.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +7 -7
- package/dist/src/alu/src/index.d.ts +1 -1
- package/dist/src/alu/src/index.js +1 -1
- package/dist/src/alu/src/parser.js +2 -6
- package/dist/src/alu/src/resonance.d.ts +13 -0
- package/dist/src/alu/src/resonance.js +41 -0
- package/dist/src/alu/test/alu.test.js +39 -0
- package/dist/src/bytes.d.ts +6 -2
- package/dist/src/bytes.js +10 -4
- package/dist/src/canon.js +44 -0
- package/dist/src/geometry.d.ts +19 -1
- package/dist/src/geometry.js +125 -141
- package/dist/src/meter.d.ts +20 -0
- package/dist/src/meter.js +21 -1
- package/dist/src/mind/articulation.js +14 -1
- package/dist/src/mind/attention.d.ts +12 -0
- package/dist/src/mind/attention.js +44 -16
- package/dist/src/mind/bridge.js +3 -3
- package/dist/src/mind/derivation.d.ts +40 -0
- package/dist/src/mind/derivation.js +34 -0
- package/dist/src/mind/graph-search.d.ts +89 -15
- package/dist/src/mind/graph-search.js +345 -174
- package/dist/src/mind/learning.js +1 -1
- package/dist/src/mind/mechanisms/cover.d.ts +19 -3
- package/dist/src/mind/mechanisms/cover.js +101 -58
- package/dist/src/mind/mechanisms/recall.js +0 -1
- package/dist/src/mind/mind.js +2 -2
- package/dist/src/mind/pipeline.d.ts +5 -1
- package/dist/src/mind/pipeline.js +175 -87
- package/dist/src/mind/primitives.d.ts +25 -5
- package/dist/src/mind/primitives.js +107 -44
- package/dist/src/mind/reasoning.d.ts +18 -4
- package/dist/src/mind/reasoning.js +445 -321
- package/dist/src/mind/recognition.js +29 -13
- package/dist/src/mind/resonance.js +1 -11
- package/dist/src/mind/traverse.d.ts +3 -3
- package/dist/src/mind/traverse.js +3 -3
- package/dist/src/mind/types.d.ts +7 -1
- package/dist/src/store-sqlite.d.ts +25 -0
- package/dist/src/store-sqlite.js +89 -1
- package/dist/src/store.d.ts +48 -4
- package/dist/src/store.js +86 -6
- package/docs/INDEX.md +18 -18
- package/docs/INVARIANTS.md +16 -16
- package/docs/architecture/bounded-reads.md +1 -1
- package/docs/architecture/caches.md +5 -4
- package/docs/architecture/closure.md +45 -5
- package/docs/architecture/cost-model.md +16 -0
- package/docs/architecture/factored-machinery.md +14 -13
- package/docs/architecture/fold-contract.md +51 -1
- package/docs/architecture/mechanism-market.md +21 -0
- package/docs/architecture/memoization.md +3 -3
- package/docs/architecture/meter.md +2 -1
- package/docs/architecture/saturation.md +12 -0
- package/docs/architecture/store.md +25 -2
- package/docs/failures/tempting-but-wrong.md +13 -2
- package/docs/harness/gates.md +12 -10
- package/docs/mechanisms/cover.md +23 -6
- package/jsr.json +1 -1
- package/package.json +1 -1
- package/src/alu/README.md +10 -2
- package/src/alu/src/index.ts +1 -0
- package/src/alu/src/parser.ts +6 -6
- package/src/alu/src/resonance.ts +42 -0
- package/src/alu/test/alu.test.ts +40 -0
- package/src/bytes.ts +13 -3
- package/src/canon.ts +40 -0
- package/src/geometry.ts +183 -154
- package/src/meter.ts +21 -1
- package/src/mind/articulation.ts +14 -2
- package/src/mind/attention.ts +47 -25
- package/src/mind/bridge.ts +3 -3
- package/src/mind/derivation.ts +77 -0
- package/src/mind/graph-search.ts +449 -221
- package/src/mind/learning.ts +1 -7
- package/src/mind/match.ts +1 -2
- package/src/mind/mechanisms/cast.ts +1 -2
- package/src/mind/mechanisms/cover.ts +149 -84
- package/src/mind/mechanisms/extraction.ts +1 -2
- package/src/mind/mechanisms/prefix-completion.ts +1 -1
- package/src/mind/mechanisms/recall.ts +1 -3
- package/src/mind/mechanisms/reference.ts +1 -1
- package/src/mind/mind.ts +5 -30
- package/src/mind/pipeline.ts +206 -102
- package/src/mind/primitives.ts +119 -43
- package/src/mind/reasoning.ts +558 -413
- package/src/mind/recognition.ts +24 -9
- package/src/mind/resonance.ts +2 -16
- package/src/mind/trace.ts +1 -1
- package/src/mind/traverse.ts +3 -3
- package/src/mind/types.ts +9 -11
- package/src/store-sqlite.ts +92 -1
- package/src/store.ts +113 -7
- package/test/105-derive-through-reports-its-refusal.test.mjs +8 -5
- package/test/106-the-join-fires.test.mjs +21 -0
- package/test/111-the-cover-assembly-is-counted.test.mjs +8 -5
- package/test/128-the-leads-somewhere-pair-agrees.test.mjs +18 -12
- package/test/136-the-two-named-limits.test.mjs +3 -2
- package/test/137-the-law-lives-once-and-below.test.mjs +21 -0
- package/test/148-exact-shortcuts-agree.test.mjs +188 -0
- package/test/149-the-closure-engine.test.mjs +138 -0
- package/test/150-the-join-is-output-sensitive.test.mjs +66 -0
- package/test/151-the-cover-pays-for-what-it-reaches.test.mjs +142 -0
- package/test/152-the-read-side-names-as-the-write-side.test.mjs +146 -0
- package/test/153-a-cheaper-bound-is-looked-at-first.test.mjs +155 -0
- package/test/24-generalization.test.mjs +32 -0
- package/test/36-bloom.test.mjs +53 -0
- package/test/37-cluster-dispersion-fusion.test.mjs +75 -0
- package/test/48-recognise-turn-connective.test.mjs +3 -2
- package/test/55-cost-meter.test.mjs +4 -4
- package/test/90-connector-read-cap.test.mjs +7 -7
package/src/alu/src/resonance.ts
CHANGED
|
@@ -104,6 +104,48 @@ export async function prefetchOpposites(
|
|
|
104
104
|
};
|
|
105
105
|
}
|
|
106
106
|
|
|
107
|
+
/** {@link prefetchOpposites} resolved ON DEMAND: run `apply` against a
|
|
108
|
+
* synchronous snapshot that answers only the opposites already resolved, and
|
|
109
|
+
* when the computation ASKED for one of `symbols` that is not yet resolved,
|
|
110
|
+
* resolve it through the host and run `apply` again — until a run asks for
|
|
111
|
+
* nothing new. The result is `apply(prefetchOpposites(resonance, symbols))`
|
|
112
|
+
* exactly: an opposite outside `symbols` reads null in both, every one inside
|
|
113
|
+
* that the computation reads is the host's answer in both, and `apply` is
|
|
114
|
+
* pure, so a run that read the same answers returns the same bytes. What it
|
|
115
|
+
* saves is every host call no computation reads: only the polymorphic
|
|
116
|
+
* inverse reads opposites, yet the eager prefetch paid one per symbol operand
|
|
117
|
+
* for ANY operation — on SEMA a halo-index query each, measured at 30-200 ms
|
|
118
|
+
* of a plain dialogue turn's parse that computed nothing. */
|
|
119
|
+
export async function withOppositesOnDemand<R>(
|
|
120
|
+
resonance: AluResonance,
|
|
121
|
+
symbols: Iterable<Uint8Array>,
|
|
122
|
+
apply: (sync: ResonanceSync) => R,
|
|
123
|
+
): Promise<R> {
|
|
124
|
+
const allowed = new Set<string>();
|
|
125
|
+
for (const bytes of symbols) allowed.add(latin1(bytes));
|
|
126
|
+
const table = new Map<string, Uint8Array | null>();
|
|
127
|
+
const asked = new Map<string, Uint8Array>();
|
|
128
|
+
const sync: ResonanceSync = {
|
|
129
|
+
opposite: (bytes: Uint8Array) => {
|
|
130
|
+
const key = latin1(bytes);
|
|
131
|
+
if (!allowed.has(key)) return null;
|
|
132
|
+
const known = table.get(key);
|
|
133
|
+
if (known !== undefined) return known;
|
|
134
|
+
asked.set(key, bytes);
|
|
135
|
+
return null;
|
|
136
|
+
},
|
|
137
|
+
recogniseOp: () => null,
|
|
138
|
+
};
|
|
139
|
+
for (;;) {
|
|
140
|
+
const out = apply(sync);
|
|
141
|
+
if (asked.size === 0) return out;
|
|
142
|
+
for (const [key, bytes] of asked) {
|
|
143
|
+
table.set(key, (await resonance.opposite(bytes)) ?? null);
|
|
144
|
+
}
|
|
145
|
+
asked.clear();
|
|
146
|
+
}
|
|
147
|
+
}
|
|
148
|
+
|
|
107
149
|
/** Pre-resolve BOTH capabilities a computation may need synchronously — the
|
|
108
150
|
* resonant opposite of a symbol (for the polymorphic inverse) AND the operation
|
|
109
151
|
* a symbol's MEANING names (for a higher-order nd op's function argument) — over
|
package/src/alu/test/alu.test.ts
CHANGED
|
@@ -9,6 +9,7 @@ import assert from "node:assert/strict";
|
|
|
9
9
|
import {
|
|
10
10
|
addBits,
|
|
11
11
|
Alu,
|
|
12
|
+
type AluHost,
|
|
12
13
|
type AluResonance,
|
|
13
14
|
asReal,
|
|
14
15
|
compareBits,
|
|
@@ -579,6 +580,45 @@ test("conceptAnchors exposes the operation vocabulary for resonant recognition",
|
|
|
579
580
|
}
|
|
580
581
|
});
|
|
581
582
|
|
|
583
|
+
test("a symbol's opposite is asked of the host only when the operation reads it", async () => {
|
|
584
|
+
// Only the polymorphic inverse reads an opposite. Resolving every symbol
|
|
585
|
+
// operand's opposite up front paid one host call per operand for ANY
|
|
586
|
+
// operation — on SEMA a halo-index query each, 30-200 ms of a plain
|
|
587
|
+
// dialogue turn's parse that computed nothing.
|
|
588
|
+
const asked: string[] = [];
|
|
589
|
+
const host: AluHost = {
|
|
590
|
+
meaningOf: async () => null,
|
|
591
|
+
continuation: async (b) => {
|
|
592
|
+
asked.push(dec(b));
|
|
593
|
+
return dec(b) === "large" ? enc("small") : null;
|
|
594
|
+
},
|
|
595
|
+
segment: (bytes) => {
|
|
596
|
+
const runs: Array<{ i: number; j: number }> = [];
|
|
597
|
+
for (let i = 0; i < bytes.length;) {
|
|
598
|
+
if (bytes[i] === 32) {
|
|
599
|
+
i++;
|
|
600
|
+
continue;
|
|
601
|
+
}
|
|
602
|
+
let j = i;
|
|
603
|
+
while (j < bytes.length && bytes[j] !== 32) j++;
|
|
604
|
+
runs.push({ i, j });
|
|
605
|
+
i = j;
|
|
606
|
+
}
|
|
607
|
+
return runs;
|
|
608
|
+
},
|
|
609
|
+
reach: Number.POSITIVE_INFINITY,
|
|
610
|
+
};
|
|
611
|
+
const u = new Alu({}, host);
|
|
612
|
+
// An operation that does not read opposites: nothing computed, nothing asked.
|
|
613
|
+
assert.deepEqual(await u.parse(enc("sqrt large")), []);
|
|
614
|
+
assert.deepEqual(asked, []);
|
|
615
|
+
// The inverse reads it: asked once, and grounded exactly as before.
|
|
616
|
+
const out = await u.parse(enc("opposite large"));
|
|
617
|
+
assert.equal(out.length, 1);
|
|
618
|
+
assert.equal(dec(out[0].bytes), "small");
|
|
619
|
+
assert.deepEqual(asked, ["large"]);
|
|
620
|
+
});
|
|
621
|
+
|
|
582
622
|
test("prefetchRecognisedOps bridges async recognition to a sync map", async () => {
|
|
583
623
|
// A stub host resonance: "the rate of change of" means a derivative.
|
|
584
624
|
const stub: AluResonance = {
|
package/src/bytes.ts
CHANGED
|
@@ -35,11 +35,21 @@ export function concat2(a: Uint8Array, b: Uint8Array): Uint8Array {
|
|
|
35
35
|
return out;
|
|
36
36
|
}
|
|
37
37
|
|
|
38
|
-
/** Latin-1 view of a byte span —
|
|
39
|
-
*
|
|
38
|
+
/** Latin-1 view of a byte span — ONE code unit per byte, so it is injective
|
|
39
|
+
* and safe as an exact cache key (every byte 0–255 maps to one code unit).
|
|
40
|
+
* Batched `String.fromCharCode`, chunked to stay within the engine's argument
|
|
41
|
+
* limit on long spans; the one definition every content-keyed memo uses.
|
|
42
|
+
* `apply` takes the typed array as its argument list directly — a spread
|
|
43
|
+
* walks it through the iterator protocol first, ~2.7× slower per short key. */
|
|
40
44
|
export function latin1(b: Uint8Array): string {
|
|
45
|
+
const n = b.length;
|
|
41
46
|
let s = "";
|
|
42
|
-
for (let
|
|
47
|
+
for (let i = 0; i < n; i += 4096) {
|
|
48
|
+
s += String.fromCharCode.apply(
|
|
49
|
+
null,
|
|
50
|
+
b.subarray(i, Math.min(i + 4096, n)) as unknown as number[],
|
|
51
|
+
);
|
|
52
|
+
}
|
|
43
53
|
return s;
|
|
44
54
|
}
|
|
45
55
|
|
package/src/canon.ts
CHANGED
|
@@ -44,6 +44,12 @@ const enc = new TextEncoder();
|
|
|
44
44
|
* deliberately conservative: punctuation, digits and word order are content
|
|
45
45
|
* and pass through untouched. */
|
|
46
46
|
export function textCanon(bytes: Uint8Array): Uint8Array {
|
|
47
|
+
if (isAscii(bytes)) return asciiCanon(bytes);
|
|
48
|
+
return unicodeCanon(bytes);
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/** The general reading — the DEFINITION of {@link textCanon}. */
|
|
52
|
+
function unicodeCanon(bytes: Uint8Array): Uint8Array {
|
|
47
53
|
const s = dec
|
|
48
54
|
.decode(bytes)
|
|
49
55
|
.normalize("NFKC")
|
|
@@ -52,6 +58,40 @@ export function textCanon(bytes: Uint8Array): Uint8Array {
|
|
|
52
58
|
return enc.encode(s);
|
|
53
59
|
}
|
|
54
60
|
|
|
61
|
+
function isAscii(bytes: Uint8Array): boolean {
|
|
62
|
+
for (let i = 0; i < bytes.length; i++) if (bytes[i] >= 0x80) return false;
|
|
63
|
+
return true;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/** {@link unicodeCanon} on ASCII input, byte for byte, without the string
|
|
67
|
+
* round trip. Exact, not approximate: NFKC is the identity on ASCII, the
|
|
68
|
+
* lowercase of ASCII is A–Z → a–z, and the ASCII members of the regex's `\s`
|
|
69
|
+
* are TAB, LF, VT, FF, CR and SPACE — so an interior run of them becomes one
|
|
70
|
+
* space and an edge run stays verbatim, exactly as the regex rewrites it.
|
|
71
|
+
* The canonicalizer runs once per probed span on the recognition and join
|
|
72
|
+
* paths, so its constant is paid thousands of times per response; test/148
|
|
73
|
+
* pins the agreement over random ASCII and over the edge cases. */
|
|
74
|
+
function asciiCanon(bytes: Uint8Array): Uint8Array {
|
|
75
|
+
const n = bytes.length;
|
|
76
|
+
const out = new Uint8Array(n);
|
|
77
|
+
let o = 0;
|
|
78
|
+
const ws = (b: number) => b === 0x20 || (b >= 0x09 && b <= 0x0d);
|
|
79
|
+
for (let i = 0; i < n;) {
|
|
80
|
+
const b = bytes[i];
|
|
81
|
+
if (!ws(b)) {
|
|
82
|
+
out[o++] = b >= 0x41 && b <= 0x5a ? b + 0x20 : b;
|
|
83
|
+
i++;
|
|
84
|
+
continue;
|
|
85
|
+
}
|
|
86
|
+
let j = i;
|
|
87
|
+
while (j < n && ws(bytes[j])) j++;
|
|
88
|
+
if (i > 0 && j < n) out[o++] = 0x20;
|
|
89
|
+
else for (let k = i; k < j; k++) out[o++] = bytes[k];
|
|
90
|
+
i = j;
|
|
91
|
+
}
|
|
92
|
+
return o === n ? out : out.slice(0, o);
|
|
93
|
+
}
|
|
94
|
+
|
|
55
95
|
/** 32-bit FNV-1a over a canonical key — the integer the store's canon index
|
|
56
96
|
* is keyed on. Same construction as the node table's content hash; a
|
|
57
97
|
* collision is resolved by verifying canon(stored) === key, never trusted. */
|
package/src/geometry.ts
CHANGED
|
@@ -204,16 +204,23 @@ export interface Grid {
|
|
|
204
204
|
// groups of `maxGroup` adjacent items into one via permute-then-add
|
|
205
205
|
// (positional seat binding), recursing until one root remains.
|
|
206
206
|
//
|
|
207
|
-
// FLAT per-level fold — one
|
|
208
|
-
//
|
|
209
|
-
//
|
|
210
|
-
//
|
|
211
|
-
//
|
|
212
|
-
//
|
|
213
|
-
//
|
|
207
|
+
// FLAT per-level fold — one loop per level (foldSlice), each group joined by
|
|
208
|
+
// joinFlat with the permute and add FUSED (`gist[d] += v[seat[d]]`, no scratch
|
|
209
|
+
// buffer), and subtree byte lengths carried incrementally on Folded (the old
|
|
210
|
+
// boundary scan re-walked subtrees every level — O(n log n)). The per-level
|
|
211
|
+
// SUPERPOSITION is byte-identical to the original recursive foldGroup: the
|
|
212
|
+
// same FP additions in the same order.
|
|
213
|
+
//
|
|
214
|
+
// THE SHAPE IS WRITTEN ONCE, OVER A FOLD ALGEBRA. Which items group under
|
|
215
|
+
// which parent is decided by the cut levels, the keyring and, inside an
|
|
216
|
+
// over-long row, the items' content key — never by what an item IS. So the
|
|
217
|
+
// grouping (foldSlice, groupByLevel) takes the two things it needs from an
|
|
218
|
+
// item, `join` and `key`, and runs unchanged over the vector fold (perception)
|
|
219
|
+
// and over the identity fold ({@link contentIdentity}), which can therefore
|
|
220
|
+
// never disagree about the tree.
|
|
214
221
|
//
|
|
215
222
|
// LINEAR fold — intermediate gists are NOT normalized; only the final root is
|
|
216
|
-
// (
|
|
223
|
+
// (`rootOf`'s single normalize). This is a deliberate change of similarity
|
|
217
224
|
// semantics from the original per-group normalize, not a cached optimization:
|
|
218
225
|
// the fold is now a pure linear operator — a superposition of positionally-
|
|
219
226
|
// bound leaf vectors — so an interior node carries its span's natural
|
|
@@ -233,95 +240,28 @@ export interface Grid {
|
|
|
233
240
|
* each extra level applies another seat permutation to the whole gist —
|
|
234
241
|
* near-identical inputs straddling such a cliff read as orthogonal
|
|
235
242
|
* (measured: 33-byte-identical prefixes at cos ≈ 0). */
|
|
236
|
-
function foldSlice(
|
|
237
|
-
|
|
238
|
-
|
|
243
|
+
function foldSlice<T>(
|
|
244
|
+
alg: FoldAlgebra<T>,
|
|
245
|
+
mg: number,
|
|
246
|
+
items: T[],
|
|
239
247
|
start: number,
|
|
240
248
|
count: number,
|
|
241
|
-
out:
|
|
249
|
+
out: T[],
|
|
242
250
|
force: boolean,
|
|
243
251
|
): void {
|
|
244
|
-
const mg = space.maxGroup;
|
|
245
|
-
const D = space.D;
|
|
246
252
|
const complete = count - (count % mg);
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
const kids = new Array<Sema>(size);
|
|
251
|
-
let len = 0;
|
|
252
|
-
for (let k = 0; k < size; k++) {
|
|
253
|
-
const f = items[at + k];
|
|
254
|
-
const slot = twoEndedSeat(space.seats.length, size, k);
|
|
255
|
-
const seat = space.seats[slot].fwd;
|
|
256
|
-
const v = f.tree.v;
|
|
257
|
-
// Fused permute-and-accumulate — same FP ops, same order as the old
|
|
258
|
-
// permuteInto + addInto pair, with no scratch buffer.
|
|
259
|
-
for (let d = 0; d < D; d++) gist[d] += v[seat[d]];
|
|
260
|
-
kids[k] = f.tree;
|
|
261
|
-
len += f.len;
|
|
262
|
-
}
|
|
263
|
-
out.push({ tree: sema(gist, null, kids), len });
|
|
264
|
-
};
|
|
265
|
-
|
|
266
|
-
for (let i = 0; i < complete; i += mg) foldAt(start + i, mg);
|
|
267
|
-
|
|
253
|
+
for (let i = 0; i < complete; i += mg) {
|
|
254
|
+
out.push(alg.join(items.slice(start + i, start + i + mg)));
|
|
255
|
+
}
|
|
268
256
|
const leftover = count - complete;
|
|
269
257
|
if (leftover === 0) return;
|
|
270
|
-
if (force && leftover >= 2)
|
|
271
|
-
|
|
272
|
-
}
|
|
273
|
-
|
|
274
|
-
function riverFold(space: Space, row: Folded[], stableBytes: number): Folded {
|
|
275
|
-
if (row.length === 0) {
|
|
276
|
-
const z = new Float32Array(space.D);
|
|
277
|
-
return { tree: sema(z, new Uint8Array(0), null), len: 0 };
|
|
278
|
-
}
|
|
279
|
-
let level = row;
|
|
280
|
-
while (level.length > 1) {
|
|
281
|
-
// Find the item index where accumulated bytes reaches stableBytes.
|
|
282
|
-
let boundary = level.length;
|
|
283
|
-
if (stableBytes > 0) {
|
|
284
|
-
let acc = 0;
|
|
285
|
-
for (let i = 0; i < level.length; i++) {
|
|
286
|
-
acc += level[i].len;
|
|
287
|
-
if (acc >= stableBytes) {
|
|
288
|
-
boundary = i + 1;
|
|
289
|
-
break;
|
|
290
|
-
}
|
|
291
|
-
}
|
|
292
|
-
}
|
|
293
|
-
|
|
294
|
-
const next: Folded[] = [];
|
|
295
|
-
if (boundary < level.length) {
|
|
296
|
-
// Prefix folds independently of the suffix — structural stability.
|
|
297
|
-
foldSlice(space, level, 0, boundary, next, true);
|
|
298
|
-
foldSlice(space, level, boundary, level.length - boundary, next, true);
|
|
299
|
-
} else {
|
|
300
|
-
foldSlice(space, level, 0, level.length, next, true);
|
|
301
|
-
}
|
|
302
|
-
level = next;
|
|
303
|
-
}
|
|
304
|
-
// LINEAR fold — this root normalize is the ONLY normalize of the entire
|
|
305
|
-
// fold; every intermediate gist stays unnormalized (see the folding
|
|
306
|
-
// header). Skipped for a single-leaf input: that root IS the shared
|
|
307
|
-
// alphabet vector (already unit), and normalizing in place would mutate the
|
|
308
|
-
// alphabet itself.
|
|
309
|
-
if (row.length > 1) normalize(level[0].tree.v);
|
|
310
|
-
return level[0];
|
|
258
|
+
if (force && leftover >= 2) {
|
|
259
|
+
out.push(alg.join(items.slice(start + complete, start + count)));
|
|
260
|
+
} else for (let i = complete; i < count; i++) out.push(items[start + i]);
|
|
311
261
|
}
|
|
312
262
|
|
|
313
263
|
// ---- public API ----
|
|
314
264
|
|
|
315
|
-
function bytesToLeaves(
|
|
316
|
-
alphabet: Alphabet,
|
|
317
|
-
bytes: Uint8Array,
|
|
318
|
-
): Folded[] {
|
|
319
|
-
return Array.from(bytes, (b, i) => {
|
|
320
|
-
const v = alphabet.vecs[b];
|
|
321
|
-
return { tree: sema(v, bytes.slice(i, i + 1), null), len: 1 };
|
|
322
|
-
});
|
|
323
|
-
}
|
|
324
|
-
|
|
325
265
|
/** CONTENT-DEFINED FOLD BOUNDARIES — where a byte stream segments, chosen by
|
|
326
266
|
* the bytes rather than by arithmetic.
|
|
327
267
|
*
|
|
@@ -384,47 +324,11 @@ function bytesToLeaves(
|
|
|
384
324
|
* Levels are read from the hash the cut was ACCEPTED at, not recomputed, so
|
|
385
325
|
* they cost nothing beyond the divisions already being done. */
|
|
386
326
|
|
|
387
|
-
// Cyclic-polynomial table for the bounded-window cut hash. Derived once from
|
|
388
|
-
// the fold's own mixing constant — no seed, no tuning. A byte contributes
|
|
389
|
-
// BUZ[b] on entering the window; the hash rotates by one per byte, so by the
|
|
390
|
-
// time that byte leaves, its contribution has travelled k places and is
|
|
391
|
-
// removed rotated by k. The rotation is taken at the use site rather than
|
|
392
|
-
// precomputed into a second table so the window width follows maxGroup
|
|
393
|
-
// instead of being frozen at one value.
|
|
394
|
-
const BUZ = new Uint32Array(256);
|
|
395
|
-
{
|
|
396
|
-
let x = 0x9e3779b9 >>> 0;
|
|
397
|
-
for (let i = 0; i < 256; i++) {
|
|
398
|
-
x = Math.imul(x ^ (x >>> 15), 2654435761) >>> 0;
|
|
399
|
-
x = (x ^ (x >>> 13)) >>> 0;
|
|
400
|
-
BUZ[i] = x;
|
|
401
|
-
}
|
|
402
|
-
}
|
|
403
|
-
|
|
404
|
-
/** BUZ rotated by the window width — what a byte's contribution has become by
|
|
405
|
-
* the time it leaves. Cached because the width follows `maxGroup`, which is
|
|
406
|
-
* fixed for a given space: built once, then a plain table lookup per byte. */
|
|
407
|
-
let buzOutTable: Uint32Array | null = null;
|
|
408
|
-
let buzOutWidth = -1;
|
|
409
|
-
function buzOut(k: number): Uint32Array {
|
|
410
|
-
if (buzOutWidth !== k || buzOutTable === null) {
|
|
411
|
-
const t = new Uint32Array(256);
|
|
412
|
-
for (let i = 0; i < 256; i++) {
|
|
413
|
-
const v = BUZ[i];
|
|
414
|
-
t[i] = ((v << k) | (v >>> (32 - k))) >>> 0;
|
|
415
|
-
}
|
|
416
|
-
buzOutTable = t;
|
|
417
|
-
buzOutWidth = k;
|
|
418
|
-
}
|
|
419
|
-
return buzOutTable;
|
|
420
|
-
}
|
|
421
|
-
|
|
422
327
|
function contentLevels(
|
|
423
328
|
space: Space,
|
|
424
329
|
bytes: Uint8Array,
|
|
425
330
|
): { cuts: number[]; levels: number[] } {
|
|
426
331
|
const W = space.maxGroup;
|
|
427
|
-
const minLen = W - 1;
|
|
428
332
|
const maxLen = space.seats.length;
|
|
429
333
|
// MEASURED AND REFUTED — making E[segment] equal W. A segment is at least
|
|
430
334
|
// `minLen` bytes and then cuts with probability p, so E[len] = minLen +
|
|
@@ -500,21 +404,20 @@ function contentLevels(
|
|
|
500
404
|
// old 0.870 0.902 0.492 0.879 0.888 0.441
|
|
501
405
|
// this 0.935 0.952 0.732 0.920 0.916 0.935
|
|
502
406
|
//
|
|
503
|
-
// The
|
|
504
|
-
//
|
|
505
|
-
// input, so the threshold fires at content-chosen positions on a gradient
|
|
407
|
+
// The register (the last k raw bytes, `h = h << 8 | byte`, mixed by the
|
|
408
|
+
// two-round avalanche below) has EXACTLY k bytes of memory and scrambles
|
|
409
|
+
// periodic input, so the threshold fires at content-chosen positions on a gradient
|
|
506
410
|
// just as it does on text — which is what leaves the `maxLen` fallback
|
|
507
411
|
// rarely engaged instead of carrying the phase. Segment lengths are
|
|
508
412
|
// unchanged in distribution (mean 5.2-7.2 against the old 5.4-6.0), so the
|
|
509
413
|
// mechanisms fitted to that distribution see the same scale.
|
|
510
414
|
//
|
|
511
|
-
// Cost
|
|
512
|
-
//
|
|
415
|
+
// Cost per byte: a shift, an OR and the avalanche's two multiplies — no
|
|
416
|
+
// table and no auxiliary structure. (An exact sliding-window
|
|
513
417
|
// minimum — winnowing — aligns slightly better still, 0.91-0.999, but its
|
|
514
418
|
// deque costs 51 MB/s against this rule's 112 and buys nothing the
|
|
515
419
|
// scrambling hash does not already give.)
|
|
516
420
|
const k = W;
|
|
517
|
-
const OUT = buzOut(k);
|
|
518
421
|
const cuts: number[] = [];
|
|
519
422
|
const levels: number[] = [];
|
|
520
423
|
const n = bytes.length;
|
|
@@ -716,7 +619,9 @@ function contentFoldSpan(
|
|
|
716
619
|
for (let i = 0; i + 1 < edges.length; i++) {
|
|
717
620
|
segs.push(flatFold(space, alphabet, span, edges[i], edges[i + 1]));
|
|
718
621
|
}
|
|
719
|
-
if (segs.length > 1)
|
|
622
|
+
if (segs.length > 1) {
|
|
623
|
+
return groupByLevel(vectorFold(space), space, segs, levels, 1);
|
|
624
|
+
}
|
|
720
625
|
return segs[0];
|
|
721
626
|
}
|
|
722
627
|
|
|
@@ -763,7 +668,7 @@ export interface ContentFold {
|
|
|
763
668
|
* streams. Verifying the bytes here would cost O(prefix) and defeat the
|
|
764
669
|
* whole point, so the obligation sits with the caller, and every caller
|
|
765
670
|
* discharges it structurally rather than by care: `perceiveDeposit` looks the
|
|
766
|
-
* entry up under `
|
|
671
|
+
* entry up under `latin1(bytes.subarray(0, L))` — the prefix's own bytes
|
|
767
672
|
* ARE the cache key — and a conversation's fold state advances only by
|
|
768
673
|
* append. A new caller that cannot make the same structural argument must
|
|
769
674
|
* pass no `prev` at all; the cold path is always correct.
|
|
@@ -792,7 +697,7 @@ export function contentFoldIncremental(
|
|
|
792
697
|
segs.push(hit ?? flatFold(space, alphabet, bytes, edges[i], edges[i + 1]));
|
|
793
698
|
}
|
|
794
699
|
const folded = segs.length > 1
|
|
795
|
-
? groupByLevel(space, segs, levels, 1)
|
|
700
|
+
? groupByLevel(vectorFold(space), space, segs, levels, 1)
|
|
796
701
|
: segs[0];
|
|
797
702
|
// THE ROOT IS NORMALIZED IN PLACE, A CACHED SEGMENT NEVER IS. With one
|
|
798
703
|
// segment — or with a grouping that passes a lone item through — `folded`
|
|
@@ -826,23 +731,25 @@ export function contentFoldIncremental(
|
|
|
826
731
|
* the cut levels are uniformly 0 and carry no signal. Identical subtrees
|
|
827
732
|
* fold to identical vectors, so the same items in the same order always
|
|
828
733
|
* choose the same split — the property the whole fold rests on. */
|
|
829
|
-
|
|
734
|
+
const ITEM_KEY_COORDS = 8;
|
|
735
|
+
function itemKey(v: ArrayLike<number>): number {
|
|
830
736
|
let h = 0x811c9dc5;
|
|
831
|
-
for (let d = 0; d <
|
|
737
|
+
for (let d = 0; d < ITEM_KEY_COORDS; d++) {
|
|
832
738
|
h = Math.imul(h ^ ((v[d] * 8192) | 0), 0x01000193) >>> 0;
|
|
833
739
|
}
|
|
834
740
|
return h >>> 0;
|
|
835
741
|
}
|
|
836
742
|
|
|
837
|
-
function groupByLevel(
|
|
743
|
+
function groupByLevel<T>(
|
|
744
|
+
alg: FoldAlgebra<T>,
|
|
838
745
|
space: Space,
|
|
839
|
-
items:
|
|
746
|
+
items: T[],
|
|
840
747
|
levels: number[],
|
|
841
748
|
level: number,
|
|
842
|
-
):
|
|
749
|
+
): T {
|
|
843
750
|
if (items.length === 1) return items[0];
|
|
844
751
|
const maxSeats = space.seats.length;
|
|
845
|
-
const groups:
|
|
752
|
+
const groups: T[] = [];
|
|
846
753
|
const groupLevels: number[] = [];
|
|
847
754
|
// Emit [from, to) as one group, splitting it at its STRONGEST interior cut
|
|
848
755
|
// whenever it would exceed the keyring.
|
|
@@ -869,7 +776,7 @@ function groupByLevel(
|
|
|
869
776
|
// equal — fall back to the ITEMS' own content. A group's gist is
|
|
870
777
|
// diverse where its cut level is not, so hashing it gives a
|
|
871
778
|
// content-determined split point where the level array has none.
|
|
872
|
-
const key =
|
|
779
|
+
const key = alg.key(items[j]);
|
|
873
780
|
if (
|
|
874
781
|
levels[j] > bestLevel ||
|
|
875
782
|
(levels[j] === bestLevel && key > bestKey)
|
|
@@ -880,12 +787,12 @@ function groupByLevel(
|
|
|
880
787
|
}
|
|
881
788
|
}
|
|
882
789
|
const part = items.slice(at, best + 1);
|
|
883
|
-
groups.push(part.length === 1 ? part[0] :
|
|
790
|
+
groups.push(part.length === 1 ? part[0] : alg.join(part));
|
|
884
791
|
groupLevels.push(levels[best]);
|
|
885
792
|
at = best + 1;
|
|
886
793
|
}
|
|
887
794
|
const slice = items.slice(at, to);
|
|
888
|
-
groups.push(slice.length === 1 ? slice[0] :
|
|
795
|
+
groups.push(slice.length === 1 ? slice[0] : alg.join(slice));
|
|
889
796
|
};
|
|
890
797
|
let start = 0;
|
|
891
798
|
for (let i = 0; i <= levels.length; i++) {
|
|
@@ -898,10 +805,10 @@ function groupByLevel(
|
|
|
898
805
|
if (groups.length === items.length) {
|
|
899
806
|
// This level split nothing — climb rather than spin.
|
|
900
807
|
return level < 24
|
|
901
|
-
? groupByLevel(space, items, levels, level + 1)
|
|
902
|
-
:
|
|
808
|
+
? groupByLevel(alg, space, items, levels, level + 1)
|
|
809
|
+
: riverFold(alg, space, items);
|
|
903
810
|
}
|
|
904
|
-
return groupByLevel(space, groups, groupLevels, level + 1);
|
|
811
|
+
return groupByLevel(alg, space, groups, groupLevels, level + 1);
|
|
905
812
|
}
|
|
906
813
|
|
|
907
814
|
/** Join a row of already-folded items as one unnormalized node — the same
|
|
@@ -924,6 +831,123 @@ function joinFlat(space: Space, items: Folded[]): Folded {
|
|
|
924
831
|
return { tree: sema(gist, null, kids), len };
|
|
925
832
|
}
|
|
926
833
|
|
|
834
|
+
/** What the fold's SHAPE asks of the items it groups — see the note at the
|
|
835
|
+
* top of the folding section. `join` makes the parent of two or more items,
|
|
836
|
+
* children bound in seat order; `key` is {@link itemKey} of the item's raw
|
|
837
|
+
* gist, the one place the shape reads an item's content. */
|
|
838
|
+
interface FoldAlgebra<T> {
|
|
839
|
+
join(items: T[]): T;
|
|
840
|
+
key(item: T): number;
|
|
841
|
+
}
|
|
842
|
+
|
|
843
|
+
/** The vector fold — perception's algebra. */
|
|
844
|
+
const vectorFolds = new WeakMap<Space, FoldAlgebra<Folded>>();
|
|
845
|
+
function vectorFold(space: Space): FoldAlgebra<Folded> {
|
|
846
|
+
let alg = vectorFolds.get(space);
|
|
847
|
+
if (alg === undefined) {
|
|
848
|
+
alg = {
|
|
849
|
+
join: (items) => joinFlat(space, items),
|
|
850
|
+
key: (item) => itemKey(item.tree.v),
|
|
851
|
+
};
|
|
852
|
+
vectorFolds.set(space, alg);
|
|
853
|
+
}
|
|
854
|
+
return alg;
|
|
855
|
+
}
|
|
856
|
+
|
|
857
|
+
/** An item of the identity fold: the node it names (null = names nothing),
|
|
858
|
+
* the byte span it covers, and its raw gist read one coordinate at a time, on
|
|
859
|
+
* demand. */
|
|
860
|
+
interface IdentityItem {
|
|
861
|
+
id: number | null;
|
|
862
|
+
from: number;
|
|
863
|
+
to: number;
|
|
864
|
+
coord: (p: number) => number;
|
|
865
|
+
key?: number;
|
|
866
|
+
}
|
|
867
|
+
|
|
868
|
+
/** The raw gist coordinate `p` of a node whose children are `kids` — the
|
|
869
|
+
* coordinate {@link flatFold}/{@link joinFlat} would have written: the same
|
|
870
|
+
* float32 additions, in the same order, from a zero start. */
|
|
871
|
+
function boundCoord(
|
|
872
|
+
space: Space,
|
|
873
|
+
n: number,
|
|
874
|
+
kid: (k: number, at: number) => number,
|
|
875
|
+
p: number,
|
|
876
|
+
): number {
|
|
877
|
+
let acc = 0;
|
|
878
|
+
for (let k = 0; k < n; k++) {
|
|
879
|
+
const seat = space.seats[twoEndedSeat(space.seats.length, n, k)].fwd;
|
|
880
|
+
acc = Math.fround(acc + kid(k, seat[p]));
|
|
881
|
+
}
|
|
882
|
+
return acc;
|
|
883
|
+
}
|
|
884
|
+
|
|
885
|
+
/** The node a byte stream's content fold NAMES — `foldTree` over
|
|
886
|
+
* {@link contentFoldSpan}'s tree, without building a single vector.
|
|
887
|
+
*
|
|
888
|
+
* A fold names a node only when every child is named, so identity needs the
|
|
889
|
+
* tree's SHAPE and the store's answer for each node — and the shape is a
|
|
890
|
+
* function of the bytes (cuts and levels) plus, inside an over-long row, each
|
|
891
|
+
* item's {@link itemKey}: eight coordinates of its raw gist. Those are read
|
|
892
|
+
* lazily through {@link boundCoord}, bit-identical to the coordinates the
|
|
893
|
+
* vector fold computes, so the grouping (the SAME {@link groupByLevel}) cannot
|
|
894
|
+
* differ. `segment(from, to)` names a level-0 segment (one flat node over
|
|
895
|
+
* single-byte atoms, or the atom itself); `branch(kids, from, to)` names the
|
|
896
|
+
* group covering [from, to) — `kids` holds null for an unnamed child, because
|
|
897
|
+
* the store names a branch by its BYTES when its children do not name it
|
|
898
|
+
* (the write side's step 1b, store.ts `intern`), so an unnamed child does not
|
|
899
|
+
* settle its ancestors. Every item's grouping key is read from its content,
|
|
900
|
+
* named or not, for the same reason. An empty stream is the caller's: its
|
|
901
|
+
* fold is the alphabet's zero-byte leaf, not a segment. */
|
|
902
|
+
export function contentIdentity(
|
|
903
|
+
space: Space,
|
|
904
|
+
alphabet: Alphabet,
|
|
905
|
+
bytes: Uint8Array,
|
|
906
|
+
segment: (from: number, to: number) => number | null,
|
|
907
|
+
branch: (
|
|
908
|
+
kids: ReadonlyArray<number | null>,
|
|
909
|
+
from: number,
|
|
910
|
+
to: number,
|
|
911
|
+
) => number | null,
|
|
912
|
+
): number | null {
|
|
913
|
+
const { cuts, levels } = contentLevels(space, bytes);
|
|
914
|
+
const edges = [0, ...cuts, bytes.length];
|
|
915
|
+
const segs: IdentityItem[] = [];
|
|
916
|
+
for (let i = 0; i + 1 < edges.length; i++) {
|
|
917
|
+
const from = edges[i], n = edges[i + 1] - from;
|
|
918
|
+
segs.push({
|
|
919
|
+
id: segment(from, edges[i + 1]),
|
|
920
|
+
from,
|
|
921
|
+
to: edges[i + 1],
|
|
922
|
+
coord: n === 1 ? (p) => alphabet.vecs[bytes[from]][p] : (p) =>
|
|
923
|
+
boundCoord(
|
|
924
|
+
space,
|
|
925
|
+
n,
|
|
926
|
+
(k, at) => alphabet.vecs[bytes[from + k]][at],
|
|
927
|
+
p,
|
|
928
|
+
),
|
|
929
|
+
});
|
|
930
|
+
}
|
|
931
|
+
if (segs.length === 1) return segs[0].id;
|
|
932
|
+
const alg: FoldAlgebra<IdentityItem> = {
|
|
933
|
+
join: (items) => {
|
|
934
|
+
const from = items[0].from, to = items[items.length - 1].to;
|
|
935
|
+
return {
|
|
936
|
+
id: branch(items.map((it) => it.id), from, to),
|
|
937
|
+
from,
|
|
938
|
+
to,
|
|
939
|
+
coord: (p) =>
|
|
940
|
+
boundCoord(space, items.length, (k, at) => items[k].coord(at), p),
|
|
941
|
+
};
|
|
942
|
+
},
|
|
943
|
+
key: (item) =>
|
|
944
|
+
item.key ??= itemKey(
|
|
945
|
+
Array.from({ length: ITEM_KEY_COORDS }, (_, d) => item.coord(d)),
|
|
946
|
+
),
|
|
947
|
+
};
|
|
948
|
+
return groupByLevel(alg, space, segs, levels, 1).id;
|
|
949
|
+
}
|
|
950
|
+
|
|
927
951
|
/** One segment as a single unnormalized node: leaf per byte, each bound into
|
|
928
952
|
* a seat derived from its position relative to BOTH segment ends.
|
|
929
953
|
*
|
|
@@ -971,14 +995,6 @@ function flatFold(
|
|
|
971
995
|
return { tree: sema(gist, null, kids), len: n };
|
|
972
996
|
}
|
|
973
997
|
|
|
974
|
-
/* * The stable-prefix segmented fold (fold-contract.md). Each segment between
|
|
975
|
-
* consecutive boundaries folds PLAINLY and independently; segment roots
|
|
976
|
-
* join left-nested, and only the final root is normalized (the linear-fold
|
|
977
|
-
* contract: one normalize per perception). A segment's own inner splits
|
|
978
|
-
* need no recursion here: a nested learnt prefix is itself an earlier
|
|
979
|
-
* boundary, so the left-nested join reproduces every intermediate learnt
|
|
980
|
-
* root ((s₀·s₁) IS the root the store learnt for the first two segments'
|
|
981
|
-
* bytes, and so on). */
|
|
982
998
|
/** A fold's ROOT: ONE normalize per perception, at the root, exactly as
|
|
983
999
|
* riverFold did — the interior stays raw (the linear-fold contract).
|
|
984
1000
|
*
|
|
@@ -995,6 +1011,14 @@ function rootOf(f: Folded): Sema {
|
|
|
995
1011
|
return f.tree;
|
|
996
1012
|
}
|
|
997
1013
|
|
|
1014
|
+
/** The stable-prefix segmented fold (fold-contract.md). Each segment between
|
|
1015
|
+
* consecutive boundaries folds PLAINLY and independently; segment roots
|
|
1016
|
+
* join left-nested, and only the final root is normalized (the linear-fold
|
|
1017
|
+
* contract: one normalize per perception). A segment's own inner splits
|
|
1018
|
+
* need no recursion here: a nested learnt prefix is itself an earlier
|
|
1019
|
+
* boundary, so the left-nested join reproduces every intermediate learnt
|
|
1020
|
+
* root ((s₀·s₁) IS the root the store learnt for the first two segments'
|
|
1021
|
+
* bytes, and so on). */
|
|
998
1022
|
function stablePrefixFold(
|
|
999
1023
|
space: Space,
|
|
1000
1024
|
alphabet: Alphabet,
|
|
@@ -1113,11 +1137,16 @@ export function riverFoldRaw(space: Space, row: Folded[]): Folded {
|
|
|
1113
1137
|
const z = new Float32Array(space.D);
|
|
1114
1138
|
return { tree: sema(z, new Uint8Array(0), null), len: 0 };
|
|
1115
1139
|
}
|
|
1116
|
-
|
|
1140
|
+
return riverFold(vectorFold(space), space, row);
|
|
1141
|
+
}
|
|
1142
|
+
|
|
1143
|
+
/** The river's fixed-arity shape over any fold algebra: groups of maxGroup,
|
|
1144
|
+
* the trailing partial group forced, level after level to one root. */
|
|
1145
|
+
function riverFold<T>(alg: FoldAlgebra<T>, space: Space, row: T[]): T {
|
|
1117
1146
|
let level = row;
|
|
1118
1147
|
while (level.length > 1) {
|
|
1119
|
-
const next:
|
|
1120
|
-
foldSlice(space, level, 0, level.length, next, true);
|
|
1148
|
+
const next: T[] = [];
|
|
1149
|
+
foldSlice(alg, space.maxGroup, level, 0, level.length, next, true);
|
|
1121
1150
|
level = next;
|
|
1122
1151
|
}
|
|
1123
1152
|
return level[0];
|