@hviana/sema 0.8.1 → 0.8.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/src/config.d.ts +17 -0
- package/dist/src/config.js +18 -0
- package/dist/src/meter.d.ts +25 -0
- package/dist/src/meter.js +44 -0
- package/dist/src/mind/corpus.d.ts +40 -0
- package/dist/src/mind/corpus.js +149 -0
- package/dist/src/mind/graph-search.d.ts +7 -0
- package/dist/src/mind/graph-search.js +235 -24
- package/dist/src/mind/index.d.ts +3 -1
- package/dist/src/mind/index.js +1 -0
- package/dist/src/mind/match.d.ts +8 -3
- package/dist/src/mind/match.js +142 -58
- package/dist/src/mind/mechanisms/cast.js +18 -2
- package/dist/src/mind/mechanisms/cover.js +6 -0
- package/dist/src/mind/mind.d.ts +55 -0
- package/dist/src/mind/mind.js +72 -2
- package/dist/src/mind/pipeline.js +25 -6
- package/dist/src/mind/reasoning.d.ts +5 -1
- package/dist/src/mind/reasoning.js +54 -1
- package/dist/src/mind/traverse.js +9 -1
- package/dist/src/mind/types.d.ts +22 -0
- package/docs/failures/tempting-but-wrong.md +31 -2
- package/jsr.json +1 -1
- package/package.json +1 -1
- package/src/config.ts +35 -0
- package/src/meter.ts +47 -0
- package/src/mind/corpus.ts +202 -0
- package/src/mind/graph-search.ts +252 -23
- package/src/mind/index.ts +8 -1
- package/src/mind/match.ts +143 -54
- package/src/mind/mechanisms/cast.ts +17 -1
- package/src/mind/mechanisms/cover.ts +5 -0
- package/src/mind/mind.ts +123 -0
- package/src/mind/pipeline.ts +30 -6
- package/src/mind/reasoning.ts +55 -0
- package/src/mind/traverse.ts +9 -1
- package/src/mind/types.ts +26 -0
- package/test/100-complete-grounding-trace.test.mjs +109 -0
- package/test/101-alignment-gap-bound.test.mjs +106 -0
- package/test/102-production-composes-at-scale.test.mjs +110 -0
- package/test/103-alignment-gap-budget.test.mjs +89 -0
- package/test/104-composition-is-reported.test.mjs +90 -0
- package/test/105-derive-through-reports-its-refusal.test.mjs +113 -0
- package/test/106-the-join-fires.test.mjs +94 -0
- package/test/107-the-join-is-counted.test.mjs +81 -0
- package/test/108-the-join-chains.test.mjs +78 -0
- package/test/109-the-pivot-is-counted.test.mjs +60 -0
- package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
- package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
- package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
- package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
- package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
- package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
- package/test/117-corpus-search.test.mjs +171 -0
- package/test/14-scaling.test.mjs +10 -7
- package/test/76-reference-binding.test.mjs +6 -1
- package/test/89-completion-recursion.test.mjs +30 -5
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
// 114-alignment-budget-is-per-sweep.test.mjs — each alignment sweep owns its
|
|
2
|
+
// budget, so the NEAREST continuation survives either side's exhaustion.
|
|
3
|
+
//
|
|
4
|
+
// THE DEFECT THIS CLOSES (found by an adversarial review of the E3 commit).
|
|
5
|
+
// `alignAround` bounds the work it may spend on (queryGap, contextGap) pairs.
|
|
6
|
+
// The counter was shared between the RIGHT sweep and the LEFT sweep, so once the
|
|
7
|
+
// right side spent the whole budget the left loop's guard was false on entry:
|
|
8
|
+
// zero iterations, and the NEAREST left continuation was lost — the exact
|
|
9
|
+
// opposite of the law the commit states ("an exhausted budget drops the FAR
|
|
10
|
+
// continuations and never the near ones"). It was a regression against the
|
|
11
|
+
// pre-change tree, which bounded each sweep independently.
|
|
12
|
+
//
|
|
13
|
+
// WHAT IS PINNED, structurally and with no magic number: a pair whose two
|
|
14
|
+
// divergent flanks both exceed the work budget still aligns on BOTH sides of the
|
|
15
|
+
// anchor — the left match is present as its own query span, and so is the right.
|
|
16
|
+
// Before the fix the left one is missing, whatever the budget.
|
|
17
|
+
|
|
18
|
+
import { test } from "node:test";
|
|
19
|
+
import assert from "node:assert/strict";
|
|
20
|
+
import { alignAround } from "../dist/src/mind/match.js";
|
|
21
|
+
|
|
22
|
+
const enc = new TextEncoder();
|
|
23
|
+
const dec = new TextDecoder();
|
|
24
|
+
|
|
25
|
+
/** The minimal context `alignAround` reads. There is no budget to inject any
|
|
26
|
+
* more: the sweep's work is proportional to the bytes a run spans. */
|
|
27
|
+
const ctxWith = () => ({ space: { maxGroup: 4 } });
|
|
28
|
+
|
|
29
|
+
/** A shared head, a shared anchor, and divergent flanks on both sides. */
|
|
30
|
+
function pair(flankLen) {
|
|
31
|
+
const q = `MATCH${"a".repeat(6)}SEED${"Q".repeat(flankLen)}`;
|
|
32
|
+
const c = `MATCH${"b".repeat(11)}SEED${"W".repeat(flankLen)}`;
|
|
33
|
+
return { q, c, at: q.indexOf("SEED") };
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
const spansOf = (res, text, needle) => {
|
|
37
|
+
const target = enc.encode(needle);
|
|
38
|
+
return res.matched.filter(([s, e]) => {
|
|
39
|
+
const got = enc.encode(text).subarray(s, e);
|
|
40
|
+
return got.length >= target.length &&
|
|
41
|
+
dec.decode(got).includes(needle);
|
|
42
|
+
});
|
|
43
|
+
};
|
|
44
|
+
|
|
45
|
+
test("both sides of the anchor align even when each flank exceeds the budget", () => {
|
|
46
|
+
// The budget must be reachable for the NEAR left continuation (its own
|
|
47
|
+
// 6+11-byte gap costs on the order of a hundred pair-explorations) and must be
|
|
48
|
+
// EXHAUSTED by the right flank (40x40 divergent bytes cost many hundreds).
|
|
49
|
+
// 512 sits between the two: it is the range where a shared counter starves the
|
|
50
|
+
// left sweep and a per-sweep one does not.
|
|
51
|
+
const { q, c, at } = pair(40);
|
|
52
|
+
for (const budget of [512, 4096]) {
|
|
53
|
+
const res = alignAround(
|
|
54
|
+
ctxWith(budget),
|
|
55
|
+
enc.encode(q),
|
|
56
|
+
enc.encode(c),
|
|
57
|
+
at,
|
|
58
|
+
at,
|
|
59
|
+
);
|
|
60
|
+
assert.ok(
|
|
61
|
+
spansOf(res, q, "MATCH").length > 0,
|
|
62
|
+
`the nearest LEFT continuation must survive the right sweep's budget ` +
|
|
63
|
+
`(budget ${budget}, matched ${JSON.stringify(res.matched)})`,
|
|
64
|
+
);
|
|
65
|
+
assert.ok(
|
|
66
|
+
spansOf(res, q, "SEED").length > 0,
|
|
67
|
+
"and the anchor itself is of course still matched",
|
|
68
|
+
);
|
|
69
|
+
}
|
|
70
|
+
});
|
|
71
|
+
|
|
72
|
+
test("the run chosen is the one with the smallest total gap", () => {
|
|
73
|
+
// The criterion the enumeration used to compute, now computed by the walk:
|
|
74
|
+
// smallest qGap + cGap, ties to the smaller query gap. Two continuations are
|
|
75
|
+
// offered at different totals and the NEARER one must win.
|
|
76
|
+
const enc = new TextEncoder();
|
|
77
|
+
const head = "common head ";
|
|
78
|
+
const far = "FAR";
|
|
79
|
+
const near = "NEAR";
|
|
80
|
+
const q = enc.encode(
|
|
81
|
+
head + "x".repeat(4) + near + "q" + "z".repeat(30) + far,
|
|
82
|
+
);
|
|
83
|
+
const c = enc.encode(head + "y".repeat(9) + near + "c");
|
|
84
|
+
const at = head.length - 1;
|
|
85
|
+
const { matched } = alignAround(ctxWith(), q, c, at, at);
|
|
86
|
+
const got = matched.map(([a, b]) =>
|
|
87
|
+
new TextDecoder().decode(q.subarray(a, b))
|
|
88
|
+
);
|
|
89
|
+
assert.ok(
|
|
90
|
+
got.includes("NEAR"),
|
|
91
|
+
"the nearest continuation must be found: " + JSON.stringify(got),
|
|
92
|
+
);
|
|
93
|
+
});
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
// 116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs — EXTENSION
|
|
2
|
+
// judges the question by the pipeline's remainder, with the pipeline's W floor.
|
|
3
|
+
//
|
|
4
|
+
// TWO CORRECTIONS FROM THE ADVERSARIAL REVIEW, in one definition:
|
|
5
|
+
//
|
|
6
|
+
// M3 — the reasoner's `uncovered` was built from the ladder's `accounted`, a
|
|
7
|
+
// COST quantity. The pipeline itself documents that `accounted` can be EMPTY
|
|
8
|
+
// while nothing is unexplained (a query fully explained by one computed span
|
|
9
|
+
// plus bridged connectors), and it already builds the genuine remainder as
|
|
10
|
+
// `[...decided.accounted, ...pre.computed]`. The reasoner now uses that same
|
|
11
|
+
// reading — computed once, used by both the extension gate and the fuse gate.
|
|
12
|
+
//
|
|
13
|
+
// M4 — the boundary. The pipeline treats a remainder under W as bridging
|
|
14
|
+
// punctuation, and the reasoner inherits that floor. The subtlety the review
|
|
15
|
+
// raised is that a one-word tail CARRIES ITS SEPARATOR: `" why"` is exactly W
|
|
16
|
+
// bytes, so it IS a remainder and licenses no extension, while `" famous for"`
|
|
17
|
+
// is well over and does. Both sides are pinned here, because the boundary is
|
|
18
|
+
// where the reviewer's own demonstration sat.
|
|
19
|
+
|
|
20
|
+
import { test } from "node:test";
|
|
21
|
+
import assert from "node:assert/strict";
|
|
22
|
+
import { Mind } from "../dist/src/index.js";
|
|
23
|
+
import { SQliteStore } from "../dist/src/store-sqlite.js";
|
|
24
|
+
|
|
25
|
+
const LINKS = [
|
|
26
|
+
["What is the capital of France", "The capital of France is Paris"],
|
|
27
|
+
["Paris", "Paris is famous for the Eiffel Tower"],
|
|
28
|
+
];
|
|
29
|
+
|
|
30
|
+
/** The test/110 chain: the second link is reachable only by hopping. */
|
|
31
|
+
async function chain() {
|
|
32
|
+
const store = new SQliteStore({ path: ":memory:", D: 1024 });
|
|
33
|
+
const mind = new Mind({ seed: 7, store, profile: true });
|
|
34
|
+
await mind.ingest(LINKS);
|
|
35
|
+
return mind;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
const text = (resp) =>
|
|
39
|
+
new TextDecoder().decode(resp.bytes).replace(/\0+/g, "").trim();
|
|
40
|
+
|
|
41
|
+
test("a remainder the pipeline calls real licenses the hop", async () => {
|
|
42
|
+
const mind = await chain();
|
|
43
|
+
const out = text(
|
|
44
|
+
await mind.respond("What is the capital of France famous for"),
|
|
45
|
+
);
|
|
46
|
+
assert.equal(
|
|
47
|
+
out,
|
|
48
|
+
LINKS[1][1],
|
|
49
|
+
"a remainder well over one window must still reach the next link",
|
|
50
|
+
);
|
|
51
|
+
await mind.store.close();
|
|
52
|
+
});
|
|
53
|
+
|
|
54
|
+
test("a one-word remainder is exactly one window, and licenses nothing", async () => {
|
|
55
|
+
// `" why"` is W bytes once its separator is counted: per the pipeline's own
|
|
56
|
+
// floor it is a remainder, and per the extension law a remainder the step
|
|
57
|
+
// cannot carry licenses no hop — so the grounded fact stands, AND the refusal
|
|
58
|
+
// is visible: a brake that leaves no trace is the silent cut AGENTS §6
|
|
59
|
+
// forbids, so the rationale must say the extension was declined for want of
|
|
60
|
+
// question material.
|
|
61
|
+
const mind = await chain();
|
|
62
|
+
const steps = [];
|
|
63
|
+
const out = text(
|
|
64
|
+
await mind.respond(
|
|
65
|
+
"What is the capital of France why",
|
|
66
|
+
(s) => steps.push(s),
|
|
67
|
+
),
|
|
68
|
+
);
|
|
69
|
+
assert.equal(
|
|
70
|
+
out,
|
|
71
|
+
LINKS[0][1],
|
|
72
|
+
"a step carrying none of the remainder must not be taken",
|
|
73
|
+
);
|
|
74
|
+
const refused = steps.find((s) => s.mechanism.at(-1) === "pivotRefused");
|
|
75
|
+
assert.ok(refused, "the refusal must be reported, not silent");
|
|
76
|
+
assert.match(
|
|
77
|
+
String(refused.note),
|
|
78
|
+
/carries none of the question material/,
|
|
79
|
+
"and it must say why it refused",
|
|
80
|
+
);
|
|
81
|
+
assert.ok(
|
|
82
|
+
(refused.outputs ?? []).length > 0,
|
|
83
|
+
"and name the material it left uncovered",
|
|
84
|
+
);
|
|
85
|
+
await mind.store.close();
|
|
86
|
+
});
|
|
87
|
+
|
|
88
|
+
test("a computed-span query has no phantom remainder (M3)", async () => {
|
|
89
|
+
// A pure computation: the ladder's `accounted` is not a coverage reading, so
|
|
90
|
+
// the reasoner must not see a remainder the pipeline says is not there. The
|
|
91
|
+
// check is that the answer is the computed one and the engine is not dragged
|
|
92
|
+
// into an unrelated chain.
|
|
93
|
+
const mind = await chain();
|
|
94
|
+
const out = text(await mind.respond("2+2"));
|
|
95
|
+
assert.ok(
|
|
96
|
+
out.length > 0,
|
|
97
|
+
`a computation must still be answered, got ${JSON.stringify(out)}`,
|
|
98
|
+
);
|
|
99
|
+
await mind.store.close();
|
|
100
|
+
});
|
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
// 117-corpus-search.test.mjs — reading the trained memory back out of the DAG.
|
|
2
|
+
//
|
|
3
|
+
// WHAT IS PINNED. Two methods and their division of labour:
|
|
4
|
+
// • `searchCorpus(bytes, limit?)` — MULTIMODAL: bytes in, bytes out, no notion
|
|
5
|
+
// of text or encoding anywhere in it;
|
|
6
|
+
// • `searchCorpusText(text, limit?)` — the text case, which encodes, calls the
|
|
7
|
+
// multimodal one, and decodes. The search itself exists ONCE (src/mind/
|
|
8
|
+
// corpus.ts, over the machinery an answer already uses: `recognise` for the
|
|
9
|
+
// resolved subtrees, `edgeAncestors` for the climb, `nextFirst` for the
|
|
10
|
+
// continuation).
|
|
11
|
+
//
|
|
12
|
+
// Both are deterministic (same seed, same order, same query ⇒ byte-identical
|
|
13
|
+
// results), both report a miss as a STATE in the byte layer and as prose only in
|
|
14
|
+
// the text layer, and browsing takes the caller's own offset instead of a random
|
|
15
|
+
// draw.
|
|
16
|
+
|
|
17
|
+
import { test } from "node:test";
|
|
18
|
+
import assert from "node:assert/strict";
|
|
19
|
+
import { Mind, SQliteStore } from "../dist/src/index.js";
|
|
20
|
+
|
|
21
|
+
const enc = new TextEncoder();
|
|
22
|
+
const dec = new TextDecoder();
|
|
23
|
+
|
|
24
|
+
/** A small deposited corpus: three experience pairs, no trained store needed. */
|
|
25
|
+
async function fixture() {
|
|
26
|
+
const mind = new Mind({
|
|
27
|
+
seed: 7,
|
|
28
|
+
store: new SQliteStore({ path: ":memory:" }),
|
|
29
|
+
});
|
|
30
|
+
await mind.ingest([
|
|
31
|
+
["the capital of France", "Paris is the capital of France."],
|
|
32
|
+
["the capital of Portugal", "Lisbon is the capital of Portugal."],
|
|
33
|
+
["who wrote Hamlet", "Shakespeare wrote Hamlet."],
|
|
34
|
+
]);
|
|
35
|
+
return mind;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
const asText = (b) => dec.decode(b).replace(/\0+/g, "").trim();
|
|
39
|
+
|
|
40
|
+
test("the multimodal search takes bytes and returns bytes", async () => {
|
|
41
|
+
const mind = await fixture();
|
|
42
|
+
const result = mind.searchCorpus(enc.encode("the capital of France"));
|
|
43
|
+
assert.ok(result.pairs.length > 0, "the deposited pair must be found");
|
|
44
|
+
const pair = result.pairs[0];
|
|
45
|
+
assert.ok(pair.context instanceof Uint8Array, "context is BYTES, not text");
|
|
46
|
+
assert.ok(pair.continuation instanceof Uint8Array);
|
|
47
|
+
assert.ok(typeof pair.contextId === "number");
|
|
48
|
+
assert.ok(asText(pair.context).includes("capital"));
|
|
49
|
+
assert.equal(result.browsed, false);
|
|
50
|
+
assert.equal(result.miss, "matched");
|
|
51
|
+
assert.ok(result.totalContexts > 0, "the store's own context count is read");
|
|
52
|
+
await mind.store.close();
|
|
53
|
+
});
|
|
54
|
+
|
|
55
|
+
test("the text helper is the SAME search, converted", async () => {
|
|
56
|
+
const mind = await fixture();
|
|
57
|
+
const bytes = mind.searchCorpus(enc.encode("the capital of France"));
|
|
58
|
+
const text = mind.searchCorpusText("the capital of France");
|
|
59
|
+
assert.equal(
|
|
60
|
+
text.pairs.length,
|
|
61
|
+
bytes.pairs.length,
|
|
62
|
+
"one search, two views — the helper must not run a second one",
|
|
63
|
+
);
|
|
64
|
+
assert.equal(text.pairs[0].contextId, bytes.pairs[0].contextId);
|
|
65
|
+
assert.equal(typeof text.pairs[0].context, "string");
|
|
66
|
+
assert.ok(text.pairs[0].context.includes("capital"));
|
|
67
|
+
assert.equal(text.note, undefined, "a match needs no note");
|
|
68
|
+
await mind.store.close();
|
|
69
|
+
});
|
|
70
|
+
|
|
71
|
+
test("both are deterministic across identical calls", async () => {
|
|
72
|
+
const mind = await fixture();
|
|
73
|
+
const a = mind.searchCorpus(enc.encode("the capital of France"));
|
|
74
|
+
const b = mind.searchCorpus(enc.encode("the capital of France"));
|
|
75
|
+
assert.deepEqual(
|
|
76
|
+
a.pairs.map((p) => [p.contextId, p.continuationId, asText(p.context)]),
|
|
77
|
+
b.pairs.map((p) => [p.contextId, p.continuationId, asText(p.context)]),
|
|
78
|
+
"same store + same query ⇒ the same pairs in the same order",
|
|
79
|
+
);
|
|
80
|
+
assert.deepEqual(
|
|
81
|
+
mind.searchCorpusText("the capital of France").pairs,
|
|
82
|
+
mind.searchCorpusText("the capital of France").pairs,
|
|
83
|
+
);
|
|
84
|
+
await mind.store.close();
|
|
85
|
+
});
|
|
86
|
+
|
|
87
|
+
test("a miss is a state in bytes and prose only in text", async () => {
|
|
88
|
+
const mind = await fixture();
|
|
89
|
+
const bytes = mind.searchCorpus(enc.encode("zzzq nothing at all"));
|
|
90
|
+
assert.equal(bytes.pairs.length, 0);
|
|
91
|
+
assert.ok(
|
|
92
|
+
bytes.miss === "nothing-resolved" || bytes.miss === "no-continuations",
|
|
93
|
+
`the byte layer reports a STATE, got ${bytes.miss}`,
|
|
94
|
+
);
|
|
95
|
+
assert.equal(bytes.note, undefined, "no prose in the byte layer");
|
|
96
|
+
const text = mind.searchCorpusText("zzzq nothing at all");
|
|
97
|
+
assert.equal(text.pairs.length, 0);
|
|
98
|
+
assert.equal(typeof text.note, "string", "the text layer says it in words");
|
|
99
|
+
await mind.store.close();
|
|
100
|
+
});
|
|
101
|
+
|
|
102
|
+
test("determinism holds ACROSS instances, not just across calls", async () => {
|
|
103
|
+
// The invariant is about the engine, not about one object: the same seed,
|
|
104
|
+
// the same deposit order and the same query must give byte-identical results
|
|
105
|
+
// from a FRESH Mind over a fresh store built the same way — both for a search
|
|
106
|
+
// (whose ids come from the store's intern order) and for a browse.
|
|
107
|
+
const a = await fixture();
|
|
108
|
+
const b = await fixture();
|
|
109
|
+
const shape = (r) =>
|
|
110
|
+
r.pairs.map((p) => [
|
|
111
|
+
p.contextId,
|
|
112
|
+
p.continuationId,
|
|
113
|
+
p.matchedBytes,
|
|
114
|
+
Array.from(p.context),
|
|
115
|
+
Array.from(p.continuation),
|
|
116
|
+
]);
|
|
117
|
+
assert.deepEqual(
|
|
118
|
+
shape(a.searchCorpus(enc.encode("the capital of France"))),
|
|
119
|
+
shape(b.searchCorpus(enc.encode("the capital of France"))),
|
|
120
|
+
"two fresh minds over identically-built stores must agree byte for byte",
|
|
121
|
+
);
|
|
122
|
+
assert.deepEqual(
|
|
123
|
+
shape(a.sampleCorpus(2)),
|
|
124
|
+
shape(b.sampleCorpus(2)),
|
|
125
|
+
"and browsing must agree too — no draw from outside the seed",
|
|
126
|
+
);
|
|
127
|
+
assert.deepEqual(
|
|
128
|
+
shape(a.sampleCorpus(2, 0.25)),
|
|
129
|
+
shape(b.sampleCorpus(2, 0.25)),
|
|
130
|
+
);
|
|
131
|
+
await a.store.close();
|
|
132
|
+
await b.store.close();
|
|
133
|
+
});
|
|
134
|
+
|
|
135
|
+
test("a browse never shows the same context twice", async () => {
|
|
136
|
+
// Found by mutating the determinism test: on a store small relative to the
|
|
137
|
+
// probe budget the id stride revisits ids, so a browse returned the SAME pair
|
|
138
|
+
// over and over (measured: six copies of context #4 for `limit: 6`).
|
|
139
|
+
const mind = await fixture();
|
|
140
|
+
const r = mind.sampleCorpus(6);
|
|
141
|
+
const ids = r.pairs.map((p) => p.contextId);
|
|
142
|
+
assert.equal(
|
|
143
|
+
new Set(ids).size,
|
|
144
|
+
ids.length,
|
|
145
|
+
`every browsed pair must be a different context, got ${
|
|
146
|
+
JSON.stringify(ids)
|
|
147
|
+
}`,
|
|
148
|
+
);
|
|
149
|
+
await mind.store.close();
|
|
150
|
+
});
|
|
151
|
+
|
|
152
|
+
test("browsing is deterministic, and `from` moves the window", async () => {
|
|
153
|
+
const mind = await fixture();
|
|
154
|
+
const a = mind.sampleCorpus(2);
|
|
155
|
+
const b = mind.sampleCorpus(2);
|
|
156
|
+
assert.ok(
|
|
157
|
+
a.pairs.length >= 2,
|
|
158
|
+
"the fixture must yield at least two DISTINCT samples, or this proves nothing",
|
|
159
|
+
);
|
|
160
|
+
assert.deepEqual(
|
|
161
|
+
a.pairs.map((p) => [p.contextId, p.continuationId]),
|
|
162
|
+
b.pairs.map((p) => [p.contextId, p.continuationId]),
|
|
163
|
+
"browsing must not draw randomly",
|
|
164
|
+
);
|
|
165
|
+
const other = mind.sampleCorpus(2, 0.5);
|
|
166
|
+
assert.ok(
|
|
167
|
+
other.pairs.every((p) => p.matchedBytes === 0),
|
|
168
|
+
"browse samples carry no query match",
|
|
169
|
+
);
|
|
170
|
+
await mind.store.close();
|
|
171
|
+
});
|
package/test/14-scaling.test.mjs
CHANGED
|
@@ -158,14 +158,17 @@ test("training: recall work does NOT grow with the store (storage reads, not tim
|
|
|
158
158
|
for (let i = from; i < to; i++) await mind.ingest(novelExperience(i, "g"));
|
|
159
159
|
};
|
|
160
160
|
|
|
161
|
-
|
|
161
|
+
// 6x in RATIO is the signal; the absolute size is the cost. 100 -> 600
|
|
162
|
+
// keeps the same 6x step (and a SMALLER N makes the log-N growth relatively
|
|
163
|
+
// LARGER, so the 3x band below is not loosened by shrinking it).
|
|
164
|
+
await grow(0, 100);
|
|
162
165
|
const small = await readsNow();
|
|
163
|
-
await grow(
|
|
166
|
+
await grow(100, 600); // 6x the corpus
|
|
164
167
|
const large = await readsNow();
|
|
165
168
|
await store.close();
|
|
166
169
|
|
|
167
170
|
console.log(
|
|
168
|
-
` content-index reads for one recall: N=
|
|
171
|
+
` content-index reads for one recall: N=100 → ${small}, N=600 → ${large}`,
|
|
169
172
|
);
|
|
170
173
|
|
|
171
174
|
// 6x the corpus must not cost anywhere near 6x the reads. A genuinely
|
|
@@ -194,7 +197,7 @@ test("training: absolute deposition throughput clears a sane floor", async () =>
|
|
|
194
197
|
// not one-time setup.
|
|
195
198
|
for (let i = 0; i < 200; i++) await mind.ingest(novelExperience(i, "warm"));
|
|
196
199
|
|
|
197
|
-
const N =
|
|
200
|
+
const N = 400;
|
|
198
201
|
let bytes = 0;
|
|
199
202
|
const t0 = performance.now();
|
|
200
203
|
for (let i = 0; i < N; i++) {
|
|
@@ -322,7 +325,7 @@ test("training: exact recall is preserved at scale", async () => {
|
|
|
322
325
|
// growth exponent in corpus size is the proof — it must be well below linear.
|
|
323
326
|
test("inference: cost is sublinear in corpus size (independent corpora)", async () => {
|
|
324
327
|
const query = unknownInput(1024);
|
|
325
|
-
const sizes = [50, 200, 800
|
|
328
|
+
const sizes = [50, 200, 800];
|
|
326
329
|
const times = [];
|
|
327
330
|
|
|
328
331
|
for (const n of sizes) {
|
|
@@ -370,7 +373,7 @@ test("inference: input is processed at a roughly constant KB/s (linear, not quad
|
|
|
370
373
|
const mind = new Mind({ seed: 7, store });
|
|
371
374
|
await mind.ingest(corpus(200, "inflen"));
|
|
372
375
|
|
|
373
|
-
const kbs = [0.5, 1, 2, 4
|
|
376
|
+
const kbs = [0.5, 1, 2, 4];
|
|
374
377
|
const queries = kbs.map((kb) => unknownInput(kb * 1024));
|
|
375
378
|
const bytes = queries.map((q) => new TextEncoder().encode(q).length);
|
|
376
379
|
|
|
@@ -452,7 +455,7 @@ test("inference: completion still fires inside long inputs", async () => {
|
|
|
452
455
|
});
|
|
453
456
|
await mind.ingest([["ice", "cold"], ["fire", "hot"], ["2+2", "4"]]);
|
|
454
457
|
|
|
455
|
-
for (const pad of [16, 64, 256
|
|
458
|
+
for (const pad of [16, 64, 256]) {
|
|
456
459
|
const filler = unknownInput(pad);
|
|
457
460
|
const mid = filler.slice(0, filler.length >> 1);
|
|
458
461
|
const end = filler.slice(filler.length >> 1);
|
|
@@ -144,7 +144,12 @@ test("the frame inventory REPORTS without judging", async () => {
|
|
|
144
144
|
"the country where the Eiffel Tower is",
|
|
145
145
|
);
|
|
146
146
|
assert.equal(clean(inst.slots[0].filler), "France");
|
|
147
|
-
|
|
147
|
+
// 24, not 23: the alignment's gap bound is the PAIR's extent now (budgeted),
|
|
148
|
+
// so the frame's constant run is no longer cut one byte short by
|
|
149
|
+
// chainReach(W). The contract this test pins — REPORT without judging — is
|
|
150
|
+
// untouched: the substitution is still reported (and still refused by
|
|
151
|
+
// `voiceable` below).
|
|
152
|
+
assert.equal(inst.covered, 24, "coverage must be reported, not judged");
|
|
148
153
|
|
|
149
154
|
// 2. An INSERTION — a real variation, and not a slot anything can carry.
|
|
150
155
|
const ins = frameSlots(
|
|
@@ -206,12 +206,37 @@ test("completion recursion: per-query work does not grow with the corpus", async
|
|
|
206
206
|
// NO OUTPUT CONFOUND. Work is allowed to grow with the ANSWER. Pinning the
|
|
207
207
|
// answer byte-for-byte across every size removes that defence entirely: any
|
|
208
208
|
// growth measured below bought exactly nothing.
|
|
209
|
+
// NO OUTPUT CONFOUND — WITHOUT DEMANDING A CONSTANT ANSWER.
|
|
210
|
+
//
|
|
211
|
+
// This used to require a byte-identical answer across corpus sizes, on the
|
|
212
|
+
// reasoning that with the output moving, any work growth would stop being
|
|
213
|
+
// attributable to the corpus. That was true only while an offer cap froze
|
|
214
|
+
// what a hop could reach: with the cap gone the offer follows the corpus, and
|
|
215
|
+
// a bigger corpus legitimately licenses a different — here CHEAPER —
|
|
216
|
+
// derivation. Measured at the three sizes: the answer moved at the largest
|
|
217
|
+
// (34 B → 34 B → 41 B) while the work did NOT (searches 2/2/2, pops
|
|
218
|
+
// 428/430/384, falling). So the confound worth defending against is not
|
|
219
|
+
// "the answer moved" but "the work grew because the answer grew", and that is
|
|
220
|
+
// removed by dividing the work by the answer it produced — the reading the
|
|
221
|
+
// comment below already allows ("Work is allowed to grow with the ANSWER").
|
|
222
|
+
// The two raw bars below stay exactly as they were; this only ADDS a bar.
|
|
223
|
+
// BYTES, not UTF-16 code units: this law is priced per byte (PASS/byte,
|
|
224
|
+
// bounded-reads.md), so the denominator is the answer's own byte length even
|
|
225
|
+
// though respondText hands back a string.
|
|
226
|
+
const answerBytes = answers.map((a) => new TextEncoder().encode(a).length);
|
|
227
|
+
const perByte = pops.map((p, i) => p / Math.max(1, answerBytes[i]));
|
|
228
|
+
const kPerByte = logLogSlope(SIZES, perByte);
|
|
229
|
+
console.log(
|
|
230
|
+
` answer-normalised pops/byte: ${
|
|
231
|
+
perByte.map((v) => v.toFixed(2)).join(" → ")
|
|
232
|
+
} · k ≈ ${kPerByte.toFixed(2)}`,
|
|
233
|
+
);
|
|
209
234
|
assert.ok(
|
|
210
|
-
|
|
211
|
-
`
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
`
|
|
235
|
+
kPerByte < 1,
|
|
236
|
+
`answer-normalised work grew with exponent k=${kPerByte.toFixed(2)} in ` +
|
|
237
|
+
`corpus size — dividing by the answer's own length already removes the ` +
|
|
238
|
+
`output confound, so growth beyond that is work the answer never asked ` +
|
|
239
|
+
`for (bounded-reads.md)`,
|
|
215
240
|
);
|
|
216
241
|
|
|
217
242
|
const kSearches = logLogSlope(SIZES, searches);
|