@hviana/sema 0.9.0 → 0.9.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +7 -7
- package/dist/src/alu/src/index.d.ts +1 -1
- package/dist/src/alu/src/index.js +1 -1
- package/dist/src/alu/src/parser.js +2 -6
- package/dist/src/alu/src/resonance.d.ts +13 -0
- package/dist/src/alu/src/resonance.js +41 -0
- package/dist/src/alu/test/alu.test.js +39 -0
- package/dist/src/bytes.d.ts +6 -2
- package/dist/src/bytes.js +10 -4
- package/dist/src/canon.js +44 -0
- package/dist/src/geometry.d.ts +19 -1
- package/dist/src/geometry.js +125 -141
- package/dist/src/meter.d.ts +20 -0
- package/dist/src/meter.js +21 -1
- package/dist/src/mind/articulation.js +14 -1
- package/dist/src/mind/attention.d.ts +12 -0
- package/dist/src/mind/attention.js +44 -16
- package/dist/src/mind/bridge.js +3 -3
- package/dist/src/mind/derivation.d.ts +40 -0
- package/dist/src/mind/derivation.js +34 -0
- package/dist/src/mind/graph-search.d.ts +89 -15
- package/dist/src/mind/graph-search.js +345 -174
- package/dist/src/mind/learning.js +1 -1
- package/dist/src/mind/mechanisms/cover.d.ts +19 -3
- package/dist/src/mind/mechanisms/cover.js +101 -58
- package/dist/src/mind/mechanisms/recall.js +0 -1
- package/dist/src/mind/mind.js +2 -2
- package/dist/src/mind/pipeline.d.ts +5 -1
- package/dist/src/mind/pipeline.js +175 -87
- package/dist/src/mind/primitives.d.ts +25 -5
- package/dist/src/mind/primitives.js +107 -44
- package/dist/src/mind/reasoning.d.ts +18 -4
- package/dist/src/mind/reasoning.js +445 -321
- package/dist/src/mind/recognition.js +29 -13
- package/dist/src/mind/resonance.js +1 -11
- package/dist/src/mind/traverse.d.ts +3 -3
- package/dist/src/mind/traverse.js +3 -3
- package/dist/src/mind/types.d.ts +7 -1
- package/dist/src/store-sqlite.d.ts +25 -0
- package/dist/src/store-sqlite.js +89 -1
- package/dist/src/store.d.ts +48 -4
- package/dist/src/store.js +86 -6
- package/docs/INDEX.md +18 -18
- package/docs/INVARIANTS.md +16 -16
- package/docs/architecture/bounded-reads.md +1 -1
- package/docs/architecture/caches.md +5 -4
- package/docs/architecture/closure.md +45 -5
- package/docs/architecture/cost-model.md +16 -0
- package/docs/architecture/factored-machinery.md +14 -13
- package/docs/architecture/fold-contract.md +51 -1
- package/docs/architecture/mechanism-market.md +21 -0
- package/docs/architecture/memoization.md +3 -3
- package/docs/architecture/meter.md +2 -1
- package/docs/architecture/saturation.md +12 -0
- package/docs/architecture/store.md +25 -2
- package/docs/failures/tempting-but-wrong.md +13 -2
- package/docs/harness/gates.md +12 -10
- package/docs/mechanisms/cover.md +23 -6
- package/jsr.json +1 -1
- package/package.json +1 -1
- package/src/alu/README.md +10 -2
- package/src/alu/src/index.ts +1 -0
- package/src/alu/src/parser.ts +6 -6
- package/src/alu/src/resonance.ts +42 -0
- package/src/alu/test/alu.test.ts +40 -0
- package/src/bytes.ts +13 -3
- package/src/canon.ts +40 -0
- package/src/geometry.ts +183 -154
- package/src/meter.ts +21 -1
- package/src/mind/articulation.ts +14 -2
- package/src/mind/attention.ts +47 -25
- package/src/mind/bridge.ts +3 -3
- package/src/mind/derivation.ts +77 -0
- package/src/mind/graph-search.ts +449 -221
- package/src/mind/learning.ts +1 -7
- package/src/mind/match.ts +1 -2
- package/src/mind/mechanisms/cast.ts +1 -2
- package/src/mind/mechanisms/cover.ts +149 -84
- package/src/mind/mechanisms/extraction.ts +1 -2
- package/src/mind/mechanisms/prefix-completion.ts +1 -1
- package/src/mind/mechanisms/recall.ts +1 -3
- package/src/mind/mechanisms/reference.ts +1 -1
- package/src/mind/mind.ts +5 -30
- package/src/mind/pipeline.ts +206 -102
- package/src/mind/primitives.ts +119 -43
- package/src/mind/reasoning.ts +558 -413
- package/src/mind/recognition.ts +24 -9
- package/src/mind/resonance.ts +2 -16
- package/src/mind/trace.ts +1 -1
- package/src/mind/traverse.ts +3 -3
- package/src/mind/types.ts +9 -11
- package/src/store-sqlite.ts +92 -1
- package/src/store.ts +113 -7
- package/test/105-derive-through-reports-its-refusal.test.mjs +8 -5
- package/test/106-the-join-fires.test.mjs +21 -0
- package/test/111-the-cover-assembly-is-counted.test.mjs +8 -5
- package/test/128-the-leads-somewhere-pair-agrees.test.mjs +18 -12
- package/test/136-the-two-named-limits.test.mjs +3 -2
- package/test/137-the-law-lives-once-and-below.test.mjs +21 -0
- package/test/148-exact-shortcuts-agree.test.mjs +188 -0
- package/test/149-the-closure-engine.test.mjs +138 -0
- package/test/150-the-join-is-output-sensitive.test.mjs +66 -0
- package/test/151-the-cover-pays-for-what-it-reaches.test.mjs +142 -0
- package/test/152-the-read-side-names-as-the-write-side.test.mjs +146 -0
- package/test/153-a-cheaper-bound-is-looked-at-first.test.mjs +155 -0
- package/test/24-generalization.test.mjs +32 -0
- package/test/36-bloom.test.mjs +53 -0
- package/test/37-cluster-dispersion-fusion.test.mjs +75 -0
- package/test/48-recognise-turn-connective.test.mjs +3 -2
- package/test/55-cost-meter.test.mjs +4 -4
- package/test/90-connector-read-cap.test.mjs +7 -7
|
@@ -11,12 +11,12 @@
|
|
|
11
11
|
import { PASS, STEP } from "./graph-search.js";
|
|
12
12
|
import { gistOf, read, resolve } from "./primitives.js";
|
|
13
13
|
import { recognise } from "./recognition.js";
|
|
14
|
-
import {
|
|
15
|
-
import { closed, remainderOf, unaccountedBytes, unexplainedSpans, windowOf, } from "./derivation.js";
|
|
14
|
+
import { fusionLayer, walkLayer } from "./reasoning.js";
|
|
15
|
+
import { closed, closeOver, remainderOf, unaccountedBytes, unexplainedSpans, windowOf, } from "./derivation.js";
|
|
16
16
|
import { rItem } from "./trace.js";
|
|
17
17
|
import { unexplainedLabel } from "./rationale.js";
|
|
18
18
|
import { hubBound } from "./traverse.js";
|
|
19
|
-
import { Precomputed } from "./pipeline-mechanism.js";
|
|
19
|
+
import { Precomputed, } from "./pipeline-mechanism.js";
|
|
20
20
|
import { coverMechanism } from "./mechanisms/cover.js";
|
|
21
21
|
import { castMechanism } from "./mechanisms/cast.js";
|
|
22
22
|
import { confluenceMechanism } from "./mechanisms/confluence.js";
|
|
@@ -26,7 +26,7 @@ import { prefixMechanism } from "./mechanisms/prefix-completion.js";
|
|
|
26
26
|
import { recallMechanism } from "./mechanisms/recall.js";
|
|
27
27
|
// Re-exports: cover's pre-resolution helpers and the ALU adapter kept
|
|
28
28
|
// importable from the pipeline module (their historical home).
|
|
29
|
-
export {
|
|
29
|
+
export { offerConcepts, offerConnectors } from "./mechanisms/cover.js";
|
|
30
30
|
export { aluToMechanism } from "./mechanisms/alu.js";
|
|
31
31
|
// ── Extension dispatch (pre-loop parse) ─────────────────────────────────────
|
|
32
32
|
async function collectComputed(ctx, mechanisms, query) {
|
|
@@ -197,12 +197,14 @@ export async function think(ctx, query, mechs) {
|
|
|
197
197
|
// below.
|
|
198
198
|
const incumbent = best;
|
|
199
199
|
const incumbentGrade = incumbent === null ? null : grade(incumbent.weight);
|
|
200
|
-
const regime =
|
|
200
|
+
const regime = worthDeclared(2 * STEP)
|
|
201
201
|
? "composition"
|
|
202
202
|
: "retrieval";
|
|
203
203
|
ctx.trace?.step("regimePrediction", [rItem(query, "query")], [], regime === "retrieval"
|
|
204
|
-
?
|
|
205
|
-
`
|
|
204
|
+
? (incumbentGrade !== null && incumbentGrade <= climbFloorGrade
|
|
205
|
+
? `retrieval regime — incumbent grade ${incumbentGrade} ≤ climb floor ${climbFloorGrade}, `
|
|
206
|
+
: `retrieval regime — grade ${bound} already reached by a mechanism run ahead, below climb floor ${climbFloorGrade}, `) +
|
|
207
|
+
`so no mechanism floored above that grade runs; CAST will not climb`
|
|
206
208
|
: `composition regime — ${incumbentGrade === null
|
|
207
209
|
? "no incumbent (nothing grounded)"
|
|
208
210
|
: `incumbent grade ${incumbentGrade}`} above climb floor ${climbFloorGrade}, so the full market and climb run`, undefined, {
|
|
@@ -210,38 +212,147 @@ export async function think(ctx, query, mechs) {
|
|
|
210
212
|
regime,
|
|
211
213
|
incumbentGrade,
|
|
212
214
|
climbFloorGrade,
|
|
215
|
+
...(bound !== Infinity ? { boundGrade: bound } : {}),
|
|
213
216
|
});
|
|
214
217
|
};
|
|
215
218
|
// Phase 3: grounding loop
|
|
216
219
|
// Per-mechanism accounting (src/meter.ts). The market's whole premise is
|
|
217
220
|
// that mechanisms compete on one cost scale — so the profiling read-out is
|
|
218
221
|
// also per-mechanism, uniformly: the loop never asks which one it holds.
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
222
|
+
//
|
|
223
|
+
// A CHEAPER BOUND IS LOOKED AT BEFORE A DEARER ONE IS PAID FOR.
|
|
224
|
+
//
|
|
225
|
+
// The declared order is the tie-break priority, and the pruning above is
|
|
226
|
+
// only as strong as the incumbent it has: a mechanism floored LOWER than the
|
|
227
|
+
// one about to invest, but declared after it, could not prune it. Measured
|
|
228
|
+
// on the 31.7M-node store: a lowercased dialogue turn (#97 of the battery)
|
|
229
|
+
// was won by recall at grade 1 — after CAST (floor grade 2) had paid the
|
|
230
|
+
// consensus climb and the weave, confluence (3) a reach climb, and extraction
|
|
231
|
+
// and reference their reads: ~10 s of a 12.6 s response, for candidates that
|
|
232
|
+
// could not win.
|
|
233
|
+
//
|
|
234
|
+
// So before mechanism `m` first-touches anything, every LATER mechanism whose
|
|
235
|
+
// bound is strictly lower runs AHEAD of it, cheapest bound first, and the
|
|
236
|
+
// lowest grade any of them reaches becomes `bound`. A mechanism whose floor
|
|
237
|
+
// grade exceeds `bound` is then skipped. THE DECISION IS UNCHANGED — the
|
|
238
|
+
// same candidate wins as in the declared order:
|
|
239
|
+
// • a mechanism `p` run ahead with best grade g bounds the final grade by
|
|
240
|
+
// g: in the declared order p either runs (its candidate is weighed) or is
|
|
241
|
+
// pruned by an incumbent already at or below p's floor ≤ g;
|
|
242
|
+
// • so a mechanism floored above `bound` has only candidates the final
|
|
243
|
+
// winner strictly outgrades — and with every candidate above `bound`
|
|
244
|
+
// dropped, every mechanism floored at or below it meets the same
|
|
245
|
+
// run-or-prune decision (`f < incumbent` iff `f < min(incumbent,
|
|
246
|
+
// bound + 1)` for f ≤ bound) and yields the same candidates;
|
|
247
|
+
// • and the winner is chosen from the candidates at or below `bound`, in
|
|
248
|
+
// declared order — `consider` replays them where they are declared.
|
|
249
|
+
// Equal-grade floors are NOT skipped (≤, not <): a mechanism declared earlier
|
|
250
|
+
// keeps the tie it would have won. Running `p` ahead is never extra work:
|
|
251
|
+
// what prunes p in the declared order is a candidate at or below p's floor,
|
|
252
|
+
// which only a mechanism floored at or below it can produce — and every such
|
|
253
|
+
// mechanism is either already run or run ahead of p.
|
|
254
|
+
//
|
|
255
|
+
// The bound is learnt by asking `floor` with a `worthRunning` that refuses:
|
|
256
|
+
// under the investment discipline (pipeline-mechanism.ts) a floor that cannot
|
|
257
|
+
// pay returns its bound UNINVESTED, so the question costs no analysis.
|
|
258
|
+
const refuse = () => false;
|
|
259
|
+
const probed = new Array(mechanisms.length);
|
|
260
|
+
const probeGrade = async (i) => {
|
|
261
|
+
if (probed[i] === undefined) {
|
|
262
|
+
const f = await mechanisms[i].floor(ctx, query, pre, refuse);
|
|
263
|
+
probed[i] = f === null ? null : grade(f);
|
|
264
|
+
}
|
|
265
|
+
return probed[i];
|
|
266
|
+
};
|
|
267
|
+
/** Per mechanism: its floor once computed, and its results once run. */
|
|
268
|
+
const floors = new Map();
|
|
269
|
+
const runs = new Map();
|
|
270
|
+
let bound = Infinity;
|
|
271
|
+
const worthAhead = (floor) => grade(floor) <
|
|
272
|
+
Math.min(best === null ? Infinity : grade(best.weight), bound);
|
|
273
|
+
const worthDeclared = (floor) => worthRunning(floor) && grade(floor) <= bound;
|
|
274
|
+
const floorOf = async (i, worth) => {
|
|
275
|
+
if (floors.has(i))
|
|
276
|
+
return floors.get(i);
|
|
277
|
+
const mech = mechanisms[i];
|
|
223
278
|
const floor = meter
|
|
224
|
-
? await meter.time(`${mech.name}.floor`, () => mech.floor(ctx, query, pre,
|
|
225
|
-
: await mech.floor(ctx, query, pre,
|
|
279
|
+
? await meter.time(`${mech.name}.floor`, () => mech.floor(ctx, query, pre, worth))
|
|
280
|
+
: await mech.floor(ctx, query, pre, worth);
|
|
226
281
|
if (meter) {
|
|
227
282
|
if (floor === null)
|
|
228
283
|
meter.mechanismSkips++;
|
|
229
284
|
else
|
|
230
285
|
meter.mechanismFloors++;
|
|
231
286
|
}
|
|
287
|
+
floors.set(i, floor);
|
|
288
|
+
return floor;
|
|
289
|
+
};
|
|
290
|
+
const runOf = async (i) => {
|
|
291
|
+
let results = runs.get(i);
|
|
292
|
+
if (results === undefined) {
|
|
293
|
+
const mech = mechanisms[i];
|
|
294
|
+
if (meter)
|
|
295
|
+
meter.mechanismRuns++;
|
|
296
|
+
results = meter
|
|
297
|
+
? await meter.time(`${mech.name}.run`, () => mech.run(ctx, query, pre))
|
|
298
|
+
: await mech.run(ctx, query, pre);
|
|
299
|
+
runs.set(i, results);
|
|
300
|
+
}
|
|
301
|
+
return results;
|
|
302
|
+
};
|
|
303
|
+
const runAhead = async (mi) => {
|
|
304
|
+
const g = await probeGrade(mi);
|
|
305
|
+
if (g === null)
|
|
306
|
+
return;
|
|
307
|
+
const ahead = [];
|
|
308
|
+
for (let j = mi + 1; j < mechanisms.length; j++) {
|
|
309
|
+
if (floors.has(j))
|
|
310
|
+
continue;
|
|
311
|
+
const gj = await probeGrade(j);
|
|
312
|
+
if (gj !== null && gj < g)
|
|
313
|
+
ahead.push([gj, j]);
|
|
314
|
+
}
|
|
315
|
+
ahead.sort((a, b) => a[0] - b[0] || a[1] - b[1]);
|
|
316
|
+
for (const [gj, j] of ahead) {
|
|
317
|
+
// Only what the declared order would also run: a bound that cannot beat
|
|
318
|
+
// what is already held is not even asked for its real floor.
|
|
319
|
+
if (!(gj < Math.min(best === null ? Infinity : grade(best.weight), bound))) {
|
|
320
|
+
continue;
|
|
321
|
+
}
|
|
322
|
+
const floor = await floorOf(j, worthAhead);
|
|
323
|
+
if (floor === null || !worthAhead(floor))
|
|
324
|
+
continue;
|
|
325
|
+
ctx.trace?.step("runAhead", [], [], `${mechanisms[j].name} runs ahead of ${mechanisms[mi].name} — its floor (grade ${grade(floor)}) is below ${mechanisms[mi].name}'s (grade ${g}), so its result bounds what ${mechanisms[mi].name} could win`);
|
|
326
|
+
for (const r of await runOf(j)) {
|
|
327
|
+
// `consider` drops an empty answer, so it bounds nothing.
|
|
328
|
+
if (r.bytes.length === 0)
|
|
329
|
+
continue;
|
|
330
|
+
bound = Math.min(bound, grade(weigh(r.accounted, r.moves)));
|
|
331
|
+
}
|
|
332
|
+
}
|
|
333
|
+
};
|
|
334
|
+
for (let mi = 0; mi < mechanisms.length; mi++) {
|
|
335
|
+
const mech = mechanisms[mi];
|
|
336
|
+
if (mi > 0) {
|
|
337
|
+
await runAhead(mi);
|
|
338
|
+
reportRegime();
|
|
339
|
+
}
|
|
340
|
+
const floor = await floorOf(mi, worthDeclared);
|
|
232
341
|
if (floor === null) {
|
|
233
342
|
ctx.trace?.step("skipMechanism", [], [], `${mech.name} skipped — structural precondition failed`);
|
|
234
343
|
continue;
|
|
235
344
|
}
|
|
345
|
+
if (grade(floor) > bound) {
|
|
346
|
+
if (meter)
|
|
347
|
+
meter.mechanismsBounded++;
|
|
348
|
+
ctx.trace?.step("skipMechanism", [], [], `${mech.name} skipped — floor ${floor} cannot beat grade ${bound}, already reached by a mechanism run ahead`);
|
|
349
|
+
continue;
|
|
350
|
+
}
|
|
236
351
|
if (!worthRunning(floor)) {
|
|
237
352
|
ctx.trace?.step("skipMechanism", [], [], `${mech.name} skipped — floor ${floor} cannot beat incumbent (grade ${grade(best.weight)})`);
|
|
238
353
|
continue;
|
|
239
354
|
}
|
|
240
|
-
|
|
241
|
-
meter.mechanismRuns++;
|
|
242
|
-
const results = meter
|
|
243
|
-
? await meter.time(`${mech.name}.run`, () => mech.run(ctx, query, pre))
|
|
244
|
-
: await mech.run(ctx, query, pre);
|
|
355
|
+
const results = await runOf(mi);
|
|
245
356
|
for (const r of results) {
|
|
246
357
|
// ONE FORMULA, EVERY CANDIDATE: the chart's derivation reports how many
|
|
247
358
|
// discrete moves it made and which bytes it could not recognise; the
|
|
@@ -352,19 +463,18 @@ export async function think(ctx, query, mechs) {
|
|
|
352
463
|
// without evidence stays owed, and a later transition pays it only by carrying
|
|
353
464
|
// it (the law reads the window; see derivation.ts). Same reading, one
|
|
354
465
|
// definition — not a second spelling of it here.
|
|
355
|
-
const
|
|
466
|
+
const priced = [
|
|
356
467
|
...decided.accounted,
|
|
357
468
|
...pre.computed.map((u) => [u.i, u.j]),
|
|
358
|
-
]
|
|
469
|
+
];
|
|
470
|
+
const explained = priced.filter(([a, b]) => windowOf([a, b], answer, query, ctx.space.maxGroup) !== null);
|
|
471
|
+
const paid = remainderOf(query.length, explained, ctx.space.maxGroup);
|
|
359
472
|
// WHAT THE CONSTRUCTION WITHHOLDS, at or above one quantum: the difference between
|
|
360
473
|
// the remainder paid in full and the remainder paid by carrying. Both readings
|
|
361
|
-
// are the law's, so the floor is applied once and in one place
|
|
362
|
-
|
|
363
|
-
...decided.accounted,
|
|
364
|
-
...pre.computed.map((u) => [u.i, u.j]),
|
|
365
|
-
], ctx.space.maxGroup);
|
|
366
|
-
const paid = remainderOf(query.length, explained, ctx.space.maxGroup);
|
|
474
|
+
// are the law's, so the floor is applied once and in one place — and the
|
|
475
|
+
// paid-in-full reading is computed only when a meter will read it.
|
|
367
476
|
if (ctx.meter) {
|
|
477
|
+
const paidInFull = remainderOf(query.length, priced, ctx.space.maxGroup);
|
|
368
478
|
ctx.meter.groundingWithheldBytes += unaccountedBytes(paid) -
|
|
369
479
|
unaccountedBytes(paidInFull);
|
|
370
480
|
}
|
|
@@ -451,73 +561,51 @@ export async function think(ctx, query, mechs) {
|
|
|
451
561
|
meter.postGroundingRemainderSpans += uncovered.length;
|
|
452
562
|
meter.postGroundingRemainderBytes += unaccountedBytes(uncovered);
|
|
453
563
|
}
|
|
454
|
-
// THE
|
|
455
|
-
//
|
|
456
|
-
//
|
|
457
|
-
//
|
|
458
|
-
//
|
|
459
|
-
//
|
|
460
|
-
//
|
|
461
|
-
//
|
|
564
|
+
// ── THE CLOSURE ENGINE ───────────────────────────────────────────────
|
|
565
|
+
//
|
|
566
|
+
// The grounding's state is CLOSED under the law by two layers, in order: the
|
|
567
|
+
// multi-hop WALK (`walkLayer`: a forward absorb or a pivot, offered one at a
|
|
568
|
+
// time) and the multi-topic FUSION (`fusionLayer`: one composed transition).
|
|
569
|
+
// Every step either layer offers goes through the same `closure` and the same
|
|
570
|
+
// law; this function no longer sequences them or gates them by hand.
|
|
571
|
+
//
|
|
572
|
+
// WHAT USED TO BE TWO HAND-WRITTEN GATES IS NOW THE LAW'S, OR THE LAYER'S:
|
|
573
|
+
// • a declared-complete grounding (`fixed`) admits no transition — the
|
|
574
|
+
// engine enters no layer for it (the law's first clause), which is the
|
|
575
|
+
// walk-skip and the fusion-skip this branch used to spell separately;
|
|
576
|
+
// • a CLOSED derivation has nothing left for a second topic to account for,
|
|
577
|
+
// so the fusion layer does not ENGAGE — read off the state the walk
|
|
578
|
+
// reached, the one fusion is actually offered against.
|
|
579
|
+
//
|
|
580
|
+
// What the fusion needs from the grounding — whether its substance is purely
|
|
581
|
+
// computed (`unclimbed`) and where it stands in the query (`primarySpans`) —
|
|
582
|
+
// is the grounding's own evidence, resolved here where both readings are in
|
|
583
|
+
// hand.
|
|
462
584
|
//
|
|
463
|
-
//
|
|
464
|
-
//
|
|
465
|
-
//
|
|
466
|
-
//
|
|
467
|
-
//
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
: await reason(ctx, query, state, preConsumed, pre, voiced);
|
|
471
|
-
const reasoned = extension ?? state;
|
|
472
|
-
// Fuse only when the query has a genuine REMAINDER no mechanism's
|
|
473
|
-
// structural evidence touched at all. `decided.accounted` alone
|
|
474
|
-
// undercounts this: it is a COST-LADDER quantity (cover.ts prices its
|
|
475
|
-
// masked/computed spans at near-zero and deliberately leaves them out of
|
|
476
|
-
// `accounted` so PASS-bridged bytes are still charged), not a coverage
|
|
477
|
-
// one — a query fully explained by one computed span plus bridged
|
|
478
|
-
// connectors can report `accounted: []` while nothing is actually left
|
|
479
|
-
// unexplained. The genuine remainder is what NEITHER the winning
|
|
480
|
-
// candidate's accounted spans NOR any recognised extension's computed
|
|
481
|
-
// span (`pre.computed` — every mechanism's parse() output, ALU included)
|
|
482
|
-
// ever touched. A remainder under one river-fold quantum (W, the same
|
|
483
|
-
// floor cover.ts's restatedSpan and the honesty-density bar above both
|
|
484
|
-
// use) is bridging punctuation/whitespace, never a second topic —
|
|
485
|
-
// observed: a single space between two fully-computed arithmetic spans
|
|
486
|
-
// ("2+2 3+3") registered as "unaccounted" and pulled in an unrelated
|
|
487
|
-
// corpus fact, corrupting "4 6" into "4 63".
|
|
488
|
-
// THE GATE ASKS THE LAW, and that is an OPTIMISATION, not a tidy-up: the state
|
|
489
|
-
// above ALREADY carries the remainder (`remainderOf`, per-span, with the W
|
|
490
|
-
// floor applied), so asking it costs nothing, while the total this line used to
|
|
491
|
-
// compute (`unaccounted(explained)`) was one more sum over the spans on every
|
|
492
|
-
// response. The two readings are the same condition, not two: the ACCOUNTING
|
|
493
|
-
// applies the same W floor the gate does, so a gap below one quantum never
|
|
494
|
-
// survives into `explained` and the total cannot reach W without some single
|
|
495
|
-
// gap reaching it. Measured over twelve constructions at W = 4 (test/136.3,
|
|
496
|
-
// which pins the equivalence and both sides of it).
|
|
497
|
-
// Whether the winning candidate's entire recognised substance is
|
|
498
|
-
// COMPUTED — every accounted span exactly a pre.computed span, nothing
|
|
499
|
-
// from a genuinely recognised/climbed site. fuseAttention's lone-root
|
|
500
|
-
// shortcut assumes a single point of attention already IS primary's own
|
|
501
|
-
// source; that assumption is exactly backwards for a pure computation
|
|
502
|
-
// (an ALU result has no anchor of its own) — see fuseAttention's
|
|
503
|
-
// `unclimbed` parameter, gated there by Attention.breadth so a
|
|
504
|
-
// coincidental echo (which this flag alone cannot distinguish) is still
|
|
505
|
-
// rejected.
|
|
585
|
+
// Whether the winning candidate's entire recognised substance is COMPUTED —
|
|
586
|
+
// every accounted span exactly a pre.computed span, nothing from a genuinely
|
|
587
|
+
// recognised/climbed site. fuseAttention's lone-root shortcut assumes a
|
|
588
|
+
// single point of attention already IS primary's own source; that assumption
|
|
589
|
+
// is exactly backwards for a pure computation (an ALU result has no anchor of
|
|
590
|
+
// its own) — gated there by Attention.breadth so a coincidental echo (which
|
|
591
|
+
// this flag alone cannot distinguish) is still rejected.
|
|
506
592
|
const unclimbed = state.accounted.length > 0 &&
|
|
507
593
|
state.accounted.every(([i, j]) => pre.computed.some((u) => u.i === i && u.j === j));
|
|
508
|
-
// Where the winning grounding stands in the query — fusion places primary
|
|
509
|
-
//
|
|
510
|
-
//
|
|
511
|
-
//
|
|
512
|
-
//
|
|
513
|
-
// read here for POSITION instead of for coverage — and resolved here, where
|
|
514
|
-
// both readings are in hand, rather than inside fuseAttention.
|
|
594
|
+
// Where the winning grounding stands in the query — fusion places primary by
|
|
595
|
+
// it. `accounted` is the cost-ladder read and is authoritative when
|
|
596
|
+
// non-empty; when it is empty the grounding is a pure COMPUTATION, whose
|
|
597
|
+
// evidence is its computed span — the cost-ladder-vs-coverage distinction
|
|
598
|
+
// `explained` above draws, read here for POSITION instead of coverage.
|
|
515
599
|
const primarySpans = state.accounted.length > 0
|
|
516
600
|
? state.accounted
|
|
517
601
|
: pre.computed.map((u) => [u.i, u.j]);
|
|
518
|
-
const fused =
|
|
519
|
-
|
|
520
|
-
|
|
602
|
+
const fused = await closeOver(state, query, ctx.space.maxGroup, [
|
|
603
|
+
walkLayer(ctx, query, preConsumed, pre, voiced),
|
|
604
|
+
{
|
|
605
|
+
...fusionLayer(ctx, query, pre, unclimbed, primarySpans),
|
|
606
|
+
engages: (d) => !closed(d),
|
|
607
|
+
},
|
|
608
|
+
], meter ? (name, walk) => meter.time(name, walk) : undefined);
|
|
521
609
|
done(fused.product,
|
|
522
610
|
// NO CLAIM ABOUT FUSION HERE. `fuseAttention` is entered whenever a
|
|
523
611
|
// remainder ≥ W exists and returns early when there is nothing to bridge, so
|
|
@@ -1,11 +1,6 @@
|
|
|
1
1
|
import { Vec } from "../vec.js";
|
|
2
2
|
import { Sema } from "../sema.js";
|
|
3
3
|
import type { Input, MindContext } from "./types.js";
|
|
4
|
-
/** The content key of a byte span — one latin1 char per byte, an exact,
|
|
5
|
-
* collision-free encoding. Spans on the perception path are query-scale
|
|
6
|
-
* (windows, regions, candidate spans), so key construction is far cheaper
|
|
7
|
-
* than the river fold it deduplicates. */
|
|
8
|
-
export declare function latin1Key(bytes: Uint8Array): string;
|
|
9
4
|
/** The {@link perceive} memo key: the span's content PLUS the boundary set it
|
|
10
5
|
* was folded under. The tree is a function of BOTH — the same bytes fold
|
|
11
6
|
* plainly with no boundaries and into a left-nested stable-prefix shape with
|
|
@@ -59,6 +54,31 @@ export declare function foldTree(ctx: MindContext, n: Sema, start: number, visit
|
|
|
59
54
|
end: number;
|
|
60
55
|
node: number | null;
|
|
61
56
|
};
|
|
57
|
+
/** The EXACT content-addressed node of a byte stream — `foldTree(perceive)`,
|
|
58
|
+
* read for identity alone.
|
|
59
|
+
*
|
|
60
|
+
* A fold names a branch only when every child is named, so identity needs the
|
|
61
|
+
* fold's SHAPE and the store's answer per node, never its vectors:
|
|
62
|
+
* {@link contentIdentity} walks the same shape (geometry.ts — one grouping
|
|
63
|
+
* rule, two algebras) and asks the store bottom-up, building no D-dimensional
|
|
64
|
+
* gist and leaving nothing in the perception memo. Each node is named the
|
|
65
|
+
* way the store's write side names it ({@link branchNaming}). `test/148` pins
|
|
66
|
+
* the agreement with the full fold over random and corpus spans. */
|
|
67
|
+
export declare function exactNode(ctx: MindContext, bytes: Uint8Array): number | null;
|
|
68
|
+
/** {@link exactNode}, with whether the span's own name was found only through
|
|
69
|
+
* its BYTES — its children named no branch, and the flat node over the same
|
|
70
|
+
* bytes did ({@link branchNaming}). That is where the exact lookup used to
|
|
71
|
+
* MISS, so it is where {@link resolve} still asks the canonical class: the
|
|
72
|
+
* class may hold the learnt member that leads somewhere, which a flat index
|
|
73
|
+
* entry need not (measured: `tonight` named an edge-less window and
|
|
74
|
+
* pre-empted the case-folded `Tonight` whose edge a composition stood on).
|
|
75
|
+
* Recognition's probes reach it through `resolve`; asking it again for the
|
|
76
|
+
* perceived tree's own byte-named groups changed none of 116 real queries and
|
|
77
|
+
* no test, so it is not asked there. */
|
|
78
|
+
export declare function exactNaming(ctx: MindContext, bytes: Uint8Array): {
|
|
79
|
+
id: number | null;
|
|
80
|
+
byBytes: boolean;
|
|
81
|
+
};
|
|
62
82
|
/** The canonical node id of a byte span: perceive it in isolation — the way
|
|
63
83
|
* training did — and recover its root bottom-up. Returns null if any part is
|
|
64
84
|
* unknown. */
|
|
@@ -2,25 +2,11 @@
|
|
|
2
2
|
//
|
|
3
3
|
// Address — bytes → node (perceive, foldTree, resolve)
|
|
4
4
|
// Read — node → bytes (read)
|
|
5
|
-
import { bytesToTree, contentFoldIncremental, gridToTree, hilbertBytes, stackGrids, } from "../geometry.js";
|
|
5
|
+
import { bytesToTree, contentFoldIncremental, contentIdentity, gridToTree, hilbertBytes, stackGrids, } from "../geometry.js";
|
|
6
6
|
import { canonHash } from "../canon.js";
|
|
7
|
-
import { bytesEqual } from "../bytes.js";
|
|
7
|
+
import { bytesEqual, concatBytes, latin1 } from "../bytes.js";
|
|
8
8
|
import { ALL } from "./types.js";
|
|
9
9
|
// ── Address: bytes → node ──────────────────────────────────────────────
|
|
10
|
-
/** The content key of a byte span — one latin1 char per byte, an exact,
|
|
11
|
-
* collision-free encoding. Spans on the perception path are query-scale
|
|
12
|
-
* (windows, regions, candidate spans), so key construction is far cheaper
|
|
13
|
-
* than the river fold it deduplicates. */
|
|
14
|
-
export function latin1Key(bytes) {
|
|
15
|
-
// Batched String.fromCharCode — avoids the O(n²) cost of repeated += on
|
|
16
|
-
// potentially-large query spans, and stays well under the ~65536 arg limit.
|
|
17
|
-
const n = bytes.length;
|
|
18
|
-
let s = "";
|
|
19
|
-
for (let i = 0; i < n; i += 4096) {
|
|
20
|
-
s += String.fromCharCode(...bytes.subarray(i, Math.min(i + 4096, n)));
|
|
21
|
-
}
|
|
22
|
-
return s;
|
|
23
|
-
}
|
|
24
10
|
/** The {@link perceive} memo key: the span's content PLUS the boundary set it
|
|
25
11
|
* was folded under. The tree is a function of BOTH — the same bytes fold
|
|
26
12
|
* plainly with no boundaries and into a left-nested stable-prefix shape with
|
|
@@ -32,7 +18,7 @@ export function latin1Key(bytes) {
|
|
|
32
18
|
* the boundary rendering is digits and commas, so no content byte can forge
|
|
33
19
|
* the split. */
|
|
34
20
|
export function perceiveKey(bytes, boundaries) {
|
|
35
|
-
const k =
|
|
21
|
+
const k = latin1(bytes);
|
|
36
22
|
return boundaries === undefined || boundaries.length === 0
|
|
37
23
|
? k
|
|
38
24
|
: k + "\u0000" + boundaries.join(",");
|
|
@@ -111,7 +97,7 @@ export function perceiveDeposit(ctx, bytes, conversational = false) {
|
|
|
111
97
|
.filter((L) => L >= 2 && L < bytes.length)
|
|
112
98
|
.sort((a, b) => b - a);
|
|
113
99
|
for (const L of lens) {
|
|
114
|
-
const hit = ctx._depositTrees.get(
|
|
100
|
+
const hit = ctx._depositTrees.get(latin1(bytes.subarray(0, L)));
|
|
115
101
|
if (hit !== undefined) {
|
|
116
102
|
prev = hit.content;
|
|
117
103
|
break;
|
|
@@ -130,7 +116,7 @@ export function perceiveDeposit(ctx, bytes, conversational = false) {
|
|
|
130
116
|
ctx._depositLens.clear();
|
|
131
117
|
ctx._depositTrees.clear();
|
|
132
118
|
}
|
|
133
|
-
ctx._depositTrees.set(
|
|
119
|
+
ctx._depositTrees.set(latin1(bytes), { content: folded.fold });
|
|
134
120
|
ctx._depositLens.add(bytes.length);
|
|
135
121
|
}
|
|
136
122
|
return folded.tree;
|
|
@@ -203,14 +189,10 @@ export function foldTree(ctx, n, start, visit) {
|
|
|
203
189
|
return { end, node };
|
|
204
190
|
}
|
|
205
191
|
let pos = start;
|
|
206
|
-
let known = true;
|
|
207
192
|
const kids = [];
|
|
208
193
|
for (const k of n.kids) {
|
|
209
194
|
const r = foldTree(ctx, k, pos, visit);
|
|
210
|
-
|
|
211
|
-
known = false;
|
|
212
|
-
else if (known)
|
|
213
|
-
kids.push(r.node);
|
|
195
|
+
kids.push(r.node);
|
|
214
196
|
pos = r.end;
|
|
215
197
|
}
|
|
216
198
|
// Same store-probe elision as the leaf case: a cached entry already names
|
|
@@ -218,17 +200,103 @@ export function foldTree(ctx, n, start, visit) {
|
|
|
218
200
|
// id need not be re-derived. Using it also keeps a warm walk's ids
|
|
219
201
|
// bit-identical to a cold walk's rather than re-deriving them from children
|
|
220
202
|
// that may themselves have come from cache.
|
|
221
|
-
const
|
|
222
|
-
? cached.id
|
|
223
|
-
:
|
|
224
|
-
|
|
225
|
-
: null;
|
|
203
|
+
const named = cached !== undefined
|
|
204
|
+
? { id: cached.id, byBytes: false }
|
|
205
|
+
: branchNaming(ctx, kids, treeBytes(n));
|
|
206
|
+
const node = named.id;
|
|
226
207
|
visit?.(n, start, pos, node);
|
|
227
208
|
if (node !== null && ctx._resolvedSubtrees) {
|
|
228
209
|
ctx._resolvedSubtrees.set(n, { id: node, len: pos - start });
|
|
229
210
|
}
|
|
230
211
|
return { end: pos, node };
|
|
231
212
|
}
|
|
213
|
+
/** A perceived subtree's bytes, its leaves in order. */
|
|
214
|
+
function treeBytes(n) {
|
|
215
|
+
const parts = [];
|
|
216
|
+
const walk = (x) => {
|
|
217
|
+
if (x.kids === null)
|
|
218
|
+
parts.push(x.leaf ?? new Uint8Array(0));
|
|
219
|
+
else
|
|
220
|
+
for (const k of x.kids)
|
|
221
|
+
walk(k);
|
|
222
|
+
};
|
|
223
|
+
walk(n);
|
|
224
|
+
return concatBytes(parts);
|
|
225
|
+
}
|
|
226
|
+
/** The EXACT content-addressed node of a byte stream — `foldTree(perceive)`,
|
|
227
|
+
* read for identity alone.
|
|
228
|
+
*
|
|
229
|
+
* A fold names a branch only when every child is named, so identity needs the
|
|
230
|
+
* fold's SHAPE and the store's answer per node, never its vectors:
|
|
231
|
+
* {@link contentIdentity} walks the same shape (geometry.ts — one grouping
|
|
232
|
+
* rule, two algebras) and asks the store bottom-up, building no D-dimensional
|
|
233
|
+
* gist and leaving nothing in the perception memo. Each node is named the
|
|
234
|
+
* way the store's write side names it ({@link branchNaming}). `test/148` pins
|
|
235
|
+
* the agreement with the full fold over random and corpus spans. */
|
|
236
|
+
export function exactNode(ctx, bytes) {
|
|
237
|
+
return exactNaming(ctx, bytes).id;
|
|
238
|
+
}
|
|
239
|
+
/** {@link exactNode}, with whether the span's own name was found only through
|
|
240
|
+
* its BYTES — its children named no branch, and the flat node over the same
|
|
241
|
+
* bytes did ({@link branchNaming}). That is where the exact lookup used to
|
|
242
|
+
* MISS, so it is where {@link resolve} still asks the canonical class: the
|
|
243
|
+
* class may hold the learnt member that leads somewhere, which a flat index
|
|
244
|
+
* entry need not (measured: `tonight` named an edge-less window and
|
|
245
|
+
* pre-empted the case-folded `Tonight` whose edge a composition stood on).
|
|
246
|
+
* Recognition's probes reach it through `resolve`; asking it again for the
|
|
247
|
+
* perceived tree's own byte-named groups changed none of 116 real queries and
|
|
248
|
+
* no test, so it is not asked there. */
|
|
249
|
+
export function exactNaming(ctx, bytes) {
|
|
250
|
+
if (bytes.length === 0) {
|
|
251
|
+
return { id: foldTree(ctx, perceive(ctx, bytes), 0).node, byBytes: false };
|
|
252
|
+
}
|
|
253
|
+
if (ctx.meter)
|
|
254
|
+
ctx.meter.identityBytes += bytes.length;
|
|
255
|
+
let byBytes = false;
|
|
256
|
+
const id = contentIdentity(ctx.space, ctx.alphabet, bytes, (from, to) => to - from === 1
|
|
257
|
+
? ctx.store.findLeaf(bytes.subarray(from, to))
|
|
258
|
+
: flatNode(ctx, bytes.subarray(from, to)), (kids, from, to) => {
|
|
259
|
+
const named = branchNaming(ctx, kids, bytes.subarray(from, to));
|
|
260
|
+
if (from === 0 && to === bytes.length)
|
|
261
|
+
byBytes = named.byBytes;
|
|
262
|
+
return named.id;
|
|
263
|
+
});
|
|
264
|
+
return { id, byBytes };
|
|
265
|
+
}
|
|
266
|
+
/** The flat node over a span's single-byte atoms — the node every deposit
|
|
267
|
+
* interns for its whole input and for each canonical window (learning.ts
|
|
268
|
+
* `deposit`, `indexSubSpans`). The store's negative filter refuses most
|
|
269
|
+
* misses without a lookup. */
|
|
270
|
+
function flatNode(ctx, span) {
|
|
271
|
+
const store = ctx.store;
|
|
272
|
+
return store.findFlatBranch
|
|
273
|
+
? store.findFlatBranch(span)
|
|
274
|
+
: store.findBranch(Array.from(span, (b) => -(b + 1)));
|
|
275
|
+
}
|
|
276
|
+
/** THE READ SIDE NAMES A BRANCH EXACTLY AS THE WRITE SIDE DID. `intern`
|
|
277
|
+
* (store.ts) names a branch by its children; when they name none, it looks up
|
|
278
|
+
* the flat node over the same bytes and REUSES it (step 1b, "same bytes, same
|
|
279
|
+
* node") — so a deposit whose fold grouped `ver` + `!` was stored with the
|
|
280
|
+
* window `ver!` as that child. Reading by the children alone could never
|
|
281
|
+
* name such a deposit again: measured on the 31.7M-node store, 8 of 80 stored
|
|
282
|
+
* dialogue turns asked verbatim resolved to nothing (a 25-byte turn, a final
|
|
283
|
+
* `?` or `!`, …) and fell to the composition path. Same order as the write
|
|
284
|
+
* side: the children first, the bytes when they name nothing — and an unnamed
|
|
285
|
+
* child does not settle it, since the write side minted that child and still
|
|
286
|
+
* reached step 1b. */
|
|
287
|
+
function branchNaming(ctx, kids, span) {
|
|
288
|
+
if (kids.every((k) => k !== null)) {
|
|
289
|
+
const id = ctx.store.findBranch(kids);
|
|
290
|
+
if (id !== null)
|
|
291
|
+
return { id, byBytes: false };
|
|
292
|
+
}
|
|
293
|
+
if (kids.length < 2)
|
|
294
|
+
return { id: null, byBytes: false };
|
|
295
|
+
const id = flatNode(ctx, span);
|
|
296
|
+
if (id !== null && ctx.meter)
|
|
297
|
+
ctx.meter.flatBranchNames++;
|
|
298
|
+
return { id, byBytes: id !== null };
|
|
299
|
+
}
|
|
232
300
|
/** The canonical node id of a byte span: perceive it in isolation — the way
|
|
233
301
|
* training did — and recover its root bottom-up. Returns null if any part is
|
|
234
302
|
* unknown. */
|
|
@@ -237,10 +305,10 @@ export function resolve(ctx, bytes) {
|
|
|
237
305
|
return null;
|
|
238
306
|
if (ctx.meter)
|
|
239
307
|
ctx.meter.resolves++;
|
|
240
|
-
const exact =
|
|
241
|
-
if (exact !== null)
|
|
308
|
+
const { id: exact, byBytes } = exactNaming(ctx, bytes);
|
|
309
|
+
if (exact !== null && !byBytes)
|
|
242
310
|
return exact;
|
|
243
|
-
return canonResolve(ctx, bytes);
|
|
311
|
+
return canonResolve(ctx, bytes) ?? exact;
|
|
244
312
|
}
|
|
245
313
|
/** Equivalence-class resolution: when the exact content-addressed lookup
|
|
246
314
|
* misses, find a stored node whose CANONICAL key equals the span's — the
|
|
@@ -258,7 +326,7 @@ export function canonResolve(ctx, bytes) {
|
|
|
258
326
|
if (bytes.length < 2)
|
|
259
327
|
return null;
|
|
260
328
|
const memo = ctx.canonMemo;
|
|
261
|
-
const memoKey = memo ?
|
|
329
|
+
const memoKey = memo ? latin1(bytes) : "";
|
|
262
330
|
if (memo) {
|
|
263
331
|
const hit = memo.get(memoKey);
|
|
264
332
|
if (hit !== undefined)
|
|
@@ -275,7 +343,7 @@ export function canonResolve(ctx, bytes) {
|
|
|
275
343
|
// skips identity rows) — the exact content-addressed lookup of the
|
|
276
344
|
// canonical bytes finds it directly.
|
|
277
345
|
if (key.length !== bytes.length || !bytesEqual(key, bytes)) {
|
|
278
|
-
const direct =
|
|
346
|
+
const direct = exactNode(ctx, key);
|
|
279
347
|
if (direct !== null)
|
|
280
348
|
return set(direct);
|
|
281
349
|
}
|
|
@@ -295,17 +363,12 @@ export function canonResolve(ctx, bytes) {
|
|
|
295
363
|
// resolved for these bytes is their FOLD — the deposit-shaped node that
|
|
296
364
|
// carries the edges and halos. Re-folding the candidate's bytes lands
|
|
297
365
|
// on exactly the node the canonical-case query would have found.
|
|
298
|
-
const folded =
|
|
366
|
+
const folded = exactNode(ctx, bytesOf);
|
|
299
367
|
const use = folded ?? id;
|
|
300
|
-
// THE ADMISSION PREDICATE,
|
|
301
|
-
//
|
|
302
|
-
//
|
|
303
|
-
|
|
304
|
-
// bar and this site would rank a node as leading on evidence the law
|
|
305
|
-
// refuses. Calling `leadsSomewhere` here is not possible — `traverse.ts`
|
|
306
|
-
// imports THIS file, so it would be a cycle — which is why the pair is
|
|
307
|
-
// spelled out rather than named.
|
|
308
|
-
const leads = store.hasNext(use) || store.hasHalo(use);
|
|
368
|
+
// THE ADMISSION PREDICATE, asked of the store that owns it (edge or halo,
|
|
369
|
+
// the halo tier carrying the mass bar). Asking `haloMass(use) > 0` instead
|
|
370
|
+
// would agree only while `minHaloMass <= 1`.
|
|
371
|
+
const leads = store.leadsSomewhere(use);
|
|
309
372
|
if (best === null || (leads && !bestLeads) ||
|
|
310
373
|
(leads === bestLeads && use < best)) {
|
|
311
374
|
best = use;
|