@hviana/sema 0.8.0 → 0.8.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +22 -1
- package/DATASETS.md +1 -1
- package/dist/example/train_base/config.js +2 -2
- package/dist/example/train_base/corpora/massive.js +1 -1
- package/dist/example/train_base/readers.js +1 -1
- package/dist/src/geometry.d.ts +10 -10
- package/dist/src/geometry.js +25 -24
- package/dist/src/meter.d.ts +4 -12
- package/dist/src/meter.js +14 -14
- package/dist/src/mind/attention.js +12 -12
- package/dist/src/mind/bridge.d.ts +8 -8
- package/dist/src/mind/bridge.js +33 -32
- package/dist/src/mind/graph-search.d.ts +0 -8
- package/dist/src/mind/graph-search.js +9 -8
- package/dist/src/mind/junction.d.ts +1 -1
- package/dist/src/mind/junction.js +8 -8
- package/dist/src/mind/learning.js +36 -35
- package/dist/src/mind/match.js +14 -13
- package/dist/src/mind/mechanisms/cover.js +13 -12
- package/dist/src/mind/mechanisms/prefix-completion.js +24 -24
- package/dist/src/mind/mechanisms/recall.js +38 -40
- package/dist/src/mind/mechanisms/reference.js +16 -16
- package/dist/src/mind/mind.d.ts +6 -7
- package/dist/src/mind/pipeline-mechanism.d.ts +10 -8
- package/dist/src/mind/pipeline-mechanism.js +25 -21
- package/dist/src/mind/pipeline.d.ts +9 -9
- package/dist/src/mind/pipeline.js +24 -23
- package/dist/src/mind/primitives.d.ts +5 -5
- package/dist/src/mind/primitives.js +5 -5
- package/dist/src/mind/recognition.d.ts +14 -13
- package/dist/src/mind/recognition.js +23 -23
- package/dist/src/mind/resonance.js +21 -21
- package/dist/src/mind/traverse.d.ts +54 -52
- package/dist/src/mind/traverse.js +74 -72
- package/dist/src/mind/types.d.ts +4 -4
- package/dist/src/store.d.ts +12 -12
- package/dist/src/store.js +12 -12
- package/docs/INDEX.md +2 -2
- package/docs/architecture/exact-vs-approximate.md +2 -1
- package/docs/architecture/fold-contract.md +1 -1
- package/docs/failures/tempting-but-wrong.md +2 -3
- package/docs/harness/gates.md +7 -7
- package/example/train_base/config.ts +2 -2
- package/example/train_base/corpora/massive.ts +1 -1
- package/example/train_base/readers.ts +1 -1
- package/jsr.json +1 -1
- package/package.json +1 -1
- package/src/geometry.ts +25 -24
- package/src/meter.ts +14 -14
- package/src/mind/attention.ts +12 -12
- package/src/mind/bridge.ts +33 -32
- package/src/mind/graph-search.ts +9 -8
- package/src/mind/junction.ts +8 -8
- package/src/mind/learning.ts +36 -35
- package/src/mind/match.ts +20 -19
- package/src/mind/mechanisms/cover.ts +13 -12
- package/src/mind/mechanisms/prefix-completion.ts +24 -24
- package/src/mind/mechanisms/recall.ts +38 -40
- package/src/mind/mechanisms/reference.ts +16 -16
- package/src/mind/mind.ts +6 -7
- package/src/mind/pipeline-mechanism.ts +25 -21
- package/src/mind/pipeline.ts +33 -32
- package/src/mind/primitives.ts +5 -5
- package/src/mind/recognition.ts +23 -23
- package/src/mind/resonance.ts +21 -21
- package/src/mind/traverse.ts +74 -72
- package/src/mind/types.ts +4 -4
- package/src/store.ts +20 -20
- package/test/08-storage.test.mjs +1 -1
- package/test/35-prefix-edge.test.mjs +1 -1
- package/test/40-choosenext-scale-guard.test.mjs +16 -17
- package/test/56-bridge-identity-admission.test.mjs +6 -6
- package/test/70-prefix-completion.test.mjs +4 -3
- package/test/72-prefix-candidate-supply.test.mjs +3 -3
- package/test/73-scaffolding-only-bridge-abstains.test.mjs +6 -6
- package/test/75-multiturn-context-optimisation.test.mjs +5 -5
- package/test/84-composed-answer-honesty.test.mjs +5 -6
- package/test/88-dependency-footprint.test.mjs +1 -1
- package/test/89-completion-recursion.test.mjs +17 -14
- package/test/90-connector-read-cap.test.mjs +10 -8
- package/test/93-regime-prediction.test.mjs +10 -10
- package/test/94-cross-region-budget.test.mjs +2 -2
- package/test/95-wide-resonance-removed.test.mjs +8 -7
- package/test/96-bytes-walk-termination.test.mjs +3 -3
package/src/mind/bridge.ts
CHANGED
|
@@ -148,23 +148,23 @@ export function dismissedKnownContent(
|
|
|
148
148
|
return false;
|
|
149
149
|
}
|
|
150
150
|
|
|
151
|
-
// The seeded aligner this file used to own now lives in the shared match
|
|
152
|
-
//
|
|
153
|
-
//
|
|
154
|
-
//
|
|
151
|
+
// The seeded aligner this file used to own now lives in the shared match family
|
|
152
|
+
// as {@link alignAround} — the frame reading (match.ts) reads the same gaps and
|
|
153
|
+
// asks the OPPOSITE question of them (see AlignGap's own doc). Two consumers,
|
|
154
|
+
// one definition (factored-machinery.md); the bridge's reading is unchanged.
|
|
155
155
|
const align = alignAround;
|
|
156
156
|
|
|
157
157
|
/** Recall's corroborated-substitution bridge — see the module comment.
|
|
158
158
|
* Returns the best bridged grounding proposal, or null. */
|
|
159
159
|
/** `proposed` is a THUNK, not a list: the bridge's own cheap gates (the
|
|
160
|
-
* two-quantum query floor and the O(|query|) stored-window anchor scan)
|
|
161
|
-
*
|
|
162
|
-
*
|
|
163
|
-
*
|
|
164
|
-
*
|
|
165
|
-
*
|
|
166
|
-
*
|
|
167
|
-
*
|
|
160
|
+
* two-quantum query floor and the O(|query|) stored-window anchor scan) decide
|
|
161
|
+
* whether ANY candidate can be aligned, and they need no proposals to do it.
|
|
162
|
+
* Resolving the caller's proposals eagerly meant recall paid its exhaustive
|
|
163
|
+
* whole-index resonance — the most expensive single act on the refusal path —
|
|
164
|
+
* for every query, including the ones whose windows the store has never seen
|
|
165
|
+
* and which the anchor scan rejects outright. Same investment discipline the
|
|
166
|
+
* mechanism floors follow (mechanism-market.md): never compute a shared
|
|
167
|
+
* analysis just to discard it. */
|
|
168
168
|
export async function substitutionBridge(
|
|
169
169
|
ctx: MindContext,
|
|
170
170
|
query: Uint8Array,
|
|
@@ -290,12 +290,12 @@ async function bridgeImpl(
|
|
|
290
290
|
);
|
|
291
291
|
return null;
|
|
292
292
|
}
|
|
293
|
-
// NO DISCRIMINATING LITERAL EVIDENCE — abstain (
|
|
294
|
-
// through the literal spans it did NOT substitute; those anchors are
|
|
295
|
-
// whole of its evidence.
|
|
296
|
-
// clamped at the √N hub bound, i.e. the window is corpus-global
|
|
297
|
-
// — the query's unsubstituted part discriminates nothing, and the
|
|
298
|
-
// substituted span is carrying the entire semantic load.
|
|
293
|
+
// NO DISCRIMINATING LITERAL EVIDENCE — abstain (INVARIANTS.md). A bridge
|
|
294
|
+
// grounds through the literal spans it did NOT substitute; those anchors are
|
|
295
|
+
// the whole of its evidence. When every one of them is SATURATED —
|
|
296
|
+
// containment clamped at the √N hub bound, i.e. the window is corpus-global
|
|
297
|
+
// scaffolding — the query's unsubstituted part discriminates nothing, and the
|
|
298
|
+
// single substituted span is carrying the entire semantic load. That is not a
|
|
299
299
|
// corroborated bridge; it is a template match, and it FABRICATES.
|
|
300
300
|
//
|
|
301
301
|
// Measured on the trained store (hubBound 571). "What is the capital of"
|
|
@@ -310,7 +310,8 @@ async function bridgeImpl(
|
|
|
310
310
|
// them silent and cannot be credited for them.
|
|
311
311
|
//
|
|
312
312
|
// This introduces NO new threshold: `bound` is the same √N reading of "hub"
|
|
313
|
-
// the anchor scan already clamps its own containment read to (
|
|
313
|
+
// the anchor scan already clamps its own containment read to (thresholds.md,
|
|
314
|
+
// commonality.md).
|
|
314
315
|
if (allWindowsAreScaffolding(ctx, query)) {
|
|
315
316
|
ctx.trace?.step(
|
|
316
317
|
"substitutionBridge",
|
|
@@ -353,8 +354,8 @@ async function bridgeImpl(
|
|
|
353
354
|
//
|
|
354
355
|
// The question every gap poses is "may the two forms differ HERE without
|
|
355
356
|
// differing in what they SAY?", and that is the discriminative-vs-
|
|
356
|
-
// scaffolding question
|
|
357
|
-
// population.
|
|
357
|
+
// scaffolding question commonality.md names, over the CORPUS-GLOBAL
|
|
358
|
+
// population. It already has one definition — `dominates(reachOf(...), N)`,
|
|
358
359
|
// the same gate confluence's filler test uses ("scaffolding never binds").
|
|
359
360
|
// Nothing new is derived here; the bar is read, not invented.
|
|
360
361
|
//
|
|
@@ -366,17 +367,17 @@ async function bridgeImpl(
|
|
|
366
367
|
// climb's own definition of non-discriminative), or it resolves to a
|
|
367
368
|
// majority of the corpus's contexts. "the process of ", " is the ".
|
|
368
369
|
//
|
|
369
|
-
// THE READING MATTERS, not just the population
|
|
370
|
-
// deliberately does NOT go through `reachOf`, which maps BOTH "saturated"
|
|
371
|
-
//
|
|
372
|
-
//
|
|
373
|
-
//
|
|
374
|
-
//
|
|
375
|
-
//
|
|
376
|
-
//
|
|
377
|
-
//
|
|
378
|
-
//
|
|
379
|
-
//
|
|
370
|
+
// THE READING MATTERS, not just the population — see commonality.md. This
|
|
371
|
+
// deliberately does NOT go through `reachOf`, which maps BOTH "saturated" and
|
|
372
|
+
// "reaches nothing" to Infinity. For IDF weighting those are the same thing
|
|
373
|
+
// (no usable identity evidence); for THIS question they are opposites — a
|
|
374
|
+
// window reaching nothing is novel content, the most discriminative material
|
|
375
|
+
// there is, and reading it as Infinity would call it scaffolding. Measured:
|
|
376
|
+
// with `reachOf`, "Is water wet?" was answered with "No, heavy water is not
|
|
377
|
+
// wet." — "heav"/"eavy" occur once, reach no edge-bearing ancestor, and were
|
|
378
|
+
// written off as filler. So an empty-rooted window is NEVER explained, and
|
|
379
|
+
// neither is an untrained one (the same principle attestedQ applies to the
|
|
380
|
+
// query side).
|
|
380
381
|
const reachMemo = sharedReachMemo(ctx);
|
|
381
382
|
const explainedSpan = (
|
|
382
383
|
bytes: Uint8Array,
|
package/src/mind/graph-search.ts
CHANGED
|
@@ -375,9 +375,10 @@ export class GraphSearch {
|
|
|
375
375
|
private readonly host: GraphSearchHost,
|
|
376
376
|
) {}
|
|
377
377
|
|
|
378
|
-
|
|
379
|
-
* rather than imported from `traverse.ts` because
|
|
380
|
-
* deliberately host-based (it holds a bare Store, never a
|
|
378
|
+
/* * The hub bound √N (bounded-reads.md) — the ONE
|
|
379
|
+
* fan-out cap, stated here rather than imported from `traverse.ts` because
|
|
380
|
+
* this module is deliberately host-based (it holds a bare Store, never a
|
|
381
|
+
* MindContext).
|
|
381
382
|
* That is the same write/read-side duplication convention canonical.ts's
|
|
382
383
|
* header documents: if the formula changes it must change in BOTH places.
|
|
383
384
|
* It is stated ONCE per side, though — the expression used to be spelled
|
|
@@ -1038,12 +1039,12 @@ export class GraphSearch {
|
|
|
1038
1039
|
const memo = this.recompleteMemo;
|
|
1039
1040
|
if (memo.has(node)) return memo.get(node) ?? null;
|
|
1040
1041
|
// Re-covering is how a PRODUCED node's bytes enter the search at all: the
|
|
1041
|
-
// cover machinery otherwise only ever sees the QUERY's spans.
|
|
1042
|
+
// cover machinery otherwise only ever sees the QUERY's spans. The recursion
|
|
1042
1043
|
// is allowed to nest — a chain IS nested completions — but it is bounded so
|
|
1043
|
-
// the work stays the ANSWER's (
|
|
1044
|
-
// guard, only ACCEPTED completions recurse, and the nested solve
|
|
1045
|
-
// the form by its own shape instead of re-recognising the
|
|
1046
|
-
// inside it.
|
|
1044
|
+
// the work stays the ANSWER's (bounded-reads.md): the stack below is the
|
|
1045
|
+
// cycle guard, only ACCEPTED completions recurse, and the nested solve
|
|
1046
|
+
// decomposes the form by its own shape instead of re-recognising the
|
|
1047
|
+
// corpus's hub forms inside it.
|
|
1047
1048
|
//
|
|
1048
1049
|
// `recompleteOpen` IS the stack of the chain being built, so MEMBERSHIP is
|
|
1049
1050
|
// the cycle guard: a node already open on this chain cannot re-enter it.
|
package/src/mind/junction.ts
CHANGED
|
@@ -208,7 +208,7 @@ function cachedContainers(
|
|
|
208
208
|
* is legitimately reached across many containing structures. Half the
|
|
209
209
|
* successful junctions would be lost.
|
|
210
210
|
*
|
|
211
|
-
*
|
|
211
|
+
* REFUTED EARLY-STOP (side-cone exhaustion, saturation.md's "real saturation"):
|
|
212
212
|
* stopping the walk the moment ONE side's upward cone is emptied is wrong,
|
|
213
213
|
* in both a hub-guarded form and a hub-flagged form. The junction test is
|
|
214
214
|
* a BYTE containment over the UNION of the two cones, and a junction can be
|
|
@@ -265,13 +265,13 @@ export function junctionContainersFrom(
|
|
|
265
265
|
d: 0,
|
|
266
266
|
}));
|
|
267
267
|
while (stack.length > 0 && out.length < bound) {
|
|
268
|
-
// BUDGET EXHAUSTION IS AN ABSTENTION, AND IT MUST BE VISIBLE
|
|
269
|
-
// walk stops with work still on the stack, the caller
|
|
270
|
-
// and falls through to a lower ladder rung —
|
|
271
|
-
// outside, from a walk that looked everywhere
|
|
272
|
-
// SHARED budget (cross-region's one k·W allowance
|
|
273
|
-
// pair can drain it, so a later pair's exact tier may
|
|
274
|
-
// this counter is the only thing that says so.
|
|
268
|
+
// BUDGET EXHAUSTION IS AN ABSTENTION, AND IT MUST BE VISIBLE
|
|
269
|
+
// (INVARIANTS.md). The walk stops with work still on the stack, the caller
|
|
270
|
+
// reads "no container" and falls through to a lower ladder rung —
|
|
271
|
+
// indistinguishable, from the outside, from a walk that looked everywhere
|
|
272
|
+
// and found nothing. With a SHARED budget (cross-region's one k·W allowance
|
|
273
|
+
// per tier) an EARLIER pair can drain it, so a later pair's exact tier may
|
|
274
|
+
// never run at all; this counter is the only thing that says so.
|
|
275
275
|
if (b.n-- <= 0) {
|
|
276
276
|
if (ctx.meter) ctx.meter.junctionBudgetExhausted++;
|
|
277
277
|
break;
|
package/src/mind/learning.ts
CHANGED
|
@@ -306,7 +306,8 @@ function constituentSketch(ctx: MindContext, id: number, k: number): number[] {
|
|
|
306
306
|
for (const g of constituentSketch(ctx, kid, k)) pool.push(g);
|
|
307
307
|
}
|
|
308
308
|
}
|
|
309
|
-
// Bottom-k by identity, then by id so ties are corpus-determined
|
|
309
|
+
// Bottom-k by identity, then by id so ties are corpus-determined
|
|
310
|
+
// (determinism.md).
|
|
310
311
|
pool.sort((a, b) => (unitPriority(a) - unitPriority(b)) || (a - b));
|
|
311
312
|
const seen = new Set<number>();
|
|
312
313
|
out = [];
|
|
@@ -368,15 +369,15 @@ function constituentSketch(ctx: MindContext, id: number, k: number): number[] {
|
|
|
368
369
|
* terms unique to that partner, which dilute but never mislead; the shared
|
|
369
370
|
* units contribute the signal.
|
|
370
371
|
*
|
|
371
|
-
* HUBS ARE THE ONE EXCLUSION, read LIMITed as `parentsFirst(n, bound+1)` —
|
|
372
|
-
*
|
|
373
|
-
*
|
|
374
|
-
*
|
|
375
|
-
*
|
|
376
|
-
*
|
|
377
|
-
*
|
|
378
|
-
*
|
|
379
|
-
*
|
|
372
|
+
* HUBS ARE THE ONE EXCLUSION, read LIMITed as `parentsFirst(n, bound+1)` — the
|
|
373
|
+
* store's own exact hub-or-not probe (a result longer than the bound means MORE
|
|
374
|
+
* than the bound), never a fan-in-sized read. A constituent with more than √N
|
|
375
|
+
* structural parents is scaffolding by bounded-reads.md's bound: " is ", "the
|
|
376
|
+
* ". Superposing it would put a term shared by every deposit into every
|
|
377
|
+
* profile, ALL halos would correlate, and the concept threshold's null model
|
|
378
|
+
* (unrelated halos at 0 ± 1/√D) that halo-sketch.md's hygiene note protects
|
|
379
|
+
* would collapse. It is still DESCENDED into — a hub chunk can contain a rare
|
|
380
|
+
* unit — but contributes nothing itself.
|
|
380
381
|
*
|
|
381
382
|
* Byte atoms are skipped in BOTH representations (a negative id and a stored
|
|
382
383
|
* kid-less node): an atom's fan-in is the alphabet's, so it can only ever
|
|
@@ -386,32 +387,32 @@ function constituentSketch(ctx: MindContext, id: number, k: number): number[] {
|
|
|
386
387
|
* analogy strength 0.3636 -> 0.2004, "no halo-tier company evidence",
|
|
387
388
|
* test/29 C1).
|
|
388
389
|
*
|
|
389
|
-
* A FUNCTION OF THE NODE AND THE CORPUS STATE — stated precisely, because
|
|
390
|
-
*
|
|
391
|
-
*
|
|
392
|
-
*
|
|
393
|
-
*
|
|
394
|
-
*
|
|
395
|
-
*
|
|
396
|
-
*
|
|
397
|
-
*
|
|
398
|
-
*
|
|
399
|
-
*
|
|
400
|
-
*
|
|
401
|
-
*
|
|
402
|
-
*
|
|
403
|
-
*
|
|
404
|
-
*
|
|
390
|
+
* A FUNCTION OF THE NODE AND THE CORPUS STATE — stated precisely, because the
|
|
391
|
+
* weaker claim is the true one. The constituents are read from the STORE, never
|
|
392
|
+
* from the depositing tree's id map: that map holds only the nodes THIS deposit
|
|
393
|
+
* newly interned, so a partner met a second time yielded a profile missing
|
|
394
|
+
* exactly those constituents, the exact-partner case fell from cosine 1 to
|
|
395
|
+
* 1/√(1+k), and the geometry stopped meaning anything. Reading the store fixes
|
|
396
|
+
* that. It does NOT make the profile permanent: the hub test reads fan-in
|
|
397
|
+
* against √N and both grow with training, so a partner poured early and again
|
|
398
|
+
* late can profile differently. That residue is confined to the hub EXCLUSION —
|
|
399
|
+
* which terms are dropped as scaffolding — and never to which units are found,
|
|
400
|
+
* because the descent itself is now order-independent. The drift is
|
|
401
|
+
* one-directional and benign: a term can only ever go from contributing to
|
|
402
|
+
* being excluded as scaffolding. Replay of a fixed training order is
|
|
403
|
+
* bit-identical, so determinism.md holds. What must not be claimed is that a
|
|
404
|
+
* node's profile is fixed for all time; it is fixed given the corpus that has
|
|
405
|
+
* been seen.
|
|
405
406
|
*
|
|
406
|
-
*
|
|
407
|
-
*
|
|
408
|
-
*
|
|
409
|
-
*
|
|
410
|
-
*
|
|
411
|
-
*
|
|
412
|
-
*
|
|
413
|
-
*
|
|
414
|
-
*
|
|
407
|
+
* THE NULL MODEL IS OTHERWISE UNTOUCHED (halo-sketch.md). Every term is still a
|
|
408
|
+
* seeded function of a NODE IDENTITY, never a gist, so no byte-similarity
|
|
409
|
+
* between partners can leak content similarity into distributional similarity.
|
|
410
|
+
* The result is normalized, so ONE episode still pours ONE unit of mass: {@link
|
|
411
|
+
* Store.haloMass} keeps counting episodes and every mass-based reading is
|
|
412
|
+
* unchanged. Two partners sharing j of k discriminating constituents meet at
|
|
413
|
+
* j/(1+k) — graded evidence, above the 1/√D noise floor and below
|
|
414
|
+
* conceptThreshold until the overlap is most of the content, which is the
|
|
415
|
+
* semantics "same company" should have.
|
|
415
416
|
*
|
|
416
417
|
* Bounded: at most {@link PROFILE_VISITS} constituents are classified, each
|
|
417
418
|
* by ONE LIMITed structural-parent read, so a pour costs O(1) reads in the
|
package/src/mind/match.ts
CHANGED
|
@@ -319,12 +319,12 @@ export function alignGraded(
|
|
|
319
319
|
//
|
|
320
320
|
// "these bytes occupy a place the corpus keeps open" (bind).
|
|
321
321
|
//
|
|
322
|
-
// Both arrive as unaligned residue.
|
|
323
|
-
//
|
|
324
|
-
//
|
|
325
|
-
//
|
|
326
|
-
//
|
|
327
|
-
//
|
|
322
|
+
// Both arrive as unaligned residue. That single missing distinction is why the
|
|
323
|
+
// substitution bridge refuses on `attestedQ`, why the cover charges PASS over a
|
|
324
|
+
// slot, and why CAST reads a filler as noise rather than as the variable it is.
|
|
325
|
+
// The family below supplies it, and it lives HERE — not in any mechanism —
|
|
326
|
+
// because it is the ordinary (matcher, projection, gate) triple of
|
|
327
|
+
// match-project.md with its three parts in their proper places:
|
|
328
328
|
//
|
|
329
329
|
// matcher alignAround + contractGap + frameSlots — bytes only, no
|
|
330
330
|
// projection, no licence. SAFE FOR EVERY CONSUMER: knowing a
|
|
@@ -1168,24 +1168,25 @@ export async function project(
|
|
|
1168
1168
|
|
|
1169
1169
|
// ── The span-shape family ───────────────────────────────────────────────────
|
|
1170
1170
|
//
|
|
1171
|
-
// "Is this answer drawn from this context?" has TWO formally distinct
|
|
1172
|
-
//
|
|
1173
|
-
//
|
|
1174
|
-
//
|
|
1175
|
-
//
|
|
1176
|
-
//
|
|
1177
|
-
//
|
|
1178
|
-
//
|
|
1179
|
-
//
|
|
1180
|
-
//
|
|
1181
|
-
//
|
|
1171
|
+
// "Is this answer drawn from this context?" has TWO formally distinct readings,
|
|
1172
|
+
// and the pair plus the anchor classifier built on them are SHARED machinery —
|
|
1173
|
+
// extraction proposes span-shaped exemplars with them, the shared
|
|
1174
|
+
// `Precomputed.spanShapedOf` container computes them, and fusion (reasoning.ts)
|
|
1175
|
+
// gates on the strict one. They lived inside mechanisms/extraction.ts, so
|
|
1176
|
+
// `pipeline-mechanism.ts` and `reasoning.ts` both had to import back OUT of a
|
|
1177
|
+
// specific mechanism — an inversion the mechanism market forbids: the shared
|
|
1178
|
+
// contract may not depend on any one mechanism (mechanism-market.md), and a
|
|
1179
|
+
// shared matcher belongs to this family (match-project.md), never to a
|
|
1180
|
+
// mechanism's private helpers. Deleting extraction must not break the shared
|
|
1181
|
+
// container, so they live here.
|
|
1182
1182
|
//
|
|
1183
1183
|
// • isSpanShaped — the OPEN reading (sparse in-order embedding).
|
|
1184
1184
|
// • containsSpan — the STRICT reading (contiguous run or resolved node).
|
|
1185
1185
|
// • skillExemplar — classify one anchor into (context, answer) using them.
|
|
1186
1186
|
//
|
|
1187
|
-
// The two readings are NOT interchangeable;
|
|
1188
|
-
// and each function's own doc states what breaks if it is
|
|
1187
|
+
// The two readings are NOT interchangeable; match-project.md pins the
|
|
1188
|
+
// distinction and each function's own doc states what breaks if it is
|
|
1189
|
+
// substituted.
|
|
1189
1190
|
|
|
1190
1191
|
/** Check whether an anchor is a span-shaped skill exemplar: it represents a
|
|
1191
1192
|
* fact whose context and answer together form a span-in-context pattern.
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
// Cover consumes recognition directly (its axioms are the query's own
|
|
5
5
|
// decomposition) plus the computed spans any parse()-bearing mechanism
|
|
6
6
|
// contributed: computed spans MASK colliding recognised sites and enter the
|
|
7
|
-
// search at zero cost ("computation always wins",
|
|
7
|
+
// search at zero cost ("computation always wins", alu.md) — which is also why
|
|
8
8
|
// cover runs FIRST in defaultMechanisms: a computed-backed cover becomes a
|
|
9
9
|
// near-zero-cost incumbent that prunes the other mechanisms through the
|
|
10
10
|
// ordinary admissible-floor check, with no extension special-case anywhere.
|
|
@@ -75,22 +75,23 @@ export async function resolveConnectors(
|
|
|
75
75
|
if (query === undefined || ctx.answeredSpans.length === 0) return true;
|
|
76
76
|
const continuations = ctx.store.nextFirst(s.payload, hubBound(ctx));
|
|
77
77
|
return !continuations.some((answer) => {
|
|
78
|
-
// PREFIX-CAPPED (
|
|
79
|
-
// occur INSIDE it, so read one byte past the query's length —
|
|
80
|
-
// detect the overflow — and reject without reconstructing the
|
|
81
|
-
// The `+ 1` is what makes the test exact rather than a
|
|
82
|
-
// result of exactly `query.length + 1` bytes is known to
|
|
83
|
-
// and anything shorter is the candidate's COMPLETE
|
|
84
|
-
// substring test below is the same test as before.
|
|
85
|
-
// probe bridge.ts:256 already uses.)
|
|
78
|
+
// PREFIX-CAPPED (bounded-reads.md): a candidate longer than the query
|
|
79
|
+
// cannot occur INSIDE it, so read one byte past the query's length —
|
|
80
|
+
// enough to detect the overflow — and reject without reconstructing the
|
|
81
|
+
// rest. The `+ 1` is what makes the test exact rather than a
|
|
82
|
+
// truncation: a result of exactly `query.length + 1` bytes is known to
|
|
83
|
+
// be too long, and anything shorter is the candidate's COMPLETE
|
|
84
|
+
// content, so the substring test below is the same test as before. (The
|
|
85
|
+
// same overflow probe bridge.ts:256 already uses.)
|
|
86
86
|
//
|
|
87
87
|
// This loop runs up to hubBound(ctx) = √N reads PER SITE, and only on a
|
|
88
88
|
// multi-turn response — `answeredSpans` is empty for a plain respond(),
|
|
89
|
-
// so the probe does not execute there.
|
|
89
|
+
// so the probe does not execute there. The cap cannot reduce the read
|
|
90
90
|
// COUNT — only a semantic change to the "already answered" test could —
|
|
91
91
|
// but it bounds each read by the query instead of by the corpus, which
|
|
92
|
-
// is what
|
|
93
|
-
// reads 4 bytes per candidate instead of the ~231 it
|
|
92
|
+
// is what bounded-reads.md asks for and what rescues a SHORT query: at
|
|
93
|
+
// 3 bytes this reads 4 bytes per candidate instead of the ~231 it
|
|
94
|
+
// averaged before.
|
|
94
95
|
const bytes = read(ctx, answer, query.length + 1);
|
|
95
96
|
return bytes.length <= query.length && indexOf(query, bytes, 0) >= 0;
|
|
96
97
|
});
|
|
@@ -1,15 +1,15 @@
|
|
|
1
1
|
// mechanisms/prefix-completion.ts — Grounding a query that IS the opening of a
|
|
2
2
|
// trained form (Grounding V).
|
|
3
3
|
//
|
|
4
|
-
// A MECHANISM, NOT A TIER.
|
|
5
|
-
//
|
|
6
|
-
// from, where placement rather than the cost ladder decided.
|
|
4
|
+
// A MECHANISM, NOT A TIER. This used to run inside recall's refusal path, in a
|
|
5
|
+
// fixed if-chain that first-match-wins — the shape CAST was refactored away
|
|
6
|
+
// from, where placement rather than the cost ladder decided. Its claim is
|
|
7
7
|
// maximal (every query byte literally matched, from offset zero, against a
|
|
8
8
|
// trained form) at one STEP, so as a market candidate it competes honestly and
|
|
9
|
-
// the decider weighs it like everything else.
|
|
9
|
+
// the decider weighs it like everything else. It is registered LAST: recall's
|
|
10
10
|
// exact self-match makes an IDENTITY claim about the query while this makes a
|
|
11
|
-
// CONTAINMENT one, and on an exact grade tie the identity claim is the
|
|
12
|
-
//
|
|
11
|
+
// CONTAINMENT one, and on an exact grade tie the identity claim is the stronger
|
|
12
|
+
// evidence — the same ordering exact-vs-approximate.md's ladders use.
|
|
13
13
|
//
|
|
14
14
|
// Its SUPPLY moved too, and further: `formsOpenedBy` (traverse.ts) answers a
|
|
15
15
|
// question about the STORE — "which trained forms does this byte run open?" —
|
|
@@ -49,12 +49,12 @@
|
|
|
49
49
|
//
|
|
50
50
|
// So this is a RETRIEVABILITY gap, not a semantic one, and the ANN is the wrong
|
|
51
51
|
// instrument for it: a proper prefix's gist cannot rank its own continuation.
|
|
52
|
-
// The repair is CONTENT-ADDRESSED (
|
|
53
|
-
// the leaf-id WINDOW index the write side already maintains
|
|
54
|
-
// trained forms does this byte run open?" in a bounded √N
|
|
55
|
-
// mechanism's first supply.
|
|
56
|
-
// second, for prefixes long enough that the gist still
|
|
57
|
-
// read, never re-issued.
|
|
52
|
+
// The repair is CONTENT-ADDRESSED (exact-vs-approximate.md) — `formsOpenedBy`
|
|
53
|
+
// (traverse.ts) reads the leaf-id WINDOW index the write side already maintains
|
|
54
|
+
// and answers "which trained forms does this byte run open?" in a bounded √N
|
|
55
|
+
// walk. That is this mechanism's first supply. The response's memoised top-k
|
|
56
|
+
// `resonance()` is the second, for prefixes long enough that the gist still
|
|
57
|
+
// ranks the form; it is read, never re-issued.
|
|
58
58
|
//
|
|
59
59
|
// AN EXHAUSTIVE ANN LIST IS NOT A SUPPLY HERE, AND WAS REMOVED. This tier once
|
|
60
60
|
// read `Precomputed.wideResonance()` — a full-index `resonate(guide, √N,
|
|
@@ -273,20 +273,20 @@ export const prefixMechanism: PipelineMechanism = {
|
|
|
273
273
|
return STEP;
|
|
274
274
|
},
|
|
275
275
|
async run(ctx, query, pre) {
|
|
276
|
-
// ONE SUPPLY PASS, not a two-tier `??`.
|
|
276
|
+
// ONE SUPPLY PASS, not a two-tier `??`. The window index (exact,
|
|
277
277
|
// content-addressed) and the response's memoised top-k (approximate) are
|
|
278
|
-
// concatenated and the three guards decide ONCE over the union.
|
|
278
|
+
// concatenated and the three guards decide ONCE over the union. A
|
|
279
279
|
// first-then-fallback chain would let the APPROXIMATE tier override the
|
|
280
|
-
// EXACT one (
|
|
281
|
-
// returns null and the fallback re-runs the guards
|
|
282
|
-
// alone — which, seeing only one of the two forms,
|
|
283
|
-
// precisely the disagreement-suppression guard 3
|
|
284
|
-
// is the exact tier's ambiguity being washed away
|
|
285
|
-
// Evaluating the union means a disagreement the
|
|
286
|
-
// be hidden by what the ANN happens to rank.
|
|
287
|
-
// response's ONE memoised top-k (
|
|
288
|
-
// path on the queries where this mechanism fires,
|
|
289
|
-
// a second index scan.
|
|
280
|
+
// EXACT one (exact-vs-approximate.md): when formsOpenedBy finds two
|
|
281
|
+
// continuations, guard 3 returns null and the fallback re-runs the guards
|
|
282
|
+
// on resonance's top-k alone — which, seeing only one of the two forms,
|
|
283
|
+
// would voice it. That is precisely the disagreement-suppression guard 3
|
|
284
|
+
// exists to prevent, and it is the exact tier's ambiguity being washed away
|
|
285
|
+
// by the approximate tier. Evaluating the union means a disagreement the
|
|
286
|
+
// window index saw can never be hidden by what the ANN happens to rank. The
|
|
287
|
+
// ANN read is the response's ONE memoised top-k (memoization.md), already
|
|
288
|
+
// paid by recall's refusal path on the queries where this mechanism fires,
|
|
289
|
+
// so reading it here is not a second index scan.
|
|
290
290
|
const ids = [
|
|
291
291
|
...formsOpenedBy(ctx, query),
|
|
292
292
|
...(await pre.resonance()).map((h) => h.id),
|
|
@@ -236,24 +236,22 @@ export async function recallByResonance(
|
|
|
236
236
|
}
|
|
237
237
|
}
|
|
238
238
|
|
|
239
|
-
// The query-relative grounding fraction, shared by tiers 2–4 — gated on
|
|
240
|
-
//
|
|
241
|
-
//
|
|
242
|
-
//
|
|
243
|
-
//
|
|
244
|
-
//
|
|
245
|
-
//
|
|
246
|
-
//
|
|
247
|
-
//
|
|
248
|
-
//
|
|
249
|
-
//
|
|
250
|
-
// √(lenG/lenQ)
|
|
251
|
-
//
|
|
252
|
-
//
|
|
253
|
-
//
|
|
254
|
-
//
|
|
255
|
-
// subtract the significance bar (3/√D, §8.3) before converting. Derived
|
|
256
|
-
// from the existing bars; never tuned.
|
|
239
|
+
// The query-relative grounding fraction, shared by tiers 2–4 — gated on the
|
|
240
|
+
// FRACTION OF THE QUERY the grounding explains, not the raw cosine. Root
|
|
241
|
+
// gists are unit vectors, but their magnitudes are recoverable from the byte
|
|
242
|
+
// lengths (‖·‖ = √len under the linear fold): cos = shared/√(lenQ·lenG), so
|
|
243
|
+
// shared/lenQ = cos·√(lenG/lenQ). The raw cosine punished honest containment
|
|
244
|
+
// — a query fully inside a longer grounded answer scored √(lenQ/lenG) and was
|
|
245
|
+
// refused — and let a long answer sharing only scaffolding pass; the
|
|
246
|
+
// query-relative fraction measures exactly what the reach bar means: how much
|
|
247
|
+
// of THE QUERY the store accounts for. Chance similarity survives the length
|
|
248
|
+
// conversion AMPLIFIED: the same √(lenG/lenQ) factor that converts an honest
|
|
249
|
+
// shared fraction into a query-relative one multiplies the estimator/chance
|
|
250
|
+
// floor too, so a long stored form (√(lenG/lenQ) ≈ 10 at 100×) lifted a
|
|
251
|
+
// noise-level cosine past the reach bar and grounded pure gibberish
|
|
252
|
+
// (observed). Only the ABOVE-CHANCE part of the similarity is evidence of
|
|
253
|
+
// shared content — subtract the significance bar (3/√D, thresholds.md) before
|
|
254
|
+
// converting. Derived from the existing bars; never tuned.
|
|
257
255
|
const sig = significanceBar(ctx.store.D);
|
|
258
256
|
const reach = reachThreshold(ctx.space.maxGroup);
|
|
259
257
|
const fracOfQuery = (cos: number, otherLen: number): number =>
|
|
@@ -411,14 +409,14 @@ export async function recallByResonance(
|
|
|
411
409
|
}
|
|
412
410
|
}
|
|
413
411
|
}
|
|
414
|
-
// 3b. Corroborated-substitution bridge — refusal-path only (bridge.ts).
|
|
415
|
-
//
|
|
416
|
-
//
|
|
417
|
-
//
|
|
418
|
-
//
|
|
419
|
-
//
|
|
420
|
-
// cluster here once made every honest refusal cost hundreds of ms
|
|
421
|
-
// of k.
|
|
412
|
+
// 3b. Corroborated-substitution bridge — refusal-path only (bridge.ts). The
|
|
413
|
+
// bridge's proposal source is the response's ONE top-k read — the same list
|
|
414
|
+
// recall already ranked above — never an exhaustive √N scan. The bridge's own
|
|
415
|
+
// candidate cap is 2·recallQueryK, so top-k proposals are exactly the budget
|
|
416
|
+
// it can consume, and every proposal is byte-verified downstream
|
|
417
|
+
// (exact-vs-approximate.md). Reuse the memoised `resonance()`; scanning every
|
|
418
|
+
// IVF cluster here once made every honest refusal cost hundreds of ms
|
|
419
|
+
// regardless of k.
|
|
422
420
|
const wideIds = async () => (await pre.resonance()).map((h) => h.id);
|
|
423
421
|
|
|
424
422
|
// Every gist-based tier has failed; before refusing, align the query
|
|
@@ -471,12 +469,12 @@ export async function recallByResonance(
|
|
|
471
469
|
// prefixCompletion runs a few lines below and carries the three guards
|
|
472
470
|
// this tier lacks — unreadable-continuation veto, sub-quantum
|
|
473
471
|
// continuation, and UNIQUENESS (distinct continuations ⇒ refuse), which
|
|
474
|
-
// is exactly what 4,300 competing values must trip.
|
|
475
|
-
//
|
|
476
|
-
//
|
|
477
|
-
// purpose — a candidate differing by case or punctuation
|
|
478
|
-
// capital of france" → "What is the capital of France?") is
|
|
479
|
-
// prefix, keeps grounding here, and is unaffected.
|
|
472
|
+
// is exactly what 4,300 competing values must trip. So this is not a new
|
|
473
|
+
// rule and not a new threshold: it is deferring a prefix decision to the
|
|
474
|
+
// tier that owns it (match-project.md, one factored machinery).
|
|
475
|
+
// Byte-strict on purpose — a candidate differing by case or punctuation
|
|
476
|
+
// ("what is the capital of france" → "What is the capital of France?") is
|
|
477
|
+
// NOT a byte prefix, keeps grounding here, and is unaffected.
|
|
480
478
|
const strictPrefix = g !== null &&
|
|
481
479
|
cBytes.length > query.length &&
|
|
482
480
|
indexOf(cBytes, query, 0) === 0;
|
|
@@ -542,15 +540,15 @@ export async function recallByResonance(
|
|
|
542
540
|
}
|
|
543
541
|
}
|
|
544
542
|
|
|
545
|
-
// The refusal/echo decision.
|
|
546
|
-
//
|
|
543
|
+
// The refusal/echo decision. The echo returns a stored form's bytes AS the
|
|
544
|
+
// answer — a near-identity claim about the query — and identity-grade
|
|
547
545
|
// decisions are never made on an estimated score ("approximate scores may
|
|
548
|
-
// rank and propose; they may never decide",
|
|
549
|
-
// overshooting the reach bar echoed a WRONG-entity neighbour
|
|
550
|
-
// Zamunda?" echoed the Armenia fact, observed).
|
|
551
|
-
// anyway to be echoed, so the decision uses their EXACT fold: one river
|
|
552
|
-
// fold of the top hit, measured in the same query-relative,
|
|
553
|
-
//
|
|
546
|
+
// rank and propose; they may never decide", exact-vs-approximate.md): the
|
|
547
|
+
// RaBitQ estimate overshooting the reach bar echoed a WRONG-entity neighbour
|
|
548
|
+
// ("capital of Zamunda?" echoed the Armenia fact, observed). The bytes are
|
|
549
|
+
// read anyway to be echoed, so the decision uses their EXACT fold: one river
|
|
550
|
+
// fold of the top hit, measured in the same query-relative, chance-corrected
|
|
551
|
+
// units as the tier above.
|
|
554
552
|
const topBytes = read(ctx, top.id);
|
|
555
553
|
const exact = topBytes.length > 0
|
|
556
554
|
? cosine(queryGist, gistOf(ctx, topBytes))
|