@hviana/sema 0.8.0 → 0.8.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +22 -1
- package/DATASETS.md +1 -1
- package/dist/example/train_base/config.js +2 -2
- package/dist/example/train_base/corpora/massive.js +1 -1
- package/dist/example/train_base/readers.js +1 -1
- package/dist/src/config.d.ts +17 -0
- package/dist/src/config.js +18 -0
- package/dist/src/geometry.d.ts +10 -10
- package/dist/src/geometry.js +25 -24
- package/dist/src/meter.d.ts +29 -12
- package/dist/src/meter.js +58 -14
- package/dist/src/mind/attention.js +12 -12
- package/dist/src/mind/bridge.d.ts +8 -8
- package/dist/src/mind/bridge.js +33 -32
- package/dist/src/mind/corpus.d.ts +40 -0
- package/dist/src/mind/corpus.js +149 -0
- package/dist/src/mind/graph-search.d.ts +7 -8
- package/dist/src/mind/graph-search.js +244 -32
- package/dist/src/mind/index.d.ts +3 -1
- package/dist/src/mind/index.js +1 -0
- package/dist/src/mind/junction.d.ts +1 -1
- package/dist/src/mind/junction.js +8 -8
- package/dist/src/mind/learning.js +36 -35
- package/dist/src/mind/match.d.ts +8 -3
- package/dist/src/mind/match.js +156 -71
- package/dist/src/mind/mechanisms/cast.js +18 -2
- package/dist/src/mind/mechanisms/cover.js +19 -12
- package/dist/src/mind/mechanisms/prefix-completion.js +24 -24
- package/dist/src/mind/mechanisms/recall.js +38 -40
- package/dist/src/mind/mechanisms/reference.js +16 -16
- package/dist/src/mind/mind.d.ts +61 -7
- package/dist/src/mind/mind.js +72 -2
- package/dist/src/mind/pipeline-mechanism.d.ts +10 -8
- package/dist/src/mind/pipeline-mechanism.js +25 -21
- package/dist/src/mind/pipeline.d.ts +9 -9
- package/dist/src/mind/pipeline.js +49 -29
- package/dist/src/mind/primitives.d.ts +5 -5
- package/dist/src/mind/primitives.js +5 -5
- package/dist/src/mind/reasoning.d.ts +5 -1
- package/dist/src/mind/reasoning.js +54 -1
- package/dist/src/mind/recognition.d.ts +14 -13
- package/dist/src/mind/recognition.js +23 -23
- package/dist/src/mind/resonance.js +21 -21
- package/dist/src/mind/traverse.d.ts +54 -52
- package/dist/src/mind/traverse.js +83 -73
- package/dist/src/mind/types.d.ts +26 -4
- package/dist/src/store.d.ts +12 -12
- package/dist/src/store.js +12 -12
- package/docs/INDEX.md +2 -2
- package/docs/architecture/exact-vs-approximate.md +2 -1
- package/docs/architecture/fold-contract.md +1 -1
- package/docs/failures/tempting-but-wrong.md +33 -5
- package/docs/harness/gates.md +7 -7
- package/example/train_base/config.ts +2 -2
- package/example/train_base/corpora/massive.ts +1 -1
- package/example/train_base/readers.ts +1 -1
- package/jsr.json +1 -1
- package/package.json +1 -1
- package/src/config.ts +35 -0
- package/src/geometry.ts +25 -24
- package/src/meter.ts +61 -14
- package/src/mind/attention.ts +12 -12
- package/src/mind/bridge.ts +33 -32
- package/src/mind/corpus.ts +202 -0
- package/src/mind/graph-search.ts +261 -31
- package/src/mind/index.ts +8 -1
- package/src/mind/junction.ts +8 -8
- package/src/mind/learning.ts +36 -35
- package/src/mind/match.ts +163 -73
- package/src/mind/mechanisms/cast.ts +17 -1
- package/src/mind/mechanisms/cover.ts +18 -12
- package/src/mind/mechanisms/prefix-completion.ts +24 -24
- package/src/mind/mechanisms/recall.ts +38 -40
- package/src/mind/mechanisms/reference.ts +16 -16
- package/src/mind/mind.ts +129 -7
- package/src/mind/pipeline-mechanism.ts +25 -21
- package/src/mind/pipeline.ts +63 -38
- package/src/mind/primitives.ts +5 -5
- package/src/mind/reasoning.ts +55 -0
- package/src/mind/recognition.ts +23 -23
- package/src/mind/resonance.ts +21 -21
- package/src/mind/traverse.ts +83 -73
- package/src/mind/types.ts +30 -4
- package/src/store.ts +20 -20
- package/test/08-storage.test.mjs +1 -1
- package/test/100-complete-grounding-trace.test.mjs +109 -0
- package/test/101-alignment-gap-bound.test.mjs +106 -0
- package/test/102-production-composes-at-scale.test.mjs +110 -0
- package/test/103-alignment-gap-budget.test.mjs +89 -0
- package/test/104-composition-is-reported.test.mjs +90 -0
- package/test/105-derive-through-reports-its-refusal.test.mjs +113 -0
- package/test/106-the-join-fires.test.mjs +94 -0
- package/test/107-the-join-is-counted.test.mjs +81 -0
- package/test/108-the-join-chains.test.mjs +78 -0
- package/test/109-the-pivot-is-counted.test.mjs +60 -0
- package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
- package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
- package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
- package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
- package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
- package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
- package/test/117-corpus-search.test.mjs +171 -0
- package/test/14-scaling.test.mjs +10 -7
- package/test/35-prefix-edge.test.mjs +1 -1
- package/test/40-choosenext-scale-guard.test.mjs +16 -17
- package/test/56-bridge-identity-admission.test.mjs +6 -6
- package/test/70-prefix-completion.test.mjs +4 -3
- package/test/72-prefix-candidate-supply.test.mjs +3 -3
- package/test/73-scaffolding-only-bridge-abstains.test.mjs +6 -6
- package/test/75-multiturn-context-optimisation.test.mjs +5 -5
- package/test/76-reference-binding.test.mjs +6 -1
- package/test/84-composed-answer-honesty.test.mjs +5 -6
- package/test/88-dependency-footprint.test.mjs +1 -1
- package/test/89-completion-recursion.test.mjs +47 -19
- package/test/90-connector-read-cap.test.mjs +10 -8
- package/test/93-regime-prediction.test.mjs +10 -10
- package/test/94-cross-region-budget.test.mjs +2 -2
- package/test/95-wide-resonance-removed.test.mjs +8 -7
- package/test/96-bytes-walk-termination.test.mjs +3 -3
|
@@ -268,7 +268,8 @@ function constituentSketch(ctx, id, k) {
|
|
|
268
268
|
pool.push(g);
|
|
269
269
|
}
|
|
270
270
|
}
|
|
271
|
-
// Bottom-k by identity, then by id so ties are corpus-determined
|
|
271
|
+
// Bottom-k by identity, then by id so ties are corpus-determined
|
|
272
|
+
// (determinism.md).
|
|
272
273
|
pool.sort((a, b) => (unitPriority(a) - unitPriority(b)) || (a - b));
|
|
273
274
|
const seen = new Set();
|
|
274
275
|
out = [];
|
|
@@ -331,15 +332,15 @@ function constituentSketch(ctx, id, k) {
|
|
|
331
332
|
* terms unique to that partner, which dilute but never mislead; the shared
|
|
332
333
|
* units contribute the signal.
|
|
333
334
|
*
|
|
334
|
-
* HUBS ARE THE ONE EXCLUSION, read LIMITed as `parentsFirst(n, bound+1)` —
|
|
335
|
-
*
|
|
336
|
-
*
|
|
337
|
-
*
|
|
338
|
-
*
|
|
339
|
-
*
|
|
340
|
-
*
|
|
341
|
-
*
|
|
342
|
-
*
|
|
335
|
+
* HUBS ARE THE ONE EXCLUSION, read LIMITed as `parentsFirst(n, bound+1)` — the
|
|
336
|
+
* store's own exact hub-or-not probe (a result longer than the bound means MORE
|
|
337
|
+
* than the bound), never a fan-in-sized read. A constituent with more than √N
|
|
338
|
+
* structural parents is scaffolding by bounded-reads.md's bound: " is ", "the
|
|
339
|
+
* ". Superposing it would put a term shared by every deposit into every
|
|
340
|
+
* profile, ALL halos would correlate, and the concept threshold's null model
|
|
341
|
+
* (unrelated halos at 0 ± 1/√D) that halo-sketch.md's hygiene note protects
|
|
342
|
+
* would collapse. It is still DESCENDED into — a hub chunk can contain a rare
|
|
343
|
+
* unit — but contributes nothing itself.
|
|
343
344
|
*
|
|
344
345
|
* Byte atoms are skipped in BOTH representations (a negative id and a stored
|
|
345
346
|
* kid-less node): an atom's fan-in is the alphabet's, so it can only ever
|
|
@@ -349,32 +350,32 @@ function constituentSketch(ctx, id, k) {
|
|
|
349
350
|
* analogy strength 0.3636 -> 0.2004, "no halo-tier company evidence",
|
|
350
351
|
* test/29 C1).
|
|
351
352
|
*
|
|
352
|
-
* A FUNCTION OF THE NODE AND THE CORPUS STATE — stated precisely, because
|
|
353
|
-
*
|
|
354
|
-
*
|
|
355
|
-
*
|
|
356
|
-
*
|
|
357
|
-
*
|
|
358
|
-
*
|
|
359
|
-
*
|
|
360
|
-
*
|
|
361
|
-
*
|
|
362
|
-
*
|
|
363
|
-
*
|
|
364
|
-
*
|
|
365
|
-
*
|
|
366
|
-
*
|
|
367
|
-
*
|
|
353
|
+
* A FUNCTION OF THE NODE AND THE CORPUS STATE — stated precisely, because the
|
|
354
|
+
* weaker claim is the true one. The constituents are read from the STORE, never
|
|
355
|
+
* from the depositing tree's id map: that map holds only the nodes THIS deposit
|
|
356
|
+
* newly interned, so a partner met a second time yielded a profile missing
|
|
357
|
+
* exactly those constituents, the exact-partner case fell from cosine 1 to
|
|
358
|
+
* 1/√(1+k), and the geometry stopped meaning anything. Reading the store fixes
|
|
359
|
+
* that. It does NOT make the profile permanent: the hub test reads fan-in
|
|
360
|
+
* against √N and both grow with training, so a partner poured early and again
|
|
361
|
+
* late can profile differently. That residue is confined to the hub EXCLUSION —
|
|
362
|
+
* which terms are dropped as scaffolding — and never to which units are found,
|
|
363
|
+
* because the descent itself is now order-independent. The drift is
|
|
364
|
+
* one-directional and benign: a term can only ever go from contributing to
|
|
365
|
+
* being excluded as scaffolding. Replay of a fixed training order is
|
|
366
|
+
* bit-identical, so determinism.md holds. What must not be claimed is that a
|
|
367
|
+
* node's profile is fixed for all time; it is fixed given the corpus that has
|
|
368
|
+
* been seen.
|
|
368
369
|
*
|
|
369
|
-
*
|
|
370
|
-
*
|
|
371
|
-
*
|
|
372
|
-
*
|
|
373
|
-
*
|
|
374
|
-
*
|
|
375
|
-
*
|
|
376
|
-
*
|
|
377
|
-
*
|
|
370
|
+
* THE NULL MODEL IS OTHERWISE UNTOUCHED (halo-sketch.md). Every term is still a
|
|
371
|
+
* seeded function of a NODE IDENTITY, never a gist, so no byte-similarity
|
|
372
|
+
* between partners can leak content similarity into distributional similarity.
|
|
373
|
+
* The result is normalized, so ONE episode still pours ONE unit of mass: {@link
|
|
374
|
+
* Store.haloMass} keeps counting episodes and every mass-based reading is
|
|
375
|
+
* unchanged. Two partners sharing j of k discriminating constituents meet at
|
|
376
|
+
* j/(1+k) — graded evidence, above the 1/√D noise floor and below
|
|
377
|
+
* conceptThreshold until the overlap is most of the content, which is the
|
|
378
|
+
* semantics "same company" should have.
|
|
378
379
|
*
|
|
379
380
|
* Bounded: at most {@link PROFILE_VISITS} constituents are classified, each
|
|
380
381
|
* by ONE LIMITed structural-parent read, so a pour costs O(1) reads in the
|
package/dist/src/mind/match.d.ts
CHANGED
|
@@ -78,9 +78,14 @@ export interface AlignGap {
|
|
|
78
78
|
}
|
|
79
79
|
/** Extend a seed match (query offset qo ↔ candidate offset co) to its maximal
|
|
80
80
|
* common run, then walk outward in both directions collecting further common
|
|
81
|
-
* runs of at least W bytes across
|
|
82
|
-
*
|
|
83
|
-
*
|
|
81
|
+
* runs of at least W bytes across mismatch gaps. Each gap's LENGTH is the
|
|
82
|
+
* pair's own extent (a gap cannot be longer than the bytes it spans) and the
|
|
83
|
+
* sweep's WORK is proportional to the bytes a run spans (the context's windows
|
|
84
|
+
* are indexed once, then the query's are walked) — the arity bound
|
|
85
|
+
* (`chainReach`) used to cap BOTH, and truncated every learned frame whose
|
|
86
|
+
* slot was longer. Each sweep owns its own budget, so an exhausted right
|
|
87
|
+
* sweep never starves the left one. Returns the matched query spans and the
|
|
88
|
+
* mismatch pairs between consecutive runs.
|
|
84
89
|
*
|
|
85
90
|
* This is the SEEDED aligner, distinct from {@link alignRuns}: that one finds
|
|
86
91
|
* every run two structures share anywhere (a weave), this one reads two
|
package/dist/src/mind/match.js
CHANGED
|
@@ -31,8 +31,8 @@
|
|
|
31
31
|
// they gate.
|
|
32
32
|
import { addInto, cosine, dot, normalize, zeros } from "../vec.js";
|
|
33
33
|
import { conceptThreshold, dominates, identityBar, significanceBar, } from "../geometry.js";
|
|
34
|
-
import { bytesEqual, indexOf } from "../bytes.js";
|
|
35
|
-
import {
|
|
34
|
+
import { bytesEqual, indexOf, latin1 } from "../bytes.js";
|
|
35
|
+
import { leafIdRun } from "./canonical.js";
|
|
36
36
|
import { foldTree, gistOf, perceive, read, resolve } from "./primitives.js";
|
|
37
37
|
import { argmaxCosine, chooseAmong, chooseNext, corpusN, edgeAncestors, guidedFirst, hubBound, hubCap, sharedReachMemo, } from "./traverse.js";
|
|
38
38
|
import { recognise, segment } from "./recognition.js";
|
|
@@ -237,9 +237,14 @@ export function alignGraded(ctx, query, contextBytes, querySites) {
|
|
|
237
237
|
}
|
|
238
238
|
/** Extend a seed match (query offset qo ↔ candidate offset co) to its maximal
|
|
239
239
|
* common run, then walk outward in both directions collecting further common
|
|
240
|
-
* runs of at least W bytes across
|
|
241
|
-
*
|
|
242
|
-
*
|
|
240
|
+
* runs of at least W bytes across mismatch gaps. Each gap's LENGTH is the
|
|
241
|
+
* pair's own extent (a gap cannot be longer than the bytes it spans) and the
|
|
242
|
+
* sweep's WORK is proportional to the bytes a run spans (the context's windows
|
|
243
|
+
* are indexed once, then the query's are walked) — the arity bound
|
|
244
|
+
* (`chainReach`) used to cap BOTH, and truncated every learned frame whose
|
|
245
|
+
* slot was longer. Each sweep owns its own budget, so an exhausted right
|
|
246
|
+
* sweep never starves the left one. Returns the matched query spans and the
|
|
247
|
+
* mismatch pairs between consecutive runs.
|
|
243
248
|
*
|
|
244
249
|
* This is the SEEDED aligner, distinct from {@link alignRuns}: that one finds
|
|
245
250
|
* every run two structures share anywhere (a weave), this one reads two
|
|
@@ -253,7 +258,21 @@ export function alignGraded(ctx, query, contextBytes, querySites) {
|
|
|
253
258
|
* window test); {@link frameSlots} takes the other reading. */
|
|
254
259
|
export function alignAround(ctx, q, c, qo, co) {
|
|
255
260
|
const W = ctx.space.maxGroup;
|
|
256
|
-
|
|
261
|
+
// THE GAP LENGTH IS THE PAIR'S OWN EXTENT; THE WORK IS BUDGETED.
|
|
262
|
+
//
|
|
263
|
+
// The sweep walks (queryGap, contextGap) pairs by ASCENDING total, so reaching
|
|
264
|
+
// a gap of size G costs about G²/2 pairs. Bounding the LENGTH by the write
|
|
265
|
+
// side's arity (`chainReach(W)` = 16) therefore truncated every learned frame
|
|
266
|
+
// whose slot is longer — measured: `bindReference` reported the cap at 18, 24,
|
|
267
|
+
// 30 and 36 bytes and `recall` answered with ANOTHER instance's filler — while
|
|
268
|
+
// removing the bound outright took the corpus-cost guard (test/89) from
|
|
269
|
+
// milliseconds to 68 seconds.
|
|
270
|
+
//
|
|
271
|
+
// Bounding the PAIRS keeps a call's cost constant however long the pair is,
|
|
272
|
+
// and the ascending order means an exhausted budget drops the FAR
|
|
273
|
+
// continuations and never the near ones — the same degradation recognition.ts
|
|
274
|
+
// documents for its canon budget. Length and work are different questions;
|
|
275
|
+
// this is the one place they were conflated.
|
|
257
276
|
// Maximal run around the seed.
|
|
258
277
|
let qs = qo, ss = co;
|
|
259
278
|
while (qs > 0 && ss > 0 && q[qs - 1] === c[ss - 1]) {
|
|
@@ -267,8 +286,66 @@ export function alignAround(ctx, q, c, qo, co) {
|
|
|
267
286
|
}
|
|
268
287
|
const matched = [[qs, qe]];
|
|
269
288
|
const gaps = [];
|
|
270
|
-
//
|
|
271
|
-
//
|
|
289
|
+
// THE SWEEP IS STRUCTURAL, NOT ENUMERATIVE.
|
|
290
|
+
//
|
|
291
|
+
// The criterion is unchanged: the next common run, MINIMUM TOTAL GAP, ties to
|
|
292
|
+
// the smaller query gap. What changed is how it is found. Enumerating
|
|
293
|
+
// (queryGap, contextGap) pairs by ascending total reaches a run at total t in
|
|
294
|
+
// about t²/2 pairs — and that quadratic shape, not the reach, was the cost
|
|
295
|
+
// problem: capping the pairs dropped reach (a legitimate 24-byte slot stopped
|
|
296
|
+
// being found), while leaving them uncapped cost 68 seconds on the corpus
|
|
297
|
+
// guard. Neither is the answer, because the answer is the algorithm.
|
|
298
|
+
//
|
|
299
|
+
// The context's windows are indexed ONCE, for lengths 1..W — W being the
|
|
300
|
+
// geometry's own unit of composition, so nothing is chosen here. Each step
|
|
301
|
+
// then walks the query's windows outward from the anchor: for a given query
|
|
302
|
+
// gap the nearest context gap that continues a run is one O(1) lookup, and the
|
|
303
|
+
// walk stops the moment the query gap alone exceeds the best total already
|
|
304
|
+
// found. So the work is proportional to the bytes the run SPANS. No budget,
|
|
305
|
+
// no cap, no number: a long slot is reached, and its price is already the
|
|
306
|
+
// ladder's (its bytes are unaccounted, so the search pays PASS per byte).
|
|
307
|
+
const index = [];
|
|
308
|
+
for (let len = 1; len <= W; len++) {
|
|
309
|
+
const m = new Map();
|
|
310
|
+
for (let o = 0; o + len <= c.length; o++) {
|
|
311
|
+
const key = latin1(c.subarray(o, o + len));
|
|
312
|
+
const at = m.get(key);
|
|
313
|
+
if (at === undefined)
|
|
314
|
+
m.set(key, [o]);
|
|
315
|
+
else
|
|
316
|
+
at.push(o);
|
|
317
|
+
}
|
|
318
|
+
index.push(m);
|
|
319
|
+
}
|
|
320
|
+
/** Smallest listed offset at or after `from`, or -1. */
|
|
321
|
+
const fromAt = (list, from) => {
|
|
322
|
+
let lo = 0, hi = list.length - 1, best = -1;
|
|
323
|
+
while (lo <= hi) {
|
|
324
|
+
const mid = (lo + hi) >> 1;
|
|
325
|
+
if (list[mid] >= from) {
|
|
326
|
+
best = list[mid];
|
|
327
|
+
hi = mid - 1;
|
|
328
|
+
}
|
|
329
|
+
else
|
|
330
|
+
lo = mid + 1;
|
|
331
|
+
}
|
|
332
|
+
return best;
|
|
333
|
+
};
|
|
334
|
+
/** Largest listed offset at or before `to`, or -1. */
|
|
335
|
+
const toAt = (list, to) => {
|
|
336
|
+
let lo = 0, hi = list.length - 1, best = -1;
|
|
337
|
+
while (lo <= hi) {
|
|
338
|
+
const mid = (lo + hi) >> 1;
|
|
339
|
+
if (list[mid] <= to) {
|
|
340
|
+
best = list[mid];
|
|
341
|
+
lo = mid + 1;
|
|
342
|
+
}
|
|
343
|
+
else
|
|
344
|
+
hi = mid - 1;
|
|
345
|
+
}
|
|
346
|
+
return best;
|
|
347
|
+
};
|
|
348
|
+
/** Length of the common run STARTING at (qi, si). */
|
|
272
349
|
const runLenAt = (qi, si) => {
|
|
273
350
|
let n = 0;
|
|
274
351
|
while (qi + n < q.length && si + n < c.length && q[qi + n] === c[si + n]) {
|
|
@@ -276,69 +353,76 @@ export function alignAround(ctx, q, c, qo, co) {
|
|
|
276
353
|
}
|
|
277
354
|
return n;
|
|
278
355
|
};
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
356
|
+
/** Length of the common run ENDING at (qi, si). */
|
|
357
|
+
const runLenBefore = (qi, si) => {
|
|
358
|
+
let n = 0;
|
|
359
|
+
while (n < qi && n < si && q[qi - 1 - n] === c[si - 1 - n])
|
|
360
|
+
n++;
|
|
361
|
+
return n;
|
|
362
|
+
};
|
|
363
|
+
/** The next run outward from an anchor, or null when the bytes run out. */
|
|
364
|
+
const nextRun = (qi, si, forward) => {
|
|
365
|
+
const qLim = forward ? q.length - qi : qi;
|
|
366
|
+
let best = null;
|
|
367
|
+
for (let gq = 0; gq < qLim; gq++) {
|
|
368
|
+
// No later query gap can beat a total already found.
|
|
369
|
+
if (best !== null && gq > best.gq + best.gs)
|
|
370
|
+
break;
|
|
371
|
+
const left = qLim - gq;
|
|
372
|
+
// A run of >= W bytes, or — when the query itself ends inside one window —
|
|
373
|
+
// the run that REACHES that end. Exactly the acceptance the sweep had.
|
|
374
|
+
const lens = left >= W ? [W] : [left];
|
|
375
|
+
for (const len of lens) {
|
|
376
|
+
const key = latin1(q.subarray(forward ? qi + gq : qi - gq - len, forward ? qi + gq + len : qi - gq));
|
|
377
|
+
const list = index[len - 1].get(key);
|
|
378
|
+
if (list === undefined)
|
|
287
379
|
continue;
|
|
288
|
-
|
|
380
|
+
const o = forward ? fromAt(list, si) : toAt(list, si - len);
|
|
381
|
+
if (o < 0)
|
|
289
382
|
continue;
|
|
290
|
-
const n =
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
383
|
+
const n = forward
|
|
384
|
+
? runLenAt(qi + gq, o)
|
|
385
|
+
: runLenBefore(qi - gq, o + len);
|
|
386
|
+
if (n < 1)
|
|
387
|
+
continue;
|
|
388
|
+
if (forward ? n >= W || qi + gq + n === q.length : n >= W || n === qi - gq) {
|
|
389
|
+
const gs = forward ? o - si : si - len - o;
|
|
390
|
+
if (best === null || gq + gs < best.gq + best.gs) {
|
|
391
|
+
best = { gq, gs, n };
|
|
296
392
|
}
|
|
297
|
-
matched.push([qi + gq, qi + gq + n]);
|
|
298
|
-
qi = qi + gq + n;
|
|
299
|
-
si = si + gs + n;
|
|
300
|
-
found = true;
|
|
301
393
|
break;
|
|
302
394
|
}
|
|
303
395
|
}
|
|
304
396
|
}
|
|
305
|
-
|
|
397
|
+
return best;
|
|
398
|
+
};
|
|
399
|
+
// RIGHT sweep.
|
|
400
|
+
let qi = qe, si = se;
|
|
401
|
+
for (;;) {
|
|
402
|
+
const step = nextRun(qi, si, true);
|
|
403
|
+
if (step === null)
|
|
306
404
|
break;
|
|
405
|
+
if (step.gq > 0 || step.gs > 0) {
|
|
406
|
+
gaps.push({ qs: qi, qe: qi + step.gq, cs: si, ce: si + step.gs });
|
|
407
|
+
}
|
|
408
|
+
matched.push([qi + step.gq, qi + step.gq + step.n]);
|
|
409
|
+
qi = qi + step.gq + step.n;
|
|
410
|
+
si = si + step.gs + step.n;
|
|
307
411
|
}
|
|
308
|
-
// LEFT sweep (mirror)
|
|
412
|
+
// LEFT sweep (mirror): an independent walk, so an exhausted right side can
|
|
413
|
+
// never starve it (pinned by test/114).
|
|
309
414
|
qi = qs;
|
|
310
415
|
si = ss;
|
|
311
416
|
for (;;) {
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
for (let gq = 0; gq <= Math.min(total, reachCap); gq++) {
|
|
315
|
-
const gs = total - gq;
|
|
316
|
-
if (gs > reachCap)
|
|
317
|
-
continue;
|
|
318
|
-
if (qi - gq <= 0 || si - gs <= 0)
|
|
319
|
-
continue;
|
|
320
|
-
// Run ENDING at (qi - gq, si - gs).
|
|
321
|
-
let n = 0;
|
|
322
|
-
while (n < qi - gq && n < si - gs &&
|
|
323
|
-
q[qi - gq - 1 - n] === c[si - gs - 1 - n]) {
|
|
324
|
-
n++;
|
|
325
|
-
}
|
|
326
|
-
if (n >= W || n === qi - gq) {
|
|
327
|
-
if (n === 0)
|
|
328
|
-
continue;
|
|
329
|
-
if (gq > 0 || gs > 0) {
|
|
330
|
-
gaps.push({ qs: qi - gq, qe: qi, cs: si - gs, ce: si });
|
|
331
|
-
}
|
|
332
|
-
matched.push([qi - gq - n, qi - gq]);
|
|
333
|
-
qi = qi - gq - n;
|
|
334
|
-
si = si - gs - n;
|
|
335
|
-
found = true;
|
|
336
|
-
break;
|
|
337
|
-
}
|
|
338
|
-
}
|
|
339
|
-
}
|
|
340
|
-
if (!found)
|
|
417
|
+
const step = nextRun(qi, si, false);
|
|
418
|
+
if (step === null)
|
|
341
419
|
break;
|
|
420
|
+
if (step.gq > 0 || step.gs > 0) {
|
|
421
|
+
gaps.push({ qs: qi - step.gq, qe: qi, cs: si - step.gs, ce: si });
|
|
422
|
+
}
|
|
423
|
+
matched.push([qi - step.gq - step.n, qi - step.gq]);
|
|
424
|
+
qi = qi - step.gq - step.n;
|
|
425
|
+
si = si - step.gs - step.n;
|
|
342
426
|
}
|
|
343
427
|
return { matched, gaps };
|
|
344
428
|
}
|
|
@@ -934,24 +1018,25 @@ export async function project(ctx, id, guide) {
|
|
|
934
1018
|
}
|
|
935
1019
|
// ── The span-shape family ───────────────────────────────────────────────────
|
|
936
1020
|
//
|
|
937
|
-
// "Is this answer drawn from this context?" has TWO formally distinct
|
|
938
|
-
//
|
|
939
|
-
//
|
|
940
|
-
//
|
|
941
|
-
//
|
|
942
|
-
//
|
|
943
|
-
//
|
|
944
|
-
//
|
|
945
|
-
//
|
|
946
|
-
//
|
|
947
|
-
//
|
|
1021
|
+
// "Is this answer drawn from this context?" has TWO formally distinct readings,
|
|
1022
|
+
// and the pair plus the anchor classifier built on them are SHARED machinery —
|
|
1023
|
+
// extraction proposes span-shaped exemplars with them, the shared
|
|
1024
|
+
// `Precomputed.spanShapedOf` container computes them, and fusion (reasoning.ts)
|
|
1025
|
+
// gates on the strict one. They lived inside mechanisms/extraction.ts, so
|
|
1026
|
+
// `pipeline-mechanism.ts` and `reasoning.ts` both had to import back OUT of a
|
|
1027
|
+
// specific mechanism — an inversion the mechanism market forbids: the shared
|
|
1028
|
+
// contract may not depend on any one mechanism (mechanism-market.md), and a
|
|
1029
|
+
// shared matcher belongs to this family (match-project.md), never to a
|
|
1030
|
+
// mechanism's private helpers. Deleting extraction must not break the shared
|
|
1031
|
+
// container, so they live here.
|
|
948
1032
|
//
|
|
949
1033
|
// • isSpanShaped — the OPEN reading (sparse in-order embedding).
|
|
950
1034
|
// • containsSpan — the STRICT reading (contiguous run or resolved node).
|
|
951
1035
|
// • skillExemplar — classify one anchor into (context, answer) using them.
|
|
952
1036
|
//
|
|
953
|
-
// The two readings are NOT interchangeable;
|
|
954
|
-
// and each function's own doc states what breaks if it is
|
|
1037
|
+
// The two readings are NOT interchangeable; match-project.md pins the
|
|
1038
|
+
// distinction and each function's own doc states what breaks if it is
|
|
1039
|
+
// substituted.
|
|
955
1040
|
/** Check whether an anchor is a span-shaped skill exemplar: it represents a
|
|
956
1041
|
* fact whose context and answer together form a span-in-context pattern.
|
|
957
1042
|
* If the anchor has a nextOf continuation, that is the answer and the anchor
|
|
@@ -16,7 +16,7 @@ import { analogyStrength, follow, project, reverseContext, sharedFrameStrengthOf
|
|
|
16
16
|
import { joinWithBridge } from "../resonance.js";
|
|
17
17
|
import { restatesQuery } from "../reasoning.js";
|
|
18
18
|
import { CONCEPT, STEP } from "../graph-search.js";
|
|
19
|
-
import {
|
|
19
|
+
import { indexOf } from "../../bytes.js";
|
|
20
20
|
import { consensusFloor, dominates } from "../../geometry.js";
|
|
21
21
|
import { unexplainedLabel, unexplainedSpans, } from "../rationale.js";
|
|
22
22
|
import { rItem, rNode } from "../trace.js";
|
|
@@ -523,7 +523,23 @@ export async function counterfactualTransfer(ctx, query, pre) {
|
|
|
523
523
|
const fwd = await follow(ctx, proj.anchor, qv);
|
|
524
524
|
if (fwd !== null && indexOf(answer, fwd, 0) < 0 &&
|
|
525
525
|
!restatesQuery(query, fwd)) {
|
|
526
|
-
|
|
526
|
+
// THROUGH THE SHARED JOINER, not a bare concatenation.
|
|
527
|
+
//
|
|
528
|
+
// `joinWithBridge` is the composition step every out-of-search assembly
|
|
529
|
+
// shares (multi-topic fusion, CAST's substitution and comparison): it
|
|
530
|
+
// asks the corpus for a learnt connector between the pieces and, on a
|
|
531
|
+
// miss, joins them BARE **and says so** — the `bridgeMiss` step (see
|
|
532
|
+
// resonance.ts). This site bypassed it, and that is the whole of the
|
|
533
|
+
// gluing the study measured: `"Steel is hard"` + `"wet"` came back as
|
|
534
|
+
// `"hardwet"`, `"eva director father"` + `"The father of…"` as
|
|
535
|
+
// `"fatherThe"` — compositions no rationale could show, because the one
|
|
536
|
+
// step that made them left no trace.
|
|
537
|
+
//
|
|
538
|
+
// Routing it through the shared joiner is the instrumentation fix that
|
|
539
|
+
// comes first: a bare join stays possible (the house rule is "joined
|
|
540
|
+
// bare, never silent") but it is now VISIBLE, and an attested connector
|
|
541
|
+
// is used when the corpus has one.
|
|
542
|
+
answer = await joinWithBridge(ctx, answer, fwd);
|
|
527
543
|
}
|
|
528
544
|
ctx.trace?.step("projectCounterfactual", [
|
|
529
545
|
rItem(filler, "filler", subj.point.anchor),
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
// Cover consumes recognition directly (its axioms are the query's own
|
|
5
5
|
// decomposition) plus the computed spans any parse()-bearing mechanism
|
|
6
6
|
// contributed: computed spans MASK colliding recognised sites and enter the
|
|
7
|
-
// search at zero cost ("computation always wins",
|
|
7
|
+
// search at zero cost ("computation always wins", alu.md) — which is also why
|
|
8
8
|
// cover runs FIRST in defaultMechanisms: a computed-backed cover becomes a
|
|
9
9
|
// near-zero-cost incumbent that prunes the other mechanisms through the
|
|
10
10
|
// ordinary admissible-floor check, with no extension special-case anywhere.
|
|
@@ -59,22 +59,23 @@ export async function resolveConnectors(ctx, sites, query) {
|
|
|
59
59
|
return true;
|
|
60
60
|
const continuations = ctx.store.nextFirst(s.payload, hubBound(ctx));
|
|
61
61
|
return !continuations.some((answer) => {
|
|
62
|
-
// PREFIX-CAPPED (
|
|
63
|
-
// occur INSIDE it, so read one byte past the query's length —
|
|
64
|
-
// detect the overflow — and reject without reconstructing the
|
|
65
|
-
// The `+ 1` is what makes the test exact rather than a
|
|
66
|
-
// result of exactly `query.length + 1` bytes is known to
|
|
67
|
-
// and anything shorter is the candidate's COMPLETE
|
|
68
|
-
// substring test below is the same test as before.
|
|
69
|
-
// probe bridge.ts:256 already uses.)
|
|
62
|
+
// PREFIX-CAPPED (bounded-reads.md): a candidate longer than the query
|
|
63
|
+
// cannot occur INSIDE it, so read one byte past the query's length —
|
|
64
|
+
// enough to detect the overflow — and reject without reconstructing the
|
|
65
|
+
// rest. The `+ 1` is what makes the test exact rather than a
|
|
66
|
+
// truncation: a result of exactly `query.length + 1` bytes is known to
|
|
67
|
+
// be too long, and anything shorter is the candidate's COMPLETE
|
|
68
|
+
// content, so the substring test below is the same test as before. (The
|
|
69
|
+
// same overflow probe bridge.ts:256 already uses.)
|
|
70
70
|
//
|
|
71
71
|
// This loop runs up to hubBound(ctx) = √N reads PER SITE, and only on a
|
|
72
72
|
// multi-turn response — `answeredSpans` is empty for a plain respond(),
|
|
73
|
-
// so the probe does not execute there.
|
|
73
|
+
// so the probe does not execute there. The cap cannot reduce the read
|
|
74
74
|
// COUNT — only a semantic change to the "already answered" test could —
|
|
75
75
|
// but it bounds each read by the query instead of by the corpus, which
|
|
76
|
-
// is what
|
|
77
|
-
// reads 4 bytes per candidate instead of the ~231 it
|
|
76
|
+
// is what bounded-reads.md asks for and what rescues a SHORT query: at
|
|
77
|
+
// 3 bytes this reads 4 bytes per candidate instead of the ~231 it
|
|
78
|
+
// averaged before.
|
|
78
79
|
const bytes = read(ctx, answer, query.length + 1);
|
|
79
80
|
return bytes.length <= query.length && indexOf(query, bytes, 0) >= 0;
|
|
80
81
|
});
|
|
@@ -82,6 +83,8 @@ export async function resolveConnectors(ctx, sites, query) {
|
|
|
82
83
|
const bridgePair = async (l, r) => {
|
|
83
84
|
if (l === r || links.has(l + "," + r))
|
|
84
85
|
return;
|
|
86
|
+
if (ctx.meter)
|
|
87
|
+
ctx.meter.coverBridges++;
|
|
85
88
|
const link = await bridge(ctx, read(ctx, l), read(ctx, r));
|
|
86
89
|
if (link !== null)
|
|
87
90
|
links.set(l + "," + r, link);
|
|
@@ -121,6 +124,10 @@ export async function resolveConnectors(ctx, sites, query) {
|
|
|
121
124
|
// plus one W-quantum of glue per joint — pass that allowance so the
|
|
122
125
|
// bridge's phrase-scale cap admits the whole learnt run.
|
|
123
126
|
const allowance = middleBytes + (m + 1) * W;
|
|
127
|
+
if (ctx.meter) {
|
|
128
|
+
ctx.meter.coverBridges++;
|
|
129
|
+
ctx.meter.coverAllowanceBytes += allowance;
|
|
130
|
+
}
|
|
124
131
|
const interior = await bridge(ctx, first.bytes, orderedNodes[m].bytes, allowance);
|
|
125
132
|
if (interior !== null)
|
|
126
133
|
links.set(key, interior);
|
|
@@ -1,15 +1,15 @@
|
|
|
1
1
|
// mechanisms/prefix-completion.ts — Grounding a query that IS the opening of a
|
|
2
2
|
// trained form (Grounding V).
|
|
3
3
|
//
|
|
4
|
-
// A MECHANISM, NOT A TIER.
|
|
5
|
-
//
|
|
6
|
-
// from, where placement rather than the cost ladder decided.
|
|
4
|
+
// A MECHANISM, NOT A TIER. This used to run inside recall's refusal path, in a
|
|
5
|
+
// fixed if-chain that first-match-wins — the shape CAST was refactored away
|
|
6
|
+
// from, where placement rather than the cost ladder decided. Its claim is
|
|
7
7
|
// maximal (every query byte literally matched, from offset zero, against a
|
|
8
8
|
// trained form) at one STEP, so as a market candidate it competes honestly and
|
|
9
|
-
// the decider weighs it like everything else.
|
|
9
|
+
// the decider weighs it like everything else. It is registered LAST: recall's
|
|
10
10
|
// exact self-match makes an IDENTITY claim about the query while this makes a
|
|
11
|
-
// CONTAINMENT one, and on an exact grade tie the identity claim is the
|
|
12
|
-
//
|
|
11
|
+
// CONTAINMENT one, and on an exact grade tie the identity claim is the stronger
|
|
12
|
+
// evidence — the same ordering exact-vs-approximate.md's ladders use.
|
|
13
13
|
//
|
|
14
14
|
// Its SUPPLY moved too, and further: `formsOpenedBy` (traverse.ts) answers a
|
|
15
15
|
// question about the STORE — "which trained forms does this byte run open?" —
|
|
@@ -49,12 +49,12 @@
|
|
|
49
49
|
//
|
|
50
50
|
// So this is a RETRIEVABILITY gap, not a semantic one, and the ANN is the wrong
|
|
51
51
|
// instrument for it: a proper prefix's gist cannot rank its own continuation.
|
|
52
|
-
// The repair is CONTENT-ADDRESSED (
|
|
53
|
-
// the leaf-id WINDOW index the write side already maintains
|
|
54
|
-
// trained forms does this byte run open?" in a bounded √N
|
|
55
|
-
// mechanism's first supply.
|
|
56
|
-
// second, for prefixes long enough that the gist still
|
|
57
|
-
// read, never re-issued.
|
|
52
|
+
// The repair is CONTENT-ADDRESSED (exact-vs-approximate.md) — `formsOpenedBy`
|
|
53
|
+
// (traverse.ts) reads the leaf-id WINDOW index the write side already maintains
|
|
54
|
+
// and answers "which trained forms does this byte run open?" in a bounded √N
|
|
55
|
+
// walk. That is this mechanism's first supply. The response's memoised top-k
|
|
56
|
+
// `resonance()` is the second, for prefixes long enough that the gist still
|
|
57
|
+
// ranks the form; it is read, never re-issued.
|
|
58
58
|
//
|
|
59
59
|
// AN EXHAUSTIVE ANN LIST IS NOT A SUPPLY HERE, AND WAS REMOVED. This tier once
|
|
60
60
|
// read `Precomputed.wideResonance()` — a full-index `resonate(guide, √N,
|
|
@@ -227,20 +227,20 @@ export const prefixMechanism = {
|
|
|
227
227
|
return STEP;
|
|
228
228
|
},
|
|
229
229
|
async run(ctx, query, pre) {
|
|
230
|
-
// ONE SUPPLY PASS, not a two-tier `??`.
|
|
230
|
+
// ONE SUPPLY PASS, not a two-tier `??`. The window index (exact,
|
|
231
231
|
// content-addressed) and the response's memoised top-k (approximate) are
|
|
232
|
-
// concatenated and the three guards decide ONCE over the union.
|
|
232
|
+
// concatenated and the three guards decide ONCE over the union. A
|
|
233
233
|
// first-then-fallback chain would let the APPROXIMATE tier override the
|
|
234
|
-
// EXACT one (
|
|
235
|
-
// returns null and the fallback re-runs the guards
|
|
236
|
-
// alone — which, seeing only one of the two forms,
|
|
237
|
-
// precisely the disagreement-suppression guard 3
|
|
238
|
-
// is the exact tier's ambiguity being washed away
|
|
239
|
-
// Evaluating the union means a disagreement the
|
|
240
|
-
// be hidden by what the ANN happens to rank.
|
|
241
|
-
// response's ONE memoised top-k (
|
|
242
|
-
// path on the queries where this mechanism fires,
|
|
243
|
-
// a second index scan.
|
|
234
|
+
// EXACT one (exact-vs-approximate.md): when formsOpenedBy finds two
|
|
235
|
+
// continuations, guard 3 returns null and the fallback re-runs the guards
|
|
236
|
+
// on resonance's top-k alone — which, seeing only one of the two forms,
|
|
237
|
+
// would voice it. That is precisely the disagreement-suppression guard 3
|
|
238
|
+
// exists to prevent, and it is the exact tier's ambiguity being washed away
|
|
239
|
+
// by the approximate tier. Evaluating the union means a disagreement the
|
|
240
|
+
// window index saw can never be hidden by what the ANN happens to rank. The
|
|
241
|
+
// ANN read is the response's ONE memoised top-k (memoization.md), already
|
|
242
|
+
// paid by recall's refusal path on the queries where this mechanism fires,
|
|
243
|
+
// so reading it here is not a second index scan.
|
|
244
244
|
const ids = [
|
|
245
245
|
...formsOpenedBy(ctx, query),
|
|
246
246
|
...(await pre.resonance()).map((h) => h.id),
|