@hviana/sema 0.8.9 → 0.9.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (111) hide show
  1. package/AGENTS.md +7 -7
  2. package/dist/src/alu/src/index.d.ts +1 -1
  3. package/dist/src/alu/src/index.js +1 -1
  4. package/dist/src/alu/src/parser.js +2 -6
  5. package/dist/src/alu/src/resonance.d.ts +13 -0
  6. package/dist/src/alu/src/resonance.js +41 -0
  7. package/dist/src/alu/test/alu.test.js +39 -0
  8. package/dist/src/bytes.d.ts +6 -2
  9. package/dist/src/bytes.js +10 -4
  10. package/dist/src/canon.js +44 -0
  11. package/dist/src/geometry.d.ts +19 -1
  12. package/dist/src/geometry.js +125 -141
  13. package/dist/src/meter.d.ts +27 -0
  14. package/dist/src/meter.js +28 -1
  15. package/dist/src/mind/articulation.js +14 -1
  16. package/dist/src/mind/attention.d.ts +12 -0
  17. package/dist/src/mind/attention.js +44 -16
  18. package/dist/src/mind/bridge.js +3 -3
  19. package/dist/src/mind/derivation.d.ts +40 -0
  20. package/dist/src/mind/derivation.js +34 -0
  21. package/dist/src/mind/graph-search.d.ts +89 -15
  22. package/dist/src/mind/graph-search.js +345 -174
  23. package/dist/src/mind/learning.js +1 -1
  24. package/dist/src/mind/mechanisms/cover.d.ts +19 -3
  25. package/dist/src/mind/mechanisms/cover.js +101 -58
  26. package/dist/src/mind/mechanisms/recall.js +0 -1
  27. package/dist/src/mind/mind.js +2 -2
  28. package/dist/src/mind/pipeline.d.ts +5 -1
  29. package/dist/src/mind/pipeline.js +175 -87
  30. package/dist/src/mind/primitives.d.ts +25 -5
  31. package/dist/src/mind/primitives.js +107 -44
  32. package/dist/src/mind/reasoning.d.ts +18 -4
  33. package/dist/src/mind/reasoning.js +445 -321
  34. package/dist/src/mind/recognition.js +55 -73
  35. package/dist/src/mind/resonance.js +1 -11
  36. package/dist/src/mind/traverse.d.ts +3 -3
  37. package/dist/src/mind/traverse.js +3 -3
  38. package/dist/src/mind/types.d.ts +7 -1
  39. package/dist/src/store-sqlite.d.ts +25 -0
  40. package/dist/src/store-sqlite.js +89 -1
  41. package/dist/src/store.d.ts +48 -4
  42. package/dist/src/store.js +86 -6
  43. package/docs/INDEX.md +18 -18
  44. package/docs/INVARIANTS.md +16 -16
  45. package/docs/architecture/bounded-reads.md +1 -1
  46. package/docs/architecture/caches.md +5 -4
  47. package/docs/architecture/closure.md +45 -5
  48. package/docs/architecture/cost-model.md +16 -0
  49. package/docs/architecture/factored-machinery.md +14 -13
  50. package/docs/architecture/fold-contract.md +51 -1
  51. package/docs/architecture/mechanism-market.md +21 -0
  52. package/docs/architecture/memoization.md +3 -3
  53. package/docs/architecture/meter.md +2 -1
  54. package/docs/architecture/saturation.md +12 -0
  55. package/docs/architecture/store.md +25 -2
  56. package/docs/failures/tempting-but-wrong.md +13 -2
  57. package/docs/harness/gates.md +12 -10
  58. package/docs/mechanisms/cover.md +23 -6
  59. package/jsr.json +1 -1
  60. package/package.json +1 -1
  61. package/src/alu/README.md +10 -2
  62. package/src/alu/src/index.ts +1 -0
  63. package/src/alu/src/parser.ts +6 -6
  64. package/src/alu/src/resonance.ts +42 -0
  65. package/src/alu/test/alu.test.ts +40 -0
  66. package/src/bytes.ts +13 -3
  67. package/src/canon.ts +40 -0
  68. package/src/geometry.ts +183 -154
  69. package/src/meter.ts +28 -1
  70. package/src/mind/articulation.ts +14 -2
  71. package/src/mind/attention.ts +47 -25
  72. package/src/mind/bridge.ts +3 -3
  73. package/src/mind/derivation.ts +77 -0
  74. package/src/mind/graph-search.ts +449 -221
  75. package/src/mind/learning.ts +1 -7
  76. package/src/mind/match.ts +1 -2
  77. package/src/mind/mechanisms/cast.ts +1 -2
  78. package/src/mind/mechanisms/cover.ts +149 -84
  79. package/src/mind/mechanisms/extraction.ts +1 -2
  80. package/src/mind/mechanisms/prefix-completion.ts +1 -1
  81. package/src/mind/mechanisms/recall.ts +1 -3
  82. package/src/mind/mechanisms/reference.ts +1 -1
  83. package/src/mind/mind.ts +5 -30
  84. package/src/mind/pipeline.ts +206 -102
  85. package/src/mind/primitives.ts +119 -43
  86. package/src/mind/reasoning.ts +558 -413
  87. package/src/mind/recognition.ts +49 -65
  88. package/src/mind/resonance.ts +2 -16
  89. package/src/mind/trace.ts +1 -1
  90. package/src/mind/traverse.ts +3 -3
  91. package/src/mind/types.ts +9 -11
  92. package/src/store-sqlite.ts +92 -1
  93. package/src/store.ts +113 -7
  94. package/test/105-derive-through-reports-its-refusal.test.mjs +8 -5
  95. package/test/106-the-join-fires.test.mjs +21 -0
  96. package/test/111-the-cover-assembly-is-counted.test.mjs +8 -5
  97. package/test/128-the-leads-somewhere-pair-agrees.test.mjs +18 -12
  98. package/test/136-the-two-named-limits.test.mjs +3 -2
  99. package/test/137-the-law-lives-once-and-below.test.mjs +21 -0
  100. package/test/148-exact-shortcuts-agree.test.mjs +188 -0
  101. package/test/149-the-closure-engine.test.mjs +138 -0
  102. package/test/150-the-join-is-output-sensitive.test.mjs +66 -0
  103. package/test/151-the-cover-pays-for-what-it-reaches.test.mjs +142 -0
  104. package/test/152-the-read-side-names-as-the-write-side.test.mjs +146 -0
  105. package/test/153-a-cheaper-bound-is-looked-at-first.test.mjs +155 -0
  106. package/test/24-generalization.test.mjs +32 -0
  107. package/test/36-bloom.test.mjs +53 -0
  108. package/test/37-cluster-dispersion-fusion.test.mjs +75 -0
  109. package/test/48-recognise-turn-connective.test.mjs +3 -2
  110. package/test/55-cost-meter.test.mjs +4 -4
  111. package/test/90-connector-read-cap.test.mjs +7 -7
@@ -5,10 +5,11 @@
5
5
  // that leads somewhere (has a continuation edge or a halo).
6
6
  // segment — leaf-parent segmentation using the geometry's own groupings.
7
7
  import { rItem } from "./trace.js";
8
- import { canonResolve, foldTree, gistOf, latin1Key, perceive, resolve, } from "./primitives.js";
8
+ import { canonResolve, foldTree, gistOf, perceive, resolve, } from "./primitives.js";
9
9
  import { atomIsHub, bearsEdge, corpusN, leadsSomewhere } from "./traverse.js";
10
10
  import { chainReach, leafIdAt, leafIdRun } from "./canonical.js";
11
11
  import { canonHash } from "../canon.js";
12
+ import { latin1 } from "../bytes.js";
12
13
  import { isChunk } from "../sema.js";
13
14
  /** Decompose a byte stream into every stored form that leads somewhere
14
15
  * (has a continuation edge or a halo). Two complementary readings:
@@ -77,7 +78,7 @@ export function recognise(ctx, bytes) {
77
78
  // not silent), so it is emitted here directly rather than only inside
78
79
  // recogniseImpl.
79
80
  if (ctx.recogniseMemo) {
80
- const key = latin1Key(bytes);
81
+ const key = latin1(bytes);
81
82
  const hit = ctx.recogniseMemo.get(key);
82
83
  if (hit !== undefined) {
83
84
  if (ctx.meter)
@@ -483,9 +484,21 @@ function recogniseImpl(ctx, bytes) {
483
484
  // encoding is the identity, so the span's bytes ARE the branch key.
484
485
  // `subarray` is a view — this allocates nothing per probe, and the
485
486
  // bloom filter answers the misses without touching the database.
486
- const flatProbe = (start, end) => store.findFlatBranch
487
- ? store.findFlatBranch(bytes.subarray(start, end))
488
- : store.findBranch(allLeafIds.slice(start, end));
487
+ //
488
+ // Through ONE span prober (Store.flatSpans), because hashing the key was
489
+ // itself the quadratic: every probe hashed its span from the start, and
490
+ // the interior pass below sweeps up to `reach` ends past each endpoint —
491
+ // O(n · reach²) bytes hashed. Measured on the 31.7M-node store, one
492
+ // composition-regime response (#97 of the battery) probed 2,079,400
493
+ // spans and hashed 251,660,406 bytes for them. The prober extends each
494
+ // start's hash instead, so the same probes, answered identically, cost
495
+ // O(n · reach).
496
+ const spans = store.flatSpans?.(bytes) ?? null;
497
+ const flatProbe = (start, end) => spans !== null
498
+ ? spans(start, end)
499
+ : store.findFlatBranch
500
+ ? store.findFlatBranch(bytes.subarray(start, end))
501
+ : store.findBranch(allLeafIds.slice(start, end));
489
502
  // THE TWO ROUTES COST DIFFERENT THINGS, SO THEY ARE PRICED SEPARATELY.
490
503
  //
491
504
  // The exact route is a bloom-gated hash over a subarray VIEW: no
@@ -518,7 +531,17 @@ function recogniseImpl(ctx, bytes) {
518
531
  // emitted, so a caller can retry a trimmed edge on the miss path only.
519
532
  if (end - start < W)
520
533
  return false;
521
- if (flatProbe(start, end) === null) {
534
+ // The byte-exact route is a BLOOM, so a non-null answer means MAYBE: it can neither
535
+ // decide (that is `resolveSpan`'s job) nor deny (that is `canonAdmits`'). Reading it
536
+ // as a YES made a false positive both drop the span and skip the decider, because the
537
+ // canon route ran only on a null. Measured on the trained corpus: 280 spans where the
538
+ // bloom claimed and the identity refused. A bloom HIT that resolves still emits
539
+ // without touching the canon route, which is what keeps the cheap path cheap.
540
+ const flat = flatProbe(start, end);
541
+ let id = flat === null ? null : resolveSpan(start, end);
542
+ if (id === null) {
543
+ if (flat !== null && ctx.meter)
544
+ ctx.meter.bloomFalsePositives++;
522
545
  if (!canonBudget) {
523
546
  if (ctx.meter)
524
547
  ctx.meter.canonProbesDenied++;
@@ -526,8 +549,8 @@ function recogniseImpl(ctx, bytes) {
526
549
  }
527
550
  if (!canonAdmits(start, end))
528
551
  return false;
552
+ id = resolveSpan(start, end);
529
553
  }
530
- const id = resolveSpan(start, end);
531
554
  if (id === null)
532
555
  return false;
533
556
  emit(start, end, id);
@@ -600,77 +623,37 @@ function recogniseImpl(ctx, bytes) {
600
623
  // keeps this off the quadratic path the budget note above describes (that
601
624
  // one had no span bound at all).
602
625
  {
603
- // The span bound is W^2, the chain's own limit, PLUS the slack the endpoint set already grants: every endpoint
604
- // sits within `radius` of a cut, so a pair that names one form may straddle cuts and still be a single form's
605
- // span. Measured on the composite fixture: W=4 (reach 16), radius 8, and the useful [7,32) is 25 bytes with its
606
- // edges 2 bytes from cuts 5 and 30 — already IN `ordered`, and excluded only by the upper bound. Both terms are
626
+ // The span bound is the chain's own limit PLUS the slack the endpoint set already grants: every endpoint sits
627
+ // within `radius` of a cut, so a pair that names one form may straddle cuts and still be a single form's span.
628
+ // Measured on the composite fixture: W=4 (reach 16), radius 8, and the useful [7,32) is 25 bytes with its edges
629
+ // 2 bytes from cuts 5 and 30 — already IN `ordered`, and excluded only by the upper bound. Both terms are
607
630
  // derived (W and the seat count); no new constant enters.
608
- const reach = chainReach(W) + 2 * radius;
631
+ //
632
+ // The chain term is W^2 * W = W^4, for a form embedded in the MIDDLE of a longer query. Measured on the
633
+ // trained corpus, the SAME real contexts in three positions: opening 12/17, MIDDLE 0/17, end 12/17. The two
634
+ // edge scans reach only prefixes and suffixes, so the interior pass is the only tier that could name it, and
635
+ // at W^2 it could not. WHAT KEEPS THIS LINEAR IS THE BOUND BEING A CONSTANT — each endpoint pairs only with
636
+ // the partners inside a fixed window, so the work stays O(n). Measured against a cut-pair enumeration with
637
+ // ±W trims (an earlier, 43-line attempt): that one reached the same 136 B at 70 888 probes, this one at
638
+ // 84 169 (x1.19 more) — and BOTH pass test/14, which gates the CLASS (linear) and not the constant. The
639
+ // simpler form won on that measurement.
640
+ const reach = chainReach(W) * W * W + 2 * radius;
609
641
  if (ctx.meter)
610
642
  ctx.meter.recogniseInteriorGaps += ordered.length;
643
+ // Each end pairs only with the starts in [end − reach, end − W], in
644
+ // ascending order — the pairs, and the order the budget is spent in,
645
+ // of the all-pairs scan, without its O(n²) enumeration.
646
+ let lo = 0;
611
647
  for (const end of ordered) {
612
- for (const start of ordered) {
613
- if (start >= end)
614
- continue;
615
- const span = end - start;
616
- if (span < W || span > reach)
617
- continue;
618
- if (ctx.meter)
619
- ctx.meter.recogniseInteriorPairs++;
620
- spend(start, end);
621
- }
622
- }
623
- // ── THE MIDDLE WAS BLIND: a CUT-PAIR probe, BOUNDED so it stays LINEAR ──────────────
624
- // A stored form embedded in the MIDDLE of a longer query was named by NO tier: the two
625
- // edge scans probe only prefixes and suffixes, and the loop above is capped at
626
- // `reach` = W^2 + 2*radius. Measured on the trained corpus, the SAME real contexts in
627
- // three positions: opening 12/17, MIDDLE 0/17, end 12/17. Both edges of such a form sit
628
- // within `radius` of CUTS, so a cut pair plus a small trim names the form exactly
629
- // (measured: a 64 B form at [27, 91) between cuts 26 and 90, trims +1/+1).
630
- //
631
- // THE BOUND IS WHAT KEEPS IT LINEAR. An unbounded cut-pair scan is O(cuts^2) probes, and
632
- // test/14 rejected exactly that (47 324 ms). Bounded by a CONSTANT span it is
633
- // O(cuts * const) = O(n), like the loop above — the constant is merely larger. The bound
634
- // is DERIVED, never pinned: `chainReach(W)` is W^2, "the deepest two-level composite the
635
- // write side's windows can spell" (canonical.ts), so `chainReach(W) * W * W` is the
636
- // four-level one; `chainReach(W) * W` is already used in bridge.ts.
637
- //
638
- // THE TRIMS ARE THE SAME DISCIPLINE THE EDGE SCANS USE — the suffix scan already probes
639
- // `spend(s, bytes.length - 1)`, so ±W is a wider version of an existing rule, not a new
640
- // threshold. Measured price/reach; all four configurations pass test/14, which gates the
641
- // CLASS (linear) and not the constant:
642
- // W^3 + ±1 -> 64 B, 8 068 probes (x2.02)
643
- // W^4 + ±W -> 136 B, 70 888 probes (x6.06) <- this one
644
- // W^3 + ±W and W^4 + ±1 are DOMINATED: each stops at the other parameter and costs more.
645
- // The loop above stays, so no candidate that produces a site today is lost. Suite
646
- // 707/707, and a differential over 24 real corpus questions is byte-identical (0 answers
647
- // changed, 0 new duplicate sites): the :258 warning that a wider bound can rediscover a
648
- // smaller subtree's content as a second, overlapping site was read and measured, and it
649
- // does not materialise here.
650
- const deepReach = chainReach(W) * W * W;
651
- for (let ci = 0; ci + 1 < startList.length; ci++) {
652
- for (let cj = ci + 1; cj < startList.length; cj++) {
653
- if (startList[cj] - startList[ci] > deepReach + 2 * radius)
648
+ while (ordered[lo] < end - reach)
649
+ lo++;
650
+ for (let i = lo; i < ordered.length; i++) {
651
+ const start = ordered[i];
652
+ if (end - start < W)
654
653
  break;
655
- // The EXACT cut pair first. Measured: with the 81 trims starting at -W the `spend`
656
- // pool ran dry (canonProbesDenied in the millions) and the canon route was then denied
657
- // to the candidates that needed it — including forms the byte-exact route SEES
658
- // (`flatProbe` true) that were still not named. Probing (ci, cj) before any trim puts
659
- // the common case in front of the famine.
660
654
  if (ctx.meter)
661
655
  ctx.meter.recogniseInteriorPairs++;
662
- spend(startList[ci], startList[cj]);
663
- for (let dl = -W; dl <= W; dl++) {
664
- for (let dr = -W; dr <= W; dr++) {
665
- const a = startList[ci] + dl;
666
- const z = startList[cj] + dr;
667
- if (a < 0 || z > bytes.length || z - a < W)
668
- continue;
669
- if (ctx.meter)
670
- ctx.meter.recogniseInteriorPairs++;
671
- spend(a, z);
672
- }
673
- }
656
+ spend(start, end);
674
657
  }
675
658
  }
676
659
  }
@@ -706,8 +689,7 @@ function recogniseImpl(ctx, bytes) {
706
689
  // arithmetic, not evidence. Removing it wholesale was measured and
707
690
  // REVERTED: it also drops legitimate multi-byte chains (the 12-byte
708
691
  // "Eiffel Tower" site vanished with it). The premise is wrong but the
709
- // trust it stood in for is real; a replacement signal is still open work.
710
- // See bench/README.md.
692
+ // trust it stood in for is real; the replacement signal follows.
711
693
  //
712
694
  // THE REPLACEMENT SIGNAL (2026-08-13): `leadsSomewhere` on the BYTE-EXACT
713
695
  // branch the chain already found. The blanket off-boundary suppression is
@@ -6,7 +6,7 @@
6
6
  import { rItem } from "./trace.js";
7
7
  import { cosine } from "../vec.js";
8
8
  import { mergeThreshold } from "../geometry.js";
9
- import { concat2, concatBytes, indexOf } from "../bytes.js";
9
+ import { concat2, concatBytes, indexOf, latin1 } from "../bytes.js";
10
10
  import { gistOf, read, resolve, walkTree } from "./primitives.js";
11
11
  import { perceive } from "./primitives.js";
12
12
  import { argmaxCosine, candidateGist, hubBound } from "./traverse.js";
@@ -107,16 +107,6 @@ function junctionEdges(ctx, left, right, maxContainer) {
107
107
  }
108
108
  return out;
109
109
  }
110
- /** A byte string as a string, ONE code unit per byte — injective, so it is
111
- * safe to build a cache key from. Chunked to keep the spread within the
112
- * engine's argument limit on long contexts. */
113
- function latin1(b) {
114
- let s = "";
115
- for (let i = 0; i < b.length; i += 4096) {
116
- s += String.fromCharCode(...b.subarray(i, i + 4096));
117
- }
118
- return s;
119
- }
120
110
  /** Per-response memo of bridge results, keyed by the response's lifecycle
121
111
  * object (ctx.climbMemo — created fresh by respond() and nulled after, so
122
112
  * entries can never outlive the read-only window they are valid in). The
@@ -58,9 +58,9 @@ export declare function atomIsHub(ctx: MindContext, contextCount: number): boole
58
58
  * it is sound as a pre-filter before a consumer that applies the full
59
59
  * predicate, and never as a replacement for it. */
60
60
  export declare function bearsEdge(ctx: MindContext, id: number): boolean;
61
- /** Whether a node LEADS SOMEWHERE — it bears a continuation edge or a halo.
62
- * The admission predicate recognition filters sites with (cover.md): a form
63
- * that
61
+ /** Whether a node LEADS SOMEWHERE — the store's admission predicate
62
+ * ({@link Store.leadsSomewhere}: edge or halo) with its edge tier memoised for
63
+ * the response. Recognition filters sites with it (cover.md): a form that
64
64
  * leads nowhere contributes nothing to any derivation. Runs once per candidate
65
65
  * span on the recognition hot path — `hasNext` is cached per response (the same
66
66
  * flat-branch ids are probed across prefix variants by canonicalChunkId).
@@ -432,9 +432,9 @@ export function atomIsHub(ctx, contextCount) {
432
432
  export function bearsEdge(ctx, id) {
433
433
  return cachedHasNext(ctx, id, getStructCache(ctx));
434
434
  }
435
- /** Whether a node LEADS SOMEWHERE — it bears a continuation edge or a halo.
436
- * The admission predicate recognition filters sites with (cover.md): a form
437
- * that
435
+ /** Whether a node LEADS SOMEWHERE — the store's admission predicate
436
+ * ({@link Store.leadsSomewhere}: edge or halo) with its edge tier memoised for
437
+ * the response. Recognition filters sites with it (cover.md): a form that
438
438
  * leads nowhere contributes nothing to any derivation. Runs once per candidate
439
439
  * span on the recognition hot path — `hasNext` is cached per response (the same
440
440
  * flat-branch ids are probed across prefix variants by canonicalChunkId).
@@ -157,7 +157,13 @@ export interface Attention {
157
157
  * a genuine further topic is named in its own distinctive wording
158
158
  * somewhere the query's scaffolding does not reach, always a SEPARATE
159
159
  * cluster from whatever else corroborates it. See
160
- * test/37-cluster-dispersion-fusion.test.mjs. */
160
+ * test/37-cluster-dispersion-fusion.test.mjs.
161
+ *
162
+ * Read from the VOTES, this is a lossy witness: a region votes once, for its
163
+ * top anchor, so a place can be lost to a tie or won through an accident.
164
+ * Fusion therefore also asks the root's CONTEXT the same question at window
165
+ * scale (reasoning.ts `sharedPlaces`) and trusts a root that either reading
166
+ * finds in two places. */
161
167
  clusters: number;
162
168
  }
163
169
  /** Both read-outs of one consensus climb. */
@@ -28,6 +28,20 @@ export declare class SQliteStore extends AbstractStore implements Store {
28
28
  private _bloom;
29
29
  /** Dedup probes answered by the filter alone this session (observability). */
30
30
  bloomSkips: number;
31
+ /** The same negative filter over the CANON index's key hashes. Recognition's
32
+ * canonical admission and the join's canonical entity scan ask `canonFind`
33
+ * once per probed span, and almost every answer is "no such key" — on an
34
+ * index that is often EMPTY (it is built only by `buildCanonIndex`).
35
+ * Loaded on first use from `canon_bloom` when its stamp matches the meta,
36
+ * else built by one sequential scan of the h column; kept exact on
37
+ * `canonAdd` (rebuilt bigger from the table, uncommitted rows included,
38
+ * when growth saturates it) and persisted with the commit that wrote the
39
+ * rows. The canon table is never deleted from, so the filter can never
40
+ * hold a false negative: a miss it reports is a miss the query would have
41
+ * returned. */
42
+ private _canonBloom;
43
+ /** The in-memory canon filter differs from the persisted one. */
44
+ private _canonBloomDirty;
31
45
  private _insertNode;
32
46
  private _insertKid;
33
47
  private _selContain;
@@ -79,6 +93,9 @@ export declare class SQliteStore extends AbstractStore implements Store {
79
93
  protected _dbGetNode(id: NodeId): NodeRec | null;
80
94
  protected _dbFindLeaf(h: number, bytes: Uint8Array): NodeId | null;
81
95
  protected _dbFindBranchByLeaf(h: number, bytes: Uint8Array): NodeId | null;
96
+ /** The node filter alone: never a false negative (every inserted hash is
97
+ * added before any probe can see it), so `false` is exact. */
98
+ protected _dbFlatMayExist(h: number, bytes: Uint8Array): boolean;
82
99
  protected _dbFindBranchByKids(h: number, packed: Uint8Array): NodeId | null;
83
100
  protected _dbInsertKid(child: NodeId, parent: NodeId): void;
84
101
  protected _dbGetParents(id: NodeId): NodeId[];
@@ -137,6 +154,14 @@ export declare class SQliteStore extends AbstractStore implements Store {
137
154
  protected _dbSetMeta(key: string, val: string): void;
138
155
  protected _dbDeleteMeta(key: string): void;
139
156
  canonAdd(h: number, id: number): void;
157
+ /** The canon filter: the persisted one when its stamp matches the meta,
158
+ * else built from the index as it stands. */
159
+ private _canonFilter;
160
+ private _canonScan;
161
+ /** What a persisted canon filter is stamped with: the incremental build's
162
+ * cursor, which every canon writer advances in the transaction it writes. */
163
+ private _canonStamp;
164
+ private _canonBloomPersist;
140
165
  canonFind(h: number): number[];
141
166
  sketchGet(id: number): number[] | null;
142
167
  sketchPut(id: number, ids: readonly number[]): void;
@@ -124,6 +124,17 @@ CREATE TABLE IF NOT EXISTS canon (
124
124
  id INTEGER NOT NULL,
125
125
  PRIMARY KEY (h, id)
126
126
  ) WITHOUT ROWID;
127
+ -- The canon index's negative filter, persisted so an open does not rescan the
128
+ -- h column (seconds on a trained store). Written in the SAME transaction as
129
+ -- the canon rows it covers, stamped with the meta 'canon.upto' of that commit;
130
+ -- a stamp that disagrees with the meta (rows a writer added without it) makes
131
+ -- the filter stale, and it is rebuilt from the table instead.
132
+ CREATE TABLE IF NOT EXISTS canon_bloom (
133
+ id INTEGER PRIMARY KEY CHECK (id = 1),
134
+ bits BLOB NOT NULL,
135
+ n INTEGER NOT NULL,
136
+ upto TEXT NOT NULL
137
+ );
127
138
  -- CONSTITUENT SKETCH (Store.sketchGet/sketchPut): the bottom-k minimal
128
139
  -- constituents of a node's subtree, k = √D, chosen by identity hash. The blob is
129
140
  -- a packed int32 little-endian run, already in hash order; an EMPTY blob is a
@@ -223,6 +234,20 @@ export class SQliteStore extends AbstractStore {
223
234
  _bloom = null;
224
235
  /** Dedup probes answered by the filter alone this session (observability). */
225
236
  bloomSkips = 0;
237
+ /** The same negative filter over the CANON index's key hashes. Recognition's
238
+ * canonical admission and the join's canonical entity scan ask `canonFind`
239
+ * once per probed span, and almost every answer is "no such key" — on an
240
+ * index that is often EMPTY (it is built only by `buildCanonIndex`).
241
+ * Loaded on first use from `canon_bloom` when its stamp matches the meta,
242
+ * else built by one sequential scan of the h column; kept exact on
243
+ * `canonAdd` (rebuilt bigger from the table, uncommitted rows included,
244
+ * when growth saturates it) and persisted with the commit that wrote the
245
+ * rows. The canon table is never deleted from, so the filter can never
246
+ * hold a false negative: a miss it reports is a miss the query would have
247
+ * returned. */
248
+ _canonBloom = null;
249
+ /** The in-memory canon filter differs from the persisted one. */
250
+ _canonBloomDirty = false;
226
251
  _insertNode = null;
227
252
  _insertKid = null;
228
253
  _selContain = null;
@@ -453,8 +478,10 @@ export class SQliteStore extends AbstractStore {
453
478
  "n = excluded.n, upto = excluded.upto").run(this._bloom.bits, this._bloom.n, this._nextId);
454
479
  }
455
480
  _dbClose() {
456
- if (this.sqlite)
481
+ if (this.sqlite) {
457
482
  this._bloomPersist();
483
+ this._canonBloomPersist();
484
+ }
458
485
  if (this.content) {
459
486
  this.content.close();
460
487
  this.content = null;
@@ -478,6 +505,8 @@ export class SQliteStore extends AbstractStore {
478
505
  _dbCommitTx() {
479
506
  if (!this._inTx || !this.sqlite)
480
507
  return;
508
+ // The canon filter commits WITH the rows it covers — see canon_bloom.
509
+ this._canonBloomPersist();
481
510
  this._inTx = false; // clear first so a throw can't wedge us mid-commit
482
511
  this.sqlite.exec("COMMIT");
483
512
  }
@@ -534,6 +563,16 @@ export class SQliteStore extends AbstractStore {
534
563
  const row = this._selFlat.get(h, bytes);
535
564
  return row ? row.id : null;
536
565
  }
566
+ /** The node filter alone: never a false negative (every inserted hash is
567
+ * added before any probe can see it), so `false` is exact. */
568
+ _dbFlatMayExist(h, bytes) {
569
+ if (this._bloom === null)
570
+ return super._dbFlatMayExist(h, bytes);
571
+ if (this._bloom.mightContain(h))
572
+ return true;
573
+ this.bloomSkips++;
574
+ return false;
575
+ }
537
576
  _dbFindBranchByKids(h, packed) {
538
577
  if (this._bloom && !this._bloom.mightContain(h)) {
539
578
  this.bloomSkips++;
@@ -855,8 +894,57 @@ export class SQliteStore extends AbstractStore {
855
894
  // bulk index build coalesces instead of paying autocommit per row.
856
895
  this._dbBeginTx();
857
896
  this._insCanon.run(h, id);
897
+ const b = this._canonFilter();
898
+ b.add(h);
899
+ // Rebuilt bigger at once, from the table as THIS connection sees it — the
900
+ // persisted filter cannot stand in, it lacks this transaction's rows.
901
+ if (b.saturated)
902
+ this._canonBloom = this._canonScan();
903
+ this._canonBloomDirty = true;
904
+ }
905
+ /** The canon filter: the persisted one when its stamp matches the meta,
906
+ * else built from the index as it stands. */
907
+ _canonFilter() {
908
+ if (this._canonBloom === null) {
909
+ const row = this.sqlite.prepare("SELECT bits, n, upto FROM canon_bloom WHERE id = 1").get();
910
+ if (row !== undefined && row.upto === this._canonStamp()) {
911
+ const b = new NodeBloom(31 - Math.clz32(row.bits.length * 8));
912
+ b.bits.set(row.bits);
913
+ b.n = row.n;
914
+ this._canonBloom = b;
915
+ }
916
+ else {
917
+ this._canonBloom = this._canonScan();
918
+ this._canonBloomDirty = true;
919
+ }
920
+ }
921
+ return this._canonBloom;
922
+ }
923
+ _canonScan() {
924
+ const b = new NodeBloom(bloomLog2For(this.canonCount()));
925
+ const scan = this.sqlite.prepare("SELECT h FROM canon");
926
+ scan.setReturnArrays(true);
927
+ for (const r of scan.iterate())
928
+ b.add(r[0]);
929
+ return b;
930
+ }
931
+ /** What a persisted canon filter is stamped with: the incremental build's
932
+ * cursor, which every canon writer advances in the transaction it writes. */
933
+ _canonStamp() {
934
+ return this._dbGetMeta("canon.upto") ?? "";
935
+ }
936
+ _canonBloomPersist() {
937
+ const b = this._canonBloom;
938
+ if (b === null || !this._canonBloomDirty || !this.sqlite)
939
+ return;
940
+ this.sqlite.prepare("INSERT INTO canon_bloom (id, bits, n, upto) VALUES (1, ?, ?, ?) " +
941
+ "ON CONFLICT(id) DO UPDATE SET bits = excluded.bits, " +
942
+ "n = excluded.n, upto = excluded.upto").run(b.bits, b.n, this._canonStamp());
943
+ this._canonBloomDirty = false;
858
944
  }
859
945
  canonFind(h) {
946
+ if (!this._canonFilter().mightContain(h))
947
+ return [];
860
948
  if (!this._selCanon) {
861
949
  this._selCanon = this.sqlite.prepare("SELECT id FROM canon WHERE h = ?");
862
950
  }
@@ -1,5 +1,5 @@
1
1
  import { Vec } from "./vec.js";
2
- import { type StoreConfig } from "./config.js";
2
+ import type { StoreConfig } from "./config.js";
3
3
  import type { Meter } from "./meter.js";
4
4
  /** A node id: a dense, non-negative integer assigned in creation order. */
5
5
  export type NodeId = number;
@@ -166,6 +166,20 @@ export interface Store {
166
166
  * bytes — the allocation-free probe span scanners use. Optional: a store
167
167
  * without it is simply probed through `findBranch`. */
168
168
  findFlatBranch?(bytes: Uint8Array): NodeId | null;
169
+ /** Whether a flat branch with these bytes MAY exist. `false` is EXACT — no
170
+ * such node exists, and a {@link findFlatBranch} would return null; `true`
171
+ * means only that a lookup is needed. The existence half of the probe,
172
+ * without its verification: a caller that will resolve the bytes anyway
173
+ * (and so verify them) asks this to refuse a miss without paying for the
174
+ * lookup a hit would repeat. Optional, like the probe it answers for. */
175
+ flatBranchMayExist?(bytes: Uint8Array): boolean;
176
+ /** A {@link findFlatBranch} over spans of ONE buffer: `probe(start, end)`
177
+ * answers exactly `findFlatBranch(bytes.subarray(start, end))`, but a span's
178
+ * content hash extends the one its start was last probed at, so a scanner
179
+ * that sweeps ends upward per start pays O(1) per probe instead of
180
+ * O(span). The buffer must not change while the prober is used.
181
+ * Optional, like the probe it accelerates. */
182
+ flatSpans?(bytes: Uint8Array): (start: number, end: number) => NodeId | null;
169
183
  /** The branch nodes that list `id` among their children — the reverse of
170
184
  * `get(id).kids`. Lets the structural DAG be climbed upward, from a
171
185
  * recognised fragment to the larger learned forms that contain it. */
@@ -314,6 +328,12 @@ export interface Store {
314
328
  * the search's fuse guard) must ask this instead: one row read, a mass
315
329
  * compare, no decode. */
316
330
  hasHalo(id: NodeId): boolean;
331
+ /** THE ADMISSION PREDICATE — whether `id` LEADS SOMEWHERE: it bears a
332
+ * continuation edge ({@link hasNext}) or a halo ({@link hasHalo}). A form
333
+ * that leads nowhere contributes nothing to any derivation. This is the ONE
334
+ * raw definition; `traverse.ts`'s `leadsSomewhere` is the same predicate with
335
+ * its edge tier memoised for the response. Two point probes at most. */
336
+ leadsSomewhere(id: NodeId): boolean;
317
337
  /** How many episode signatures were poured into `id`'s halo — the DIRECT
318
338
  * measure of distributional evidence (each training pair pours once, so
319
339
  * repetition counts, unlike {@link prevCount}, which counts DISTINCT
@@ -507,6 +527,9 @@ export declare abstract class AbstractStore implements Store {
507
527
  /** Exact-content dedup: content-key → node id. Intrinsic compression. */
508
528
  protected readonly _leafKey: BoundedMap<string, NodeId>;
509
529
  protected readonly _branchKey: BoundedMap<string, NodeId>;
530
+ /** {@link findFlatBranch}'s hits, keyed by the bytes themselves (latin1 —
531
+ * exact, unlike the hash keys above). */
532
+ protected readonly _flatKey: BoundedMap<string, NodeId>;
510
533
  /** Reconstructed-bytes read cache (regenerable), keyed by node id. */
511
534
  protected readonly _bytesCache: BoundedMap<NodeId, Uint8Array>;
512
535
  /** contentLen memo — content is immutable, so entries never invalidate. */
@@ -643,10 +666,29 @@ export declare abstract class AbstractStore implements Store {
643
666
  * answers most of those with no I/O at all, so the allocations dominated.
644
667
  *
645
668
  * Pass a subarray: it is a view, so a caller scanning spans of a query
646
- * allocates nothing per probe. Deliberately NOT memoized — its callers
647
- * probe many spans that miss, and a key string per probe is the cost this
648
- * exists to remove. */
669
+ * allocates nothing per probe that misses. HITS ARE MEMOIZED, MISSES ARE
670
+ * NOT: the negative filter answers first, so a span that is not stored
671
+ * builds no key; one that may be pays a key and then usually skips the
672
+ * lookup — the spans the identity fold names are segments, a few bytes
673
+ * each, asked over and over by every span that contains them. A node, once
674
+ * minted, keeps its bytes and its id, so a cached hit never goes stale. */
649
675
  findFlatBranch(bytes: Uint8Array): NodeId | null;
676
+ /** {@link Store.flatSpans}. Each start keeps the hash of the span it was
677
+ * last probed to, plus the hash one byte short of it; a probe ending at or
678
+ * past that end extends it, one ending one byte short of it reuses the
679
+ * second, and any other probe hashes from the start. `hashOf` is FNV-1a,
680
+ * a left fold over the bytes, so an extended hash IS the span's hash —
681
+ * same key, same filter answer, same lookup. */
682
+ flatSpans(bytes: Uint8Array): (start: number, end: number) => NodeId | null;
683
+ /** {@link findFlatBranch} past the hash: `h` must be `hashOf(bytes)`. */
684
+ private _findFlatHashed;
685
+ /** {@link Store.flatBranchMayExist} — the backend's negative filter when it
686
+ * keeps one ({@link _dbFlatMayExist}), else the lookup itself. */
687
+ flatBranchMayExist(bytes: Uint8Array): boolean;
688
+ /** Default: no filter, so anything MAY exist and the lookup decides. A
689
+ * backend with a negative filter over node hashes overrides this to answer
690
+ * from the filter alone. */
691
+ protected _dbFlatMayExist(_h: number, _bytes: Uint8Array): boolean;
650
692
  findBranch(kids: NodeId[]): NodeId | null;
651
693
  parents(id: NodeId): NodeId[];
652
694
  parentsFirst(id: NodeId, limit: number): NodeId[];
@@ -792,6 +834,8 @@ export declare abstract class AbstractStore implements Store {
792
834
  /** {@link Store.hasHalo} — MUST mirror {@link halo}'s null condition
793
835
  * exactly (row present AND mass ≥ minHaloMass), minus the decode. */
794
836
  hasHalo(id: NodeId): boolean;
837
+ /** {@link Store.leadsSomewhere} — edge first (the cheaper, commoner probe). */
838
+ leadsSomewhere(id: NodeId): boolean;
795
839
  pourHalo(id: NodeId, add: Vec): Promise<void>;
796
840
  resonateHalo(v: Vec, k: number): Promise<Hit[]>;
797
841
  private pending;