@hviana/sema 0.8.9 → 0.9.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (111) hide show
  1. package/AGENTS.md +7 -7
  2. package/dist/src/alu/src/index.d.ts +1 -1
  3. package/dist/src/alu/src/index.js +1 -1
  4. package/dist/src/alu/src/parser.js +2 -6
  5. package/dist/src/alu/src/resonance.d.ts +13 -0
  6. package/dist/src/alu/src/resonance.js +41 -0
  7. package/dist/src/alu/test/alu.test.js +39 -0
  8. package/dist/src/bytes.d.ts +6 -2
  9. package/dist/src/bytes.js +10 -4
  10. package/dist/src/canon.js +44 -0
  11. package/dist/src/geometry.d.ts +19 -1
  12. package/dist/src/geometry.js +125 -141
  13. package/dist/src/meter.d.ts +27 -0
  14. package/dist/src/meter.js +28 -1
  15. package/dist/src/mind/articulation.js +14 -1
  16. package/dist/src/mind/attention.d.ts +12 -0
  17. package/dist/src/mind/attention.js +44 -16
  18. package/dist/src/mind/bridge.js +3 -3
  19. package/dist/src/mind/derivation.d.ts +40 -0
  20. package/dist/src/mind/derivation.js +34 -0
  21. package/dist/src/mind/graph-search.d.ts +89 -15
  22. package/dist/src/mind/graph-search.js +345 -174
  23. package/dist/src/mind/learning.js +1 -1
  24. package/dist/src/mind/mechanisms/cover.d.ts +19 -3
  25. package/dist/src/mind/mechanisms/cover.js +101 -58
  26. package/dist/src/mind/mechanisms/recall.js +0 -1
  27. package/dist/src/mind/mind.js +2 -2
  28. package/dist/src/mind/pipeline.d.ts +5 -1
  29. package/dist/src/mind/pipeline.js +175 -87
  30. package/dist/src/mind/primitives.d.ts +25 -5
  31. package/dist/src/mind/primitives.js +107 -44
  32. package/dist/src/mind/reasoning.d.ts +18 -4
  33. package/dist/src/mind/reasoning.js +445 -321
  34. package/dist/src/mind/recognition.js +55 -73
  35. package/dist/src/mind/resonance.js +1 -11
  36. package/dist/src/mind/traverse.d.ts +3 -3
  37. package/dist/src/mind/traverse.js +3 -3
  38. package/dist/src/mind/types.d.ts +7 -1
  39. package/dist/src/store-sqlite.d.ts +25 -0
  40. package/dist/src/store-sqlite.js +89 -1
  41. package/dist/src/store.d.ts +48 -4
  42. package/dist/src/store.js +86 -6
  43. package/docs/INDEX.md +18 -18
  44. package/docs/INVARIANTS.md +16 -16
  45. package/docs/architecture/bounded-reads.md +1 -1
  46. package/docs/architecture/caches.md +5 -4
  47. package/docs/architecture/closure.md +45 -5
  48. package/docs/architecture/cost-model.md +16 -0
  49. package/docs/architecture/factored-machinery.md +14 -13
  50. package/docs/architecture/fold-contract.md +51 -1
  51. package/docs/architecture/mechanism-market.md +21 -0
  52. package/docs/architecture/memoization.md +3 -3
  53. package/docs/architecture/meter.md +2 -1
  54. package/docs/architecture/saturation.md +12 -0
  55. package/docs/architecture/store.md +25 -2
  56. package/docs/failures/tempting-but-wrong.md +13 -2
  57. package/docs/harness/gates.md +12 -10
  58. package/docs/mechanisms/cover.md +23 -6
  59. package/jsr.json +1 -1
  60. package/package.json +1 -1
  61. package/src/alu/README.md +10 -2
  62. package/src/alu/src/index.ts +1 -0
  63. package/src/alu/src/parser.ts +6 -6
  64. package/src/alu/src/resonance.ts +42 -0
  65. package/src/alu/test/alu.test.ts +40 -0
  66. package/src/bytes.ts +13 -3
  67. package/src/canon.ts +40 -0
  68. package/src/geometry.ts +183 -154
  69. package/src/meter.ts +28 -1
  70. package/src/mind/articulation.ts +14 -2
  71. package/src/mind/attention.ts +47 -25
  72. package/src/mind/bridge.ts +3 -3
  73. package/src/mind/derivation.ts +77 -0
  74. package/src/mind/graph-search.ts +449 -221
  75. package/src/mind/learning.ts +1 -7
  76. package/src/mind/match.ts +1 -2
  77. package/src/mind/mechanisms/cast.ts +1 -2
  78. package/src/mind/mechanisms/cover.ts +149 -84
  79. package/src/mind/mechanisms/extraction.ts +1 -2
  80. package/src/mind/mechanisms/prefix-completion.ts +1 -1
  81. package/src/mind/mechanisms/recall.ts +1 -3
  82. package/src/mind/mechanisms/reference.ts +1 -1
  83. package/src/mind/mind.ts +5 -30
  84. package/src/mind/pipeline.ts +206 -102
  85. package/src/mind/primitives.ts +119 -43
  86. package/src/mind/reasoning.ts +558 -413
  87. package/src/mind/recognition.ts +49 -65
  88. package/src/mind/resonance.ts +2 -16
  89. package/src/mind/trace.ts +1 -1
  90. package/src/mind/traverse.ts +3 -3
  91. package/src/mind/types.ts +9 -11
  92. package/src/store-sqlite.ts +92 -1
  93. package/src/store.ts +113 -7
  94. package/test/105-derive-through-reports-its-refusal.test.mjs +8 -5
  95. package/test/106-the-join-fires.test.mjs +21 -0
  96. package/test/111-the-cover-assembly-is-counted.test.mjs +8 -5
  97. package/test/128-the-leads-somewhere-pair-agrees.test.mjs +18 -12
  98. package/test/136-the-two-named-limits.test.mjs +3 -2
  99. package/test/137-the-law-lives-once-and-below.test.mjs +21 -0
  100. package/test/148-exact-shortcuts-agree.test.mjs +188 -0
  101. package/test/149-the-closure-engine.test.mjs +138 -0
  102. package/test/150-the-join-is-output-sensitive.test.mjs +66 -0
  103. package/test/151-the-cover-pays-for-what-it-reaches.test.mjs +142 -0
  104. package/test/152-the-read-side-names-as-the-write-side.test.mjs +146 -0
  105. package/test/153-a-cheaper-bound-is-looked-at-first.test.mjs +155 -0
  106. package/test/24-generalization.test.mjs +32 -0
  107. package/test/36-bloom.test.mjs +53 -0
  108. package/test/37-cluster-dispersion-fusion.test.mjs +75 -0
  109. package/test/48-recognise-turn-connective.test.mjs +3 -2
  110. package/test/55-cost-meter.test.mjs +4 -4
  111. package/test/90-connector-read-cap.test.mjs +7 -7
@@ -11,13 +11,13 @@ import {
11
11
  canonResolve,
12
12
  foldTree,
13
13
  gistOf,
14
- latin1Key,
15
14
  perceive,
16
15
  resolve,
17
16
  } from "./primitives.js";
18
17
  import { atomIsHub, bearsEdge, corpusN, leadsSomewhere } from "./traverse.js";
19
18
  import { chainReach, leafIdAt, leafIdRun } from "./canonical.js";
20
19
  import { canonHash } from "../canon.js";
20
+ import { latin1 } from "../bytes.js";
21
21
  import { isChunk, type Sema } from "../sema.js";
22
22
  import type { Leaf, Site } from "./graph-search.js";
23
23
 
@@ -91,7 +91,7 @@ export function recognise(
91
91
  // not silent), so it is emitted here directly rather than only inside
92
92
  // recogniseImpl.
93
93
  if (ctx.recogniseMemo) {
94
- const key = latin1Key(bytes);
94
+ const key = latin1(bytes);
95
95
  const hit = ctx.recogniseMemo.get(key);
96
96
  if (hit !== undefined) {
97
97
  if (ctx.meter) ctx.meter.recogniseHits++;
@@ -495,8 +495,20 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
495
495
  // encoding is the identity, so the span's bytes ARE the branch key.
496
496
  // `subarray` is a view — this allocates nothing per probe, and the
497
497
  // bloom filter answers the misses without touching the database.
498
+ //
499
+ // Through ONE span prober (Store.flatSpans), because hashing the key was
500
+ // itself the quadratic: every probe hashed its span from the start, and
501
+ // the interior pass below sweeps up to `reach` ends past each endpoint —
502
+ // O(n · reach²) bytes hashed. Measured on the 31.7M-node store, one
503
+ // composition-regime response (#97 of the battery) probed 2,079,400
504
+ // spans and hashed 251,660,406 bytes for them. The prober extends each
505
+ // start's hash instead, so the same probes, answered identically, cost
506
+ // O(n · reach).
507
+ const spans = store.flatSpans?.(bytes) ?? null;
498
508
  const flatProbe = (start: number, end: number): number | null =>
499
- store.findFlatBranch
509
+ spans !== null
510
+ ? spans(start, end)
511
+ : store.findFlatBranch
500
512
  ? store.findFlatBranch(bytes.subarray(start, end))
501
513
  : store.findBranch(allLeafIds.slice(start, end));
502
514
  // THE TWO ROUTES COST DIFFERENT THINGS, SO THEY ARE PRICED SEPARATELY.
@@ -534,14 +546,23 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
534
546
  // pass below spends the same budget on those pairs. Returns whether it
535
547
  // emitted, so a caller can retry a trimmed edge on the miss path only.
536
548
  if (end - start < W) return false;
537
- if (flatProbe(start, end) === null) {
549
+ // The byte-exact route is a BLOOM, so a non-null answer means MAYBE: it can neither
550
+ // decide (that is `resolveSpan`'s job) nor deny (that is `canonAdmits`'). Reading it
551
+ // as a YES made a false positive both drop the span and skip the decider, because the
552
+ // canon route ran only on a null. Measured on the trained corpus: 280 spans where the
553
+ // bloom claimed and the identity refused. A bloom HIT that resolves still emits
554
+ // without touching the canon route, which is what keeps the cheap path cheap.
555
+ const flat = flatProbe(start, end);
556
+ let id = flat === null ? null : resolveSpan(start, end);
557
+ if (id === null) {
558
+ if (flat !== null && ctx.meter) ctx.meter.bloomFalsePositives++;
538
559
  if (!canonBudget) {
539
560
  if (ctx.meter) ctx.meter.canonProbesDenied++;
540
561
  return false;
541
562
  }
542
563
  if (!canonAdmits(start, end)) return false;
564
+ id = resolveSpan(start, end);
543
565
  }
544
- const id = resolveSpan(start, end);
545
566
  if (id === null) return false;
546
567
  emit(start, end, id);
547
568
  return true;
@@ -610,71 +631,35 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
610
631
  // keeps this off the quadratic path the budget note above describes (that
611
632
  // one had no span bound at all).
612
633
  {
613
- // The span bound is W^2, the chain's own limit, PLUS the slack the endpoint set already grants: every endpoint
614
- // sits within `radius` of a cut, so a pair that names one form may straddle cuts and still be a single form's
615
- // span. Measured on the composite fixture: W=4 (reach 16), radius 8, and the useful [7,32) is 25 bytes with its
616
- // edges 2 bytes from cuts 5 and 30 — already IN `ordered`, and excluded only by the upper bound. Both terms are
634
+ // The span bound is the chain's own limit PLUS the slack the endpoint set already grants: every endpoint sits
635
+ // within `radius` of a cut, so a pair that names one form may straddle cuts and still be a single form's span.
636
+ // Measured on the composite fixture: W=4 (reach 16), radius 8, and the useful [7,32) is 25 bytes with its edges
637
+ // 2 bytes from cuts 5 and 30 — already IN `ordered`, and excluded only by the upper bound. Both terms are
617
638
  // derived (W and the seat count); no new constant enters.
618
- const reach = chainReach(W) + 2 * radius;
639
+ //
640
+ // The chain term is W^2 * W = W^4, for a form embedded in the MIDDLE of a longer query. Measured on the
641
+ // trained corpus, the SAME real contexts in three positions: opening 12/17, MIDDLE 0/17, end 12/17. The two
642
+ // edge scans reach only prefixes and suffixes, so the interior pass is the only tier that could name it, and
643
+ // at W^2 it could not. WHAT KEEPS THIS LINEAR IS THE BOUND BEING A CONSTANT — each endpoint pairs only with
644
+ // the partners inside a fixed window, so the work stays O(n). Measured against a cut-pair enumeration with
645
+ // ±W trims (an earlier, 43-line attempt): that one reached the same 136 B at 70 888 probes, this one at
646
+ // 84 169 (x1.19 more) — and BOTH pass test/14, which gates the CLASS (linear) and not the constant. The
647
+ // simpler form won on that measurement.
648
+ const reach = chainReach(W) * W * W + 2 * radius;
619
649
  if (ctx.meter) ctx.meter.recogniseInteriorGaps += ordered.length;
650
+ // Each end pairs only with the starts in [end − reach, end − W], in
651
+ // ascending order — the pairs, and the order the budget is spent in,
652
+ // of the all-pairs scan, without its O(n²) enumeration.
653
+ let lo = 0;
620
654
  for (const end of ordered) {
621
- for (const start of ordered) {
622
- if (start >= end) continue;
623
- const span = end - start;
624
- if (span < W || span > reach) continue;
655
+ while (ordered[lo] < end - reach) lo++;
656
+ for (let i = lo; i < ordered.length; i++) {
657
+ const start = ordered[i];
658
+ if (end - start < W) break;
625
659
  if (ctx.meter) ctx.meter.recogniseInteriorPairs++;
626
660
  spend(start, end);
627
661
  }
628
662
  }
629
- // ── THE MIDDLE WAS BLIND: a CUT-PAIR probe, BOUNDED so it stays LINEAR ──────────────
630
- // A stored form embedded in the MIDDLE of a longer query was named by NO tier: the two
631
- // edge scans probe only prefixes and suffixes, and the loop above is capped at
632
- // `reach` = W^2 + 2*radius. Measured on the trained corpus, the SAME real contexts in
633
- // three positions: opening 12/17, MIDDLE 0/17, end 12/17. Both edges of such a form sit
634
- // within `radius` of CUTS, so a cut pair plus a small trim names the form exactly
635
- // (measured: a 64 B form at [27, 91) between cuts 26 and 90, trims +1/+1).
636
- //
637
- // THE BOUND IS WHAT KEEPS IT LINEAR. An unbounded cut-pair scan is O(cuts^2) probes, and
638
- // test/14 rejected exactly that (47 324 ms). Bounded by a CONSTANT span it is
639
- // O(cuts * const) = O(n), like the loop above — the constant is merely larger. The bound
640
- // is DERIVED, never pinned: `chainReach(W)` is W^2, "the deepest two-level composite the
641
- // write side's windows can spell" (canonical.ts), so `chainReach(W) * W * W` is the
642
- // four-level one; `chainReach(W) * W` is already used in bridge.ts.
643
- //
644
- // THE TRIMS ARE THE SAME DISCIPLINE THE EDGE SCANS USE — the suffix scan already probes
645
- // `spend(s, bytes.length - 1)`, so ±W is a wider version of an existing rule, not a new
646
- // threshold. Measured price/reach; all four configurations pass test/14, which gates the
647
- // CLASS (linear) and not the constant:
648
- // W^3 + ±1 -> 64 B, 8 068 probes (x2.02)
649
- // W^4 + ±W -> 136 B, 70 888 probes (x6.06) <- this one
650
- // W^3 + ±W and W^4 + ±1 are DOMINATED: each stops at the other parameter and costs more.
651
- // The loop above stays, so no candidate that produces a site today is lost. Suite
652
- // 707/707, and a differential over 24 real corpus questions is byte-identical (0 answers
653
- // changed, 0 new duplicate sites): the :258 warning that a wider bound can rediscover a
654
- // smaller subtree's content as a second, overlapping site was read and measured, and it
655
- // does not materialise here.
656
- const deepReach = chainReach(W) * W * W;
657
- for (let ci = 0; ci + 1 < startList.length; ci++) {
658
- for (let cj = ci + 1; cj < startList.length; cj++) {
659
- if (startList[cj] - startList[ci] > deepReach + 2 * radius) break;
660
- // The EXACT cut pair first. Measured: with the 81 trims starting at -W the `spend`
661
- // pool ran dry (canonProbesDenied in the millions) and the canon route was then denied
662
- // to the candidates that needed it — including forms the byte-exact route SEES
663
- // (`flatProbe` true) that were still not named. Probing (ci, cj) before any trim puts
664
- // the common case in front of the famine.
665
- if (ctx.meter) ctx.meter.recogniseInteriorPairs++;
666
- spend(startList[ci], startList[cj]);
667
- for (let dl = -W; dl <= W; dl++) {
668
- for (let dr = -W; dr <= W; dr++) {
669
- const a = startList[ci] + dl;
670
- const z = startList[cj] + dr;
671
- if (a < 0 || z > bytes.length || z - a < W) continue;
672
- if (ctx.meter) ctx.meter.recogniseInteriorPairs++;
673
- spend(a, z);
674
- }
675
- }
676
- }
677
- }
678
663
  }
679
664
  }
680
665
  }
@@ -710,8 +695,7 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
710
695
  // arithmetic, not evidence. Removing it wholesale was measured and
711
696
  // REVERTED: it also drops legitimate multi-byte chains (the 12-byte
712
697
  // "Eiffel Tower" site vanished with it). The premise is wrong but the
713
- // trust it stood in for is real; a replacement signal is still open work.
714
- // See bench/README.md.
698
+ // trust it stood in for is real; the replacement signal follows.
715
699
  //
716
700
  // THE REPLACEMENT SIGNAL (2026-08-13): `leadsSomewhere` on the BYTE-EXACT
717
701
  // branch the chain already found. The blanket off-boundary suppression is
@@ -3,12 +3,11 @@
3
3
  // Address → Resonate → filter by Traverse/Read predicates → transform.
4
4
  // Used by bridge, recallByResonance, pivotInto, meaningOf.
5
5
  // (The graded locate() matcher formerly here lives in match.ts.)
6
- import { rItem, rNode } from "./trace.js";
7
- import { decodeText } from "./rationale.js";
6
+ import { rItem } from "./trace.js";
8
7
 
9
8
  import { cosine, Vec } from "../vec.js";
10
9
  import { mergeThreshold } from "../geometry.js";
11
- import { concat2, concatBytes, indexOf } from "../bytes.js";
10
+ import { concat2, concatBytes, indexOf, latin1 } from "../bytes.js";
12
11
  import type { MindContext } from "./types.js";
13
12
  import { gistOf, read, resolve, walkTree } from "./primitives.js";
14
13
  import { perceive } from "./primitives.js";
@@ -17,8 +16,6 @@ import {
17
16
  cachedRead,
18
17
  type Junction,
19
18
  junctionContainers,
20
- junctionContainersFrom,
21
- junctionSeeds,
22
19
  junctionSynonyms,
23
20
  walkCache,
24
21
  } from "./junction.js";
@@ -128,17 +125,6 @@ function junctionEdges(
128
125
  return out;
129
126
  }
130
127
 
131
- /** A byte string as a string, ONE code unit per byte — injective, so it is
132
- * safe to build a cache key from. Chunked to keep the spread within the
133
- * engine's argument limit on long contexts. */
134
- function latin1(b: Uint8Array): string {
135
- let s = "";
136
- for (let i = 0; i < b.length; i += 4096) {
137
- s += String.fromCharCode(...b.subarray(i, i + 4096));
138
- }
139
- return s;
140
- }
141
-
142
128
  /** Per-response memo of bridge results, keyed by the response's lifecycle
143
129
  * object (ctx.climbMemo — created fresh by respond() and nulled after, so
144
130
  * entries can never outlive the read-only window they are valid in). The
package/src/mind/trace.ts CHANGED
@@ -6,7 +6,7 @@
6
6
 
7
7
  import type { MindContext } from "./types.js";
8
8
  import { read } from "./primitives.js";
9
- import type { DerivationItem, DerivationStep } from "./graph-search.js";
9
+ import type { DerivationStep } from "./graph-search.js";
10
10
  import { decodeText } from "./rationale.js";
11
11
  import type { RationaleItem } from "./rationale.js";
12
12
 
@@ -488,9 +488,9 @@ export function bearsEdge(ctx: MindContext, id: number): boolean {
488
488
  return cachedHasNext(ctx, id, getStructCache(ctx));
489
489
  }
490
490
 
491
- /** Whether a node LEADS SOMEWHERE — it bears a continuation edge or a halo.
492
- * The admission predicate recognition filters sites with (cover.md): a form
493
- * that
491
+ /** Whether a node LEADS SOMEWHERE — the store's admission predicate
492
+ * ({@link Store.leadsSomewhere}: edge or halo) with its edge tier memoised for
493
+ * the response. Recognition filters sites with it (cover.md): a form that
494
494
  * leads nowhere contributes nothing to any derivation. Runs once per candidate
495
495
  * span on the recognition hot path — `hasNext` is cached per response (the same
496
496
  * flat-branch ids are probed across prefix variants by canonicalChunkId).
package/src/mind/types.ts CHANGED
@@ -10,15 +10,7 @@ import type { Space } from "../sema.js";
10
10
  import type { Alphabet } from "../alphabet.js";
11
11
  import type { MindConfig } from "../config.js";
12
12
  import type { Meter } from "../meter.js";
13
- import type {
14
- ComputedResult,
15
- DerivationItem,
16
- DerivationStep,
17
- GraphSearch,
18
- Leaf,
19
- Seg,
20
- Site,
21
- } from "./graph-search.js";
13
+ import type { GraphSearch, Leaf, Seg, Site } from "./graph-search.js";
22
14
  import type { Rationale } from "./rationale.js";
23
15
  import type { ContentFold, Grid } from "../geometry.js";
24
16
 
@@ -33,7 +25,7 @@ export interface DepositCacheEntry {
33
25
  /** The plain content fold's reusable segment state. */
34
26
  content: ContentFold;
35
27
  }
36
- import { bytesEqual, concatBytes, indexOf } from "../bytes.js";
28
+ import { bytesEqual, concatBytes } from "../bytes.js";
37
29
  import { restates } from "./derivation.js";
38
30
  import { dominates } from "../geometry.js";
39
31
 
@@ -199,7 +191,13 @@ export interface Attention {
199
191
  * a genuine further topic is named in its own distinctive wording
200
192
  * somewhere the query's scaffolding does not reach, always a SEPARATE
201
193
  * cluster from whatever else corroborates it. See
202
- * test/37-cluster-dispersion-fusion.test.mjs. */
194
+ * test/37-cluster-dispersion-fusion.test.mjs.
195
+ *
196
+ * Read from the VOTES, this is a lossy witness: a region votes once, for its
197
+ * top anchor, so a place can be lost to a tie or won through an accident.
198
+ * Fusion therefore also asks the root's CONTEXT the same question at window
199
+ * scale (reasoning.ts `sharedPlaces`) and trusts a root that either reading
200
+ * finds in two places. */
203
201
  clusters: number;
204
202
  }
205
203
 
@@ -146,6 +146,17 @@ CREATE TABLE IF NOT EXISTS canon (
146
146
  id INTEGER NOT NULL,
147
147
  PRIMARY KEY (h, id)
148
148
  ) WITHOUT ROWID;
149
+ -- The canon index's negative filter, persisted so an open does not rescan the
150
+ -- h column (seconds on a trained store). Written in the SAME transaction as
151
+ -- the canon rows it covers, stamped with the meta 'canon.upto' of that commit;
152
+ -- a stamp that disagrees with the meta (rows a writer added without it) makes
153
+ -- the filter stale, and it is rebuilt from the table instead.
154
+ CREATE TABLE IF NOT EXISTS canon_bloom (
155
+ id INTEGER PRIMARY KEY CHECK (id = 1),
156
+ bits BLOB NOT NULL,
157
+ n INTEGER NOT NULL,
158
+ upto TEXT NOT NULL
159
+ );
149
160
  -- CONSTITUENT SKETCH (Store.sketchGet/sketchPut): the bottom-k minimal
150
161
  -- constituents of a node's subtree, k = √D, chosen by identity hash. The blob is
151
162
  -- a packed int32 little-endian run, already in hash order; an EMPTY blob is a
@@ -251,6 +262,20 @@ export class SQliteStore extends AbstractStore implements Store {
251
262
  private _bloom: NodeBloom | null = null;
252
263
  /** Dedup probes answered by the filter alone this session (observability). */
253
264
  bloomSkips = 0;
265
+ /** The same negative filter over the CANON index's key hashes. Recognition's
266
+ * canonical admission and the join's canonical entity scan ask `canonFind`
267
+ * once per probed span, and almost every answer is "no such key" — on an
268
+ * index that is often EMPTY (it is built only by `buildCanonIndex`).
269
+ * Loaded on first use from `canon_bloom` when its stamp matches the meta,
270
+ * else built by one sequential scan of the h column; kept exact on
271
+ * `canonAdd` (rebuilt bigger from the table, uncommitted rows included,
272
+ * when growth saturates it) and persisted with the commit that wrote the
273
+ * rows. The canon table is never deleted from, so the filter can never
274
+ * hold a false negative: a miss it reports is a miss the query would have
275
+ * returned. */
276
+ private _canonBloom: NodeBloom | null = null;
277
+ /** The in-memory canon filter differs from the persisted one. */
278
+ private _canonBloomDirty = false;
254
279
 
255
280
  private _insertNode: any = null;
256
281
  private _insertKid: any = null;
@@ -509,7 +534,10 @@ export class SQliteStore extends AbstractStore implements Store {
509
534
  }
510
535
 
511
536
  protected _dbClose(): void {
512
- if (this.sqlite) this._bloomPersist();
537
+ if (this.sqlite) {
538
+ this._bloomPersist();
539
+ this._canonBloomPersist();
540
+ }
513
541
  if (this.content) {
514
542
  this.content.close();
515
543
  this.content = null;
@@ -534,6 +562,8 @@ export class SQliteStore extends AbstractStore implements Store {
534
562
 
535
563
  protected _dbCommitTx(): void {
536
564
  if (!this._inTx || !this.sqlite) return;
565
+ // The canon filter commits WITH the rows it covers — see canon_bloom.
566
+ this._canonBloomPersist();
537
567
  this._inTx = false; // clear first so a throw can't wedge us mid-commit
538
568
  this.sqlite.exec("COMMIT");
539
569
  }
@@ -615,6 +645,15 @@ export class SQliteStore extends AbstractStore implements Store {
615
645
  return row ? row.id : null;
616
646
  }
617
647
 
648
+ /** The node filter alone: never a false negative (every inserted hash is
649
+ * added before any probe can see it), so `false` is exact. */
650
+ protected override _dbFlatMayExist(h: number, bytes: Uint8Array): boolean {
651
+ if (this._bloom === null) return super._dbFlatMayExist(h, bytes);
652
+ if (this._bloom.mightContain(h)) return true;
653
+ this.bloomSkips++;
654
+ return false;
655
+ }
656
+
618
657
  protected _dbFindBranchByKids(
619
658
  h: number,
620
659
  packed: Uint8Array,
@@ -1053,9 +1092,61 @@ export class SQliteStore extends AbstractStore implements Store {
1053
1092
  // bulk index build coalesces instead of paying autocommit per row.
1054
1093
  this._dbBeginTx();
1055
1094
  this._insCanon.run(h, id);
1095
+ const b = this._canonFilter();
1096
+ b.add(h);
1097
+ // Rebuilt bigger at once, from the table as THIS connection sees it — the
1098
+ // persisted filter cannot stand in, it lacks this transaction's rows.
1099
+ if (b.saturated) this._canonBloom = this._canonScan();
1100
+ this._canonBloomDirty = true;
1101
+ }
1102
+
1103
+ /** The canon filter: the persisted one when its stamp matches the meta,
1104
+ * else built from the index as it stands. */
1105
+ private _canonFilter(): NodeBloom {
1106
+ if (this._canonBloom === null) {
1107
+ const row = this.sqlite!.prepare(
1108
+ "SELECT bits, n, upto FROM canon_bloom WHERE id = 1",
1109
+ ).get() as { bits: Uint8Array; n: number; upto: string } | undefined;
1110
+ if (row !== undefined && row.upto === this._canonStamp()) {
1111
+ const b = new NodeBloom(31 - Math.clz32(row.bits.length * 8));
1112
+ b.bits.set(row.bits);
1113
+ b.n = row.n;
1114
+ this._canonBloom = b;
1115
+ } else {
1116
+ this._canonBloom = this._canonScan();
1117
+ this._canonBloomDirty = true;
1118
+ }
1119
+ }
1120
+ return this._canonBloom;
1121
+ }
1122
+
1123
+ private _canonScan(): NodeBloom {
1124
+ const b = new NodeBloom(bloomLog2For(this.canonCount()));
1125
+ const scan = this.sqlite!.prepare("SELECT h FROM canon");
1126
+ scan.setReturnArrays(true);
1127
+ for (const r of scan.iterate() as IterableIterator<[number]>) b.add(r[0]);
1128
+ return b;
1129
+ }
1130
+
1131
+ /** What a persisted canon filter is stamped with: the incremental build's
1132
+ * cursor, which every canon writer advances in the transaction it writes. */
1133
+ private _canonStamp(): string {
1134
+ return this._dbGetMeta("canon.upto") ?? "";
1135
+ }
1136
+
1137
+ private _canonBloomPersist(): void {
1138
+ const b = this._canonBloom;
1139
+ if (b === null || !this._canonBloomDirty || !this.sqlite) return;
1140
+ this.sqlite.prepare(
1141
+ "INSERT INTO canon_bloom (id, bits, n, upto) VALUES (1, ?, ?, ?) " +
1142
+ "ON CONFLICT(id) DO UPDATE SET bits = excluded.bits, " +
1143
+ "n = excluded.n, upto = excluded.upto",
1144
+ ).run(b.bits, b.n, this._canonStamp());
1145
+ this._canonBloomDirty = false;
1056
1146
  }
1057
1147
 
1058
1148
  canonFind(h: number): number[] {
1149
+ if (!this._canonFilter().mightContain(h)) return [];
1059
1150
  if (!this._selCanon) {
1060
1151
  this._selCanon = this.sqlite!.prepare(
1061
1152
  "SELECT id FROM canon WHERE h = ?",
package/src/store.ts CHANGED
@@ -29,9 +29,10 @@
29
29
  // essentials required for SQLite + VectorDatabase communication.
30
30
 
31
31
  import { addInto, copy, dot, normalize, Vec } from "./vec.js";
32
- import { DEFAULT_CONFIG, type StoreConfig } from "./config.js";
32
+ import type { StoreConfig } from "./config.js";
33
33
  import { identityBar } from "./geometry.js";
34
34
  import type { Meter } from "./meter.js";
35
+ import { latin1 } from "./bytes.js";
35
36
 
36
37
  /** A node id: a dense, non-negative integer assigned in creation order. */
37
38
  export type NodeId = number;
@@ -340,6 +341,20 @@ export interface Store {
340
341
  * bytes — the allocation-free probe span scanners use. Optional: a store
341
342
  * without it is simply probed through `findBranch`. */
342
343
  findFlatBranch?(bytes: Uint8Array): NodeId | null;
344
+ /** Whether a flat branch with these bytes MAY exist. `false` is EXACT — no
345
+ * such node exists, and a {@link findFlatBranch} would return null; `true`
346
+ * means only that a lookup is needed. The existence half of the probe,
347
+ * without its verification: a caller that will resolve the bytes anyway
348
+ * (and so verify them) asks this to refuse a miss without paying for the
349
+ * lookup a hit would repeat. Optional, like the probe it answers for. */
350
+ flatBranchMayExist?(bytes: Uint8Array): boolean;
351
+ /** A {@link findFlatBranch} over spans of ONE buffer: `probe(start, end)`
352
+ * answers exactly `findFlatBranch(bytes.subarray(start, end))`, but a span's
353
+ * content hash extends the one its start was last probed at, so a scanner
354
+ * that sweeps ends upward per start pays O(1) per probe instead of
355
+ * O(span). The buffer must not change while the prober is used.
356
+ * Optional, like the probe it accelerates. */
357
+ flatSpans?(bytes: Uint8Array): (start: number, end: number) => NodeId | null;
343
358
  /** The branch nodes that list `id` among their children — the reverse of
344
359
  * `get(id).kids`. Lets the structural DAG be climbed upward, from a
345
360
  * recognised fragment to the larger learned forms that contain it. */
@@ -499,6 +514,12 @@ export interface Store {
499
514
  * the search's fuse guard) must ask this instead: one row read, a mass
500
515
  * compare, no decode. */
501
516
  hasHalo(id: NodeId): boolean;
517
+ /** THE ADMISSION PREDICATE — whether `id` LEADS SOMEWHERE: it bears a
518
+ * continuation edge ({@link hasNext}) or a halo ({@link hasHalo}). A form
519
+ * that leads nowhere contributes nothing to any derivation. This is the ONE
520
+ * raw definition; `traverse.ts`'s `leadsSomewhere` is the same predicate with
521
+ * its edge tier memoised for the response. Two point probes at most. */
522
+ leadsSomewhere(id: NodeId): boolean;
502
523
  /** How many episode signatures were poured into `id`'s halo — the DIRECT
503
524
  * measure of distributional evidence (each training pair pours once, so
504
525
  * repetition counts, unlike {@link prevCount}, which counts DISTINCT
@@ -662,11 +683,14 @@ export function unpackKids(blob: Uint8Array): NodeId[] {
662
683
 
663
684
  /** 32-bit FNV-1a of a byte blob — the integer content hash `idx_node_h` keys
664
685
  * on. Collisions are resolved by verifying the stored blob, never trusted. */
686
+ const FNV_OFFSET = 0x811c9dc5 >>> 0;
687
+ const FNV_PRIME = 0x01000193;
688
+
665
689
  function hashOf(bytes: Uint8Array): number {
666
- let h = 0x811c9dc5 >>> 0;
690
+ let h = FNV_OFFSET;
667
691
  for (let i = 0; i < bytes.length; i++) {
668
692
  h ^= bytes[i];
669
- h = Math.imul(h, 0x01000193) >>> 0;
693
+ h = Math.imul(h, FNV_PRIME) >>> 0;
670
694
  }
671
695
  return h >>> 0;
672
696
  }
@@ -957,6 +981,9 @@ export abstract class AbstractStore implements Store {
957
981
  /** Exact-content dedup: content-key → node id. Intrinsic compression. */
958
982
  protected readonly _leafKey: BoundedMap<string, NodeId>;
959
983
  protected readonly _branchKey: BoundedMap<string, NodeId>;
984
+ /** {@link findFlatBranch}'s hits, keyed by the bytes themselves (latin1 —
985
+ * exact, unlike the hash keys above). */
986
+ protected readonly _flatKey: BoundedMap<string, NodeId>;
960
987
  /** Reconstructed-bytes read cache (regenerable), keyed by node id. */
961
988
  protected readonly _bytesCache: BoundedMap<NodeId, Uint8Array>;
962
989
  /** contentLen memo — content is immutable, so entries never invalidate. */
@@ -1063,6 +1090,12 @@ export abstract class AbstractStore implements Store {
1063
1090
  this.compactEveryNWrites = config.compactEveryNWrites;
1064
1091
  this._leafKey = new BoundedMap(config.dedupCacheMax);
1065
1092
  this._branchKey = new BoundedMap(config.dedupCacheMax);
1093
+ this._flatKey = new BoundedMap(
1094
+ config.dedupCacheMax,
1095
+ undefined,
1096
+ "lru",
1097
+ "clock",
1098
+ );
1066
1099
  this._bytesCache = new BoundedMap(
1067
1100
  config.bytesCacheMax,
1068
1101
  (v) => v.byteLength,
@@ -1400,12 +1433,80 @@ export abstract class AbstractStore implements Store {
1400
1433
  * answers most of those with no I/O at all, so the allocations dominated.
1401
1434
  *
1402
1435
  * Pass a subarray: it is a view, so a caller scanning spans of a query
1403
- * allocates nothing per probe. Deliberately NOT memoized — its callers
1404
- * probe many spans that miss, and a key string per probe is the cost this
1405
- * exists to remove. */
1436
+ * allocates nothing per probe that misses. HITS ARE MEMOIZED, MISSES ARE
1437
+ * NOT: the negative filter answers first, so a span that is not stored
1438
+ * builds no key; one that may be pays a key and then usually skips the
1439
+ * lookup — the spans the identity fold names are segments, a few bytes
1440
+ * each, asked over and over by every span that contains them. A node, once
1441
+ * minted, keeps its bytes and its id, so a cached hit never goes stale. */
1406
1442
  findFlatBranch(bytes: Uint8Array): NodeId | null {
1443
+ return this._findFlatHashed(hashOf(bytes), bytes);
1444
+ }
1445
+
1446
+ /** {@link Store.flatSpans}. Each start keeps the hash of the span it was
1447
+ * last probed to, plus the hash one byte short of it; a probe ending at or
1448
+ * past that end extends it, one ending one byte short of it reuses the
1449
+ * second, and any other probe hashes from the start. `hashOf` is FNV-1a,
1450
+ * a left fold over the bytes, so an extended hash IS the span's hash —
1451
+ * same key, same filter answer, same lookup. */
1452
+ flatSpans(bytes: Uint8Array): (start: number, end: number) => NodeId | null {
1453
+ const n = bytes.length;
1454
+ const runEnd = new Int32Array(n + 1).fill(-1);
1455
+ const runH = new Uint32Array(n + 1);
1456
+ const shortH = new Uint32Array(n + 1);
1457
+ return (start, end) => {
1458
+ let h: number;
1459
+ let from: number;
1460
+ const e0 = runEnd[start];
1461
+ if (e0 >= 0 && e0 <= end) {
1462
+ h = runH[start];
1463
+ from = e0;
1464
+ } else if (e0 >= 0 && e0 - 1 === end && end > start) {
1465
+ h = shortH[start];
1466
+ from = end;
1467
+ } else {
1468
+ h = FNV_OFFSET;
1469
+ from = start;
1470
+ }
1471
+ let prev = h;
1472
+ for (let i = from; i < end; i++) {
1473
+ prev = h;
1474
+ h = Math.imul(h ^ bytes[i], FNV_PRIME) >>> 0;
1475
+ }
1476
+ if (from < end) {
1477
+ runEnd[start] = end;
1478
+ runH[start] = h;
1479
+ shortH[start] = prev;
1480
+ }
1481
+ return this._findFlatHashed(h, bytes.subarray(start, end));
1482
+ };
1483
+ }
1484
+
1485
+ /** {@link findFlatBranch} past the hash: `h` must be `hashOf(bytes)`. */
1486
+ private _findFlatHashed(h: number, bytes: Uint8Array): NodeId | null {
1407
1487
  if (this.meter) this.meter.branchLookups++;
1408
- return this._dbFindBranchByLeaf(hashOf(bytes), bytes);
1488
+ if (!this._dbFlatMayExist(h, bytes)) return null;
1489
+ const key = bytes.length <= DEDUP_KEY_MAX ? latin1(bytes) : null;
1490
+ if (key !== null) {
1491
+ const hit = this._flatKey.get(key);
1492
+ if (hit !== undefined) return hit;
1493
+ }
1494
+ const id = this._dbFindBranchByLeaf(h, bytes);
1495
+ if (id !== null && key !== null) this._flatKey.set(key, id);
1496
+ return id;
1497
+ }
1498
+
1499
+ /** {@link Store.flatBranchMayExist} — the backend's negative filter when it
1500
+ * keeps one ({@link _dbFlatMayExist}), else the lookup itself. */
1501
+ flatBranchMayExist(bytes: Uint8Array): boolean {
1502
+ return this._dbFlatMayExist(hashOf(bytes), bytes);
1503
+ }
1504
+
1505
+ /** Default: no filter, so anything MAY exist and the lookup decides. A
1506
+ * backend with a negative filter over node hashes overrides this to answer
1507
+ * from the filter alone. */
1508
+ protected _dbFlatMayExist(_h: number, _bytes: Uint8Array): boolean {
1509
+ return true;
1409
1510
  }
1410
1511
 
1411
1512
  findBranch(kids: NodeId[]): NodeId | null {
@@ -2310,6 +2411,11 @@ export abstract class AbstractStore implements Store {
2310
2411
  return r !== null && r.mass >= this.minHaloMass;
2311
2412
  }
2312
2413
 
2414
+ /** {@link Store.leadsSomewhere} — edge first (the cheaper, commoner probe). */
2415
+ leadsSomewhere(id: NodeId): boolean {
2416
+ return this.hasNext(id) || this.hasHalo(id);
2417
+ }
2418
+
2313
2419
  async pourHalo(id: NodeId, add: Vec): Promise<void> {
2314
2420
  await this._ensureReady();
2315
2421
  // A node with a halo is a genuine resonance target — the consensus climb
@@ -52,10 +52,13 @@ const misses = (steps) =>
52
52
 
53
53
  test("the join's refusal is reported, naming the candidate and tail it tried", async () => {
54
54
  const mind = await chain();
55
- // The THREE-relation query: the second join is still refused (the rule
56
- // concludes terminal — the study's other half), so this is where the refusal
57
- // is observable. The two-relation one now JOINS (measured, and pinned by
58
- // test/99's spec), which is why this test moved here.
55
+ // The THREE-relation query: the chain joins through to Stockholm, and on the
56
+ // way the fact it stands on ("…Gustaf Molander is Sweden.") offers its other
57
+ // entity, "gustaf molander", with the remaining tail " capital" — a key no
58
+ // deposit learnt, so that join is refused, and that refusal is observable.
59
+ // A join is priced only for a fact a derivation STANDS ON (graph-search.ts,
60
+ // solve), so the refusal reported is that fact's — not one from a fact the
61
+ // exploration merely reached.
59
62
  const steps = [];
60
63
  await mind.respond("eva director country capital", (s) => steps.push(s));
61
64
  const got = misses(steps);
@@ -63,7 +66,7 @@ test("the join's refusal is reported, naming the candidate and tail it tried", a
63
66
  const named = got.some((s) => {
64
67
  const parts = (s.inputs ?? []).map((i) => String(i.text));
65
68
  return parts.some((t) => t.includes("gustaf molander")) &&
66
- parts.some((t) => t.includes("country"));
69
+ parts.some((t) => t.includes("capital"));
67
70
  });
68
71
  assert.ok(named, "the report must name the candidate and the tail it tried");
69
72
  // …and WHERE the candidate came from: a refusal that names only bytes leaves
@@ -92,3 +92,24 @@ test("a query with no tail is answered directly, not by a join", async () => {
92
92
  );
93
93
  await mind.store.close();
94
94
  });
95
+
96
+ test("the asker's wording that already holds the answer's form is not spliced into it", async () => {
97
+ // Articulation revoices an answer form in the asker's words when the two
98
+ // keep the same company (halo). `eva` and `eva director` keep the SAME
99
+ // company here — both lead to the same fact — but the asker's wording holds
100
+ // the form plus a further word: splicing it in is not a re-voicing, it adds
101
+ // the asker's other words to the answer. The DAG cannot see the containment
102
+ // (the fold cuts `eva director` as `eva d|irector`), so it is read off the
103
+ // bytes. Observed before: "The director of eva director is Gustaf Molander."
104
+ const lower = "The director of eva is Gustaf Molander.";
105
+ const store = new SQliteStore({ path: ":memory:" });
106
+ const mind = new Mind({ seed: 7, store });
107
+ await mind.ingest([
108
+ ["eva", lower],
109
+ ["eva director", lower],
110
+ ["gustaf molander", F2],
111
+ ["gustaf molander country", F2],
112
+ ]);
113
+ assert.equal(text(await mind.respond("eva director")).trim(), lower);
114
+ await mind.store.close();
115
+ });