@hviana/sema 0.8.9 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -184,6 +184,13 @@ export declare class Meter {
184
184
  * route was priced out. Counted where the fact happens (the `!canonBudget`
185
185
  * refusal), not where the probe is called. */
186
186
  canonProbesDenied: number;
187
+ /** Spans the BLOOM claimed and the exact identity then refused — so `probe` used to
188
+ * drop the span AND skip the decider. `findFlatBranch` is bloom-gated on purpose (a
189
+ * miss costs no database read), but a non-null answer means MAYBE; treating it as YES
190
+ * is what made the canon route run only `if (flatProbe === null)`. Counted where the
191
+ * fact happens (a non-null bloom whose `resolveSpan` is null), so the price of letting
192
+ * the canon route see those spans again is a measurement rather than a guess. */
193
+ bloomFalsePositives: number;
187
194
  /** The pipeline's remainder AT THE DECISION POINT, in bytes: what the grounded
188
195
  * answer plus the pre-computed spans left unexplained, after the same W floor
189
196
  * the fuse gate uses. This is the quantity that licenses (or refuses) the
package/dist/src/meter.js CHANGED
@@ -194,6 +194,13 @@ export class Meter {
194
194
  * route was priced out. Counted where the fact happens (the `!canonBudget`
195
195
  * refusal), not where the probe is called. */
196
196
  canonProbesDenied = 0;
197
+ /** Spans the BLOOM claimed and the exact identity then refused — so `probe` used to
198
+ * drop the span AND skip the decider. `findFlatBranch` is bloom-gated on purpose (a
199
+ * miss costs no database read), but a non-null answer means MAYBE; treating it as YES
200
+ * is what made the canon route run only `if (flatProbe === null)`. Counted where the
201
+ * fact happens (a non-null bloom whose `resolveSpan` is null), so the price of letting
202
+ * the canon route see those spans again is a measurement rather than a guess. */
203
+ bloomFalsePositives = 0;
197
204
  /** The pipeline's remainder AT THE DECISION POINT, in bytes: what the grounded
198
205
  * answer plus the pre-computed spans left unexplained, after the same W floor
199
206
  * the fuse gate uses. This is the quantity that licenses (or refuses) the
@@ -518,7 +518,17 @@ function recogniseImpl(ctx, bytes) {
518
518
  // emitted, so a caller can retry a trimmed edge on the miss path only.
519
519
  if (end - start < W)
520
520
  return false;
521
- if (flatProbe(start, end) === null) {
521
+ // The byte-exact route is a BLOOM, so a non-null answer means MAYBE: it can neither
522
+ // decide (that is `resolveSpan`'s job) nor deny (that is `canonAdmits`'). Reading it
523
+ // as a YES made a false positive both drop the span and skip the decider, because the
524
+ // canon route ran only on a null. Measured on the trained corpus: 280 spans where the
525
+ // bloom claimed and the identity refused. A bloom HIT that resolves still emits
526
+ // without touching the canon route, which is what keeps the cheap path cheap.
527
+ const flat = flatProbe(start, end);
528
+ let id = flat === null ? null : resolveSpan(start, end);
529
+ if (id === null) {
530
+ if (flat !== null && ctx.meter)
531
+ ctx.meter.bloomFalsePositives++;
522
532
  if (!canonBudget) {
523
533
  if (ctx.meter)
524
534
  ctx.meter.canonProbesDenied++;
@@ -526,8 +536,8 @@ function recogniseImpl(ctx, bytes) {
526
536
  }
527
537
  if (!canonAdmits(start, end))
528
538
  return false;
539
+ id = resolveSpan(start, end);
529
540
  }
530
- const id = resolveSpan(start, end);
531
541
  if (id === null)
532
542
  return false;
533
543
  emit(start, end, id);
@@ -600,12 +610,21 @@ function recogniseImpl(ctx, bytes) {
600
610
  // keeps this off the quadratic path the budget note above describes (that
601
611
  // one had no span bound at all).
602
612
  {
603
- // The span bound is W^2, the chain's own limit, PLUS the slack the endpoint set already grants: every endpoint
604
- // sits within `radius` of a cut, so a pair that names one form may straddle cuts and still be a single form's
605
- // span. Measured on the composite fixture: W=4 (reach 16), radius 8, and the useful [7,32) is 25 bytes with its
606
- // edges 2 bytes from cuts 5 and 30 — already IN `ordered`, and excluded only by the upper bound. Both terms are
613
+ // The span bound is the chain's own limit PLUS the slack the endpoint set already grants: every endpoint sits
614
+ // within `radius` of a cut, so a pair that names one form may straddle cuts and still be a single form's span.
615
+ // Measured on the composite fixture: W=4 (reach 16), radius 8, and the useful [7,32) is 25 bytes with its edges
616
+ // 2 bytes from cuts 5 and 30 — already IN `ordered`, and excluded only by the upper bound. Both terms are
607
617
  // derived (W and the seat count); no new constant enters.
608
- const reach = chainReach(W) + 2 * radius;
618
+ //
619
+ // The chain term is W^2 * W = W^4, for a form embedded in the MIDDLE of a longer query. Measured on the
620
+ // trained corpus, the SAME real contexts in three positions: opening 12/17, MIDDLE 0/17, end 12/17. The two
621
+ // edge scans reach only prefixes and suffixes, so the interior pass is the only tier that could name it, and
622
+ // at W^2 it could not. WHAT KEEPS THIS LINEAR IS THE BOUND BEING A CONSTANT — each endpoint pairs only with
623
+ // the partners inside a fixed window, so the work stays O(n). Measured against a cut-pair enumeration with
624
+ // ±W trims (an earlier, 43-line attempt): that one reached the same 136 B at 70 888 probes, this one at
625
+ // 84 169 (x1.19 more) — and BOTH pass test/14, which gates the CLASS (linear) and not the constant. The
626
+ // simpler form won on that measurement.
627
+ const reach = chainReach(W) * W * W + 2 * radius;
609
628
  if (ctx.meter)
610
629
  ctx.meter.recogniseInteriorGaps += ordered.length;
611
630
  for (const end of ordered) {
@@ -620,59 +639,6 @@ function recogniseImpl(ctx, bytes) {
620
639
  spend(start, end);
621
640
  }
622
641
  }
623
- // ── THE MIDDLE WAS BLIND: a CUT-PAIR probe, BOUNDED so it stays LINEAR ──────────────
624
- // A stored form embedded in the MIDDLE of a longer query was named by NO tier: the two
625
- // edge scans probe only prefixes and suffixes, and the loop above is capped at
626
- // `reach` = W^2 + 2*radius. Measured on the trained corpus, the SAME real contexts in
627
- // three positions: opening 12/17, MIDDLE 0/17, end 12/17. Both edges of such a form sit
628
- // within `radius` of CUTS, so a cut pair plus a small trim names the form exactly
629
- // (measured: a 64 B form at [27, 91) between cuts 26 and 90, trims +1/+1).
630
- //
631
- // THE BOUND IS WHAT KEEPS IT LINEAR. An unbounded cut-pair scan is O(cuts^2) probes, and
632
- // test/14 rejected exactly that (47 324 ms). Bounded by a CONSTANT span it is
633
- // O(cuts * const) = O(n), like the loop above — the constant is merely larger. The bound
634
- // is DERIVED, never pinned: `chainReach(W)` is W^2, "the deepest two-level composite the
635
- // write side's windows can spell" (canonical.ts), so `chainReach(W) * W * W` is the
636
- // four-level one; `chainReach(W) * W` is already used in bridge.ts.
637
- //
638
- // THE TRIMS ARE THE SAME DISCIPLINE THE EDGE SCANS USE — the suffix scan already probes
639
- // `spend(s, bytes.length - 1)`, so ±W is a wider version of an existing rule, not a new
640
- // threshold. Measured price/reach; all four configurations pass test/14, which gates the
641
- // CLASS (linear) and not the constant:
642
- // W^3 + ±1 -> 64 B, 8 068 probes (x2.02)
643
- // W^4 + ±W -> 136 B, 70 888 probes (x6.06) <- this one
644
- // W^3 + ±W and W^4 + ±1 are DOMINATED: each stops at the other parameter and costs more.
645
- // The loop above stays, so no candidate that produces a site today is lost. Suite
646
- // 707/707, and a differential over 24 real corpus questions is byte-identical (0 answers
647
- // changed, 0 new duplicate sites): the :258 warning that a wider bound can rediscover a
648
- // smaller subtree's content as a second, overlapping site was read and measured, and it
649
- // does not materialise here.
650
- const deepReach = chainReach(W) * W * W;
651
- for (let ci = 0; ci + 1 < startList.length; ci++) {
652
- for (let cj = ci + 1; cj < startList.length; cj++) {
653
- if (startList[cj] - startList[ci] > deepReach + 2 * radius)
654
- break;
655
- // The EXACT cut pair first. Measured: with the 81 trims starting at -W the `spend`
656
- // pool ran dry (canonProbesDenied in the millions) and the canon route was then denied
657
- // to the candidates that needed it — including forms the byte-exact route SEES
658
- // (`flatProbe` true) that were still not named. Probing (ci, cj) before any trim puts
659
- // the common case in front of the famine.
660
- if (ctx.meter)
661
- ctx.meter.recogniseInteriorPairs++;
662
- spend(startList[ci], startList[cj]);
663
- for (let dl = -W; dl <= W; dl++) {
664
- for (let dr = -W; dr <= W; dr++) {
665
- const a = startList[ci] + dl;
666
- const z = startList[cj] + dr;
667
- if (a < 0 || z > bytes.length || z - a < W)
668
- continue;
669
- if (ctx.meter)
670
- ctx.meter.recogniseInteriorPairs++;
671
- spend(a, z);
672
- }
673
- }
674
- }
675
- }
676
642
  }
677
643
  }
678
644
  }
package/jsr.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "$schema": "https://jsr.io/schema/config-file.v1.json",
3
3
  "name": "@hviana/sema",
4
- "version": "0.8.9",
4
+ "version": "0.9.0",
5
5
  "exports": "./src/index.ts"
6
6
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@hviana/sema",
3
- "version": "0.8.9",
3
+ "version": "0.9.0",
4
4
  "description": "Sema: a non-parametric, instance-based reasoning system.",
5
5
  "repository": {
6
6
  "type": "git",
package/src/meter.ts CHANGED
@@ -252,6 +252,13 @@ export class Meter {
252
252
  * route was priced out. Counted where the fact happens (the `!canonBudget`
253
253
  * refusal), not where the probe is called. */
254
254
  canonProbesDenied = 0;
255
+ /** Spans the BLOOM claimed and the exact identity then refused — so `probe` used to
256
+ * drop the span AND skip the decider. `findFlatBranch` is bloom-gated on purpose (a
257
+ * miss costs no database read), but a non-null answer means MAYBE; treating it as YES
258
+ * is what made the canon route run only `if (flatProbe === null)`. Counted where the
259
+ * fact happens (a non-null bloom whose `resolveSpan` is null), so the price of letting
260
+ * the canon route see those spans again is a measurement rather than a guess. */
261
+ bloomFalsePositives = 0;
255
262
  /** The pipeline's remainder AT THE DECISION POINT, in bytes: what the grounded
256
263
  * answer plus the pre-computed spans left unexplained, after the same W floor
257
264
  * the fuse gate uses. This is the quantity that licenses (or refuses) the
@@ -534,14 +534,23 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
534
534
  // pass below spends the same budget on those pairs. Returns whether it
535
535
  // emitted, so a caller can retry a trimmed edge on the miss path only.
536
536
  if (end - start < W) return false;
537
- if (flatProbe(start, end) === null) {
537
+ // The byte-exact route is a BLOOM, so a non-null answer means MAYBE: it can neither
538
+ // decide (that is `resolveSpan`'s job) nor deny (that is `canonAdmits`'). Reading it
539
+ // as a YES made a false positive both drop the span and skip the decider, because the
540
+ // canon route ran only on a null. Measured on the trained corpus: 280 spans where the
541
+ // bloom claimed and the identity refused. A bloom HIT that resolves still emits
542
+ // without touching the canon route, which is what keeps the cheap path cheap.
543
+ const flat = flatProbe(start, end);
544
+ let id = flat === null ? null : resolveSpan(start, end);
545
+ if (id === null) {
546
+ if (flat !== null && ctx.meter) ctx.meter.bloomFalsePositives++;
538
547
  if (!canonBudget) {
539
548
  if (ctx.meter) ctx.meter.canonProbesDenied++;
540
549
  return false;
541
550
  }
542
551
  if (!canonAdmits(start, end)) return false;
552
+ id = resolveSpan(start, end);
543
553
  }
544
- const id = resolveSpan(start, end);
545
554
  if (id === null) return false;
546
555
  emit(start, end, id);
547
556
  return true;
@@ -610,12 +619,21 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
610
619
  // keeps this off the quadratic path the budget note above describes (that
611
620
  // one had no span bound at all).
612
621
  {
613
- // The span bound is W^2, the chain's own limit, PLUS the slack the endpoint set already grants: every endpoint
614
- // sits within `radius` of a cut, so a pair that names one form may straddle cuts and still be a single form's
615
- // span. Measured on the composite fixture: W=4 (reach 16), radius 8, and the useful [7,32) is 25 bytes with its
616
- // edges 2 bytes from cuts 5 and 30 — already IN `ordered`, and excluded only by the upper bound. Both terms are
622
+ // The span bound is the chain's own limit PLUS the slack the endpoint set already grants: every endpoint sits
623
+ // within `radius` of a cut, so a pair that names one form may straddle cuts and still be a single form's span.
624
+ // Measured on the composite fixture: W=4 (reach 16), radius 8, and the useful [7,32) is 25 bytes with its edges
625
+ // 2 bytes from cuts 5 and 30 — already IN `ordered`, and excluded only by the upper bound. Both terms are
617
626
  // derived (W and the seat count); no new constant enters.
618
- const reach = chainReach(W) + 2 * radius;
627
+ //
628
+ // The chain term is W^2 * W = W^4, for a form embedded in the MIDDLE of a longer query. Measured on the
629
+ // trained corpus, the SAME real contexts in three positions: opening 12/17, MIDDLE 0/17, end 12/17. The two
630
+ // edge scans reach only prefixes and suffixes, so the interior pass is the only tier that could name it, and
631
+ // at W^2 it could not. WHAT KEEPS THIS LINEAR IS THE BOUND BEING A CONSTANT — each endpoint pairs only with
632
+ // the partners inside a fixed window, so the work stays O(n). Measured against a cut-pair enumeration with
633
+ // ±W trims (an earlier, 43-line attempt): that one reached the same 136 B at 70 888 probes, this one at
634
+ // 84 169 (x1.19 more) — and BOTH pass test/14, which gates the CLASS (linear) and not the constant. The
635
+ // simpler form won on that measurement.
636
+ const reach = chainReach(W) * W * W + 2 * radius;
619
637
  if (ctx.meter) ctx.meter.recogniseInteriorGaps += ordered.length;
620
638
  for (const end of ordered) {
621
639
  for (const start of ordered) {
@@ -626,55 +644,6 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
626
644
  spend(start, end);
627
645
  }
628
646
  }
629
- // ── THE MIDDLE WAS BLIND: a CUT-PAIR probe, BOUNDED so it stays LINEAR ──────────────
630
- // A stored form embedded in the MIDDLE of a longer query was named by NO tier: the two
631
- // edge scans probe only prefixes and suffixes, and the loop above is capped at
632
- // `reach` = W^2 + 2*radius. Measured on the trained corpus, the SAME real contexts in
633
- // three positions: opening 12/17, MIDDLE 0/17, end 12/17. Both edges of such a form sit
634
- // within `radius` of CUTS, so a cut pair plus a small trim names the form exactly
635
- // (measured: a 64 B form at [27, 91) between cuts 26 and 90, trims +1/+1).
636
- //
637
- // THE BOUND IS WHAT KEEPS IT LINEAR. An unbounded cut-pair scan is O(cuts^2) probes, and
638
- // test/14 rejected exactly that (47 324 ms). Bounded by a CONSTANT span it is
639
- // O(cuts * const) = O(n), like the loop above — the constant is merely larger. The bound
640
- // is DERIVED, never pinned: `chainReach(W)` is W^2, "the deepest two-level composite the
641
- // write side's windows can spell" (canonical.ts), so `chainReach(W) * W * W` is the
642
- // four-level one; `chainReach(W) * W` is already used in bridge.ts.
643
- //
644
- // THE TRIMS ARE THE SAME DISCIPLINE THE EDGE SCANS USE — the suffix scan already probes
645
- // `spend(s, bytes.length - 1)`, so ±W is a wider version of an existing rule, not a new
646
- // threshold. Measured price/reach; all four configurations pass test/14, which gates the
647
- // CLASS (linear) and not the constant:
648
- // W^3 + ±1 -> 64 B, 8 068 probes (x2.02)
649
- // W^4 + ±W -> 136 B, 70 888 probes (x6.06) <- this one
650
- // W^3 + ±W and W^4 + ±1 are DOMINATED: each stops at the other parameter and costs more.
651
- // The loop above stays, so no candidate that produces a site today is lost. Suite
652
- // 707/707, and a differential over 24 real corpus questions is byte-identical (0 answers
653
- // changed, 0 new duplicate sites): the :258 warning that a wider bound can rediscover a
654
- // smaller subtree's content as a second, overlapping site was read and measured, and it
655
- // does not materialise here.
656
- const deepReach = chainReach(W) * W * W;
657
- for (let ci = 0; ci + 1 < startList.length; ci++) {
658
- for (let cj = ci + 1; cj < startList.length; cj++) {
659
- if (startList[cj] - startList[ci] > deepReach + 2 * radius) break;
660
- // The EXACT cut pair first. Measured: with the 81 trims starting at -W the `spend`
661
- // pool ran dry (canonProbesDenied in the millions) and the canon route was then denied
662
- // to the candidates that needed it — including forms the byte-exact route SEES
663
- // (`flatProbe` true) that were still not named. Probing (ci, cj) before any trim puts
664
- // the common case in front of the famine.
665
- if (ctx.meter) ctx.meter.recogniseInteriorPairs++;
666
- spend(startList[ci], startList[cj]);
667
- for (let dl = -W; dl <= W; dl++) {
668
- for (let dr = -W; dr <= W; dr++) {
669
- const a = startList[ci] + dl;
670
- const z = startList[cj] + dr;
671
- if (a < 0 || z > bytes.length || z - a < W) continue;
672
- if (ctx.meter) ctx.meter.recogniseInteriorPairs++;
673
- spend(a, z);
674
- }
675
- }
676
- }
677
- }
678
647
  }
679
648
  }
680
649
  }