@hviana/sema 0.8.8 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/src/meter.d.ts +7 -0
- package/dist/src/meter.js +7 -0
- package/dist/src/mind/recognition.js +26 -7
- package/jsr.json +1 -1
- package/package.json +1 -1
- package/src/meter.ts +7 -0
- package/src/mind/recognition.ts +25 -7
- package/test/186-embedded-middle.test.mjs +101 -0
package/dist/src/meter.d.ts
CHANGED
|
@@ -184,6 +184,13 @@ export declare class Meter {
|
|
|
184
184
|
* route was priced out. Counted where the fact happens (the `!canonBudget`
|
|
185
185
|
* refusal), not where the probe is called. */
|
|
186
186
|
canonProbesDenied: number;
|
|
187
|
+
/** Spans the BLOOM claimed and the exact identity then refused — so `probe` used to
|
|
188
|
+
* drop the span AND skip the decider. `findFlatBranch` is bloom-gated on purpose (a
|
|
189
|
+
* miss costs no database read), but a non-null answer means MAYBE; treating it as YES
|
|
190
|
+
* is what made the canon route run only `if (flatProbe === null)`. Counted where the
|
|
191
|
+
* fact happens (a non-null bloom whose `resolveSpan` is null), so the price of letting
|
|
192
|
+
* the canon route see those spans again is a measurement rather than a guess. */
|
|
193
|
+
bloomFalsePositives: number;
|
|
187
194
|
/** The pipeline's remainder AT THE DECISION POINT, in bytes: what the grounded
|
|
188
195
|
* answer plus the pre-computed spans left unexplained, after the same W floor
|
|
189
196
|
* the fuse gate uses. This is the quantity that licenses (or refuses) the
|
package/dist/src/meter.js
CHANGED
|
@@ -194,6 +194,13 @@ export class Meter {
|
|
|
194
194
|
* route was priced out. Counted where the fact happens (the `!canonBudget`
|
|
195
195
|
* refusal), not where the probe is called. */
|
|
196
196
|
canonProbesDenied = 0;
|
|
197
|
+
/** Spans the BLOOM claimed and the exact identity then refused — so `probe` used to
|
|
198
|
+
* drop the span AND skip the decider. `findFlatBranch` is bloom-gated on purpose (a
|
|
199
|
+
* miss costs no database read), but a non-null answer means MAYBE; treating it as YES
|
|
200
|
+
* is what made the canon route run only `if (flatProbe === null)`. Counted where the
|
|
201
|
+
* fact happens (a non-null bloom whose `resolveSpan` is null), so the price of letting
|
|
202
|
+
* the canon route see those spans again is a measurement rather than a guess. */
|
|
203
|
+
bloomFalsePositives = 0;
|
|
197
204
|
/** The pipeline's remainder AT THE DECISION POINT, in bytes: what the grounded
|
|
198
205
|
* answer plus the pre-computed spans left unexplained, after the same W floor
|
|
199
206
|
* the fuse gate uses. This is the quantity that licenses (or refuses) the
|
|
@@ -518,7 +518,17 @@ function recogniseImpl(ctx, bytes) {
|
|
|
518
518
|
// emitted, so a caller can retry a trimmed edge on the miss path only.
|
|
519
519
|
if (end - start < W)
|
|
520
520
|
return false;
|
|
521
|
-
|
|
521
|
+
// The byte-exact route is a BLOOM, so a non-null answer means MAYBE: it can neither
|
|
522
|
+
// decide (that is `resolveSpan`'s job) nor deny (that is `canonAdmits`'). Reading it
|
|
523
|
+
// as a YES made a false positive both drop the span and skip the decider, because the
|
|
524
|
+
// canon route ran only on a null. Measured on the trained corpus: 280 spans where the
|
|
525
|
+
// bloom claimed and the identity refused. A bloom HIT that resolves still emits
|
|
526
|
+
// without touching the canon route, which is what keeps the cheap path cheap.
|
|
527
|
+
const flat = flatProbe(start, end);
|
|
528
|
+
let id = flat === null ? null : resolveSpan(start, end);
|
|
529
|
+
if (id === null) {
|
|
530
|
+
if (flat !== null && ctx.meter)
|
|
531
|
+
ctx.meter.bloomFalsePositives++;
|
|
522
532
|
if (!canonBudget) {
|
|
523
533
|
if (ctx.meter)
|
|
524
534
|
ctx.meter.canonProbesDenied++;
|
|
@@ -526,8 +536,8 @@ function recogniseImpl(ctx, bytes) {
|
|
|
526
536
|
}
|
|
527
537
|
if (!canonAdmits(start, end))
|
|
528
538
|
return false;
|
|
539
|
+
id = resolveSpan(start, end);
|
|
529
540
|
}
|
|
530
|
-
const id = resolveSpan(start, end);
|
|
531
541
|
if (id === null)
|
|
532
542
|
return false;
|
|
533
543
|
emit(start, end, id);
|
|
@@ -600,12 +610,21 @@ function recogniseImpl(ctx, bytes) {
|
|
|
600
610
|
// keeps this off the quadratic path the budget note above describes (that
|
|
601
611
|
// one had no span bound at all).
|
|
602
612
|
{
|
|
603
|
-
// The span bound is
|
|
604
|
-
//
|
|
605
|
-
//
|
|
606
|
-
//
|
|
613
|
+
// The span bound is the chain's own limit PLUS the slack the endpoint set already grants: every endpoint sits
|
|
614
|
+
// within `radius` of a cut, so a pair that names one form may straddle cuts and still be a single form's span.
|
|
615
|
+
// Measured on the composite fixture: W=4 (reach 16), radius 8, and the useful [7,32) is 25 bytes with its edges
|
|
616
|
+
// 2 bytes from cuts 5 and 30 — already IN `ordered`, and excluded only by the upper bound. Both terms are
|
|
607
617
|
// derived (W and the seat count); no new constant enters.
|
|
608
|
-
|
|
618
|
+
//
|
|
619
|
+
// The chain term is W^2 * W = W^4, for a form embedded in the MIDDLE of a longer query. Measured on the
|
|
620
|
+
// trained corpus, the SAME real contexts in three positions: opening 12/17, MIDDLE 0/17, end 12/17. The two
|
|
621
|
+
// edge scans reach only prefixes and suffixes, so the interior pass is the only tier that could name it, and
|
|
622
|
+
// at W^2 it could not. WHAT KEEPS THIS LINEAR IS THE BOUND BEING A CONSTANT — each endpoint pairs only with
|
|
623
|
+
// the partners inside a fixed window, so the work stays O(n). Measured against a cut-pair enumeration with
|
|
624
|
+
// ±W trims (an earlier, 43-line attempt): that one reached the same 136 B at 70 888 probes, this one at
|
|
625
|
+
// 84 169 (x1.19 more) — and BOTH pass test/14, which gates the CLASS (linear) and not the constant. The
|
|
626
|
+
// simpler form won on that measurement.
|
|
627
|
+
const reach = chainReach(W) * W * W + 2 * radius;
|
|
609
628
|
if (ctx.meter)
|
|
610
629
|
ctx.meter.recogniseInteriorGaps += ordered.length;
|
|
611
630
|
for (const end of ordered) {
|
package/jsr.json
CHANGED
package/package.json
CHANGED
package/src/meter.ts
CHANGED
|
@@ -252,6 +252,13 @@ export class Meter {
|
|
|
252
252
|
* route was priced out. Counted where the fact happens (the `!canonBudget`
|
|
253
253
|
* refusal), not where the probe is called. */
|
|
254
254
|
canonProbesDenied = 0;
|
|
255
|
+
/** Spans the BLOOM claimed and the exact identity then refused — so `probe` used to
|
|
256
|
+
* drop the span AND skip the decider. `findFlatBranch` is bloom-gated on purpose (a
|
|
257
|
+
* miss costs no database read), but a non-null answer means MAYBE; treating it as YES
|
|
258
|
+
* is what made the canon route run only `if (flatProbe === null)`. Counted where the
|
|
259
|
+
* fact happens (a non-null bloom whose `resolveSpan` is null), so the price of letting
|
|
260
|
+
* the canon route see those spans again is a measurement rather than a guess. */
|
|
261
|
+
bloomFalsePositives = 0;
|
|
255
262
|
/** The pipeline's remainder AT THE DECISION POINT, in bytes: what the grounded
|
|
256
263
|
* answer plus the pre-computed spans left unexplained, after the same W floor
|
|
257
264
|
* the fuse gate uses. This is the quantity that licenses (or refuses) the
|
package/src/mind/recognition.ts
CHANGED
|
@@ -534,14 +534,23 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
|
|
|
534
534
|
// pass below spends the same budget on those pairs. Returns whether it
|
|
535
535
|
// emitted, so a caller can retry a trimmed edge on the miss path only.
|
|
536
536
|
if (end - start < W) return false;
|
|
537
|
-
|
|
537
|
+
// The byte-exact route is a BLOOM, so a non-null answer means MAYBE: it can neither
|
|
538
|
+
// decide (that is `resolveSpan`'s job) nor deny (that is `canonAdmits`'). Reading it
|
|
539
|
+
// as a YES made a false positive both drop the span and skip the decider, because the
|
|
540
|
+
// canon route ran only on a null. Measured on the trained corpus: 280 spans where the
|
|
541
|
+
// bloom claimed and the identity refused. A bloom HIT that resolves still emits
|
|
542
|
+
// without touching the canon route, which is what keeps the cheap path cheap.
|
|
543
|
+
const flat = flatProbe(start, end);
|
|
544
|
+
let id = flat === null ? null : resolveSpan(start, end);
|
|
545
|
+
if (id === null) {
|
|
546
|
+
if (flat !== null && ctx.meter) ctx.meter.bloomFalsePositives++;
|
|
538
547
|
if (!canonBudget) {
|
|
539
548
|
if (ctx.meter) ctx.meter.canonProbesDenied++;
|
|
540
549
|
return false;
|
|
541
550
|
}
|
|
542
551
|
if (!canonAdmits(start, end)) return false;
|
|
552
|
+
id = resolveSpan(start, end);
|
|
543
553
|
}
|
|
544
|
-
const id = resolveSpan(start, end);
|
|
545
554
|
if (id === null) return false;
|
|
546
555
|
emit(start, end, id);
|
|
547
556
|
return true;
|
|
@@ -610,12 +619,21 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
|
|
|
610
619
|
// keeps this off the quadratic path the budget note above describes (that
|
|
611
620
|
// one had no span bound at all).
|
|
612
621
|
{
|
|
613
|
-
// The span bound is
|
|
614
|
-
//
|
|
615
|
-
//
|
|
616
|
-
//
|
|
622
|
+
// The span bound is the chain's own limit PLUS the slack the endpoint set already grants: every endpoint sits
|
|
623
|
+
// within `radius` of a cut, so a pair that names one form may straddle cuts and still be a single form's span.
|
|
624
|
+
// Measured on the composite fixture: W=4 (reach 16), radius 8, and the useful [7,32) is 25 bytes with its edges
|
|
625
|
+
// 2 bytes from cuts 5 and 30 — already IN `ordered`, and excluded only by the upper bound. Both terms are
|
|
617
626
|
// derived (W and the seat count); no new constant enters.
|
|
618
|
-
|
|
627
|
+
//
|
|
628
|
+
// The chain term is W^2 * W = W^4, for a form embedded in the MIDDLE of a longer query. Measured on the
|
|
629
|
+
// trained corpus, the SAME real contexts in three positions: opening 12/17, MIDDLE 0/17, end 12/17. The two
|
|
630
|
+
// edge scans reach only prefixes and suffixes, so the interior pass is the only tier that could name it, and
|
|
631
|
+
// at W^2 it could not. WHAT KEEPS THIS LINEAR IS THE BOUND BEING A CONSTANT — each endpoint pairs only with
|
|
632
|
+
// the partners inside a fixed window, so the work stays O(n). Measured against a cut-pair enumeration with
|
|
633
|
+
// ±W trims (an earlier, 43-line attempt): that one reached the same 136 B at 70 888 probes, this one at
|
|
634
|
+
// 84 169 (x1.19 more) — and BOTH pass test/14, which gates the CLASS (linear) and not the constant. The
|
|
635
|
+
// simpler form won on that measurement.
|
|
636
|
+
const reach = chainReach(W) * W * W + 2 * radius;
|
|
619
637
|
if (ctx.meter) ctx.meter.recogniseInteriorGaps += ordered.length;
|
|
620
638
|
for (const end of ordered) {
|
|
621
639
|
for (const start of ordered) {
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
// 186-embedded-middle.test.mjs — the CUT-PAIR probe of `recognise`: a stored form in the MIDDLE of a query.
|
|
2
|
+
//
|
|
3
|
+
// THE CAPABILITY. A stored form embedded in the MIDDLE of a longer question is named, and its continuation is
|
|
4
|
+
// answered. No tier reached it before: the two edge scans probe only prefixes and suffixes (`spend(0, prefixes[i])`,
|
|
5
|
+
// `spend(s, bytes.length)`), and the interior pass was capped at `reach` = W^2 + 2*radius. Measured on the trained
|
|
6
|
+
// corpus, the SAME real contexts in three positions: opening 12/17, MIDDLE 0/17, end 12/17.
|
|
7
|
+
//
|
|
8
|
+
// HOW. `startList` already holds the content-defined cuts, and both edges of such a form sit within `radius` of a
|
|
9
|
+
// cut. The probe therefore pairs CUTS — the exact pair first, then a ±W trim — under a constant span bound derived
|
|
10
|
+
// from W (`chainReach(W) * W * W`, never a pinned number). The bound is what keeps it linear: an unbounded cut-pair
|
|
11
|
+
// scan is O(cuts^2) and test/14 rejects it.
|
|
12
|
+
//
|
|
13
|
+
// WHAT THIS FILE PINS: the recognition and the ANSWER at all three positions, that an unstored form is still refused
|
|
14
|
+
// (the negative control that keeps the middle assertions from passing for the wrong reason), and the determinism of
|
|
15
|
+
// the reading. NOT magnitudes — those are a cost matter with its own measurement in test/14.
|
|
16
|
+
import test from "node:test";
|
|
17
|
+
import assert from "node:assert/strict";
|
|
18
|
+
import { Mind } from "../dist/src/index.js";
|
|
19
|
+
import { SQliteStore } from "../dist/src/store-sqlite.js";
|
|
20
|
+
import { recognise } from "../dist/src/mind/recognition.js";
|
|
21
|
+
|
|
22
|
+
const dec = new TextDecoder();
|
|
23
|
+
// 64 B, so the form is TWICE `reach` (W^2 + 2*radius) — the dead zone the cut-pair probe exists to close.
|
|
24
|
+
const FORM = "the painter was born in Verano and the river runs through Verano";
|
|
25
|
+
const CONTINUATION = "that region is Calenta";
|
|
26
|
+
const OTHER = "the harbour master";
|
|
27
|
+
const OTHER_CONTINUATION = "the harbour is at Kestrel";
|
|
28
|
+
|
|
29
|
+
async function mind() {
|
|
30
|
+
const m = new Mind({
|
|
31
|
+
seed: 7,
|
|
32
|
+
store: new SQliteStore({ path: ":memory:" }),
|
|
33
|
+
profile: true,
|
|
34
|
+
});
|
|
35
|
+
await m.ingest([
|
|
36
|
+
[FORM, CONTINUATION],
|
|
37
|
+
[OTHER, OTHER_CONTINUATION],
|
|
38
|
+
]);
|
|
39
|
+
return m;
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
function named(m, query, form) {
|
|
43
|
+
const bytes = new TextEncoder().encode(query);
|
|
44
|
+
return recognise(m, bytes).sites.some((s) =>
|
|
45
|
+
dec.decode(bytes.subarray(s.start, s.end)) === form
|
|
46
|
+
);
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
test("recognise(): a stored form EMBEDDED IN THE MIDDLE of a longer query is named", async () => {
|
|
50
|
+
const m = await mind();
|
|
51
|
+
const query = `I ask: ${OTHER}, ${FORM}, and the note is filed.`;
|
|
52
|
+
assert.equal(named(m, query, FORM), true);
|
|
53
|
+
});
|
|
54
|
+
|
|
55
|
+
test("the middle form is ANSWERED with its own continuation, not the surrounding fact's", async () => {
|
|
56
|
+
const m = await mind();
|
|
57
|
+
const answer = String(
|
|
58
|
+
await m.respondText(
|
|
59
|
+
`I ask: ${OTHER}, ${FORM}, and the note is filed.`,
|
|
60
|
+
() => {},
|
|
61
|
+
),
|
|
62
|
+
);
|
|
63
|
+
assert.match(answer, /Calenta/);
|
|
64
|
+
});
|
|
65
|
+
|
|
66
|
+
test("the same form at the OPENING and at the END is still named — the edge tiers did not regress", async () => {
|
|
67
|
+
const m = await mind();
|
|
68
|
+
assert.equal(named(m, `${FORM}, and the note is filed.`, FORM), true);
|
|
69
|
+
assert.equal(named(m, `I ask: ${OTHER}, and ${FORM}.`, FORM), true);
|
|
70
|
+
});
|
|
71
|
+
|
|
72
|
+
test("an UNSTORED form is still refused in the middle — the assertion above is not vacuous", async () => {
|
|
73
|
+
const m = await mind();
|
|
74
|
+
// Same LENGTH by construction (one word swapped), and never deposited: the middle assertions must not
|
|
75
|
+
// pass merely because some longer span happens to cover the region.
|
|
76
|
+
const unstored = FORM.replace("painter", "sculpto");
|
|
77
|
+
assert.equal(
|
|
78
|
+
unstored.length,
|
|
79
|
+
FORM.length,
|
|
80
|
+
"the control must be the same size as the form",
|
|
81
|
+
);
|
|
82
|
+
assert.notEqual(unstored, FORM);
|
|
83
|
+
assert.equal(
|
|
84
|
+
named(m, `I ask: ${OTHER}, ${unstored}, and the note is filed.`, unstored),
|
|
85
|
+
false,
|
|
86
|
+
);
|
|
87
|
+
});
|
|
88
|
+
|
|
89
|
+
test("the reading is deterministic, and the interior counter is wired", async () => {
|
|
90
|
+
const first = await mind();
|
|
91
|
+
const second = await mind();
|
|
92
|
+
const query = `I ask: ${OTHER}, ${FORM}, and the note is filed.`;
|
|
93
|
+
const a = String(await first.respondText(query, () => {}));
|
|
94
|
+
const b = String(await second.respondText(query, () => {}));
|
|
95
|
+
assert.equal(a, b);
|
|
96
|
+
const counters = first.lastCost?.counters ?? {};
|
|
97
|
+
assert.ok(
|
|
98
|
+
(counters.recogniseInteriorPairs ?? 0) > 0,
|
|
99
|
+
"recogniseInteriorPairs must be wired",
|
|
100
|
+
);
|
|
101
|
+
});
|