@hviana/sema 0.8.1 → 0.8.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/src/config.d.ts +17 -0
- package/dist/src/config.js +18 -0
- package/dist/src/meter.d.ts +25 -0
- package/dist/src/meter.js +44 -0
- package/dist/src/mind/corpus.d.ts +40 -0
- package/dist/src/mind/corpus.js +149 -0
- package/dist/src/mind/graph-search.d.ts +7 -0
- package/dist/src/mind/graph-search.js +235 -24
- package/dist/src/mind/index.d.ts +3 -1
- package/dist/src/mind/index.js +1 -0
- package/dist/src/mind/match.d.ts +8 -3
- package/dist/src/mind/match.js +142 -58
- package/dist/src/mind/mechanisms/cast.js +18 -2
- package/dist/src/mind/mechanisms/cover.js +6 -0
- package/dist/src/mind/mind.d.ts +55 -0
- package/dist/src/mind/mind.js +72 -2
- package/dist/src/mind/pipeline.js +25 -6
- package/dist/src/mind/reasoning.d.ts +5 -1
- package/dist/src/mind/reasoning.js +54 -1
- package/dist/src/mind/traverse.js +9 -1
- package/dist/src/mind/types.d.ts +22 -0
- package/docs/failures/tempting-but-wrong.md +31 -2
- package/jsr.json +1 -1
- package/package.json +1 -1
- package/src/config.ts +35 -0
- package/src/meter.ts +47 -0
- package/src/mind/corpus.ts +202 -0
- package/src/mind/graph-search.ts +252 -23
- package/src/mind/index.ts +8 -1
- package/src/mind/match.ts +143 -54
- package/src/mind/mechanisms/cast.ts +17 -1
- package/src/mind/mechanisms/cover.ts +5 -0
- package/src/mind/mind.ts +123 -0
- package/src/mind/pipeline.ts +30 -6
- package/src/mind/reasoning.ts +55 -0
- package/src/mind/traverse.ts +9 -1
- package/src/mind/types.ts +26 -0
- package/test/100-complete-grounding-trace.test.mjs +109 -0
- package/test/101-alignment-gap-bound.test.mjs +106 -0
- package/test/102-production-composes-at-scale.test.mjs +110 -0
- package/test/103-alignment-gap-budget.test.mjs +89 -0
- package/test/104-composition-is-reported.test.mjs +90 -0
- package/test/105-derive-through-reports-its-refusal.test.mjs +113 -0
- package/test/106-the-join-fires.test.mjs +94 -0
- package/test/107-the-join-is-counted.test.mjs +81 -0
- package/test/108-the-join-chains.test.mjs +78 -0
- package/test/109-the-pivot-is-counted.test.mjs +60 -0
- package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
- package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
- package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
- package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
- package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
- package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
- package/test/117-corpus-search.test.mjs +171 -0
- package/test/14-scaling.test.mjs +10 -7
- package/test/76-reference-binding.test.mjs +6 -1
- package/test/89-completion-recursion.test.mjs +30 -5
package/dist/src/config.d.ts
CHANGED
|
@@ -100,6 +100,23 @@ export interface MindConfig {
|
|
|
100
100
|
seed: number;
|
|
101
101
|
recallQueryK: number;
|
|
102
102
|
haloQueryK: number;
|
|
103
|
+
/** Corpus reading (see src/mind/corpus.ts): results per call, resolved
|
|
104
|
+
* nodes climbed from, contexts requested per climb, probes used to stride
|
|
105
|
+
* the id space when browsing, bytes of each side a preview keeps, and the
|
|
106
|
+
* smallest deposited note browsing will show. Capacities and budgets only —
|
|
107
|
+
* the one material floor (a resolved node must account for W bytes) is
|
|
108
|
+
* derived from the geometry, not declared here. */
|
|
109
|
+
corpusLimitMax: number;
|
|
110
|
+
corpusClimbs: number;
|
|
111
|
+
corpusContextsPerClimb: number;
|
|
112
|
+
corpusSampleProbes: number;
|
|
113
|
+
corpusPreviewBytes: number;
|
|
114
|
+
corpusSampleFloorBytes: number;
|
|
115
|
+
/** Items one rationale step may ITEMISE (the whole field is still counted in
|
|
116
|
+
* the step's note). A capacity of the rationale, not of recall: sharing
|
|
117
|
+
* `recallQueryK` meant `new Mind({recallQueryK: 100000})` un-bounded the very
|
|
118
|
+
* payload the bound exists for (found by an adversarial review). */
|
|
119
|
+
rationaleSampleK: number;
|
|
103
120
|
normalizeEpsilon: number;
|
|
104
121
|
cosineEpsilon: number;
|
|
105
122
|
alu: AluConfig;
|
package/dist/src/config.js
CHANGED
|
@@ -5,6 +5,13 @@ export const DEFAULT_CONFIG = {
|
|
|
5
5
|
seed: 42,
|
|
6
6
|
recallQueryK: 12,
|
|
7
7
|
haloQueryK: 12,
|
|
8
|
+
rationaleSampleK: 12,
|
|
9
|
+
corpusLimitMax: 24,
|
|
10
|
+
corpusClimbs: 24,
|
|
11
|
+
corpusContextsPerClimb: 6,
|
|
12
|
+
corpusSampleProbes: 6000,
|
|
13
|
+
corpusPreviewBytes: 220,
|
|
14
|
+
corpusSampleFloorBytes: 12,
|
|
8
15
|
normalizeEpsilon: 1e-12,
|
|
9
16
|
cosineEpsilon: 1e-12,
|
|
10
17
|
alu: {
|
|
@@ -44,6 +51,17 @@ export function resolveConfig(opts = {}) {
|
|
|
44
51
|
seed: opts.seed ?? DEFAULT_CONFIG.seed,
|
|
45
52
|
recallQueryK: opts.recallQueryK ?? DEFAULT_CONFIG.recallQueryK,
|
|
46
53
|
haloQueryK: opts.haloQueryK ?? DEFAULT_CONFIG.haloQueryK,
|
|
54
|
+
rationaleSampleK: opts.rationaleSampleK ?? DEFAULT_CONFIG.rationaleSampleK,
|
|
55
|
+
corpusLimitMax: opts.corpusLimitMax ?? DEFAULT_CONFIG.corpusLimitMax,
|
|
56
|
+
corpusClimbs: opts.corpusClimbs ?? DEFAULT_CONFIG.corpusClimbs,
|
|
57
|
+
corpusContextsPerClimb: opts.corpusContextsPerClimb ??
|
|
58
|
+
DEFAULT_CONFIG.corpusContextsPerClimb,
|
|
59
|
+
corpusSampleProbes: opts.corpusSampleProbes ??
|
|
60
|
+
DEFAULT_CONFIG.corpusSampleProbes,
|
|
61
|
+
corpusPreviewBytes: opts.corpusPreviewBytes ??
|
|
62
|
+
DEFAULT_CONFIG.corpusPreviewBytes,
|
|
63
|
+
corpusSampleFloorBytes: opts.corpusSampleFloorBytes ??
|
|
64
|
+
DEFAULT_CONFIG.corpusSampleFloorBytes,
|
|
47
65
|
normalizeEpsilon: opts.normalizeEpsilon ?? DEFAULT_CONFIG.normalizeEpsilon,
|
|
48
66
|
cosineEpsilon: opts.cosineEpsilon ?? DEFAULT_CONFIG.cosineEpsilon,
|
|
49
67
|
alu: {
|
package/dist/src/meter.d.ts
CHANGED
|
@@ -152,6 +152,31 @@ export declare class Meter {
|
|
|
152
152
|
mechanismRuns: number;
|
|
153
153
|
/** Candidates the decider weighed. */
|
|
154
154
|
candidates: number;
|
|
155
|
+
/** `deriveThrough` yielded — a fact was reached through the subject the query
|
|
156
|
+
* never named. */
|
|
157
|
+
joinFired: number;
|
|
158
|
+
/** Refused: no key names the entity and the tail together. (A key that
|
|
159
|
+
* resolves but leads nowhere is not "refused" — it is not the relation, so
|
|
160
|
+
* the scan simply moves on; there is no counter for a case the loop cannot
|
|
161
|
+
* reach.) */
|
|
162
|
+
joinNoKey: number;
|
|
163
|
+
/** Refused: the fact contains no entity that leads anywhere. */
|
|
164
|
+
joinNoEntity: number;
|
|
165
|
+
/** Times the reasoner pivoted on a span its answer contains and stepped
|
|
166
|
+
* across that fact. */
|
|
167
|
+
pivotSteps: number;
|
|
168
|
+
/** `bridge` calls the cover makes assembling connectors (pairwise + n-ary). */
|
|
169
|
+
coverBridges: number;
|
|
170
|
+
/** Continuations a CHAIN hop offered the search. Bounded by the question
|
|
171
|
+
* (`ceil(queryLen / W)`) rather than by the corpus's fan-out — measured on a
|
|
172
|
+
* hub of degree 1083, offering every continuation grew the chart to 3113 outs
|
|
173
|
+
* and cost a 270 MB peak / 256 MB OOM for a two-word question. */
|
|
174
|
+
chainOffers: number;
|
|
175
|
+
/** Σ byte-allowance the n-ary interior passes those bridges. The allowance
|
|
176
|
+
* is `middleBytes + (m + 1) * W` — every intermediate answer's bytes plus
|
|
177
|
+
* one window of glue per joint — so it is the quantity that grows with a hub
|
|
178
|
+
* query's answers, and the first thing to read when the peak moves. */
|
|
179
|
+
coverAllowanceBytes: number;
|
|
155
180
|
private readonly _phases;
|
|
156
181
|
private readonly _t0;
|
|
157
182
|
/** Every work counter's current value, by name — the snapshot `time`
|
package/dist/src/meter.js
CHANGED
|
@@ -149,6 +149,50 @@ export class Meter {
|
|
|
149
149
|
mechanismRuns = 0;
|
|
150
150
|
/** Candidates the decider weighed. */
|
|
151
151
|
candidates = 0;
|
|
152
|
+
// ── Graph search: the fact join (DIRECTION) ─────────────────────────────
|
|
153
|
+
//
|
|
154
|
+
// The join's outcome was observable ONLY through the rationale, and the
|
|
155
|
+
// rationale PERTURBS the search (measured: appending text to a refusal note
|
|
156
|
+
// changed a traced answer). These four counters are the untraced view — the
|
|
157
|
+
// same surface every other work counter uses, incremented where the decision
|
|
158
|
+
// is made, never behind a trace guard.
|
|
159
|
+
/** `deriveThrough` yielded — a fact was reached through the subject the query
|
|
160
|
+
* never named. */
|
|
161
|
+
joinFired = 0;
|
|
162
|
+
/** Refused: no key names the entity and the tail together. (A key that
|
|
163
|
+
* resolves but leads nowhere is not "refused" — it is not the relation, so
|
|
164
|
+
* the scan simply moves on; there is no counter for a case the loop cannot
|
|
165
|
+
* reach.) */
|
|
166
|
+
joinNoKey = 0;
|
|
167
|
+
/** Refused: the fact contains no entity that leads anywhere. */
|
|
168
|
+
joinNoEntity = 0;
|
|
169
|
+
// ── Mind: the multi-hop pivot (EXTENSION) ───────────────────────────────
|
|
170
|
+
//
|
|
171
|
+
// `pivotStep` was observable only through the rationale, and the rationale
|
|
172
|
+
// perturbs the search (measured). How far the reasoner hopped is a
|
|
173
|
+
// BEHAVIOUR, so it needs an untraced view: one counter, incremented where the
|
|
174
|
+
// step is emitted.
|
|
175
|
+
/** Times the reasoner pivoted on a span its answer contains and stepped
|
|
176
|
+
* across that fact. */
|
|
177
|
+
pivotSteps = 0;
|
|
178
|
+
// ── Mind: the cover's connector assembly (LIMIT) ────────────────────────
|
|
179
|
+
//
|
|
180
|
+
// The cover's `run` is 91% of a hub query's time (`"Hello."`: 2.7 s of 3.0 s)
|
|
181
|
+
// and holds its ~270 MB peak, and none of it was countable: `searchPushes`
|
|
182
|
+
// and `candidates` do not see the connector assembly. These two counters are
|
|
183
|
+
// the untraced view of it.
|
|
184
|
+
/** `bridge` calls the cover makes assembling connectors (pairwise + n-ary). */
|
|
185
|
+
coverBridges = 0;
|
|
186
|
+
/** Continuations a CHAIN hop offered the search. Bounded by the question
|
|
187
|
+
* (`ceil(queryLen / W)`) rather than by the corpus's fan-out — measured on a
|
|
188
|
+
* hub of degree 1083, offering every continuation grew the chart to 3113 outs
|
|
189
|
+
* and cost a 270 MB peak / 256 MB OOM for a two-word question. */
|
|
190
|
+
chainOffers = 0;
|
|
191
|
+
/** Σ byte-allowance the n-ary interior passes those bridges. The allowance
|
|
192
|
+
* is `middleBytes + (m + 1) * W` — every intermediate answer's bytes plus
|
|
193
|
+
* one window of glue per joint — so it is the quantity that grows with a hub
|
|
194
|
+
* query's answers, and the first thing to read when the peak moves. */
|
|
195
|
+
coverAllowanceBytes = 0;
|
|
152
196
|
// ── Phases ──────────────────────────────────────────────────────────────
|
|
153
197
|
_phases = new Map();
|
|
154
198
|
_t0 = performance.now();
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
import type { MindContext } from "./types.js";
|
|
2
|
+
/** One stored experience pair, as bytes. */
|
|
3
|
+
export interface CorpusPair {
|
|
4
|
+
context: Uint8Array;
|
|
5
|
+
continuation: Uint8Array;
|
|
6
|
+
contextId: number;
|
|
7
|
+
continuationId: number;
|
|
8
|
+
/** Bytes of the query this pair was matched on — 0 when browsing. */
|
|
9
|
+
matchedBytes: number;
|
|
10
|
+
/** True when the stored bytes ran past the declared preview capacity. */
|
|
11
|
+
contextTruncated: boolean;
|
|
12
|
+
continuationTruncated: boolean;
|
|
13
|
+
}
|
|
14
|
+
/** Why a search produced no pairs. A STATE, so a caller's own layer can say it
|
|
15
|
+
* in its own words — the byte layer does not speak. */
|
|
16
|
+
export type CorpusMiss = "matched" | "nothing-resolved" | "no-continuations";
|
|
17
|
+
export interface CorpusResult {
|
|
18
|
+
pairs: CorpusPair[];
|
|
19
|
+
/** Query subtrees that content-addressed to a real stored node. */
|
|
20
|
+
resolved: number;
|
|
21
|
+
/** Distinct edge-bearing contexts the climb reached. */
|
|
22
|
+
reached: number;
|
|
23
|
+
/** Distinct contexts that carry a learnt continuation, store-wide. */
|
|
24
|
+
totalContexts: number;
|
|
25
|
+
/** True when these are browse samples rather than search results. */
|
|
26
|
+
browsed: boolean;
|
|
27
|
+
miss: CorpusMiss;
|
|
28
|
+
}
|
|
29
|
+
/** Which stored notes does this query reach? BYTES in, BYTES out.
|
|
30
|
+
*
|
|
31
|
+
* Exact content addressing through the machinery that already exists: the
|
|
32
|
+
* query's recognised sites are the resolved subtrees, the climb goes up from
|
|
33
|
+
* the biggest first, and a pair is a context that carries a continuation. */
|
|
34
|
+
export declare function searchCorpus(ctx: MindContext, queryBytes: Uint8Array, limit?: number): CorpusResult;
|
|
35
|
+
/** Browse real pairs, striding the id space so the sample is spread rather than
|
|
36
|
+
* one local cluster. DETERMINISTIC: `from` is the caller's own offset, so
|
|
37
|
+
* browsing twice with different offsets shows different notes without a random
|
|
38
|
+
* draw (the demo drew `Math.random()`, which the engine cannot do — same seed,
|
|
39
|
+
* same order, same query must mean the same answer). */
|
|
40
|
+
export declare function sampleCorpus(ctx: MindContext, limit?: number, from?: number): CorpusResult;
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
// corpus.ts — read the trained memory back out of the DAG, AS DATA.
|
|
2
|
+
//
|
|
3
|
+
// A trained experience pair IS one continuation edge: `src` is the context that
|
|
4
|
+
// was deposited, `dst` is what the mind learnt follows it. Reading them back
|
|
5
|
+
// uses the store's own structure and its own indexes — no auxiliary index is
|
|
6
|
+
// built, nothing is written, and NOTHING HERE KNOWS ABOUT TEXT: this layer takes
|
|
7
|
+
// bytes and returns bytes. The text case is one helper on the Mind
|
|
8
|
+
// (`searchCorpusText`), which encodes, calls this, and decodes.
|
|
9
|
+
//
|
|
10
|
+
// WHERE EACH STAGE COMES FROM (ported from the demo's `explore.ts`, which
|
|
11
|
+
// hand-rolled its own resolution):
|
|
12
|
+
//
|
|
13
|
+
// 1. PERCEIVE, CONTENT-ADDRESS and ADMIT the query — `recognise()`, the SAME
|
|
14
|
+
// machinery an answer goes through. It returns the sites: the query spans
|
|
15
|
+
// that content-addressed to a stored node that can lead somewhere. The
|
|
16
|
+
// demo's own recursive `findLeaf`/`findBranch` walk was a second
|
|
17
|
+
// implementation of exactly this, and it is NOT ported.
|
|
18
|
+
// 2. CLIMB the structural `kid` table from each resolved site to the
|
|
19
|
+
// edge-bearing contexts above it (`edgeAncestors`, traverse.ts), weighting
|
|
20
|
+
// each context by how much query content reached it.
|
|
21
|
+
// 3. READ the continuation off the edge table (`nextFirst`).
|
|
22
|
+
//
|
|
23
|
+
// COST is set by how much of the QUERY resolves, never by the size of the
|
|
24
|
+
// store: the sites are what recognition already found, the climb is bounded by
|
|
25
|
+
// the declared `corpusClimbs`/`corpusContextsPerClimb`, and every store call is
|
|
26
|
+
// a point probe or a capped read. All work is accounted by the store's own
|
|
27
|
+
// meter hooks when a response's meter is open — there is no second instrument.
|
|
28
|
+
//
|
|
29
|
+
// WHAT THIS IS NOT. Exact content addressing, not fuzzy keyword search: a query
|
|
30
|
+
// shares results with a stored note when it shares actual chunk-aligned content
|
|
31
|
+
// with it. An arbitrary mid-word fragment resolves to nothing, and the honest
|
|
32
|
+
// answer there is "nothing matched" — which is why `sampleCorpus` exists, and
|
|
33
|
+
// why the miss is reported as a STATE rather than as prose (the text helper
|
|
34
|
+
// turns it into words).
|
|
35
|
+
import { recognise } from "./recognition.js";
|
|
36
|
+
import { edgeAncestors } from "./traverse.js";
|
|
37
|
+
/** A context node as a pair, or null when it carries no continuation. */
|
|
38
|
+
function pairOf(ctx, id, matchedBytes) {
|
|
39
|
+
const outs = ctx.store.nextFirst(id, 1);
|
|
40
|
+
if (outs.length === 0)
|
|
41
|
+
return null;
|
|
42
|
+
const cap = ctx.cfg.corpusPreviewBytes;
|
|
43
|
+
const context = ctx.store.bytesPrefix(id, cap + 1);
|
|
44
|
+
const continuation = ctx.store.bytesPrefix(outs[0], cap + 1);
|
|
45
|
+
const contextTruncated = context.length > cap;
|
|
46
|
+
const continuationTruncated = continuation.length > cap;
|
|
47
|
+
return {
|
|
48
|
+
context: contextTruncated ? context.subarray(0, cap) : context,
|
|
49
|
+
continuation: continuationTruncated
|
|
50
|
+
? continuation.subarray(0, cap)
|
|
51
|
+
: continuation,
|
|
52
|
+
contextId: id,
|
|
53
|
+
continuationId: outs[0],
|
|
54
|
+
matchedBytes,
|
|
55
|
+
contextTruncated,
|
|
56
|
+
continuationTruncated,
|
|
57
|
+
};
|
|
58
|
+
}
|
|
59
|
+
/** Which stored notes does this query reach? BYTES in, BYTES out.
|
|
60
|
+
*
|
|
61
|
+
* Exact content addressing through the machinery that already exists: the
|
|
62
|
+
* query's recognised sites are the resolved subtrees, the climb goes up from
|
|
63
|
+
* the biggest first, and a pair is a context that carries a continuation. */
|
|
64
|
+
export function searchCorpus(ctx, queryBytes, limit) {
|
|
65
|
+
const store = ctx.store;
|
|
66
|
+
const want = Math.max(1, Math.min(limit ?? ctx.cfg.corpusContextsPerClimb, ctx.cfg.corpusLimitMax));
|
|
67
|
+
// A resolved subtree must account for at least one window (W): a single
|
|
68
|
+
// character resolves against almost any store and means nothing. W is the
|
|
69
|
+
// mind's own line between chance and evidence — derived, never declared.
|
|
70
|
+
const floor = ctx.space.maxGroup;
|
|
71
|
+
const resolved = recognise(ctx, queryBytes).sites
|
|
72
|
+
.map((s) => ({ id: s.payload, len: store.contentLen(s.payload, 512) }))
|
|
73
|
+
.filter((r) => r.len >= floor);
|
|
74
|
+
// Biggest first, lowest id breaking ties: a clause is evidence, a character is
|
|
75
|
+
// noise, and equal evidence must not be decided by iteration order.
|
|
76
|
+
const byLength = resolved
|
|
77
|
+
.sort((a, b) => b.len - a.len || a.id - b.id)
|
|
78
|
+
.slice(0, ctx.cfg.corpusClimbs);
|
|
79
|
+
// Weight each context by how much query content reached it.
|
|
80
|
+
const weight = new Map();
|
|
81
|
+
for (const { id, len } of byLength) {
|
|
82
|
+
for (const root of edgeAncestors(ctx, id, ctx.cfg.corpusContextsPerClimb).roots) {
|
|
83
|
+
weight.set(root, (weight.get(root) ?? 0) + len);
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
const pairs = [];
|
|
87
|
+
const ranked = [...weight.entries()].sort((a, b) => b[1] - a[1] || a[0] - b[0]);
|
|
88
|
+
for (const [id, w] of ranked) {
|
|
89
|
+
if (pairs.length >= want)
|
|
90
|
+
break;
|
|
91
|
+
const pair = pairOf(ctx, id, w);
|
|
92
|
+
if (pair)
|
|
93
|
+
pairs.push(pair);
|
|
94
|
+
}
|
|
95
|
+
return {
|
|
96
|
+
pairs,
|
|
97
|
+
resolved: byLength.length,
|
|
98
|
+
reached: weight.size,
|
|
99
|
+
totalContexts: store.edgeSourceCount(),
|
|
100
|
+
browsed: false,
|
|
101
|
+
miss: pairs.length > 0
|
|
102
|
+
? "matched"
|
|
103
|
+
: weight.size === 0
|
|
104
|
+
? "nothing-resolved"
|
|
105
|
+
: "no-continuations",
|
|
106
|
+
};
|
|
107
|
+
}
|
|
108
|
+
/** Browse real pairs, striding the id space so the sample is spread rather than
|
|
109
|
+
* one local cluster. DETERMINISTIC: `from` is the caller's own offset, so
|
|
110
|
+
* browsing twice with different offsets shows different notes without a random
|
|
111
|
+
* draw (the demo drew `Math.random()`, which the engine cannot do — same seed,
|
|
112
|
+
* same order, same query must mean the same answer). */
|
|
113
|
+
export function sampleCorpus(ctx, limit, from = 0) {
|
|
114
|
+
const store = ctx.store;
|
|
115
|
+
const want = Math.max(1, Math.min(limit ?? ctx.cfg.corpusContextsPerClimb, ctx.cfg.corpusLimitMax));
|
|
116
|
+
const total = store.nodeCount();
|
|
117
|
+
const probes = ctx.cfg.corpusSampleProbes;
|
|
118
|
+
const floorBytes = ctx.cfg.corpusSampleFloorBytes;
|
|
119
|
+
const pairs = [];
|
|
120
|
+
// EACH CONTEXT AT MOST ONCE. Striding the id space revisits ids when the
|
|
121
|
+
// store is small relative to the probe budget (measured: a 160-node store
|
|
122
|
+
// returned the SAME pair six times for `limit: 6`), and a browse that repeats
|
|
123
|
+
// itself is not a browse. The demo had the same hole; it is invisible only on
|
|
124
|
+
// a store far larger than the probe budget.
|
|
125
|
+
const seen = new Set();
|
|
126
|
+
for (let i = 0; i < probes && pairs.length < want && total > 0; i++) {
|
|
127
|
+
const slot = (i / probes + from) % 1;
|
|
128
|
+
const id = Math.floor(slot * total);
|
|
129
|
+
if (seen.has(id))
|
|
130
|
+
continue;
|
|
131
|
+
if (!store.has(id) || !store.hasNext(id))
|
|
132
|
+
continue;
|
|
133
|
+
if (store.contentLen(id, floorBytes) < floorBytes)
|
|
134
|
+
continue;
|
|
135
|
+
const pair = pairOf(ctx, id, 0);
|
|
136
|
+
if (pair) {
|
|
137
|
+
seen.add(id);
|
|
138
|
+
pairs.push(pair);
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
return {
|
|
142
|
+
pairs,
|
|
143
|
+
resolved: 0,
|
|
144
|
+
reached: pairs.length,
|
|
145
|
+
totalContexts: store.edgeSourceCount(),
|
|
146
|
+
browsed: true,
|
|
147
|
+
miss: pairs.length > 0 ? "matched" : "no-continuations",
|
|
148
|
+
};
|
|
149
|
+
}
|
|
@@ -162,6 +162,13 @@ export declare class GraphSearch {
|
|
|
162
162
|
* recursive completion), and chooseNext (distributional-evidence edge
|
|
163
163
|
* disambiguation when a recognised form has multiple continuations). */
|
|
164
164
|
host: GraphSearchHost);
|
|
165
|
+
/** The nodes the QUERY canonically names — the same identity the store's keys
|
|
166
|
+
* were written through. A byte-exact test is not enough: the query writes
|
|
167
|
+
* `Eiffel Tower country` and the deposited node is `eiffel tower country`, so
|
|
168
|
+
* a join that filters the query's own subject by RAW bytes re-admits it —
|
|
169
|
+
* measured: that is the trap's wrong answer (`The capital of Eiffel Tower
|
|
170
|
+
* country is Berlin.`). Cached by query identity, because the search is
|
|
171
|
+
* reused across responses. */
|
|
165
172
|
private hubBound;
|
|
166
173
|
/** Explore the Sema graph for the lightest cover of the query and return its
|
|
167
174
|
* chosen spans left-to-right — WITH the derivation's total weight (the g
|