@hviana/sema 0.9.2 → 0.9.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/src/meter.d.ts +6 -0
- package/dist/src/meter.js +6 -0
- package/dist/src/mind/evidence.d.ts +6 -0
- package/dist/src/mind/evidence.js +19 -7
- package/dist/src/mind/mechanisms/cast.d.ts +1 -1
- package/dist/src/mind/mechanisms/cast.js +73 -12
- package/dist/src/mind/mechanisms/cover.js +4 -20
- package/dist/src/mind/mechanisms/recall.js +84 -8
- package/dist/src/mind/mechanisms/reference.js +25 -0
- package/dist/src/mind/mind.d.ts +1 -0
- package/dist/src/mind/pipeline-mechanism.js +15 -1
- package/dist/src/mind/pipeline.js +5 -4
- package/dist/src/mind/reasoning.js +28 -16
- package/dist/src/mind/resonance.d.ts +3 -2
- package/dist/src/mind/resonance.js +3 -2
- package/dist/src/mind/traverse.d.ts +52 -0
- package/dist/src/mind/traverse.js +334 -7
- package/dist/src/mind/types.d.ts +4 -1
- package/docs/INDEX.md +21 -17
- package/docs/INVARIANTS.md +1 -1
- package/docs/PHILOSOPHY.md +317 -0
- package/docs/architecture/evidence.md +172 -29
- package/jsr.json +1 -1
- package/package.json +1 -1
- package/src/meter.ts +6 -0
- package/src/mind/evidence.ts +22 -4
- package/src/mind/mechanisms/cast.ts +77 -3
- package/src/mind/mechanisms/cover.ts +6 -22
- package/src/mind/mechanisms/recall.ts +107 -9
- package/src/mind/mechanisms/reference.ts +24 -1
- package/src/mind/mind.ts +3 -1
- package/src/mind/pipeline-mechanism.ts +15 -1
- package/src/mind/pipeline.ts +5 -4
- package/src/mind/reasoning.ts +33 -15
- package/src/mind/resonance.ts +3 -2
- package/src/mind/traverse.ts +338 -6
- package/src/mind/types.ts +6 -2
- package/test/154-the-question-names-the-step.test.mjs +88 -1
- package/test/155-the-relation-read-off-another-instance.test.mjs +376 -0
- package/test/29-counterfactual.test.mjs +21 -11
- package/test/76-reference-binding.test.mjs +32 -0
|
@@ -0,0 +1,317 @@
|
|
|
1
|
+
# Philosophy — The Path of Information
|
|
2
|
+
|
|
3
|
+
`docs/INDEX.md` routes the laws. This file follows one piece of information from
|
|
4
|
+
the moment it enters Sema to the moment it is used, and says at each step what
|
|
5
|
+
happens, why it must happen that way, and what it makes possible. One principle
|
|
6
|
+
runs through the whole path:
|
|
7
|
+
|
|
8
|
+
> **Keep exactly what was given. Build ways of finding it. Decide only on what
|
|
9
|
+
> was kept. Charge for what was not found.**
|
|
10
|
+
|
|
11
|
+
Every structure below is either something kept (exact, it may decide) or a way
|
|
12
|
+
of finding (approximate, it may only propose). Holding that distinction is what
|
|
13
|
+
lets a reader predict how any part of Sema behaves.
|
|
14
|
+
|
|
15
|
+
## 1. Information enters
|
|
16
|
+
|
|
17
|
+
**What enters.** A deposit is a pair, a context and what followed it, or a bare
|
|
18
|
+
experience. That is the only relation Sema assumes of the world: _this came
|
|
19
|
+
after that_. It imposes no schema, no entities and no types. Everything else
|
|
20
|
+
must be discovered from what follows what.
|
|
21
|
+
|
|
22
|
+
**As bytes.** There is no tokenizer, no vocabulary and no language. A modality
|
|
23
|
+
is only a reading order: text is read as written, and a grid along a Hilbert
|
|
24
|
+
curve, which keeps most of the plane's locality in the stream. Whatever can be
|
|
25
|
+
read as a stream can be learnt.
|
|
26
|
+
|
|
27
|
+
**Into a tree, cut by the content itself.** A rolling hash runs over the bytes.
|
|
28
|
+
Where it vanishes, a cut falls, and the more digits it vanishes on, the higher
|
|
29
|
+
the cut's level. Higher cuts nest inside lower ones, so the stream folds into a
|
|
30
|
+
tree of about `W` children per node (`W` = 4 by default), level by level
|
|
31
|
+
(`fold-contract.md`). There are three reasons for this shape:
|
|
32
|
+
|
|
33
|
+
- **Content-defined cuts make identity independent of position.** A phrase is
|
|
34
|
+
cut the same way wherever it occurs, so in two different deposits it is the
|
|
35
|
+
same subtree. After a shift, 99.7% of cuts survive, against 14.3% for a fixed
|
|
36
|
+
grid.
|
|
37
|
+
- **A tree, not a flat list.** Recurrence happens at every scale: a word, a
|
|
38
|
+
phrase, a sentence. The hierarchy makes every scale a node, so recurrence is
|
|
39
|
+
caught wherever it happens, and an edit touches only the path from the change
|
|
40
|
+
to the root.
|
|
41
|
+
- **Nothing is imposed that a later question could not reproduce.** No turn
|
|
42
|
+
boundaries, no metadata: a question will be cut by the same rule from its own
|
|
43
|
+
bytes, and any cut it could not reproduce would make the same thing two
|
|
44
|
+
things. When the two sides disagreed, alignment went quadratic (5.2M cells
|
|
45
|
+
against 0) and turns asked verbatim stopped resolving to themselves.
|
|
46
|
+
|
|
47
|
+
**Interned by content.** Every subtree becomes a node named by what it contains:
|
|
48
|
+
a branch by its children, a short run by its bytes. The same subtree in a
|
|
49
|
+
thousand deposits is one node with a thousand parents, so the store is a DAG,
|
|
50
|
+
not a forest. Every window of `W−1` and `W` bytes is also indexed as a node of
|
|
51
|
+
its own and linked to the chunks that contain it, so a span can be found
|
|
52
|
+
whatever the fold did around it. Teaching the same thing twice creates no
|
|
53
|
+
structure.
|
|
54
|
+
|
|
55
|
+
**Linked by succession.** The context's root gets an edge to the continuation's
|
|
56
|
+
root. Each suffix of the context that is already a known form inherits that edge
|
|
57
|
+
too, and inside a bare experience the parts are chained in order. An edge is a
|
|
58
|
+
unique pair, so how many contexts lead to a continuation counts distinct
|
|
59
|
+
contexts, not repetitions.
|
|
60
|
+
|
|
61
|
+
**What this structure can answer, exactly:**
|
|
62
|
+
|
|
63
|
+
- **Is this stored?** Fold it and look the name up. One byte apart is another
|
|
64
|
+
name.
|
|
65
|
+
- **What contains this, and what is it part of?** Climb the parents and
|
|
66
|
+
containers.
|
|
67
|
+
- **What followed this?** Read its edges.
|
|
68
|
+
- **How common is this?** Count the contexts that reach it: a minority
|
|
69
|
+
discriminates, and what reaches nearly everything is scaffolding
|
|
70
|
+
(`commonality.md`).
|
|
71
|
+
- **What of a question is already known?** Recognition decomposes the question
|
|
72
|
+
into every stored form inside it that leads somewhere, with a bounded number
|
|
73
|
+
of probes per byte.
|
|
74
|
+
|
|
75
|
+
**What it cannot answer:** what is _near_. A content address destroys locality
|
|
76
|
+
on purpose, because that is what makes it a name. `colour` and `colours` are as
|
|
77
|
+
far apart, by name, as `colour` and `Zanzibar`. Everything from here on exists
|
|
78
|
+
because of that gap.
|
|
79
|
+
|
|
80
|
+
## 2. Two ways of being near
|
|
81
|
+
|
|
82
|
+
Every node also gets vectors in a high-dimensional space (a vector-symbolic
|
|
83
|
+
architecture: Plate 1995; Kanerva 2009). There are two of them, because there
|
|
84
|
+
are two independent ways for things to be near.
|
|
85
|
+
|
|
86
|
+
**The gist: nearness of form.** Each byte value has a vector from an alphabet
|
|
87
|
+
built by refinement (16 coarse directions, then 64, then 256), so neighbouring
|
|
88
|
+
values resemble each other. A node's gist binds each child to its position, by a
|
|
89
|
+
fixed permutation per seat, counted from both ends so that growth at one edge
|
|
90
|
+
keeps the other edge's coordinates, and then adds them up. The fold is linear,
|
|
91
|
+
so the cosine of two gists reads how many bytes they share in place. The gist
|
|
92
|
+
exists at birth: it is computed from the bytes alone. It puts `colour` near
|
|
93
|
+
`colours`.
|
|
94
|
+
|
|
95
|
+
**The halo: nearness of use.** Each time a fact is deposited, two pours happen:
|
|
96
|
+
|
|
97
|
+
- every newly seen part of the context receives the continuation's signature, in
|
|
98
|
+
the seat for _what it led to_;
|
|
99
|
+
- the continuation receives each part's signature, in the seat for _what led to
|
|
100
|
+
it_.
|
|
101
|
+
|
|
102
|
+
A signature is keyed on a node's identity, never on its gist. Read as whole
|
|
103
|
+
partners, company would almost never recur, since whole deposits rarely repeat.
|
|
104
|
+
So each partner contributes its own signature together with a sketch of its
|
|
105
|
+
constituents: `√D` of them, the capacity of a superposition before each term
|
|
106
|
+
falls below noise, with scaffolding excluded (`halo-sketch.md`). Two words that
|
|
107
|
+
never meet can therefore still share the _kinds_ of things around them. Each
|
|
108
|
+
episode adds one unit, by addition, so order is forgotten and proportion kept.
|
|
109
|
+
The halo is earned by experience. It puts `colour` near `hue`.
|
|
110
|
+
|
|
111
|
+
**Why two, and why kept apart.** Form and use are different axes of meaning, and
|
|
112
|
+
neither predicts the other. The halo is built from identities and never from
|
|
113
|
+
gists, so that resemblance of spelling cannot leak into resemblance of use. When
|
|
114
|
+
the two agree, it is corroboration, not echo.
|
|
115
|
+
|
|
116
|
+
**Why vectors at all, and why they never decide.** A vector restores the
|
|
117
|
+
locality the name destroyed. An index over the vectors (RaBitQ-IVF) finds the
|
|
118
|
+
near ones among millions in one query. But a vector is a lossy summary, and its
|
|
119
|
+
score is an estimate. So the space proposes and the structure decides
|
|
120
|
+
(`exact-vs-approximate.md`). The bars are set against the space's own chance:
|
|
121
|
+
random vectors are nearly orthogonal, with noise `1/√D`, so a resemblance counts
|
|
122
|
+
at `3/√D`, and two halos share a concept at `0.5 + 0.5/√D` (`thresholds.md`).
|
|
123
|
+
The single exception is the store's near-merge at deposit, bounded to one window
|
|
124
|
+
of difference.
|
|
125
|
+
|
|
126
|
+
**Two memories.** The deposits are a record: verbatim, enumerable, each with
|
|
127
|
+
provenance. The halos are a statistic accumulated over that record. This is the
|
|
128
|
+
episodic/semantic distinction (Tulving 1972) with fixed roles: **the statistic
|
|
129
|
+
proposes, the record decides.** A halo can say that two things are of a kind; it
|
|
130
|
+
never says they are the same thing, and it never establishes a fact.
|
|
131
|
+
|
|
132
|
+
**How the three readings are ordered.** Where Sema looks for a span, it tries
|
|
133
|
+
exact bytes first, then the halo's role, then the gist (`match.ts`). A form
|
|
134
|
+
counts as knowledge only if it _leads somewhere_: it has a continuation, or it
|
|
135
|
+
has a halo. Something that never led anywhere is not used.
|
|
136
|
+
|
|
137
|
+
**What the vectors add to reasoning.** They add three things:
|
|
138
|
+
|
|
139
|
+
- **Paraphrase:** a question one byte off its stored twin is found by its gist.
|
|
140
|
+
- **Substitution:** a form with no continuation of its own may borrow a halo
|
|
141
|
+
sibling's, at the price of a concept hop.
|
|
142
|
+
- **Imagination, bounded:** binding stored parts gives the gist of a whole never
|
|
143
|
+
seen, which can be searched for. That search is the most approximate tier of
|
|
144
|
+
all, and is gated hardest, because nothing about it is contained in bytes.
|
|
145
|
+
|
|
146
|
+
## 3. A question arrives
|
|
147
|
+
|
|
148
|
+
The question is folded by the same rule as every deposit, and the identity fold
|
|
149
|
+
names each part by asking the store as it folds. So perceiving a question is
|
|
150
|
+
already recognising what of it memory holds; representation and search begin as
|
|
151
|
+
one act. Recognition returns the question's _sites_: every stored form inside it
|
|
152
|
+
that leads somewhere, each with its exact identity.
|
|
153
|
+
|
|
154
|
+
## 4. Attention, built on the structure
|
|
155
|
+
|
|
156
|
+
The sites say what the question contains. They do not say what the question is
|
|
157
|
+
_about_, meaning which stored contexts its parts point to together. That is
|
|
158
|
+
attention's job, the consensus climb (`attention.ts`).
|
|
159
|
+
|
|
160
|
+
1. **Regions.** Every node of the question's tree, and every recognised site, is
|
|
161
|
+
a region.
|
|
162
|
+
2. **Lookup.** A region that is a stored form looks itself up exactly. Any other
|
|
163
|
+
region searches by gist, and pays a margin: it must beat the best rival
|
|
164
|
+
conclusion by the noise floor, scaled by how much of it is not stored.
|
|
165
|
+
3. **Climb.** Each region climbs the DAG, through parents and containers, to the
|
|
166
|
+
stored contexts it reaches. The climb stops at a named saturation when a node
|
|
167
|
+
reaches more than `√N` contexts, because past that point reading further
|
|
168
|
+
cannot discriminate (`saturation.md`).
|
|
169
|
+
4. **Weigh by rarity.** A region that reaches `c` of `N` contexts weighs
|
|
170
|
+
`ln(N/c)`. This is inverse document frequency, read over structure.
|
|
171
|
+
5. **Agree.** Independent regions add, pooled in the (+, +) semiring of the same
|
|
172
|
+
deduction engine the search uses.
|
|
173
|
+
6. **Join.** When two regions point to different contexts, the stored whole that
|
|
174
|
+
contains both is sought, by exact junction first, then through halo synonyms,
|
|
175
|
+
then by an imagined whole. Exact joint evidence explains the separate votes
|
|
176
|
+
away.
|
|
177
|
+
|
|
178
|
+
The output is a set of points of attention: stored contexts where independent
|
|
179
|
+
parts of the question agree, ranked.
|
|
180
|
+
|
|
181
|
+
Compared with a transformer's attention, which is soft content addressing over a
|
|
182
|
+
context window:
|
|
183
|
+
|
|
184
|
+
- **The keys are stored contexts,** reached by containment across the whole
|
|
185
|
+
memory, not tokens in a window. This is cross-attention, from the question
|
|
186
|
+
into memory.
|
|
187
|
+
- **The weights are rarity,** derived and not learned.
|
|
188
|
+
- **The votes are absolute.** A softmax must attend somewhere, because its
|
|
189
|
+
weights sum to one. Sema's anchors count only above floors derived from `D`
|
|
190
|
+
and `N`, so attention can come back empty.
|
|
191
|
+
- **Nothing is blended.** Mixing values would produce bytes nobody deposited.
|
|
192
|
+
Attention selects, and does not mix.
|
|
193
|
+
- **Exact evidence outranks resemblance.** Only exact evidence may explain a
|
|
194
|
+
vote away.
|
|
195
|
+
|
|
196
|
+
Attention is also spent, not assumed. The climb is the shared analysis the
|
|
197
|
+
market defers until no cheaper bound can rule it out (`mechanism-market.md`).
|
|
198
|
+
|
|
199
|
+
**Why this is the right notion of relevance.** Every judgement of relevance in
|
|
200
|
+
Sema divides a population into what it shares (frame) and what varies (filler).
|
|
201
|
+
Commonality and discrimination are the two sides of that one cut, and the answer
|
|
202
|
+
depends on the population. Sema reads three populations and never confuses them:
|
|
203
|
+
the corpus, the cohort of structures aligned to this question, and the
|
|
204
|
+
containers of a window (`commonality.md`). The rarity weighting is the graded
|
|
205
|
+
form of the cut over the corpus. Attention's deeper role is to _choose the
|
|
206
|
+
population_: its points are the cohort the next steps cut.
|
|
207
|
+
|
|
208
|
+
## 5. How it is used
|
|
209
|
+
|
|
210
|
+
**Many ways of thinking.** Each grounding mechanism reads the same material
|
|
211
|
+
differently:
|
|
212
|
+
|
|
213
|
+
- `cover` composes the question from its sites and follows their edges, by graph
|
|
214
|
+
search.
|
|
215
|
+
- CAST aligns the attention points to the question and cuts that cohort into
|
|
216
|
+
frame and filler. It then carries structure between them: substitution,
|
|
217
|
+
redirection, comparison.
|
|
218
|
+
- `confluence` intersects what independent anchors share, as long as it is not
|
|
219
|
+
scaffolding.
|
|
220
|
+
- `extraction` reads out a located frame.
|
|
221
|
+
- `reference` learns a frame from worked examples and voices its slot with the
|
|
222
|
+
asker's own bytes.
|
|
223
|
+
- `recall` takes the nearest stored form by gist.
|
|
224
|
+
- `prefix-completion` completes a known beginning.
|
|
225
|
+
- The ALU computes, and what it computes is authoritative.
|
|
226
|
+
|
|
227
|
+
**One price.** Every candidate weighs `moves + PASS·unaccounted`, and `PASS` per
|
|
228
|
+
byte of unexplained question outweighs any move (`cost-model.md`). The price is
|
|
229
|
+
not confidence. It is how much of the question remains unaccounted for, so the
|
|
230
|
+
winner is the reading that explains the most. When nothing accounts for the
|
|
231
|
+
question, silence wins, and silence is a first-class answer. An answer that is
|
|
232
|
+
only near says so.
|
|
233
|
+
|
|
234
|
+
**Reasoning that answers to the question.** The winner may be extended, step by
|
|
235
|
+
step, along succession. A step is admitted only if it closes the derivation,
|
|
236
|
+
moves to structure not yet consumed, or carries what the question still owes
|
|
237
|
+
(`closure.md`). A step counts as _named_ when the question, together with the
|
|
238
|
+
node the step stands on, witnesses one of the contexts that establish it
|
|
239
|
+
(`evidence.md`). A step the question did not name is paid from its remaining
|
|
240
|
+
debt. The question owns the inference, so a chain cannot wander off to a fact
|
|
241
|
+
nobody asked for.
|
|
242
|
+
|
|
243
|
+
**Generalization, read and never stored.** When no context is named outright,
|
|
244
|
+
attention's points may hold another instance of the question.
|
|
245
|
+
`Where was Peter Jackson born?` shares the frame of
|
|
246
|
+
`Where was the director of film Beat Girl born?`. Peter Jackson's fact is also
|
|
247
|
+
established by `Peter Jackson place of birth`, which is how the corpus spells
|
|
248
|
+
the relation. The node at hand, `Edmond T. Gréville`, is put into that frame,
|
|
249
|
+
and the result is looked up by content.
|
|
250
|
+
|
|
251
|
+
This is anti-unification (Plotkin 1970): keep what two instances share, put a
|
|
252
|
+
variable where they differ. It is admitted only under three conditions:
|
|
253
|
+
|
|
254
|
+
- the variable is a thing the store knows;
|
|
255
|
+
- two instances agree;
|
|
256
|
+
- what the corpus files under one filler is not credited to the frame.
|
|
257
|
+
|
|
258
|
+
The third condition comes from a real failure. Two people born in Wellington
|
|
259
|
+
once gave an unknown `Zorblax` the same birthplace (`test/76`), Goodman's (1955)
|
|
260
|
+
accidental generalization. The frame is never stored, and neither is any answer.
|
|
261
|
+
A conclusion kept as a deposit would become evidence for itself.
|
|
262
|
+
|
|
263
|
+
**Every answer replays.** The derivation is a hyperpath in an AND/OR hypergraph
|
|
264
|
+
whose axioms are stored nodes (`src/derive/`). Its trace replays byte for byte,
|
|
265
|
+
and ties are broken by the order of teaching, never by chance
|
|
266
|
+
(`determinism.md`). A correction erases nothing. It is a further deposit, and it
|
|
267
|
+
prevails by evidence.
|
|
268
|
+
|
|
269
|
+
## 6. The path, seen whole
|
|
270
|
+
|
|
271
|
+
| Step | What it adds | Kept or finding | It may |
|
|
272
|
+
| ----------------- | ---------------------------------------------- | --------------- | ---------------- |
|
|
273
|
+
| bytes, fold | one tree per stream, cut by content | kept | decide identity |
|
|
274
|
+
| DAG, edges | identity, parthood, succession, commonality | kept | decide |
|
|
275
|
+
| gist | nearness of form | finding | propose |
|
|
276
|
+
| halo | nearness of use | finding | propose |
|
|
277
|
+
| recognition | what of the question is known | kept | decide |
|
|
278
|
+
| attention | where the question's parts agree; a population | finding | propose |
|
|
279
|
+
| mechanisms | readings of the same material | both | offer candidates |
|
|
280
|
+
| price and closure | what remains unexplained; what may follow | kept | decide |
|
|
281
|
+
|
|
282
|
+
Representation, search and reasoning are separate modules, but not separate
|
|
283
|
+
ideas. One fold serves deposit and question. One identity serves every lookup.
|
|
284
|
+
One price serves every mechanism. One law admits every step. Everything that
|
|
285
|
+
finds is kept out of every decision, and everything that decides rests on what
|
|
286
|
+
was given.
|
|
287
|
+
|
|
288
|
+
## 7. Hypotheses the path invites
|
|
289
|
+
|
|
290
|
+
Each of these follows from a gap visible on the path, and none is a plan.
|
|
291
|
+
|
|
292
|
+
- **One cut.** Frame against filler is computed in several places: the three
|
|
293
|
+
commonality measures, `frameSlots`, `coInstanceFiller` and CAST's `depth[]`.
|
|
294
|
+
Is it one operation read over different populations, or several, as
|
|
295
|
+
`commonality.md` holds?
|
|
296
|
+
- **Form, use and identity together.** Two forms with near halos that can stand
|
|
297
|
+
for each other in every stored frame — Leibniz's substitution _salva veritate_
|
|
298
|
+
— would be one referent. That is aliases (`Clara Novello` and
|
|
299
|
+
`Clara Anastasia Novello`) seen by the record, with the halo proposing.
|
|
300
|
+
- **Imagination as hypothesis.** Binding stored parts proposes wholes never
|
|
301
|
+
seen, and today it only joins regions. Proposals made this way, then checked
|
|
302
|
+
by content, are abduction with a guaranteed verifier.
|
|
303
|
+
- **Derivations as instances.** Anti-unify derivations, not texts, and `father`
|
|
304
|
+
twice reads as `grandfather` wherever an instance shows it. Derived material
|
|
305
|
+
must never count as a new context.
|
|
306
|
+
- **Richer company.** The halo has two seats, _what it led to_ and _what led to
|
|
307
|
+
it_. What other relations of use would a seat capture, and what would they let
|
|
308
|
+
attention see?
|
|
309
|
+
- **Other streams.** Whatever has a reading order that preserves locality can
|
|
310
|
+
enter by the same path.
|
|
311
|
+
|
|
312
|
+
## References
|
|
313
|
+
|
|
314
|
+
Goodman (1955), _Fact, Fiction, and Forecast_ · Kanerva (2009), _Cognitive
|
|
315
|
+
Computation_ 1(2) · Plate (1995), _IEEE TNN_ 6(3) · Plotkin (1970), _Machine
|
|
316
|
+
Intelligence_ 5 · Tulving (1972), "Episodic and semantic memory", in
|
|
317
|
+
_Organization of Memory_.
|
|
@@ -2,8 +2,11 @@
|
|
|
2
2
|
|
|
3
3
|
> **Law:** a stored form is identified by the material at hand when every one of
|
|
4
4
|
> its bytes lies in a W-window that material holds — in any order, at any place,
|
|
5
|
-
> and wherever each piece of the material came from. A
|
|
6
|
-
>
|
|
5
|
+
> and wherever each piece of the material came from. A form witnessed in every
|
|
6
|
+
> byte but one contiguous span of content is ANOTHER INSTANCE of the same
|
|
7
|
+
> question: it says what the question's relation is, never what it asks about. A
|
|
8
|
+
> step the question did not name has to be paid for by material the question
|
|
9
|
+
> still owes.
|
|
7
10
|
|
|
8
11
|
## The operation — `src/mind/evidence.ts`
|
|
9
12
|
|
|
@@ -13,7 +16,8 @@
|
|
|
13
16
|
(which finds containers). It is exact, deterministic and linear: one index per
|
|
14
17
|
source and one probe per window of the form. A window is credited to the LAST
|
|
15
18
|
source that holds it, so source 0 (the question) is credited only with what
|
|
16
|
-
nothing else at hand supplies. A form shorter than W is never witnessed.
|
|
19
|
+
nothing else at hand supplies. A form shorter than W is never witnessed. The
|
|
20
|
+
form's own unwitnessed bytes come back as its `residue`.
|
|
17
21
|
|
|
18
22
|
## Where the material comes from
|
|
19
23
|
|
|
@@ -35,15 +39,119 @@ question nor the first hop's fact, only by both. Measured over 5,236 held-out
|
|
|
35
39
|
2WikiMultihopQA compositional questions, such a context is wholly witnessed by
|
|
36
40
|
the question alone 69 times, and by the question plus the first hop 2,153 times.
|
|
37
41
|
|
|
42
|
+
An establishing context is a DEPOSITED context: a predecessor with structural
|
|
43
|
+
parents or containers is a span interned inside bigger forms, and it inherits
|
|
44
|
+
their edges. Witnessing such a fragment named every fact it is a piece of (the
|
|
45
|
+
fragment `born?` named nine strangers' birthplaces on the 2Wiki fixture with
|
|
46
|
+
one-hop questions), so it never counts — the same structural predicate
|
|
47
|
+
`pivotInto` reads.
|
|
48
|
+
|
|
49
|
+
## Another instance of the question — `coInstanceFiller`, `byCoInstance`
|
|
50
|
+
|
|
51
|
+
The question rarely spells the relation the way the corpus does.
|
|
52
|
+
`When was the
|
|
53
|
+
director of film Jinpa born?` reaches `Pema Tseden`, whose fact is
|
|
54
|
+
established by `Pema Tseden date of birth`: no window joins `born` to
|
|
55
|
+
`date of birth`, so nothing is witnessed. What joins them is another instance of
|
|
56
|
+
the question.
|
|
57
|
+
|
|
58
|
+
A **co-instance** is a stored context that shares the question's FRAME, its
|
|
59
|
+
opening and its close byte for byte under the response's equivalence, around ONE
|
|
60
|
+
different filler, and whose filler is a thing the corpus knows: a stored context
|
|
61
|
+
with continuations of its own. `Where was Peter Jackson born?`, read against
|
|
62
|
+
`Where was the director of film Beat Girl born?`, shares `Where was` and `born?`
|
|
63
|
+
and leaves `Peter Jackson`. Three conditions make the reading exact, and each
|
|
64
|
+
was measured necessary:
|
|
65
|
+
|
|
66
|
+
- **Order.** A frame is not a bag of windows. Read order-free,
|
|
67
|
+
`Lyon is a city in France` passes for an instance of
|
|
68
|
+
`what if the capital of France were Lyon?`, which NAMES Lyon, and a
|
|
69
|
+
coincidental window inside a filler splits it.
|
|
70
|
+
- **The filler is an entity.**
|
|
71
|
+
`Explain how photosynthesis converts sunlight
|
|
72
|
+
into chemical energy.` shares
|
|
73
|
+
a frame with `Explain how photosynthesis
|
|
74
|
+
works.`, but its slot holds a
|
|
75
|
+
description of the frame's own subject, and it answers the question. The
|
|
76
|
+
filler is the longest stored context opening where the slot opens (up to one
|
|
77
|
+
window earlier: the `T` of `Taika` can sit in the question's `as t`).
|
|
78
|
+
- **The frame is unsaid.** Every frame window must be held by the material, so
|
|
79
|
+
on the walk a frame a product already said cannot name a second step.
|
|
80
|
+
`Who is the father of …?` names the second hop of
|
|
81
|
+
`Who is the father of the director of film Beat Girl?` and not the
|
|
82
|
+
grandfather. Scaffolding windows are exempt, because they are nobody's
|
|
83
|
+
evidence (`is` in `Which country Leo Mittler is from?`).
|
|
84
|
+
|
|
85
|
+
The co-instance's continuation is established by other contexts too
|
|
86
|
+
(`Peter Jackson place of birth`), and the one holding the filler spells the
|
|
87
|
+
relation the corpus's way: a RELATION FRAME, `` · `place of birth`. A frame
|
|
88
|
+
counts only where two co-instances spell it alike. One alignment agrees with
|
|
89
|
+
nothing, which is the bar `reference` holds a frame to (`MIN_INSTANCES`). The
|
|
90
|
+
exact tier's second reading (`byCoInstance`, run only when no establishing
|
|
91
|
+
context is witnessed outright) puts the node the derivation stands on in each
|
|
92
|
+
frame and looks the result up by content: `Edmond T. Gréville place of birth`
|
|
93
|
+
exists, so its continuation is named. Neither the equivalence nor the frame is
|
|
94
|
+
stored as a unit, and nothing is approximate.
|
|
95
|
+
|
|
96
|
+
**The proposals are the consensus climb's.** The climb already scores the stored
|
|
97
|
+
forms the question's regions reach, and the co-instances are among its points.
|
|
98
|
+
Its scored anchors are published on the question when it runs, and the tier
|
|
99
|
+
reads the `chainReach(W)` most corroborated of them. It climbs nothing of its
|
|
100
|
+
own. A first version climbed every question window itself and cost the 116-query
|
|
101
|
+
battery +16% climb visits for zero namings. Two consequences follow, both
|
|
102
|
+
pinned:
|
|
103
|
+
|
|
104
|
+
- A pick made before the climb read less evidence than one made after, so the
|
|
105
|
+
response's pick memo is cleared when the points arrive. Traced and untraced
|
|
106
|
+
responses agree.
|
|
107
|
+
- Recall's argument binding, which runs before the climb, asks for it first when
|
|
108
|
+
the argument has several continuations and the question names none outright.
|
|
109
|
+
That is exactly where the choice would otherwise be blind.
|
|
110
|
+
|
|
111
|
+
A question the corpus holds verbatim is its own instance and reads no frames.
|
|
112
|
+
Frames are read once per question (and once per walk material), and every node
|
|
113
|
+
pays only the exact lookups.
|
|
114
|
+
|
|
115
|
+
The same reading refuses the co-instance as an ANSWER, because its own
|
|
116
|
+
continuation speaks of its own filler. Recall's consensus-anchor tier does not
|
|
117
|
+
ground it: `Where was the performer of song God (John Lennon Song) born?`
|
|
118
|
+
elected `Where was Nicki Minaj (Nicki Minaj Song) born?`. Fusion does not fuse
|
|
119
|
+
it as a further topic. A CAST comparison does not take it as its dominant: it
|
|
120
|
+
set `Where was Shakira (Shakira Song) born?` against `John Lennon` and voiced
|
|
121
|
+
Shakira's birthplace.
|
|
122
|
+
|
|
123
|
+
## Arguments the question holds under the response's equivalence
|
|
124
|
+
|
|
125
|
+
Recognition matches bytes and the query fold's own boundaries. 2Wiki title-cases
|
|
126
|
+
its questions (`… of film Man At Bath work at?` for the stored `Man at Bath`),
|
|
127
|
+
and a title inside a question lines up with none of the fold's cuts. When
|
|
128
|
+
recall's argument binding finds no recognised argument, the stored forms the
|
|
129
|
+
consensus climb reaches become candidates if the question holds them under the
|
|
130
|
+
response's equivalence (`canonHeldPoints`: canonical containment,
|
|
131
|
+
offset-preserving, so the span is the question's own). The approximate climb
|
|
132
|
+
proposes and exact containment decides. This is read after the clean-resonance
|
|
133
|
+
tier, so a near-identical question pays no climb for it. The canonical-form
|
|
134
|
+
index it relies on is the one the trainer builds after training
|
|
135
|
+
(`buildCanonIndex`, progress.ts).
|
|
136
|
+
|
|
137
|
+
A fragment that answers other questions is no argument and no independent piece
|
|
138
|
+
of the question either. It is set aside BEFORE the binding looks for one maximal
|
|
139
|
+
argument, so `director` no longer cancels `Man at Bath`. The same predicate
|
|
140
|
+
gates CAST's redirection: the substitute `ong)?`, the tail of every
|
|
141
|
+
`… (… Song)?` question, voiced a stranger's birthplace.
|
|
142
|
+
|
|
38
143
|
## Its consumers
|
|
39
144
|
|
|
40
|
-
| Where
|
|
41
|
-
|
|
|
42
|
-
| `chooseNext` (traverse.ts)
|
|
43
|
-
| `askedEvidence` (traverse.ts)
|
|
44
|
-
| `preConsumed` (pipeline.ts)
|
|
45
|
-
| The walk (reasoning.ts)
|
|
46
|
-
|
|
|
145
|
+
| Where | What it decides |
|
|
146
|
+
| --------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
147
|
+
| `chooseNext` (traverse.ts) | **The exact tier.** It names the continuation one of whose establishing contexts (other than the node) is witnessed by the question plus the node, with the question supplying at least one window the node does not. It ranks first among the readings; the distributional ladder decides only when nothing is named, or among continuations named equally. |
|
|
148
|
+
| `askedEvidence` (traverse.ts) | The question spans that named a pick. A mechanism projecting through the pick accounts for them, because evidence travels (mechanism-market.md). Recall's argument binding uses this. |
|
|
149
|
+
| `preConsumed` (pipeline.ts) | When a grounding does not declare `used`, what it spoke for is the forms inside its answer that the question already holds. The entity the answer added stays pivotable. |
|
|
150
|
+
| The walk (reasoning.ts) | A pivot is NAMED when one of its continuation's establishing contexts is witnessed by what of the question no product has said yet, plus the pivot. A named pivot MOVES. An unnamed one is offered only while the derivation still owes something, and the law then decides by carrying. No grounding owns the next step: a CAST grounding used to move through any term inside its answer, unasked (`Who is the director of The Jerk?` walked on to Carl Reiner's citizenship). |
|
|
151
|
+
| `answersOtherQuestions` (traverse.ts) — cover sites and recall's argument binding | A FRAGMENT is a form that sits inside other forms, has several continuations, and leaves at least one window of the question beyond it. It answers other questions, so it leads somewhere for this question only when the question names one of its continuations. Recall's binding read the same fragment (`director`) and voiced the most-poured stranger's fact. |
|
|
152
|
+
| `byCoInstance` (traverse.ts) | The exact tier's second reading: the relation read off another instance of the question, transferred to the node by an exact substitution. |
|
|
153
|
+
| Recall's consensus anchor, fusion roots | A co-instance of the question is never voiced as its answer, nor fused as a further topic. |
|
|
154
|
+
| CAST comparison (mechanisms/cast.ts) | A comparison voices two structures, so the question must evidence the analog with a window of its own: neither inside the dominant's runs nor scaffolding. `father of Frederick II?` evidenced `Peter III of Aragon father` only with `father` and `of`, and the comparison glued that bare question onto the answer. |
|
|
47
155
|
|
|
48
156
|
A cover span made of nothing but scaffolding (every window a hub,
|
|
49
157
|
`scaffoldSpans`) is not accounted. The form recognised there, such as the song
|
|
@@ -85,24 +193,44 @@ memoized per node per question.
|
|
|
85
193
|
|
|
86
194
|
## Measured
|
|
87
195
|
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
196
|
+
2Wiki held-out fixture: the evidence triples of 300 validation rows, deposited
|
|
197
|
+
as `wiki2.ts` does, with the canonical-form index built as the trainer builds
|
|
198
|
+
it, and asked the rows' own composed questions. Same evidence "plus instances":
|
|
199
|
+
261 one-hop questions about OTHER entities, built from other validation rows by
|
|
200
|
+
replacing the first hop's phrase with its referent
|
|
201
|
+
(`Where was the director of film X born?` → `Where was Óskar Jónasson born?`)
|
|
202
|
+
and answered with the second hop's fact. No test entity has one.
|
|
203
|
+
|
|
204
|
+
| Compositional correct (of 133) | Witnessing only (655b192) | Final |
|
|
205
|
+
| ------------------------------ | ------------------------- | ----- |
|
|
206
|
+
| Fixture | 52 | 56 |
|
|
207
|
+
| Plus instances | 41 | 89 |
|
|
208
|
+
|
|
209
|
+
The first hop is now right on 114 and 109 of the 133. Of the first hops still
|
|
210
|
+
wrong, most name the entity differently from the corpus (`Clara Novello` for
|
|
211
|
+
`Clara Anastasia Novello`). That is an alias, which no byte reading recovers.
|
|
212
|
+
|
|
213
|
+
Removing CAST's unasked step cost six right answers on the fixture without
|
|
214
|
+
instances: second hops that were right paraphrases taken without evidence
|
|
215
|
+
(`work at`, `is from`, `When was … born`). The same licence produced the Carl
|
|
216
|
+
Reiner extension on the trained store. Without an instance, nothing names those
|
|
217
|
+
relations, and the chain stops at the first hop. The answers gained plus
|
|
218
|
+
instances are those relations: `born` → place of birth, `Why did … die` → cause
|
|
219
|
+
of death, `study`/`graduate from` → educated at, `work at` → employer, `is from`
|
|
220
|
+
→ country of citizenship.
|
|
221
|
+
|
|
222
|
+
The inference count stays low. Two hops of `father` from a question that says
|
|
223
|
+
`father` once, as in `paternal grandfather`, are not composed: the store holds
|
|
224
|
+
no evidence that `grandfather` composes `father` twice.
|
|
225
|
+
|
|
226
|
+
On the 31.7M-node store's 116-query battery, two dialogue answers changed
|
|
227
|
+
(neither was correct before). The deterministic read counters rose about 4%
|
|
228
|
+
(`bytesRead`, `ancestorVisits`, `edgeProbes`), with ANN queries up 1%. CPU on
|
|
229
|
+
this workstation varies by ±15% between back-to-back runs of one build, and
|
|
230
|
+
paired runs read 163 s against 152 s and 171 s, then 223 s against 203 s under
|
|
231
|
+
outside load. The co-instance tier read 190 forms over the battery and named
|
|
232
|
+
nothing there: on that store a frame's windows are scaffolding, and the one-hop
|
|
233
|
+
questions it would need are not in the corpus.
|
|
106
234
|
|
|
107
235
|
## Pins
|
|
108
236
|
|
|
@@ -110,4 +238,19 @@ noise. The deterministic read counters rose about 20% (`bytesRead`,
|
|
|
110
238
|
most-poured one. The second hop is named by the question plus the introduced
|
|
111
239
|
entity, whatever grounded the first hop. A question that names no further step
|
|
112
240
|
is not extended. A fragment voices none of its continuations unless the
|
|
113
|
-
question names one
|
|
241
|
+
question names one, in the cover and in recall's argument binding. A
|
|
242
|
+
comparison needs two things named. An argument held only under the response's
|
|
243
|
+
equivalence binds. Each assertion was verified by mutation.
|
|
244
|
+
- `test/155` — the relation read off another instance of the frame; absent
|
|
245
|
+
without two instances that agree; refused for a partly shared frame; never
|
|
246
|
+
voiced as the answer; a frame a product already said names no further step; a
|
|
247
|
+
description in the slot is no other instance; traced and untraced responses
|
|
248
|
+
agree; fusion does not fuse a co-instance, nor does a comparison take one as
|
|
249
|
+
its dominant. Verified by mutation.
|
|
250
|
+
- `test/76` — a fact the corpus files under an instance's filler is no carriage:
|
|
251
|
+
`reference` does not construct one for a new referent.
|
|
252
|
+
- `test/29` C3 — a further hop inside a comparison's seat waits to be asked.
|
|
253
|
+
- Unpinned, measured only at fixture scale. Excluding fragments from
|
|
254
|
+
establishing contexts is worth 16 answers on the fixture with one-hop
|
|
255
|
+
questions. Refusing a fragment as CAST's redirection substitute is worth four.
|
|
256
|
+
Fragments with inherited edges appear only at that scale.
|
package/jsr.json
CHANGED
package/package.json
CHANGED
package/src/meter.ts
CHANGED
|
@@ -275,6 +275,12 @@ export class Meter {
|
|
|
275
275
|
* came back at the √N cap, or its predecessor budget ran out before every
|
|
276
276
|
* continuation was asked about — the distributional ladder decided. */
|
|
277
277
|
askedReadsSaturated = 0;
|
|
278
|
+
/** Picks named by a CO-INSTANCE of the question — another instance of its
|
|
279
|
+
* frame whose continuation carries the relation over to this node by an
|
|
280
|
+
* exact substitution (traverse.ts, `byCoInstance`). */
|
|
281
|
+
coInstanceNamings = 0;
|
|
282
|
+
/** Proposed co-instances that tier read and witnessed. */
|
|
283
|
+
coInstanceReads = 0;
|
|
278
284
|
/** Cover sites dropped as FRAGMENTS whose several continuations the question
|
|
279
285
|
* names none of (mechanisms/cover.ts). */
|
|
280
286
|
unaskedFragments = 0;
|
package/src/mind/evidence.ts
CHANGED
|
@@ -63,6 +63,12 @@ export interface Witnessing {
|
|
|
63
63
|
spans: Array<[number, number]>;
|
|
64
64
|
/** Bytes of source 0 inside `spans`. */
|
|
65
65
|
bytes: number;
|
|
66
|
+
/** The form's own bytes no source witnesses, as merged spans of the FORM —
|
|
67
|
+
* empty exactly when `complete`. ONE residue span is the shape of a
|
|
68
|
+
* co-instance: the same frame around a different filler (`When was Peter
|
|
69
|
+
* Jackson born?` read against `When was the director of film Jinpa
|
|
70
|
+
* born?` leaves `Peter Jackson`). */
|
|
71
|
+
residue: Array<[number, number]>;
|
|
66
72
|
}
|
|
67
73
|
|
|
68
74
|
/** Witness `form` against `sources` (their window indexes, same order). A
|
|
@@ -73,8 +79,14 @@ export function witness(
|
|
|
73
79
|
indexes: ReadonlyArray<WindowIndex>,
|
|
74
80
|
W: number,
|
|
75
81
|
): Witnessing {
|
|
76
|
-
|
|
77
|
-
|
|
82
|
+
if (form.length < W || indexes.length === 0) {
|
|
83
|
+
return {
|
|
84
|
+
complete: false,
|
|
85
|
+
spans: [],
|
|
86
|
+
bytes: 0,
|
|
87
|
+
residue: form.length > 0 ? [[0, form.length]] : [],
|
|
88
|
+
};
|
|
89
|
+
}
|
|
78
90
|
const covered = new Uint8Array(form.length);
|
|
79
91
|
const own: Array<[number, number]> = [];
|
|
80
92
|
for (let o = 0; o + W <= form.length; o++) {
|
|
@@ -93,7 +105,13 @@ export function witness(
|
|
|
93
105
|
covered.fill(1, o, o + W);
|
|
94
106
|
if (from === 0) own.push([at, at + W]);
|
|
95
107
|
}
|
|
96
|
-
|
|
108
|
+
const residue: Array<[number, number]> = [];
|
|
109
|
+
for (let i = 0; i < form.length; i++) {
|
|
110
|
+
if (covered[i]) continue;
|
|
111
|
+
const last = residue[residue.length - 1];
|
|
112
|
+
if (last !== undefined && last[1] === i) last[1] = i + 1;
|
|
113
|
+
else residue.push([i, i + 1]);
|
|
114
|
+
}
|
|
97
115
|
own.sort((a, b) => a[0] - b[0]);
|
|
98
116
|
const spans: Array<[number, number]> = [];
|
|
99
117
|
for (const [s, e] of own) {
|
|
@@ -103,5 +121,5 @@ export function witness(
|
|
|
103
121
|
}
|
|
104
122
|
let bytes = 0;
|
|
105
123
|
for (const [s, e] of spans) bytes += e - s;
|
|
106
|
-
return { complete:
|
|
124
|
+
return { complete: residue.length === 0, spans, bytes, residue };
|
|
107
125
|
}
|