pi-canon 0.1.2 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,16 +1,20 @@
1
1
  /* Surfacing: tool calls stage the articles governing what they touch; the staged
2
- lines flush as ONE message per turn, once per article per session, under a hard
3
- budget. Capsules first; when the budget is spent, pointers only. One message per
2
+ lines flush as ONE message per turn, once per article per session. One message per
4
3
  turn matters: pi's steering queue drains one message per provider round trip, so
5
- a message per tool call would buy each nudge its own extra LLM call. */
4
+ a message per tool call would buy each nudge its own extra LLM call.
5
+
6
+ Nothing here is bounded by a character count. A session budget used to cap the
7
+ capsule text and degrade the overflow to bare pointers, and it was deleted in 2.0
8
+ (Shane, 2026-08-12): the constant was a guess at a policy nobody had measured, and
9
+ it decided what an agent got to see. What replaces it is measurement. Every
10
+ surfaced line records what it cost, so context taken can be read against relevance
11
+ after the fact instead of a constant ruling on it in advance. */
6
12
 
7
13
  import { appendFileSync, existsSync } from "node:fs";
8
14
  import { dirname, isAbsolute, join } from "node:path";
15
+ import { intentQuery, NONE, residue, userIntent, type IntentTurn, type Retriever } from "./retrieval.ts";
9
16
  import type { CanonStore } from "./store.ts";
10
17
 
11
- export const SESSION_BUDGET_CHARS = 4000;
12
- const MESSAGE_CHARS = 2000;
13
-
14
18
  /* Observability, env-gated and inert otherwise: PI_CANON_TRACE=<file> appends one
15
19
  JSON line per surfacing decision, so a harness can audit the staged -> flushed ->
16
20
  seen funnel instead of guessing at it. */
@@ -26,6 +30,79 @@ function trace(kind: string, data: Record<string, unknown>): void {
26
30
 
27
31
  const PATHLIKE = /(?:^|[\s"'`=:,([{])(\/?[\w.@-]+(?:\/[\w.@-]+)+)/g;
28
32
 
33
+ /* Presence marks -----------------------------------------------------------------
34
+
35
+ "Seen" used to mean "we sent it once", which is only the same thing as "the agent
36
+ can see it" in a session where nothing ever leaves the window. Once anything folds
37
+ or compacts, the two come apart, and the agent is not aware of what was folded
38
+ away. So seen is checked against the projection rather than remembered.
39
+
40
+ A mark is a normalized slice of what the article put in the window, and the caller
41
+ passes exactly that: the LINE for a surfaced article, capsule plus body for a read.
42
+ Passing anything else is the one way to break this, because presence then tests for
43
+ text that was never shown. Normalizing both sides to lowercase alphanumerics survives
44
+ JSON escaping, whitespace rewrapping, and quoting differences between however the
45
+ projection is rendered and however we wrote it. A short mark is not distinctive enough
46
+ to test, so it is never expired; failing to expire only costs a re-surface that does
47
+ not happen, while a false expiry would spam the window. A surfaced line always carries
48
+ its own address and so is always long enough; only a read of a very small article is
49
+ not.
50
+
51
+ A mark has two parts and BOTH must be in the projection.
52
+
53
+ IDENTITY is the article's own address, which is in the window whichever way the
54
+ article got there: the surfaced line reads "path: capsule" and a read prints the
55
+ address as its title. LIVENESS is the tail of whatever actually entered, the caller
56
+ passing the whole of it. A surfaced line is its capsule and is held to the capsule;
57
+ a read is capsule plus body and is held to the body.
58
+
59
+ Both parts are needed because either alone is wrong in a way that matters. Identity
60
+ alone cannot tell a one-line nudge from the full article, which is the defect that
61
+ started this (Codex, 2026-08-12): an article read in full stayed present on the
62
+ strength of its surviving capsule while the body holding the rule had folded away.
63
+ Liveness alone collides, because two articles sharing a common ending share a tail,
64
+ and the survivor then keeps the other marked present (Codex, 2026-08-13). Addresses
65
+ are unique, so requiring both closes that.
66
+
67
+ Tail rather than head for liveness, because the two ways content leaves a window are
68
+ not symmetric: a fold takes the whole message, a truncation takes the end first, so a
69
+ head mark survives exactly the loss it is supposed to report. */
70
+ const MARK_CHARS = 120;
71
+ const MARK_MINIMUM = 24;
72
+
73
+ /* How many RANKED articles may ride one message. Not a relevance threshold: see
74
+ retrieve(). Three because a nudge is read or it is not, and the addressed lines it
75
+ shares the message with are the ones that were certain. */
76
+ const RETRIEVED_PER_TURN = 3;
77
+
78
+ function fingerprint(text: string): string {
79
+ return text.toLowerCase().replace(/[^a-z0-9]+/g, " ").trim();
80
+ }
81
+
82
+ /* The projection's actual text, gathered from the message structure rather than from
83
+ JSON.stringify of it. Stringifying introduced escapes that are not in anyone's text:
84
+ a newline arrived as the two characters \ and n, and n is a letter, so the two sides
85
+ disagreed at every line break. Erasing escapes afterwards fixed that and broke
86
+ something else, making an article containing a literal backslash-n fingerprint
87
+ identically to one without it (Codex, 2026-08-13), which is a false PRESENCE and so
88
+ the expensive direction: the article is gone and nothing re-surfaces it. Reading the
89
+ strings directly means no escape is ever introduced and none has to be erased. */
90
+ function projectionText(messages: unknown[]): string {
91
+ const out: string[] = [];
92
+ const seen = new Set<unknown>();
93
+ const walk = (value: unknown): void => {
94
+ if (typeof value === "string") {
95
+ out.push(value);
96
+ } else if (value && typeof value === "object") {
97
+ if (seen.has(value)) return; /* a cyclic projection is still readable */
98
+ seen.add(value);
99
+ for (const inner of Array.isArray(value) ? value : Object.values(value)) walk(inner);
100
+ }
101
+ };
102
+ walk(messages);
103
+ return out.join("\n");
104
+ }
105
+
29
106
  /* A store and the directory whose assets it governs. The project is the first,
30
107
  unnamed mount; named mounts are outside directories (a data lake, a shared
31
108
  corpus) whose articles address as name:path. */
@@ -38,45 +115,148 @@ export interface Mount {
38
115
  export class Surfacer {
39
116
  private mounts: Mount[];
40
117
  private seen = new Set<string>();
118
+ /* What to look for in the projection to decide an article is still visible. Absent
119
+ for an article whose entered text is too short to test, which is never expired. */
120
+ private marks = new Map<string, { id: string; tail: string }>();
41
121
  private pendingUpdates = new Set<string>();
42
- private staged = new Map<string, { capsule: string; stamp: string; asset: string }>();
43
- private spent = 0;
122
+ private staged = new Map<string, { capsule: string; stamp: string; asset: string; score?: number }>();
123
+ private retriever: Retriever;
124
+ /* The intent the residue was last ranked against, so an unchanged question does not keep
125
+ buying more guesses every turn. See retrieve(). */
126
+ private lastQuery = "";
127
+ private resurface: boolean;
128
+ /* This turn's tool calls, cleared when it flushes: what the agent is doing right now,
129
+ and nothing older. Recency is structural here rather than a weighting. */
130
+ private intent: IntentTurn[] = [];
131
+ /* The user's own words, refreshed from each projection. Kept apart from the tool
132
+ calls because it has a different lifetime: a question stays the question across the
133
+ turns spent answering it, while a tool call is spent the moment it flushes. */
134
+ private spoken: IntentTurn[] = [];
135
+ /* What each currently present article cost the window, kept for the same reason the
136
+ budget was removed: the analysis wants context taken beside relevance. */
137
+ private cost = new Map<string, number>();
138
+ private surfacedEver = new Set<string>();
139
+ /* What the last flush and settle committed on the assumption the message would be
140
+ delivered. Kept so a failed send can be undone rather than silently believed. */
141
+ private lastFlush = new Map<string, { capsule: string; stamp: string; asset: string; score?: number }>();
142
+ private lastNudge: string[] = [];
143
+
144
+ /* How far the best must beat the rest of the same query before anything rides. See
145
+ retrieve(). The default is an operating point priced by a 120-cell study rather than
146
+ picked: at 1.4 a session kept every rule fact the uncut channel delivered at a ninth
147
+ of the suggestion volume, and a store with nothing relevant never reached it. 1 turns
148
+ the cutoff off. */
149
+ private standout: number;
44
150
 
45
- constructor(mounts: Mount[]) {
151
+ constructor(mounts: Mount[], retriever: Retriever = NONE, resurface = true, standout = 1.4) {
46
152
  this.mounts = mounts;
153
+ this.retriever = retriever;
154
+ this.resurface = resurface;
155
+ this.standout = standout;
47
156
  }
48
157
 
49
158
  private get project(): Mount {
50
159
  return this.mounts[0];
51
160
  }
52
161
 
53
- private mountFor(asset: string): Mount {
162
+ /* The mount an asset lives in, and the asset made absolute. Absolute because the
163
+ store strips its mount directory only as a leading prefix: handed a
164
+ project-relative path that reaches into a named mount, it would keep the whole
165
+ path and resolve a longer, wrong address inside that mount. The project mount is
166
+ indifferent, since project-relative IS its address space either way. */
167
+ private locate(asset: string): { mount: Mount; absolute: string } {
54
168
  const path = asset.replace(/\\/g, "/");
55
169
  const absolute = isAbsolute(path) ? path : join(this.project.dir, path);
56
170
  for (const mount of this.mounts.slice(1)) {
57
- if (absolute === mount.dir || absolute.startsWith(`${mount.dir}/`)) return mount;
171
+ if (absolute === mount.dir || absolute.startsWith(`${mount.dir}/`)) return { mount, absolute };
58
172
  }
59
- return this.project;
173
+ return { mount: this.project, absolute };
60
174
  }
61
175
 
62
- markSeen(path: string): void {
176
+ /* `entered` is everything the caller just put in the window for this article, not a
177
+ mark: the tail of it becomes the mark, so a caller that sent the body is held to
178
+ the body and one that sent only a capsule is held to the capsule. */
179
+ markSeen(path: string, entered?: string): void {
63
180
  if (this.staged.has(path)) trace("withdrawn", { path });
64
181
  this.seen.add(path);
182
+ this.remember(path, entered);
65
183
  this.staged.delete(path);
184
+ /* Counted like a flushed line. The session budget was deleted in favour of measuring
185
+ what context is actually taken, so a full read that recorded neither its cost nor
186
+ its having happened left the measurement reporting present=1 at chars=0 (Codex,
187
+ 2026-08-13). A read is the largest thing this package ever puts in a window. */
188
+ if (entered) {
189
+ this.cost.set(path, entered.length);
190
+ this.surfacedEver.add(path);
191
+ trace("entered", { path, chars: entered.length, via: "read" });
192
+ }
66
193
  }
67
194
 
68
- markUpdated(path: string): void {
69
- this.markSeen(path);
195
+ private remember(path: string, text: string | undefined): void {
196
+ const print = fingerprint(text ?? "");
197
+ const id = fingerprint(path);
198
+ if (print.length >= MARK_MINIMUM && id) this.marks.set(path, { id, tail: print.slice(-MARK_CHARS) });
199
+ else this.marks.delete(path);
200
+ }
201
+
202
+ /* The live projection, as the provider is about to receive it. Every article whose
203
+ mark is no longer in it has left the agent's window and stops counting as seen, so
204
+ the next touch of its asset surfaces it again.
205
+
206
+ Two honest limits. This reads whatever the projection holds when pi-canon's
207
+ handler runs, so if another extension folds after us we observe its previous
208
+ state and lag by a turn; a lagging expiry is a late re-surface, not a wrong one.
209
+ And a digested or summarized block does not carry the capsule, which is the
210
+ intended reading: a digest of a line about an article is not the article.
211
+
212
+ Never called means never expired, which is exactly 1.0 behavior, so a harness that
213
+ does not report a projection loses the mechanism and nothing else. */
214
+ observe(messages: unknown): void {
215
+ if (!Array.isArray(messages)) return;
216
+ const projection = fingerprint(projectionText(messages));
217
+ /* Read before the expiry check and independently of it: the projection is the only
218
+ place the user's own words are visible, and a run with resurface off still wants
219
+ them for the query. Presence is what the switch governs, not observation. */
220
+ this.spoken = userIntent(messages);
221
+ if (!this.resurface) return;
222
+ for (const path of [...this.seen]) {
223
+ const mark = this.marks.get(path);
224
+ if (!mark || (projection.includes(mark.id) && projection.includes(mark.tail))) continue;
225
+ this.seen.delete(path);
226
+ this.marks.delete(path);
227
+ this.cost.delete(path);
228
+ trace("departed", { path });
229
+ }
230
+ }
231
+
232
+ /* Authoring an article is a way of having it in the window, so it takes `entered`
233
+ for the same reason a read does. Passing nothing here was a real bug (Codex,
234
+ 2026-08-13): the mark was cleared, observe() skips a seen path with no mark, and the
235
+ article then stayed present for the rest of the session and could never re-surface.
236
+ Only what the write itself carried counts: a capsule-only write leaves the stored
237
+ body unseen, so marking the whole article present would be a claim about text the
238
+ agent never received. */
239
+ markUpdated(path: string, entered?: string): void {
240
+ this.markSeen(path, entered);
70
241
  this.pendingUpdates.delete(path);
71
242
  }
72
243
 
73
- get stats(): { surfaced: number; spent: number } {
74
- return { surfaced: this.seen.size, spent: this.spent };
244
+ /* surfaced counts every article this session ever put in the window; present counts
245
+ the ones still in it, and chars what those are currently occupying. They diverge
246
+ exactly when something folded an article away, which is the whole point. */
247
+ get stats(): { surfaced: number; present: number; chars: number } {
248
+ let chars = 0;
249
+ for (const value of this.cost.values()) chars += value;
250
+ return { surfaced: this.surfacedEver.size, present: this.seen.size, chars };
75
251
  }
76
252
 
77
253
  /* Candidate asset paths in a tool call: string values that are paths, and path
78
254
  shaped tokens inside them. A candidate needs to exist, or to have an existing
79
- parent, so a file about to be created still surfaces its governing article. */
255
+ parent, so a file about to be created still surfaces its governing article. Two
256
+ stated edges of that rule. A touch is a claim of attention, not of effect: a path
257
+ that merely rides a payload counts, and so does a call some later hook blocks,
258
+ because this runs before execution. And a NEW file at the project root has no
259
+ parent segment in its path, so it is not seen until it exists. */
80
260
  pathsIn(input: unknown): string[] {
81
261
  const found = new Set<string>();
82
262
  const consider = (candidate: string) => {
@@ -99,11 +279,240 @@ export class Surfacer {
99
279
  return [...found];
100
280
  }
101
281
 
282
+ /* This turn's intent, one entry per tool call. Kept separate from `collect` because
283
+ they answer different questions: collect asks what asset was touched, which the
284
+ spine answers by address, and this asks what the agent is trying to do, which is
285
+ the only thing an unaddressed article can be ranked against. */
286
+ noteIntent(toolName: unknown, input: unknown): void {
287
+ if (typeof toolName === "string" && toolName) this.intent.push({ toolName, input });
288
+ }
289
+
290
+ /* The residue, rebuilt only when the store moved under it.
291
+
292
+ This used to rebuild every turn on the grounds that the residue is small by
293
+ construction. Measured, that justifies the wrong quantity: residue() reads and stats
294
+ every article in the store before it filters any of them, so the cost tracks the STORE.
295
+ 5,000 articles cost 242ms a turn when all of them are residue and 214ms when only 50
296
+ are. At 20,000 it is 891ms, every turn, on the path a provider round trip is waiting on.
297
+
298
+ The stated reason not to cache was real and is answered rather than ignored: `updated`
299
+ has day granularity, so an article rewritten in the same session keeps its stamp and any
300
+ key built from it serves a stale ranking for the rest of the run. store.signature() keys
301
+ on mtimeMs and size instead, which move on every write, and costs 114ms where the
302
+ rebuild costs 891ms. */
303
+ private residueCache?: { signature: string; candidates: ReturnType<typeof residue> };
304
+ private reindex = true;
305
+
306
+ private candidates(store: CanonStore, dir: string): ReturnType<typeof residue> {
307
+ const signature = store.signature();
308
+ if (this.residueCache?.signature === signature) return this.residueCache.candidates;
309
+ const candidates = residue(store, dir);
310
+ this.residueCache = { signature, candidates };
311
+ /* The retriever's index is built from these, so it is stale for exactly as long as
312
+ they are. One flag, set here, cleared where the index is rebuilt. */
313
+ this.reindex = true;
314
+ trace("residue-rebuilt", { candidates: candidates.length });
315
+ return candidates;
316
+ }
317
+
318
+ /* Rank the residue against this turn's intent and stage what the query touched.
319
+
320
+ `score > 0` is not a tuned cutoff. With BM25 normalized against its saturation
321
+ ceiling it means "at least one query term appears in this article at all", which is
322
+ a property of the query rather than a constant someone picked. `standout` is the
323
+ tuned one. A 120-cell study priced it on a corpus with something worth finding in
324
+ its residue, and 1.4 held every fact the uncut channel delivered at a ninth of the
325
+ volume; that is the default, and the caller moves it against their own trace.
326
+
327
+ There used to be an absolute threshold here, on the grounds that a study session was
328
+ handed 28 ranked lines and opened 5, and the scores of the opened and the ignored
329
+ overlapped but separated. That is the precondition a cutoff needs, and the shape it
330
+ was given was wrong. Read back across two studies the same cutoff had to be 0.25 on
331
+ one corpus and 0.03 on the other, and read WITHIN one session it moved by a factor
332
+ of four with nothing but how much the agent happened to say that turn. It was never
333
+ one quantity being tuned to three values. What it was really doing was silencing
334
+ whole queries rather than trimming tails, 82% of what it removed at its operating
335
+ point, so it is now written as the thing it was doing, in a unit that ports.
336
+
337
+ What IS bounded is how many articles ride one message, and that is a different thing
338
+ from a cutoff. A cutoff rules on relevance; this rules on transport. Sharing
339
+ one token with the query is enough to score above zero, so a residue of fifty rule
340
+ articles and a query saying "export" stages fifty lines, and every one of them is
341
+ unrequested context the agent never asked to spend. The address spine is exempt
342
+ because an addressed article is a certainty and the agent touched its asset; ranked
343
+ candidates are guesses, and a guess does not get to fill the window. Nothing is
344
+ discarded: what does not fit is still eligible next turn, and the trace records what
345
+ was held back, so the cutoff question stays answerable from data. */
346
+ retrieve(): void {
347
+ if (this.retriever === NONE) return;
348
+ /* Oldest first, so intentQuery's newest-first walk reads in true order: this turn's
349
+ tool calls lead, the question that prompted them follows. */
350
+ const turns = [...this.spoken, ...this.intent];
351
+ if (!turns.length) return;
352
+ const { store, dir } = this.project;
353
+ const candidates = this.candidates(store, dir);
354
+ if (!candidates.length) return;
355
+ /* Indexed only when the store actually changed; see candidates(). */
356
+ if (this.reindex) {
357
+ this.retriever.index?.(candidates);
358
+ this.reindex = false;
359
+ }
360
+ const query = intentQuery(turns);
361
+ if (!query.trim()) return;
362
+ /* A new ranked article is justified by new intent, never by another turn passing.
363
+
364
+ The per-message cap bounds how much rides one message; on its own it did not bound
365
+ what a session spends. `seen` keeps a flushed path from returning but does nothing to
366
+ stop the NEXT three being released against the very same query, and user speech
367
+ persists in the projection while flush clears only tool intent, so an unchanged
368
+ question released three more articles every turn until the residue ran out. That
369
+ serialises the fan-out rather than bounding it (Codex, 2026-08-13). The two together
370
+ are the bound: three per message, and nothing further until the agent's intent
371
+ actually moves. */
372
+ if (query === this.lastQuery) return;
373
+ this.lastQuery = query;
374
+ let scores: Map<string, number>;
375
+ try {
376
+ scores = this.retriever.score(query, candidates);
377
+ } catch (error) {
378
+ trace("retrieval-failed", { retriever: this.retriever.name, error: String(error) });
379
+ return; /* a retriever that throws must never break the turn */
380
+ }
381
+ const scored = candidates
382
+ .map((candidate) => ({ candidate, score: scores.get(candidate.path) as number }))
383
+ .filter((entry) => typeof entry.score === "number" && entry.score > 0)
384
+ .sort((a, b) => b.score - a.score);
385
+ if (!scored.length) return;
386
+ const ranked = scored
387
+ .filter((entry) => !this.seen.has(entry.candidate.path) && !this.staged.has(entry.candidate.path))
388
+ /* Once a session, and never again. `seen` alone says "not while it is still in the
389
+ window", which lets a guess the agent already declined come back the moment the
390
+ window rolls past it. An address may resurface, because a fresh touch means the
391
+ agent is working on that asset again and no longer has the article. A GUESS may
392
+ not: nothing new happened, the agent was offered it and passed, and asking twice
393
+ is what teaches a reader to stop looking.
394
+
395
+ Measured before it was changed: a build whose ranked line did not contain the
396
+ capsule re-offered 38% of its suggestions, 151 of 393 in one arm, because
397
+ presence was tested against text that had never been delivered. That was a bug
398
+ in a study build, but the only reason it could express itself as a repeat at all
399
+ is that nothing here said once. */
400
+ .filter((entry) => !this.surfacedEver.has(entry.candidate.path));
401
+ if (!ranked.length) return;
402
+ /* Is the best thing left here worth a line, or did the query merely brush the whole
403
+ residue at once? Measured against the crowd this same query raised rather than
404
+ against a number, for the reason in the option's own comment: a score is a fraction
405
+ of the query's idf mass, so it moves with how much the agent said this turn and with
406
+ how alike the corpus is, and a constant that is right on one project is wrong on the
407
+ next by a multiple.
408
+
409
+ The crowd is the best article that will NOT ride: rank RETRIEVED_PER_TURN + 1, the
410
+ one the cap is already about to leave behind. So the question is "does the best beat
411
+ what we were not going to send anyway", which needs no constant of its own and cannot
412
+ be set inconsistently with the cap.
413
+
414
+ Both terms come from what is still ELIGIBLE, after the articles already offered this
415
+ session are taken out, because the question is whether to spend a line on what is
416
+ left rather than on what was already delivered. Computed over the whole ranking
417
+ instead, it barely moves: the best and fourth-best answers to a task the agent is
418
+ still working on are the same articles turn after turn, so every turn of a study
419
+ session reported 1.81 to 2.00 whether it had anything new to offer or not. Against
420
+ the eligible set the same sessions separated, 1.68 to 1.81 on the rankings that
421
+ carried a decisive article and 1.00 to 1.28 on the rankings that did not, which is
422
+ the session going quiet as it uses up what was worth saying.
423
+
424
+ It is measured near the top of the ranking rather than at a quantile of it because a
425
+ real query is long. An agent's turn touches nearly the whole residue, 377 of 378
426
+ articles in a study session, so a tenth of the way down is deep in the mass sharing
427
+ one common word, and the ratio to it reports the shape of the corpus rather than
428
+ anything about this query: ordinary queries reached 2.64 to 3.28 there and the query
429
+ that had something to find reached 3.10, inside that range rather than above it.
430
+
431
+ The cost of tying it to the cap is that four articles genuinely relevant at once
432
+ silence each other. That is the same bet the cap already makes, and it is bounded the
433
+ same way: what is not sent stays eligible next turn.
434
+
435
+ Fewer eligible than the cap is the case with no crowd at all. On a store that has
436
+ not been drawn down they ride, because being one of a handful of articles in the
437
+ residue that share a word with what the agent is doing is the strongest form of
438
+ standing out, not the weakest.
439
+
440
+ The eligible tail alone fails on one regime, measured on a real 33-article store:
441
+ once a session has consumed most of what the store had to say, the leftovers are a
442
+ tail of near-zero scores, so the ratio over them explodes onto junk (a 4.79 standout
443
+ on a 0.101-score best, with the ratio anti-correlating with relevance) and the
444
+ no-crowd rule above becomes a free ride for scores of 0.002. Small and drained look
445
+ identical from the eligible set; they differ in what was already delivered. So the
446
+ crowd takes a floor at the best CONSUMED responder, the strongest article this same
447
+ query raised among those already delivered this session: to spend a line, the best
448
+ thing left must beat what the query would have re-raised if it could. A genuinely
449
+ new topic clears that floor, because the old articles score weakly on its query; a
450
+ drained tail does not, because the leftovers score below the delivered on every
451
+ query. A fresh store has consumed nothing and keeps the free ride. The floor exists
452
+ only while the cutoff does: an explicit 1 is the no-cutoff measurement setting and
453
+ stays the 1.0 behavior exactly, drained or not. */
454
+ const eligiblePaths = new Set(ranked.map((entry) => entry.candidate.path));
455
+ const consumed = scored.find((entry) => !eligiblePaths.has(entry.candidate.path));
456
+ const tail = ranked.length > RETRIEVED_PER_TURN ? ranked[RETRIEVED_PER_TURN].score : 0;
457
+ const crowd = this.standout > 1 ? Math.max(tail, consumed?.score ?? 0) : tail;
458
+ const reached = crowd > 0 ? ranked[0].score / crowd : Infinity;
459
+ const passed = reached >= this.standout;
460
+ /* Every ranking, not only the ones that were cut. The number that says a cutoff is set
461
+ too high is the one it silently removed, and the number that says it is set too low
462
+ is the ratio the queries reached anyway; a trace that only records refusals can
463
+ report the first and never the second. Both readings are needed to place it, and
464
+ neither survives being inferred from the scores that rode, because the crowd they
465
+ were measured against is not in those lines. */
466
+ trace("ranked", {
467
+ standout: this.standout,
468
+ reached: reached === Infinity ? null : reached,
469
+ responders: scored.length,
470
+ eligible: ranked.length,
471
+ passed,
472
+ });
473
+ if (!passed) return;
474
+ /* What is already staged counts against the cap.
475
+
476
+ undoFlush restages an undelivered message, and restaged entries are excluded from
477
+ the candidate pool by the `staged` check above, so they were invisible to the budget
478
+ and the next turn added a full three on top of them: three, six, nine, twelve ranked
479
+ lines over four failed deliveries (workflow review, 2026-08-13). The cap is on what
480
+ one message carries, so it has to count everything that message will carry, not just
481
+ what this call contributed. */
482
+ const already = [...this.staged.values()].filter((entry) => entry.score !== undefined).length;
483
+ const room = Math.max(0, RETRIEVED_PER_TURN - already);
484
+ for (const { candidate, score } of ranked.slice(0, room)) {
485
+ this.staged.set(candidate.path, {
486
+ capsule: candidate.capsule,
487
+ stamp: candidate.updated ? ` (updated ${candidate.updated})` : "",
488
+ asset: candidate.path,
489
+ score,
490
+ });
491
+ trace("retrieved", {
492
+ path: candidate.path,
493
+ retriever: this.retriever.name,
494
+ score,
495
+ /* So a run can report how much of what it ranked was a rule on purpose. */
496
+ declared: candidate.declared,
497
+ });
498
+ }
499
+ const held = ranked.slice(room);
500
+ if (held.length) {
501
+ trace("retrieval-held", {
502
+ count: held.length,
503
+ /* The best score that did NOT ride this turn, against the worst that did: the
504
+ pair that says whether the cap ever cut anything worth carrying. */
505
+ bestHeld: held[0].score,
506
+ worstSent: room ? ranked[room - 1].score : null,
507
+ });
508
+ }
509
+ }
510
+
102
511
  /* Stage each newly touched governing article. Nothing is sent or spent here. */
103
512
  collect(assets: string[]): void {
104
513
  for (const asset of assets) {
105
- const mount = this.mountFor(asset);
106
- const article = mount.store.resolve(asset, mount.dir);
514
+ const { mount, absolute } = this.locate(asset);
515
+ const article = mount.store.resolve(absolute, mount.dir);
107
516
  if (!article) continue;
108
517
  const key = mount.name ? `${mount.name}:${article.path}` : article.path;
109
518
  this.pendingUpdates.add(key);
@@ -114,41 +523,93 @@ export class Surfacer {
114
523
  }
115
524
  }
116
525
 
117
- /* Everything staged since the last flush, as one bounded message. The budget is
118
- charged here, not at staging, so a nudge withdrawn by markSeen costs nothing;
119
- articles count as seen only once their line is part of a flushed message.
120
- Overflow stays staged for the next turn. */
526
+ /* Everything staged since the last flush, as one message. Nothing is held back and
527
+ nothing is truncated: an article whose governing asset this turn touched either
528
+ surfaces whole or does not surface. Cost is recorded per line rather than charged
529
+ against an allowance, so a nudge withdrawn by markSeen still costs nothing and the
530
+ funnel stays auditable. An article with no capsule surfaces as a pointer, which is
531
+ the only remaining reason a line is not the capsule text. */
121
532
  flush(): string | undefined {
533
+ this.intent = [];
534
+ this.lastFlush.clear();
122
535
  if (!this.staged.size) return undefined;
536
+ /* Addressed articles first, in the order they were touched, because the address is
537
+ a certainty and nothing ranked should push it down the message. Retrieved ones
538
+ follow, best score first. */
539
+ const order = [...this.staged.entries()].sort((a, b) => {
540
+ const left = a[1].score, right = b[1].score;
541
+ if (left === undefined && right === undefined) return 0;
542
+ if (left === undefined) return -1;
543
+ if (right === undefined) return 1;
544
+ return right - left;
545
+ });
123
546
  const lines: string[] = [];
124
- let size = 0;
125
- for (const [path, entry] of this.staged) {
126
- const useCapsule = entry.capsule && this.spent + entry.capsule.length <= SESSION_BUDGET_CHARS;
127
- const line = useCapsule
547
+ for (const [path, entry] of order) {
548
+ const line = entry.capsule
128
549
  ? `${path}${entry.stamp}: ${entry.capsule}`
129
550
  : `${path}${entry.stamp}: article exists. Read it before relying on ${entry.asset}.`;
130
- if (lines.length && size + line.length > MESSAGE_CHARS) {
131
- lines.push(`${this.staged.size} more staged; they surface next turn.`);
132
- break;
133
- }
134
- if (useCapsule) this.spent += entry.capsule.length;
135
- size += line.length;
136
551
  lines.push(line);
552
+ this.cost.set(path, line.length);
553
+ this.surfacedEver.add(path);
554
+ trace("surfaced", {
555
+ path,
556
+ chars: line.length,
557
+ capsule: Boolean(entry.capsule),
558
+ /* The pair the analysis wants: what it cost, and how relevant it was thought to
559
+ be. null is the address, which was never ranked and never needed to be. */
560
+ score: entry.score ?? null,
561
+ via: entry.score === undefined ? "address" : this.retriever.name,
562
+ });
137
563
  this.seen.add(path);
564
+ /* The LINE, which is what actually entered the window. Remembering the capsule
565
+ instead was right only while every line happened to contain its capsule, and two
566
+ cases break that. An article with no capsule surfaces as a pointer, so the mark was
567
+ built from an empty string, fell under MARK_MINIMUM, and left the article seen for
568
+ the rest of the session: it could never surface again however long ago it left the
569
+ window. A build whose ranked line did not carry the capsule hit the mirror image,
570
+ testing for text that had never been shown and reading as departed every turn.
571
+ Both are the same mistake, which is testing presence against something other than
572
+ what was delivered.
573
+
574
+ This also retires the workaround the short-capsule case needed. A line always
575
+ carries its own address, so it always clears MARK_MINIMUM: `Cache.` fingerprints to
576
+ 5 characters and its line to 39. The guard still earns its place on the other call
577
+ site, where a read of a tiny article really can be too short to test. */
578
+ this.remember(path, line);
579
+ this.lastFlush.set(path, entry);
138
580
  this.staged.delete(path);
139
581
  }
140
582
  const plural = lines.length > 1 ? "s" : "";
141
- trace("flushed", { lines: lines.length, spent: this.spent });
583
+ trace("flushed", { lines: lines.length, chars: this.stats.chars });
142
584
  return (
143
585
  `[pi-canon] Governing article${plural} for what this turn touches. Read the full article with ` +
144
586
  `pi_canon before depending on details; update it after real changes.\n${lines.join("\n")}`
145
587
  );
146
588
  }
147
589
 
590
+ /* Put back everything the last flush and settle committed. Both mark their work done
591
+ before the message is handed to pi, because the message is built from that work; if
592
+ the send then fails, the agent never saw the nudge and the state is a lie. Undoing
593
+ restages the lines and restores the reminders, so the next turn tries again. */
594
+ undoFlush(): void {
595
+ for (const [path, entry] of this.lastFlush) {
596
+ this.staged.set(path, entry);
597
+ this.seen.delete(path);
598
+ this.marks.delete(path);
599
+ this.cost.delete(path);
600
+ this.surfacedEver.delete(path);
601
+ }
602
+ this.lastFlush.clear();
603
+ for (const path of this.lastNudge) this.pendingUpdates.add(path);
604
+ this.lastNudge = [];
605
+ trace("delivery-undone", {});
606
+ }
607
+
148
608
  /* The write-after half of the doctrine: every governing article touched since its
149
609
  last update draws one reminder, then the slate clears for the next batch. */
150
610
  settleNudge(): string | undefined {
151
611
  const stale = [...this.pendingUpdates];
612
+ this.lastNudge = stale;
152
613
  this.pendingUpdates.clear();
153
614
  if (!stale.length) return undefined;
154
615
  trace("settle-nudge", { paths: stale });