pi-canon 0.1.2 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +27 -11
- package/extensions/canon.ts +104 -13
- package/extensions/lib/lint.ts +128 -2
- package/extensions/lib/retrieval.ts +355 -0
- package/extensions/lib/store.ts +209 -17
- package/extensions/lib/surfacing.ts +496 -35
- package/extensions/lib/tool.ts +197 -17
- package/package.json +3 -2
|
@@ -0,0 +1,355 @@
|
|
|
1
|
+
/* Retrieval: the seam for knowledge the spine cannot address.
|
|
2
|
+
|
|
3
|
+
The spine answers every asset-scoped question deterministically and for free, and
|
|
4
|
+
nothing here changes that. Retrieval runs over the RESIDUE, the articles whose
|
|
5
|
+
address matches no asset, which is the category 1.0 named and left open ("knowledge
|
|
6
|
+
filed off the asset path never surfaces"). That scoping is what makes an expensive
|
|
7
|
+
ranker affordable: the corpus to rank is what has no home, not the whole store, and
|
|
8
|
+
it stays small precisely because everything with a home is answered without ranking.
|
|
9
|
+
|
|
10
|
+
Two built-ins ship. `none` is 1.0 exactly and is the experiment's control. `lexical`
|
|
11
|
+
is BM25 with nothing but the standard library. Anything that needs a model is
|
|
12
|
+
supplied by the caller as a function, so this package never grows a torch, ONNX or
|
|
13
|
+
network dependency and never decides which model you run.
|
|
14
|
+
|
|
15
|
+
Deliberately absent: any threshold. A retriever returns SCORES, and what to do with
|
|
16
|
+
them is measured rather than assumed. A constant deciding what an agent gets to see
|
|
17
|
+
is the mistake the session budget already made once. */
|
|
18
|
+
|
|
19
|
+
import { existsSync, readdirSync } from "node:fs";
|
|
20
|
+
import { dirname, join } from "node:path";
|
|
21
|
+
import type { CanonStore } from "./store.ts";
|
|
22
|
+
|
|
23
|
+
export interface Candidate {
|
|
24
|
+
path: string;
|
|
25
|
+
capsule: string;
|
|
26
|
+
body: string;
|
|
27
|
+
updated: string;
|
|
28
|
+
/* Whether this article SAID it is a cross-cutting rule, rather than being inferred
|
|
29
|
+
into the corpus by having no asset. See residue below. */
|
|
30
|
+
declared: boolean;
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
export interface Retriever {
|
|
34
|
+
name: string;
|
|
35
|
+
/* Called off the hot path whenever the residue changes. Free to be a no-op. */
|
|
36
|
+
index?(candidates: Candidate[]): void;
|
|
37
|
+
/* Higher is more relevant. A path missing from the result scored nothing. */
|
|
38
|
+
score(query: string, candidates: Candidate[]): Map<string, number>;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/* ---------------------------------------------------------------------------------
|
|
42
|
+
The query: intent, never evidence.
|
|
43
|
+
|
|
44
|
+
pi-fold measured this on memex and it is the single most expensive thing to get
|
|
45
|
+
wrong: a live window of 29,369 characters carried 29,244 characters of raw tool
|
|
46
|
+
output and 125 characters of the agent's own reasoning, so a lexical scorer was
|
|
47
|
+
matching against `read` and `shell` dumps and every retrieval number it produced was
|
|
48
|
+
that one defect. Tool RESULTS are therefore excluded outright.
|
|
49
|
+
|
|
50
|
+
The agent's own prose is excluded for a second, independent reason. Accordion's
|
|
51
|
+
relevance-signals exploration showed that a signal read off text the agent already
|
|
52
|
+
produced is retrospective: the text was generated while the article was absent, so
|
|
53
|
+
scoring against it measures agreement with the path already taken rather than need.
|
|
54
|
+
Where the agent went wrong for lack of an article, its own words argue against it.
|
|
55
|
+
|
|
56
|
+
What survives both objections is intent. A user message is exogenous and is a
|
|
57
|
+
statement of what is wanted. A tool call is agent-authored but is a forward
|
|
58
|
+
declaration of an action rather than reasoning about content, so the circularity
|
|
59
|
+
objection does not reach it; its name plus first meaningful argument is the whole
|
|
60
|
+
signal, and the rest is payload.
|
|
61
|
+
---------------------------------------------------------------------------------- */
|
|
62
|
+
|
|
63
|
+
const INTENT_ARGUMENT_KEYS = ["path", "file_path", "pattern", "query", "command", "description"];
|
|
64
|
+
const INTENT_ARGUMENT_CHARS = 200;
|
|
65
|
+
/* How much of the window's user speech counts as the current question. Newest first and
|
|
66
|
+
bounded rather than the whole window, because the position control is the strongest
|
|
67
|
+
number either package has on this: same window, oldest half MRR 0.0485 with 4 targets
|
|
68
|
+
in the top 5, newest half 0.1071 with 10. */
|
|
69
|
+
const USER_INTENT_CHARS = 1500;
|
|
70
|
+
/* What stands in for the middle of a message too long to carry whole. Visible on purpose:
|
|
71
|
+
a query built from two ends of a message should read as such. */
|
|
72
|
+
const ELISION = "\n...\n";
|
|
73
|
+
|
|
74
|
+
export interface IntentTurn {
|
|
75
|
+
role?: unknown;
|
|
76
|
+
content?: unknown;
|
|
77
|
+
toolName?: unknown;
|
|
78
|
+
input?: unknown;
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/* Pi carries content as a string or as typed parts; only text parts are speech. */
|
|
82
|
+
function messageText(content: unknown): string {
|
|
83
|
+
if (typeof content === "string") return content;
|
|
84
|
+
if (!Array.isArray(content)) return "";
|
|
85
|
+
const text: string[] = [];
|
|
86
|
+
for (const part of content) {
|
|
87
|
+
if (part && typeof part === "object" && (part as Record<string, unknown>).type === "text") {
|
|
88
|
+
text.push(String((part as Record<string, unknown>).text ?? ""));
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
return text.join("\n");
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/* The user's own words out of the projection, newest first and bounded.
|
|
95
|
+
|
|
96
|
+
This is the best query either finding endorses without qualification: exogenous, so
|
|
97
|
+
the circularity objection cannot reach it, and a statement of what is wanted rather
|
|
98
|
+
than evidence of what was found.
|
|
99
|
+
|
|
100
|
+
pi-canon's own steered nudges arrive as messages too, and ranking articles against
|
|
101
|
+
the text of previous article nudges would be a feedback loop that scores an article
|
|
102
|
+
highly for having been surfaced already. They are excluded by both their customType
|
|
103
|
+
and their visible prefix, because only one of the two survives every delivery path. */
|
|
104
|
+
export function userIntent(messages: unknown): IntentTurn[] {
|
|
105
|
+
if (!Array.isArray(messages)) return [];
|
|
106
|
+
const picked: IntentTurn[] = [];
|
|
107
|
+
let chars = 0;
|
|
108
|
+
for (let i = messages.length - 1; i >= 0 && chars < USER_INTENT_CHARS; i -= 1) {
|
|
109
|
+
const message = messages[i] as Record<string, unknown> | null;
|
|
110
|
+
if (!message || typeof message !== "object") continue;
|
|
111
|
+
if (message.role !== "user") continue;
|
|
112
|
+
if (message.customType === "pi-canon") continue;
|
|
113
|
+
const text = messageText(message.content).trim();
|
|
114
|
+
if (!text || text.startsWith("[pi-canon]")) continue;
|
|
115
|
+
/* Both ends of an over-long message, never one.
|
|
116
|
+
|
|
117
|
+
This kept the head, was changed to keep the tail on the argument that the position
|
|
118
|
+
effect should apply within a message as it does across the window, and neither is
|
|
119
|
+
right. The measured effect is about position in the WINDOW, and "fix X, here are the
|
|
120
|
+
logs" is at least as common as a trailing ask (Codex, 2026-08-13, whose example was
|
|
121
|
+
the review brief it was reading, where tail-only would have discarded every numbered
|
|
122
|
+
question and ranked on the closing section alone). Neither end is reliably the ask,
|
|
123
|
+
so when a message will not fit, keep the opening and the closing and drop the
|
|
124
|
+
middle, which is payload in both shapes. */
|
|
125
|
+
const room = USER_INTENT_CHARS - chars;
|
|
126
|
+
/* The elision marker is part of the budget, not on top of it, or the bound is not a
|
|
127
|
+
bound: half plus half plus the marker came out over USER_INTENT_CHARS. */
|
|
128
|
+
const half = Math.max(0, Math.floor((room - ELISION.length) / 2));
|
|
129
|
+
const bounded = text.length <= room ? text
|
|
130
|
+
: `${text.slice(0, half)}${ELISION}${text.slice(-half)}`;
|
|
131
|
+
chars += bounded.length;
|
|
132
|
+
picked.push({ role: "user", content: bounded });
|
|
133
|
+
}
|
|
134
|
+
/* Reversed back to oldest first, so the caller can append this turn's tool calls
|
|
135
|
+
after them and intentQuery's newest-first walk still reads in true order. */
|
|
136
|
+
return picked.reverse();
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
function firstArgument(input: unknown): string {
|
|
140
|
+
if (!input || typeof input !== "object") return "";
|
|
141
|
+
for (const key of INTENT_ARGUMENT_KEYS) {
|
|
142
|
+
const value = (input as Record<string, unknown>)[key];
|
|
143
|
+
if (typeof value === "string" && value.trim()) return value.trim().slice(0, INTENT_ARGUMENT_CHARS);
|
|
144
|
+
}
|
|
145
|
+
return "";
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
/* Newest first, because relevance to now sits at the tail: replaying 56 real retrieval
|
|
149
|
+
targets, pi-fold measured MRR 0.1071 with 10 targets in the top 5 for the newest half
|
|
150
|
+
of a window against 0.0485 and 4 for the oldest half of the SAME window, which makes
|
|
151
|
+
it a position effect rather than a length effect. */
|
|
152
|
+
export function intentQuery(turns: readonly IntentTurn[]): string {
|
|
153
|
+
const pieces: string[] = [];
|
|
154
|
+
for (let i = turns.length - 1; i >= 0; i -= 1) {
|
|
155
|
+
const turn = turns[i];
|
|
156
|
+
if (turn?.toolName) {
|
|
157
|
+
const argument = firstArgument(turn.input);
|
|
158
|
+
pieces.push(argument ? `${String(turn.toolName)} ${argument}` : String(turn.toolName));
|
|
159
|
+
continue;
|
|
160
|
+
}
|
|
161
|
+
if (turn?.role === "user") {
|
|
162
|
+
const text = messageText(turn.content).trim();
|
|
163
|
+
if (text) pieces.push(text);
|
|
164
|
+
}
|
|
165
|
+
}
|
|
166
|
+
return pieces.join("\n");
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
/* ---------------------------------------------------------------------------------
|
|
170
|
+
BM25, standard library only.
|
|
171
|
+
|
|
172
|
+
Reimplemented rather than lifted from pi-fold. The two packages publish
|
|
173
|
+
independently, and a verbatim copy between them rebuilds exactly the drift problem
|
|
174
|
+
pi-fold deliberately retired. What crosses is the lessons, above and below, not the
|
|
175
|
+
code.
|
|
176
|
+
---------------------------------------------------------------------------------- */
|
|
177
|
+
|
|
178
|
+
const K1 = 1.5;
|
|
179
|
+
const B = 0.75;
|
|
180
|
+
const MIN_TOKEN = 4;
|
|
181
|
+
const STOPWORDS = new Set([
|
|
182
|
+
"about", "after", "again", "against", "because", "been", "before", "being", "between",
|
|
183
|
+
"both", "cannot", "could", "does", "doing", "down", "during", "each", "from", "further",
|
|
184
|
+
"have", "having", "here", "into", "itself", "just", "more", "most", "once", "only",
|
|
185
|
+
"other", "over", "same", "should", "some", "such", "than", "that", "their", "them",
|
|
186
|
+
"then", "there", "these", "they", "this", "those", "through", "under", "until", "very",
|
|
187
|
+
"were", "what", "when", "where", "which", "while", "will", "with", "would", "your",
|
|
188
|
+
]);
|
|
189
|
+
|
|
190
|
+
/* Tokens WITH repetition: BM25 reads term frequency, and deduping here would silently
|
|
191
|
+
flatten every tf to 1. */
|
|
192
|
+
export function tokens(value: string): string[] {
|
|
193
|
+
const out: string[] = [];
|
|
194
|
+
for (const raw of value.toLowerCase().split(/[^a-z0-9_]+/)) {
|
|
195
|
+
const token = raw.replace(/^_+|_+$/g, "");
|
|
196
|
+
if (token.length < MIN_TOKEN || STOPWORDS.has(token)) continue;
|
|
197
|
+
out.push(token);
|
|
198
|
+
}
|
|
199
|
+
return out;
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
interface Indexed {
|
|
203
|
+
length: number;
|
|
204
|
+
frequencies: Map<string, number>;
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
export class LexicalRetriever implements Retriever {
|
|
208
|
+
readonly name = "lexical";
|
|
209
|
+
private documents = new Map<string, Indexed>();
|
|
210
|
+
private documentFrequency = new Map<string, number>();
|
|
211
|
+
private averageLength = 0;
|
|
212
|
+
|
|
213
|
+
index(candidates: Candidate[]): void {
|
|
214
|
+
this.documents = new Map();
|
|
215
|
+
this.documentFrequency = new Map();
|
|
216
|
+
let total = 0;
|
|
217
|
+
for (const candidate of candidates) {
|
|
218
|
+
const frequencies = new Map<string, number>();
|
|
219
|
+
const terms = tokens(`${candidate.path} ${candidate.capsule} ${candidate.body}`);
|
|
220
|
+
for (const term of terms) frequencies.set(term, (frequencies.get(term) ?? 0) + 1);
|
|
221
|
+
this.documents.set(candidate.path, { length: terms.length, frequencies });
|
|
222
|
+
total += terms.length;
|
|
223
|
+
for (const term of frequencies.keys()) {
|
|
224
|
+
this.documentFrequency.set(term, (this.documentFrequency.get(term) ?? 0) + 1);
|
|
225
|
+
}
|
|
226
|
+
}
|
|
227
|
+
this.averageLength = candidates.length ? total / candidates.length : 0;
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
private idf(term: string): number {
|
|
231
|
+
const n = this.documentFrequency.get(term) ?? 0;
|
|
232
|
+
if (!n) return 0;
|
|
233
|
+
return Math.log(1 + (this.documents.size - n + 0.5) / (n + 0.5));
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
/* Divide by what a document scores when every query term is saturated in it. BM25's
|
|
237
|
+
raw score is unbounded and its scale moves with the query, so the same number means
|
|
238
|
+
different things run to run; against the ceiling it is an absolute 0..1 reading,
|
|
239
|
+
which is what makes a score comparable across turns and worth recording beside the
|
|
240
|
+
context it cost. */
|
|
241
|
+
private ceiling(terms: ReadonlySet<string>): number {
|
|
242
|
+
let ceiling = 0;
|
|
243
|
+
for (const term of terms) ceiling += this.idf(term) * (K1 + 1);
|
|
244
|
+
return ceiling;
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
score(query: string, candidates: Candidate[]): Map<string, number> {
|
|
248
|
+
const out = new Map<string, number>();
|
|
249
|
+
const terms = new Set(tokens(query));
|
|
250
|
+
if (!terms.size || !this.documents.size) return out;
|
|
251
|
+
const ceiling = this.ceiling(terms);
|
|
252
|
+
if (ceiling <= 0) return out;
|
|
253
|
+
for (const candidate of candidates) {
|
|
254
|
+
const document = this.documents.get(candidate.path);
|
|
255
|
+
if (!document?.length) continue;
|
|
256
|
+
let score = 0;
|
|
257
|
+
for (const term of terms) {
|
|
258
|
+
const frequency = document.frequencies.get(term);
|
|
259
|
+
if (!frequency) continue;
|
|
260
|
+
const denominator =
|
|
261
|
+
frequency + K1 * (1 - B + B * (document.length / (this.averageLength || 1)));
|
|
262
|
+
score += this.idf(term) * ((frequency * (K1 + 1)) / denominator);
|
|
263
|
+
}
|
|
264
|
+
if (score > 0) out.set(candidate.path, score / ceiling);
|
|
265
|
+
}
|
|
266
|
+
return out;
|
|
267
|
+
}
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
export const NONE: Retriever = {
|
|
271
|
+
name: "none",
|
|
272
|
+
score: () => new Map(),
|
|
273
|
+
};
|
|
274
|
+
|
|
275
|
+
export type RetrievalOption = "none" | "lexical" | Retriever;
|
|
276
|
+
|
|
277
|
+
export function buildRetriever(option: RetrievalOption | undefined): Retriever {
|
|
278
|
+
if (option === undefined || option === "none") return NONE;
|
|
279
|
+
if (option === "lexical") return new LexicalRetriever();
|
|
280
|
+
if (typeof option === "object" && option && typeof option.score === "function") {
|
|
281
|
+
if (typeof option.name !== "string" || !option.name.trim()) {
|
|
282
|
+
throw new Error("pi-canon: a retrieval object needs a name, so runs can be told apart.");
|
|
283
|
+
}
|
|
284
|
+
return option;
|
|
285
|
+
}
|
|
286
|
+
throw new Error(
|
|
287
|
+
`pi-canon: retrieval must be "none", "lexical", or an object with a name and a score function; got ${
|
|
288
|
+
typeof option === "string" ? `"${option}"` : typeof option
|
|
289
|
+
}.`,
|
|
290
|
+
);
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
/* The residue: articles the spine can never surface, because surfacing is triggered by
|
|
294
|
+
touching an asset and there is nothing here to touch. Everything else is already
|
|
295
|
+
answered for free and must not be ranked, or retrieval would compete with the address
|
|
296
|
+
instead of completing it.
|
|
297
|
+
|
|
298
|
+
Membership is still decided by having no asset, and that is deliberate: an article
|
|
299
|
+
unreachable by address has exactly one mechanism left, and dropping it from that
|
|
300
|
+
mechanism would lose it outright. But the set has always held two populations, and
|
|
301
|
+
until now nothing could tell them apart (Codex, 2026-08-12). One is the deliberate
|
|
302
|
+
cross-cutting rule the doctrine asks for, filed at an address naming the rule. The
|
|
303
|
+
other is an accident: a typo in an address, or an article whose asset was deleted
|
|
304
|
+
under it. Defining the corpus only by what it is not made those the same thing.
|
|
305
|
+
|
|
306
|
+
So an article may DECLARE itself with `scope: rule`, and every candidate carries
|
|
307
|
+
whether it did. Nothing here filters on it, because a declaration the agent forgot
|
|
308
|
+
must not cost it the only mechanism that can reach it. What the flag buys is honesty:
|
|
309
|
+
a declared article is a rule on purpose, an undeclared one is a question, and the two
|
|
310
|
+
stop being counted as one number. */
|
|
311
|
+
export function residue(store: CanonStore, dir: string): Candidate[] {
|
|
312
|
+
const out: Candidate[] = [];
|
|
313
|
+
for (const path of store.list()) {
|
|
314
|
+
const article = store.read(path);
|
|
315
|
+
if (!article) continue;
|
|
316
|
+
/* Declared first, so a rule stays reachable even if a file later appears at its
|
|
317
|
+
address. Membership was decided by the filesystem alone, which meant a
|
|
318
|
+
deliberate cross-cutting rule silently dropped out of the only mechanism that
|
|
319
|
+
reaches it the moment something coincided with its name (Sol Pro, 2026-08-13).
|
|
320
|
+
Undeclared and off the asset path still qualifies, so forgetting the flag stays
|
|
321
|
+
fail-open. */
|
|
322
|
+
if (article.scope !== RULE_SCOPE && governsAnAsset(dir, path)) continue;
|
|
323
|
+
out.push({
|
|
324
|
+
path,
|
|
325
|
+
capsule: article.capsule,
|
|
326
|
+
body: article.body,
|
|
327
|
+
updated: article.updated,
|
|
328
|
+
declared: article.scope === RULE_SCOPE,
|
|
329
|
+
});
|
|
330
|
+
}
|
|
331
|
+
return out;
|
|
332
|
+
}
|
|
333
|
+
|
|
334
|
+
/* The one declared value. A second would be a taxonomy, and nothing has asked for one. */
|
|
335
|
+
export const RULE_SCOPE = "rule";
|
|
336
|
+
|
|
337
|
+
/* An address governs an asset when something on disk normalizes back to it. Addresses
|
|
338
|
+
drop the extension, so `src/core/config` has to match `src/core/config.ts` as well as
|
|
339
|
+
a bare `src/core/config` directory or file, which is one readdir of the parent rather
|
|
340
|
+
than a walk of the tree. A path that cannot be read is treated as governing nothing,
|
|
341
|
+
which puts the article in the residue: over-including costs a ranking slot, while
|
|
342
|
+
under-including would silently drop knowledge from the only mechanism that can reach
|
|
343
|
+
it. */
|
|
344
|
+
export function governsAnAsset(dir: string, path: string): boolean {
|
|
345
|
+
const full = join(dir, path);
|
|
346
|
+
if (existsSync(full)) return true;
|
|
347
|
+
const base = path.slice(path.lastIndexOf("/") + 1);
|
|
348
|
+
try {
|
|
349
|
+
return readdirSync(dirname(full)).some(
|
|
350
|
+
(entry) => entry === base || (entry.startsWith(`${base}.`) && !entry.slice(base.length + 1).includes(".")),
|
|
351
|
+
);
|
|
352
|
+
} catch {
|
|
353
|
+
return false;
|
|
354
|
+
}
|
|
355
|
+
}
|
package/extensions/lib/store.ts
CHANGED
|
@@ -4,19 +4,22 @@
|
|
|
4
4
|
so the tree stays hand editable and Obsidian readable with no parser dependency;
|
|
5
5
|
keys this package does not own are carried through writes untouched. */
|
|
6
6
|
|
|
7
|
-
import { existsSync, mkdirSync, readdirSync, readFileSync, writeFileSync } from "node:fs";
|
|
7
|
+
import { appendFileSync, existsSync, mkdirSync, readdirSync, readFileSync, statSync, writeFileSync } from "node:fs";
|
|
8
8
|
import { dirname, join } from "node:path";
|
|
9
9
|
|
|
10
10
|
export interface Article {
|
|
11
11
|
path: string;
|
|
12
12
|
capsule: string;
|
|
13
13
|
updated: string;
|
|
14
|
+
/* Declared scope. "rule" means this article names a cross-cutting rule and is not
|
|
15
|
+
expected to govern an asset; empty means the address is the claim, as usual. */
|
|
16
|
+
scope: string;
|
|
14
17
|
extra: string[];
|
|
15
18
|
body: string;
|
|
16
19
|
}
|
|
17
20
|
|
|
18
21
|
const FRONT_MATTER = /^---\r?\n([\s\S]*?)\r?\n---(?:\r?\n|$)/;
|
|
19
|
-
const OWNED_KEYS = new Set(["capsule", "updated"]);
|
|
22
|
+
const OWNED_KEYS = new Set(["capsule", "updated", "scope"]);
|
|
20
23
|
|
|
21
24
|
/* Keep an address inside the tree: .. resolves against its own segments, so
|
|
22
25
|
a/b/../c means a/c, and clamps at the root, so nothing ever escapes. */
|
|
@@ -30,6 +33,16 @@ function contain(path: string): string {
|
|
|
30
33
|
return parts.join("/");
|
|
31
34
|
}
|
|
32
35
|
|
|
36
|
+
/* The same address with nothing dropped. Needed to tell an asset path from an address
|
|
37
|
+
that is already canonical: normalize drops one extension, so putting the article
|
|
38
|
+
address src/core/config.test back through it yields src/core/config, a different
|
|
39
|
+
article. Callers that accept an address from an agent must not canonicalise twice. */
|
|
40
|
+
export function contained(path: string, cwd = ""): string {
|
|
41
|
+
let out = path.trim().replace(/\\/g, "/");
|
|
42
|
+
if (cwd && (out === cwd || out.startsWith(`${cwd}/`))) out = out.slice(cwd.length);
|
|
43
|
+
return contain(out);
|
|
44
|
+
}
|
|
45
|
+
|
|
33
46
|
/* An asset address: relative to the project, contained, file extension dropped so
|
|
34
47
|
src/core/config.ts shares its article's address. The drop happens once, here at
|
|
35
48
|
the boundary; the store itself never drops again, or config.test would lose its
|
|
@@ -91,6 +104,34 @@ function unscalar(value: string): string {
|
|
|
91
104
|
return value;
|
|
92
105
|
}
|
|
93
106
|
|
|
107
|
+
/* Which collision an entry was: name.md is the first, name-2.md the second. A burst of
|
|
108
|
+
entries lands inside one millisecond and shares a stamp, so this is what actually
|
|
109
|
+
separates them, and it is the same counter that wrote the file. */
|
|
110
|
+
/* One journal entry as the index carries it: name, the instant it recorded, its subjects.
|
|
111
|
+
Short keys because this file grows one line per entry forever and is never read by a
|
|
112
|
+
human; the shape is documented here instead. */
|
|
113
|
+
interface JournalRow {
|
|
114
|
+
n: string;
|
|
115
|
+
a: string;
|
|
116
|
+
s: string[];
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
function sequenceOf(name: string): number {
|
|
120
|
+
return Number(/-(\d+)\.md$/.exec(name)?.[1] ?? 1);
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/* An inline array of scalars, each optionally quoted. Splitting on every comma broke
|
|
124
|
+
any value that legitimately contained one. */
|
|
125
|
+
function subjectList(raw: string): string[] {
|
|
126
|
+
const inner = raw.trim().replace(/^\[|\]$/g, "");
|
|
127
|
+
const out: string[] = [];
|
|
128
|
+
for (const match of inner.matchAll(/"((?:[^"\\]|\\.)*)"|([^,]+)/g)) {
|
|
129
|
+
const value = match[1] !== undefined ? unscalar(`"${match[1]}"`) : (match[2] ?? "").trim();
|
|
130
|
+
if (value) out.push(value);
|
|
131
|
+
}
|
|
132
|
+
return out;
|
|
133
|
+
}
|
|
134
|
+
|
|
94
135
|
function today(): string {
|
|
95
136
|
return new Date().toISOString().slice(0, 10);
|
|
96
137
|
}
|
|
@@ -99,6 +140,7 @@ function serialize(article: Article): string {
|
|
|
99
140
|
const meta = [
|
|
100
141
|
article.capsule ? `capsule: ${scalar(article.capsule)}` : "",
|
|
101
142
|
`updated: ${article.updated}`,
|
|
143
|
+
article.scope ? `scope: ${scalar(article.scope)}` : "",
|
|
102
144
|
...article.extra,
|
|
103
145
|
].filter(Boolean).join("\n");
|
|
104
146
|
return `---\n${meta}\n---\n${article.body.trimEnd()}\n`;
|
|
@@ -134,6 +176,7 @@ export class CanonStore {
|
|
|
134
176
|
extra,
|
|
135
177
|
capsule: typeof meta.capsule === "string" ? meta.capsule : "",
|
|
136
178
|
updated: typeof meta.updated === "string" ? meta.updated : "",
|
|
179
|
+
scope: typeof meta.scope === "string" ? meta.scope : "",
|
|
137
180
|
};
|
|
138
181
|
}
|
|
139
182
|
|
|
@@ -163,7 +206,36 @@ export class CanonStore {
|
|
|
163
206
|
return walk(this.articlesDir, "").sort();
|
|
164
207
|
}
|
|
165
208
|
|
|
166
|
-
|
|
209
|
+
/* What the store looks like right now, cheaply enough to ask every turn.
|
|
210
|
+
|
|
211
|
+
Retrieval re-reads and re-parses every article once a turn to rebuild the residue, and
|
|
212
|
+
the comment justifying that says the residue is small by construction. Measured, the
|
|
213
|
+
cost tracks the STORE and not the residue: 5,000 articles cost 242ms a turn when all of
|
|
214
|
+
them are residue and 214ms when only 50 are, because every path is read and stat-ed
|
|
215
|
+
before anything is filtered. At 20,000 articles it is 891ms, on every turn.
|
|
216
|
+
|
|
217
|
+
Statting instead of reading is 8x cheaper, 114ms against 891ms at 20,000. mtimeMs is the
|
|
218
|
+
reason this works where the obvious key does not: `updated` has day granularity, so an
|
|
219
|
+
article rewritten in the same session carries the same stamp and any cache built on it
|
|
220
|
+
serves a stale ranking for the rest of the run. Milliseconds do not have that problem,
|
|
221
|
+
and size catches a same-millisecond rewrite of a different length. */
|
|
222
|
+
signature(): string {
|
|
223
|
+
const parts: string[] = [];
|
|
224
|
+
const walk = (dir: string, prefix: string): void => {
|
|
225
|
+
if (!existsSync(dir)) return;
|
|
226
|
+
for (const entry of readdirSync(dir, { withFileTypes: true })) {
|
|
227
|
+
if (entry.isDirectory()) walk(join(dir, entry.name), `${prefix}${entry.name}/`);
|
|
228
|
+
else if (entry.name.endsWith(".md")) {
|
|
229
|
+
const stats = statSync(join(dir, entry.name));
|
|
230
|
+
parts.push(`${prefix}${entry.name}:${stats.mtimeMs}:${stats.size}`);
|
|
231
|
+
}
|
|
232
|
+
}
|
|
233
|
+
};
|
|
234
|
+
walk(this.articlesDir, "");
|
|
235
|
+
return parts.sort().join("|");
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
write(path: string, fields: { capsule?: string; body?: string; scope?: string }): Article {
|
|
167
239
|
path = contain(path);
|
|
168
240
|
const prior = this.read(path);
|
|
169
241
|
/* Agents sometimes paste a whole file as the body, front matter included; stored
|
|
@@ -178,6 +250,7 @@ export class CanonStore {
|
|
|
178
250
|
path,
|
|
179
251
|
capsule: (fields.capsule ?? prior?.capsule ?? "").replace(/\s*\n\s*/g, " ").trim(),
|
|
180
252
|
updated: today(),
|
|
253
|
+
scope: (fields.scope ?? prior?.scope ?? "").trim(),
|
|
181
254
|
extra: prior?.extra ?? [],
|
|
182
255
|
body,
|
|
183
256
|
};
|
|
@@ -194,12 +267,24 @@ export class CanonStore {
|
|
|
194
267
|
const slug =
|
|
195
268
|
(entry.slug ?? "entry").toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/(^-|-$)/g, "") || "entry";
|
|
196
269
|
mkdirSync(this.journalDir, { recursive: true });
|
|
197
|
-
|
|
198
|
-
|
|
270
|
+
/* An explicit instant, because the filename cannot carry one. Entries are named by
|
|
271
|
+
date plus slug with -2, -3 for collisions, and sorting those lexicographically puts
|
|
272
|
+
-10 before -2 and the unsuffixed name after every suffixed one, so "the newest
|
|
273
|
+
three" was not the newest three (Codex, 2026-08-13). Subjects are quoted through
|
|
274
|
+
the same scalar() as everything else, so an address containing a comma survives
|
|
275
|
+
the round trip instead of splitting into two. */
|
|
276
|
+
const stamp = new Date().toISOString();
|
|
277
|
+
const front = [
|
|
278
|
+
entry.subject?.length ? `subject: [${entry.subject.map(scalar).join(", ")}]` : "",
|
|
279
|
+
`logged: ${stamp}`,
|
|
280
|
+
].filter(Boolean).join("\n");
|
|
281
|
+
const text = `---\n${front}\n---\n${entry.body.trimEnd()}\n`;
|
|
199
282
|
for (let n = 1; ; n += 1) {
|
|
200
|
-
const
|
|
283
|
+
const name = `${today()}-${slug}${n > 1 ? `-${n}` : ""}.md`;
|
|
284
|
+
const file = join(this.journalDir, name);
|
|
201
285
|
try {
|
|
202
286
|
writeFileSync(file, text, { flag: "wx" });
|
|
287
|
+
this.note({ n: name, a: stamp, s: entry.subject ?? [] });
|
|
203
288
|
return file;
|
|
204
289
|
} catch (error) {
|
|
205
290
|
if ((error as NodeJS.ErrnoException).code !== "EEXIST") throw error;
|
|
@@ -207,6 +292,59 @@ export class CanonStore {
|
|
|
207
292
|
}
|
|
208
293
|
}
|
|
209
294
|
|
|
295
|
+
/* Append the new entry to both the loaded index and the file on disk, so the count stays
|
|
296
|
+
matched and the next session loads instead of rebuilding. Appending only when the index
|
|
297
|
+
is already loaded would leave the file one row short of the directory and force a
|
|
298
|
+
rebuild every session that journals without reading. */
|
|
299
|
+
private note(row: JournalRow): void {
|
|
300
|
+
try {
|
|
301
|
+
if (existsSync(this.indexFile)) {
|
|
302
|
+
if (this.index) this.index.push(row);
|
|
303
|
+
appendFileSync(this.indexFile, JSON.stringify(row) + "\n");
|
|
304
|
+
return;
|
|
305
|
+
}
|
|
306
|
+
/* No index yet. Drop what is cached and rebuild from the directory, which already
|
|
307
|
+
holds this entry, so the file is created complete rather than one row short. */
|
|
308
|
+
this.index = undefined;
|
|
309
|
+
this.rows();
|
|
310
|
+
} catch {
|
|
311
|
+
/* Losing the append costs a rebuild next session, never a failed write. */
|
|
312
|
+
}
|
|
313
|
+
}
|
|
314
|
+
|
|
315
|
+
/* Every journal entry, with its body. Distinct from the index, which deliberately holds
|
|
316
|
+
no bodies so an article read never opens a journal file: this reads them all, and is
|
|
317
|
+
therefore only for the agent-solicited path, never the hot one. A journal entry has no
|
|
318
|
+
address, so what scopes it is its instant and the subjects it names. */
|
|
319
|
+
journalEntries(): { name: string; logged: string; subjects: string[]; body: string }[] {
|
|
320
|
+
let names: string[];
|
|
321
|
+
try {
|
|
322
|
+
names = readdirSync(this.journalDir).filter((name) => name.endsWith(".md"));
|
|
323
|
+
} catch {
|
|
324
|
+
return [];
|
|
325
|
+
}
|
|
326
|
+
const out = [];
|
|
327
|
+
for (const name of names.sort()) {
|
|
328
|
+
try {
|
|
329
|
+
const text = readFileSync(join(this.journalDir, name), "utf8");
|
|
330
|
+
/* The same two regexes CanonStore.row uses, deliberately. Journal front matter is
|
|
331
|
+
written as `subject: [a, b]`, which the generic parser does not return under
|
|
332
|
+
`meta`, so reading it a second way here would let search and the journal index
|
|
333
|
+
disagree about what an entry names. */
|
|
334
|
+
out.push({
|
|
335
|
+
name,
|
|
336
|
+
logged: /^logged:\s*(.*)$/m.exec(text)?.[1]?.trim() || name.slice(0, 10),
|
|
337
|
+
subjects: subjectList(/^subject:\s*(.*)$/m.exec(text)?.[1] ?? ""),
|
|
338
|
+
body: text.replace(/^---\n[\s\S]*?\n---\n/, ""),
|
|
339
|
+
});
|
|
340
|
+
} catch {
|
|
341
|
+
/* An unreadable entry is skipped, never fatal: search is a convenience and one
|
|
342
|
+
bad file must not make the whole journal unsearchable. */
|
|
343
|
+
}
|
|
344
|
+
}
|
|
345
|
+
return out;
|
|
346
|
+
}
|
|
347
|
+
|
|
210
348
|
journalCount(): number {
|
|
211
349
|
try {
|
|
212
350
|
return readdirSync(this.journalDir).filter((name) => name.endsWith(".md")).length;
|
|
@@ -215,20 +353,74 @@ export class CanonStore {
|
|
|
215
353
|
}
|
|
216
354
|
}
|
|
217
355
|
|
|
218
|
-
/*
|
|
219
|
-
|
|
220
|
-
|
|
356
|
+
/* One row per entry: filename, the instant it recorded, the addresses it names.
|
|
357
|
+
Everything journalMentions needs, so a read never opens a journal file. */
|
|
358
|
+
private index: JournalRow[] | undefined;
|
|
359
|
+
|
|
360
|
+
private get indexFile(): string {
|
|
361
|
+
return join(this.root, ".journal-index.jsonl");
|
|
362
|
+
}
|
|
363
|
+
|
|
364
|
+
private static row(dir: string, name: string): JournalRow {
|
|
365
|
+
const text = readFileSync(join(dir, name), "utf8");
|
|
366
|
+
/* Hand written and pre-2.0 entries have no stamp; the date in the name is the
|
|
367
|
+
best available and still orders them against each other. */
|
|
368
|
+
return {
|
|
369
|
+
n: name,
|
|
370
|
+
a: /^logged:\s*(.*)$/m.exec(text)?.[1]?.trim() || name.slice(0, 10),
|
|
371
|
+
s: subjectList(/^subject:\s*(.*)$/m.exec(text)?.[1] ?? ""),
|
|
372
|
+
};
|
|
373
|
+
}
|
|
374
|
+
|
|
375
|
+
/* The index, checked against one directory listing and no file reads.
|
|
376
|
+
|
|
377
|
+
Holding the cache without checking was wrong: a second CanonStore on the same root, or
|
|
378
|
+
this one after an entry arrives from outside, answers from a snapshot that has since
|
|
379
|
+
moved. One readdir per call is the cheap validation, and it is still the whole point,
|
|
380
|
+
because what this replaced read every entry on every article read.
|
|
381
|
+
|
|
382
|
+
What a count cannot see is an entry whose subject line is edited in place: the count
|
|
383
|
+
matches and the stale row stands. That is the deliberate trade. Catching it means
|
|
384
|
+
opening every entry, which is the scan this exists to remove, and the cost of being
|
|
385
|
+
wrong is one filename missing from a hint list that only invites digging. */
|
|
386
|
+
private rows(): JournalRow[] {
|
|
387
|
+
const count = this.journalCount();
|
|
388
|
+
if (this.index && this.index.length === count) return this.index;
|
|
389
|
+
let loaded: JournalRow[] | undefined;
|
|
221
390
|
try {
|
|
222
|
-
|
|
223
|
-
.
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
return subject.replace(/^\[|\]$/g, "").split(",").some((s) => s.trim() === path);
|
|
227
|
-
})
|
|
228
|
-
.sort();
|
|
391
|
+
loaded = readFileSync(this.indexFile, "utf8")
|
|
392
|
+
.split("\n")
|
|
393
|
+
.filter(Boolean)
|
|
394
|
+
.map((line) => JSON.parse(line) as JournalRow);
|
|
229
395
|
} catch {
|
|
230
|
-
|
|
396
|
+
loaded = undefined;
|
|
231
397
|
}
|
|
398
|
+
if (loaded && loaded.length === count) return (this.index = loaded);
|
|
399
|
+
let rebuilt: JournalRow[] = [];
|
|
400
|
+
try {
|
|
401
|
+
rebuilt = readdirSync(this.journalDir)
|
|
402
|
+
.filter((name) => name.endsWith(".md"))
|
|
403
|
+
.map((name) => CanonStore.row(this.journalDir, name));
|
|
404
|
+
} catch {
|
|
405
|
+
return (this.index = []);
|
|
406
|
+
}
|
|
407
|
+
try {
|
|
408
|
+
mkdirSync(this.root, { recursive: true });
|
|
409
|
+
writeFileSync(this.indexFile, rebuilt.map((r) => JSON.stringify(r)).join("\n") + "\n");
|
|
410
|
+
} catch {
|
|
411
|
+
/* An unwritable index costs a rebuild next session, never a failed read. */
|
|
412
|
+
}
|
|
413
|
+
return (this.index = rebuilt);
|
|
414
|
+
}
|
|
415
|
+
|
|
416
|
+
/* Journal entries whose subject names this address: the index a read surfaces
|
|
417
|
+
so the agent can dig into event history when it wants more than current truth. */
|
|
418
|
+
/* Oldest first, by the instant the entry recorded rather than by its filename. */
|
|
419
|
+
journalMentions(path: string): string[] {
|
|
420
|
+
return this.rows()
|
|
421
|
+
.filter((row) => row.s.includes(path))
|
|
422
|
+
.sort((a, b) => (a.a === b.a ? sequenceOf(a.n) - sequenceOf(b.n) : a.a.localeCompare(b.a)))
|
|
423
|
+
.map((row) => row.n);
|
|
232
424
|
}
|
|
233
425
|
|
|
234
426
|
map(under = ""): string {
|