@sayknow-cli/coding-agent 0.5.25 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/CHANGELOG.md +36 -1
  2. package/dist/types/config/settings-schema.d.ts +51 -5
  3. package/dist/types/config/task-model-specialties.d.ts +55 -0
  4. package/dist/types/decisions/keyword-learning.d.ts +61 -0
  5. package/dist/types/decisions/llm-backend.d.ts +13 -1
  6. package/dist/types/decisions/prompt-triage.d.ts +42 -0
  7. package/dist/types/decisions/skill-routing.d.ts +41 -6
  8. package/dist/types/decisions/task-routing.d.ts +96 -11
  9. package/dist/types/hooks/native-prompt-routing.d.ts +21 -0
  10. package/dist/types/hooks/native-skill-hook.d.ts +3 -0
  11. package/dist/types/hooks/skill-keywords.d.ts +9 -0
  12. package/dist/types/hooks/skill-state.d.ts +20 -3
  13. package/dist/types/hooks/ui-skill-keywords.d.ts +15 -0
  14. package/dist/types/i18n/messages/en.d.ts +15 -0
  15. package/dist/types/lsp/index.d.ts +1 -1
  16. package/dist/types/lsp/types.d.ts +1 -1
  17. package/dist/types/modes/components/model-selector.d.ts +11 -0
  18. package/dist/types/sdk/session.d.ts +3 -13
  19. package/dist/types/session/agent-session.d.ts +8 -0
  20. package/dist/types/session/auth-storage-discovery.d.ts +13 -0
  21. package/dist/types/task/index.d.ts +1 -1
  22. package/dist/types/task/receipt.d.ts +2 -0
  23. package/dist/types/task/types.d.ts +114 -18
  24. package/dist/types/tools/browser.d.ts +2 -2
  25. package/dist/types/tools/subagent.d.ts +2 -2
  26. package/package.json +7 -7
  27. package/scripts/eval-skill-routing.ts +37 -12
  28. package/src/config/settings-schema.ts +64 -12
  29. package/src/config/task-model-specialties.ts +131 -0
  30. package/src/decisions/index.ts +8 -2
  31. package/src/decisions/keyword-learning.ts +678 -0
  32. package/src/decisions/llm-backend.ts +213 -67
  33. package/src/decisions/prompt-triage.ts +163 -0
  34. package/src/decisions/skill-routing.ts +39 -56
  35. package/src/decisions/task-routing.ts +382 -66
  36. package/src/decisions/typesafe-backend.ts +3 -0
  37. package/src/hooks/native-prompt-routing.ts +190 -0
  38. package/src/hooks/native-skill-hook.ts +21 -12
  39. package/src/hooks/skill-keywords.ts +9 -0
  40. package/src/hooks/skill-state.ts +41 -10
  41. package/src/hooks/ui-skill-keywords.ts +67 -10
  42. package/src/i18n/messages/de.ts +16 -0
  43. package/src/i18n/messages/en.ts +16 -0
  44. package/src/i18n/messages/es.ts +16 -0
  45. package/src/i18n/messages/fr.ts +16 -0
  46. package/src/i18n/messages/ja.ts +16 -0
  47. package/src/i18n/messages/ko.ts +16 -0
  48. package/src/i18n/messages/zh.ts +16 -0
  49. package/src/internal-urls/docs-index.generated.ts +1 -1
  50. package/src/main.ts +1 -1
  51. package/src/modes/components/model-selector.ts +275 -34
  52. package/src/modes/controllers/selector-controller.ts +50 -2
  53. package/src/modes/shared/agent-wire/command-dispatch.ts +1 -1
  54. package/src/prompts/tools/task.md +1 -0
  55. package/src/sdk/session.ts +5 -82
  56. package/src/session/agent-session.ts +137 -37
  57. package/src/session/auth-storage-discovery.ts +83 -0
  58. package/src/slash-commands/builtin-registry.ts +11 -9
  59. package/src/task/index.ts +98 -40
  60. package/src/task/receipt.ts +3 -0
  61. package/src/task/types.ts +44 -0
@@ -0,0 +1,678 @@
1
+ /**
2
+ * Self-populating keyword table.
3
+ *
4
+ * The hand-written table in `hooks/skill-keywords.ts` has eighteen literal
5
+ * strings in it and recalled 0/9 on Korean prompts, because nobody can
6
+ * enumerate Korean particle and ending combinations by hand. The semantic stage
7
+ * covers that gap, but it pays a model call every time — including for a
8
+ * phrasing the same user has already routed the same way five times.
9
+ *
10
+ * This module closes the loop: every semantic routing answer is observed here,
11
+ * discriminative two-stem patterns are mined out of it, and a pattern that has
12
+ * proved itself is promoted into the deterministic stage. The next prompt
13
+ * carrying it is answered for free.
14
+ *
15
+ * **What makes this safe.** A keyword fires with full authority and no
16
+ * confidence to fall back on, so a wrong promotion silently switches on the
17
+ * mutation guard and the Stop hook for a user who asked for neither. Every rule
18
+ * below exists to make a wrong promotion hard:
19
+ *
20
+ * - A pattern needs the *same* answer on two or more **distinct** prompts.
21
+ * Repeating one prompt verbatim proves nothing and is fingerprinted out.
22
+ * - One contradiction destroys it. Appearing in a prompt routed to a different
23
+ * workflow — or to no workflow at all — blocks the pattern permanently, since
24
+ * a phrase that shows up on both sides is by definition not discriminative.
25
+ * - Only a calibrated, confident answer counts for the fast promotion path;
26
+ * an ordinal answer from a constrained LLM needs an extra observation.
27
+ * - Two stems, ordered, with a bounded gap. One stem is a word, and words are
28
+ * ambiguous; a whole sentence is a cache key, not a rule.
29
+ *
30
+ * **What is stored.** Stems and fingerprint hashes, never prompt text.
31
+ *
32
+ * **Which languages.** Whatever the user writes in. Words are found with the
33
+ * platform's Unicode segmenter, so a Thai, Chinese or Japanese prompt with no
34
+ * spaces mines the same two-stem rules a Korean or English one does, and the
35
+ * rule fires on the same script it was mined from. Nothing here knows which
36
+ * language a prompt is in and nothing needs to.
37
+ */
38
+ import { createHash } from "node:crypto";
39
+ import { rename } from "node:fs/promises";
40
+ import * as path from "node:path";
41
+ import { getAgentDir, logger } from "@sayknow-cli/utils";
42
+ import type { SkillKeywordDefinition } from "../hooks/skill-keywords";
43
+ import { SKC_SKILL_KEYWORD_DEFINITIONS } from "../hooks/skill-keywords";
44
+ import type { CanonicalSkcWorkflowSkill } from "../skill-state/active-state";
45
+
46
+ export const LEARNED_KEYWORD_STORE_VERSION = 1;
47
+
48
+ /** Distinct prompts a pattern must survive before it fires on its own. */
49
+ const PROMOTION_OBSERVATIONS_CALIBRATED = 2;
50
+ /**
51
+ * An ordinal confidence from a constrained LLM ranks, it does not measure, so
52
+ * the threshold below is meaningless for it. One more independent prompt is the
53
+ * only evidence that is still worth anything.
54
+ */
55
+ const PROMOTION_OBSERVATIONS_ORDINAL = 3;
56
+ /** Below this, a calibrated backend is telling you it is guessing. */
57
+ const MIN_PROMOTION_CONFIDENCE = 0.9;
58
+
59
+ /** Characters between the two stems. Wide enough for a particle and an adverb. */
60
+ const MAX_STEM_GAP = 16;
61
+ /** Only the opening of a prompt carries intent; the rest is payload. */
62
+ const MAX_MINED_CHARS = 400;
63
+ /** A long prompt would otherwise mine hundreds of pairs, one of which will be noise. */
64
+ const MAX_SHINGLES_PER_PROMPT = 48;
65
+
66
+ const MAX_CANDIDATES = 512;
67
+ const MAX_ENTRIES = 128;
68
+ /** Enough to prove "distinct prompts" without growing the file per user. */
69
+ const MAX_FINGERPRINTS = 8;
70
+ /** A candidate nobody has reinforced in a month is noise from a one-off session. */
71
+ const CANDIDATE_TTL_MS = 30 * 24 * 60 * 60 * 1000;
72
+
73
+ /** Control character 0x01: cannot occur in a mined stem, so the join is unambiguous. */
74
+ const STEM_SEPARATOR = String.fromCharCode(1);
75
+
76
+ /**
77
+ * English function words. They pass the length filter, pair with anything, and
78
+ * would produce patterns like "the … code" that match every prompt ever sent.
79
+ *
80
+ * English only, on purpose: it is the one language whose vocabulary every
81
+ * prompt shares (identifiers, file names, tool output). Function words of any
82
+ * other language stop at the contradiction rule instead — the first ordinary
83
+ * prompt in that language blocks them, since it carries them too.
84
+ */
85
+ const ENGLISH_STOPWORDS = new Set([
86
+ "the",
87
+ "and",
88
+ "but",
89
+ "for",
90
+ "with",
91
+ "this",
92
+ "that",
93
+ "there",
94
+ "here",
95
+ "from",
96
+ "into",
97
+ "about",
98
+ "you",
99
+ "your",
100
+ "our",
101
+ "can",
102
+ "could",
103
+ "should",
104
+ "would",
105
+ "will",
106
+ "please",
107
+ "just",
108
+ "have",
109
+ "has",
110
+ "not",
111
+ "are",
112
+ "was",
113
+ "were",
114
+ "its",
115
+ "they",
116
+ "them",
117
+ "then",
118
+ "than",
119
+ ]);
120
+
121
+ /**
122
+ * Korean particles and endings, stripped from the tail of a token.
123
+ *
124
+ * This is not morphology — it is the smallest thing that makes the three forms
125
+ * of one noun mine the same stem, which is the entire reason the literal table
126
+ * could not express Korean.
127
+ */
128
+ const KOREAN_TAIL_CHARS = new Set([
129
+ "은",
130
+ "는",
131
+ "이",
132
+ "가",
133
+ "을",
134
+ "를",
135
+ "에",
136
+ "서",
137
+ "로",
138
+ "와",
139
+ "과",
140
+ "도",
141
+ "만",
142
+ "의",
143
+ "야",
144
+ "요",
145
+ "죠",
146
+ "줘",
147
+ "자",
148
+ "고",
149
+ "해",
150
+ "했",
151
+ "한",
152
+ "할",
153
+ "함",
154
+ "다",
155
+ "니",
156
+ "까",
157
+ "며",
158
+ "면",
159
+ "지",
160
+ "랑",
161
+ "나",
162
+ "어",
163
+ "아",
164
+ "게",
165
+ "라",
166
+ "습",
167
+ // Syllables that only ever appear inside multi-character particles
168
+ // ("부터", "까지", "으로", "에서"). Without them the same noun mines two
169
+ // different stems and the two prompts never agree on anything.
170
+ "부",
171
+ "터",
172
+ "까",
173
+ "지",
174
+ "으",
175
+ ]);
176
+
177
+ const HANGUL_PATTERN = /\p{Script=Hangul}/u;
178
+ /**
179
+ * Scripts a regex engine finds no word boundary in. Korean glues its particles
180
+ * onto the noun; Chinese, Japanese and Thai write no spaces at all. A stem from
181
+ * one of these is matched as a bare substring, and one character of it is too
182
+ * little to call a word.
183
+ */
184
+ const UNBOUNDED_SCRIPT_PATTERN =
185
+ /[\p{Script=Hangul}\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Thai}\p{Script=Lao}\p{Script=Khmer}\p{Script=Myanmar}]/u;
186
+ const SINGLE_HAN_PATTERN = /^\p{Script=Han}$/u;
187
+ /**
188
+ * Japanese writes grammar in hiragana and content in kanji or katakana, so a
189
+ * token that is hiragana through and through is a particle, an ending or a
190
+ * politeness marker — "して", "ください", "から". Measured live: without this,
191
+ * two ralplan prompts promoted "って … 承認", and "って" is in every casual
192
+ * request. The script-level rule is the closest thing Japanese has to the
193
+ * English stoplist, and it needs no vocabulary.
194
+ */
195
+ const HIRAGANA_ONLY_PATTERN = /^[\p{Script=Hiragana}ー]+$/u;
196
+ const LETTER_PATTERN = /\p{L}/u;
197
+ /**
198
+ * Splits a segmenter word on the punctuation it keeps inside one: `foo.ts`, `my_var`,
199
+ * `3.5`. Marks stay in — Thai vowels and Devanagari matras are combining characters,
200
+ * not letters, and cutting a word at each of them leaves nothing recognisable.
201
+ */
202
+ const WORD_INNER_BREAK_PATTERN = /[^\p{L}\p{M}\p{N}']+/u;
203
+ const WORD_SEGMENTER = new Intl.Segmenter(undefined, { granularity: "word" });
204
+
205
+ interface StoredPattern {
206
+ /** Two stems joined by {@link STEM_SEPARATOR}. */
207
+ readonly stems: string;
208
+ readonly skill: CanonicalSkcWorkflowSkill;
209
+ /** Distinct prompts that produced this answer. */
210
+ positives: number;
211
+ /** Fingerprints already counted, so one prompt cannot promote itself. */
212
+ fingerprints: string[];
213
+ firstSeen: number;
214
+ lastSeen: number;
215
+ }
216
+
217
+ interface LearnedKeywordStore {
218
+ version: number;
219
+ /** Promoted: these fire deterministically. */
220
+ entries: StoredPattern[];
221
+ /** Seen once, or seen with an ordinal answer. Not firing yet. */
222
+ candidates: StoredPattern[];
223
+ /** Proved ambiguous. Never reconsidered — that is the point. */
224
+ blocked: string[];
225
+ }
226
+
227
+ function emptyStore(): LearnedKeywordStore {
228
+ return { version: LEARNED_KEYWORD_STORE_VERSION, entries: [], candidates: [], blocked: [] };
229
+ }
230
+
231
+ let storePathOverride: string | undefined;
232
+ let cache: LearnedKeywordStore | undefined;
233
+
234
+ /** Test seam. Also lets a host keep the table out of the user's home. */
235
+ export function setLearnedKeywordStorePath(value: string | undefined): void {
236
+ storePathOverride = value;
237
+ cache = undefined;
238
+ }
239
+
240
+ export function resetLearnedKeywordCache(): void {
241
+ cache = undefined;
242
+ }
243
+
244
+ /**
245
+ * Where the learned table lives.
246
+ *
247
+ * User-global on purpose: a phrasing you use is a phrasing you use, and making it
248
+ * per-repository would mean relearning the same rule in every checkout.
249
+ * `SKC_LEARNED_KEYWORDS_PATH` redirects it, so a test driving a real session under
250
+ * a temp agent dir cannot write into the developer's actual home.
251
+ */
252
+ export function getLearnedKeywordStorePath(): string {
253
+ const override = storePathOverride ?? process.env.SKC_LEARNED_KEYWORDS_PATH?.trim();
254
+ return override || path.join(getAgentDir(), "learned-skill-keywords.json");
255
+ }
256
+
257
+ // ---------------------------------------------------------------------------
258
+ // Mining
259
+ // ---------------------------------------------------------------------------
260
+
261
+ function stemKorean(token: string): string {
262
+ let stem = token;
263
+ // Three strips, floored at two syllables. That is what it takes to collapse a
264
+ // stacked particle ("계획서부터" -> "계획") onto the same stem as the plain one
265
+ // ("계획서를" -> "계획"), and over-stripping is the safe direction: the compiled
266
+ // pattern matches the raw prompt as a substring, so a shorter stem costs
267
+ // precision, which the two-prompt agreement rule then filters, while a longer
268
+ // one costs recall, which nothing recovers.
269
+ for (let strip = 0; strip < 3; strip++) {
270
+ if (stem.length <= 2) break;
271
+ const tail = stem[stem.length - 1];
272
+ if (!tail || !KOREAN_TAIL_CHARS.has(tail)) break;
273
+ stem = stem.slice(0, -1);
274
+ }
275
+ return stem;
276
+ }
277
+
278
+ /**
279
+ * Break text into words in whatever script it is written.
280
+ *
281
+ * `Intl.Segmenter` carries the ICU dictionaries, which is what makes a Thai or
282
+ * Japanese prompt with no spaces come apart into words at all. Its Chinese
283
+ * dictionary misses most technical compounds and hands back one character at
284
+ * a time ("文档" → "文", "档"), so a run of single characters is re-joined into
285
+ * character bigrams — the standard fallback for CJK without a dictionary, and it
286
+ * recovers "文档", "重构" and "模块" at the cost of a few cross-word pairs that
287
+ * the two-prompt rule never promotes.
288
+ */
289
+ function segmentWords(text: string): string[] {
290
+ const words: string[] = [];
291
+ let hanRun = "";
292
+ const flushHanRun = () => {
293
+ if (hanRun.length === 1) words.push(hanRun);
294
+ for (let i = 0; i + 1 < hanRun.length; i++) words.push(hanRun.slice(i, i + 2));
295
+ hanRun = "";
296
+ };
297
+ for (const segment of WORD_SEGMENTER.segment(text)) {
298
+ if (!segment.isWordLike) {
299
+ flushHanRun();
300
+ continue;
301
+ }
302
+ if (SINGLE_HAN_PATTERN.test(segment.segment)) {
303
+ hanRun += segment.segment;
304
+ continue;
305
+ }
306
+ flushHanRun();
307
+ for (const part of segment.segment.split(WORD_INNER_BREAK_PATTERN)) if (part) words.push(part);
308
+ }
309
+ flushHanRun();
310
+ return words;
311
+ }
312
+
313
+ /** Normalize to the token stream both mining and fingerprinting agree on. */
314
+ function tokenize(text: string): string[] {
315
+ const tokens: string[] = [];
316
+ for (const raw of segmentWords(text.slice(0, MAX_MINED_CHARS).toLowerCase())) {
317
+ // A path or a bare number is payload, not intent, and would mine a pattern
318
+ // that only ever matches this one repository.
319
+ if (raw.length > 24 || !LETTER_PATTERN.test(raw)) continue;
320
+ if (HANGUL_PATTERN.test(raw)) {
321
+ const stem = stemKorean(raw);
322
+ if (stem.length >= 2) tokens.push(stem);
323
+ continue;
324
+ }
325
+ if (UNBOUNDED_SCRIPT_PATTERN.test(raw)) {
326
+ if (raw.length >= 2 && !HIRAGANA_ONLY_PATTERN.test(raw)) tokens.push(raw);
327
+ continue;
328
+ }
329
+ if (raw.length < 3 || ENGLISH_STOPWORDS.has(raw)) continue;
330
+ tokens.push(raw);
331
+ }
332
+ return tokens;
333
+ }
334
+
335
+ /**
336
+ * Ordered stem pairs within a three-token window.
337
+ *
338
+ * The gap is what makes a learned pattern generalize where the literal table
339
+ * cannot: a prompt that says "make me a plan document" mines `plan … document`,
340
+ * which then also matches "make the plan into a document" — the phrasing that
341
+ * made the hand-written entry miss.
342
+ */
343
+ export function mineShingles(text: string): string[] {
344
+ const tokens = tokenize(text);
345
+ const shingles: string[] = [];
346
+ const seen = new Set<string>();
347
+ for (let i = 0; i < tokens.length; i++) {
348
+ for (let j = i + 1; j <= i + 2 && j < tokens.length; j++) {
349
+ const a = tokens[i];
350
+ const b = tokens[j];
351
+ if (!a || !b || a === b) continue;
352
+ const key = `${a}${STEM_SEPARATOR}${b}`;
353
+ if (seen.has(key)) continue;
354
+ seen.add(key);
355
+ shingles.push(key);
356
+ if (shingles.length >= MAX_SHINGLES_PER_PROMPT) return shingles;
357
+ }
358
+ }
359
+ return shingles;
360
+ }
361
+
362
+ function fingerprint(text: string): string {
363
+ return createHash("sha1").update(tokenize(text).join(" ")).digest("hex").slice(0, 12);
364
+ }
365
+
366
+ // ---------------------------------------------------------------------------
367
+ // Pattern compilation
368
+ // ---------------------------------------------------------------------------
369
+
370
+ function escapeRegex(value: string): string {
371
+ return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
372
+ }
373
+
374
+ function stemToPattern(stem: string): string {
375
+ const escaped = escapeRegex(stem);
376
+ // A stem from a script without visible word boundaries can only be matched as
377
+ // a bare substring. Every other script gets a real left boundary so "plan"
378
+ // does not match "explanation" — in any alphabet, hence the Unicode classes.
379
+ // The right side stays open on purpose: it is what lets one stem cover every
380
+ // inflection of the same word.
381
+ return UNBOUNDED_SCRIPT_PATTERN.test(stem) ? escaped : `(?<![\\p{L}\\p{N}_])${escaped}`;
382
+ }
383
+
384
+ function compilePattern(stems: string): RegExp | null {
385
+ const [a, b] = stems.split(STEM_SEPARATOR);
386
+ if (!a || !b) return null;
387
+ // Sentence terminators bound the gap: two stems on opposite sides of a full
388
+ // stop are two different thoughts, not one intent. Fullwidth forms included,
389
+ // since that is what a CJK keyboard produces.
390
+ return new RegExp(`${stemToPattern(a)}[^.?!。?!\\n]{0,${MAX_STEM_GAP}}${stemToPattern(b)}`, "iu");
391
+ }
392
+
393
+ function displayKeyword(stems: string): string {
394
+ return stems.split(STEM_SEPARATOR).join(" … ");
395
+ }
396
+
397
+ const MAX_HARDCODED_PRIORITY = Math.max(...SKC_SKILL_KEYWORD_DEFINITIONS.map(definition => definition.priority));
398
+
399
+ /**
400
+ * Always negative, so a learned pattern loses every tie to an enumerated one
401
+ * while keeping the relative order between workflows. A human who wrote a
402
+ * keyword outranks a rule mined from the user's own history, every time.
403
+ */
404
+ function learnedPriorityFor(skill: CanonicalSkcWorkflowSkill): number {
405
+ const base = SKC_SKILL_KEYWORD_DEFINITIONS.find(definition => definition.skill === skill)?.priority ?? 0;
406
+ return base - MAX_HARDCODED_PRIORITY - 1;
407
+ }
408
+
409
+ // ---------------------------------------------------------------------------
410
+ // Store IO
411
+ // ---------------------------------------------------------------------------
412
+
413
+ function sanitize(raw: unknown): LearnedKeywordStore {
414
+ if (!raw || typeof raw !== "object") return emptyStore();
415
+ const value = raw as Partial<LearnedKeywordStore>;
416
+ if (value.version !== LEARNED_KEYWORD_STORE_VERSION) return emptyStore();
417
+ const patterns = (input: unknown): StoredPattern[] =>
418
+ Array.isArray(input)
419
+ ? input.filter(
420
+ (item): item is StoredPattern =>
421
+ typeof item === "object" &&
422
+ item !== null &&
423
+ typeof (item as StoredPattern).stems === "string" &&
424
+ (item as StoredPattern).stems.includes(STEM_SEPARATOR) &&
425
+ typeof (item as StoredPattern).skill === "string" &&
426
+ Array.isArray((item as StoredPattern).fingerprints),
427
+ )
428
+ : [];
429
+ return {
430
+ version: LEARNED_KEYWORD_STORE_VERSION,
431
+ entries: patterns(value.entries).slice(0, MAX_ENTRIES),
432
+ candidates: patterns(value.candidates).slice(0, MAX_CANDIDATES),
433
+ blocked: Array.isArray(value.blocked) ? value.blocked.filter(item => typeof item === "string") : [],
434
+ };
435
+ }
436
+
437
+ async function readStore(): Promise<LearnedKeywordStore> {
438
+ if (cache) return cache;
439
+ try {
440
+ const file = Bun.file(getLearnedKeywordStorePath());
441
+ cache = (await file.exists()) ? sanitize(await file.json()) : emptyStore();
442
+ } catch (error) {
443
+ // A corrupt or unreadable table must never take down a user's turn; losing
444
+ // what was learned is the correct price.
445
+ logger.debug("decisions/keyword-learning: unreadable store, starting empty", { error: String(error) });
446
+ cache = emptyStore();
447
+ }
448
+ return cache;
449
+ }
450
+
451
+ async function writeStore(store: LearnedKeywordStore): Promise<void> {
452
+ cache = store;
453
+ const target = getLearnedKeywordStorePath();
454
+ const temp = `${target}.${process.pid}.tmp`;
455
+ try {
456
+ // Temp + rename: a concurrent session reading mid-write must never see a
457
+ // half-written file and conclude the table is corrupt.
458
+ await Bun.write(temp, JSON.stringify(store));
459
+ await rename(temp, target);
460
+ } catch (error) {
461
+ logger.debug("decisions/keyword-learning: store write failed", { error: String(error) });
462
+ }
463
+ }
464
+
465
+ // ---------------------------------------------------------------------------
466
+ // Public surface
467
+ // ---------------------------------------------------------------------------
468
+
469
+ /**
470
+ * Promoted patterns, in the shape the deterministic stage consumes.
471
+ *
472
+ * Priority sits one below the hand-written entry for the same workflow: when a
473
+ * learned pattern and an enumerated keyword disagree, the human wins.
474
+ */
475
+ export async function loadLearnedKeywordDefinitions(): Promise<SkillKeywordDefinition[]> {
476
+ const store = await readStore();
477
+ const definitions: SkillKeywordDefinition[] = [];
478
+ for (const entry of store.entries) {
479
+ const pattern = compilePattern(entry.stems);
480
+ if (!pattern) continue;
481
+ definitions.push({
482
+ keyword: displayKeyword(entry.stems),
483
+ skill: entry.skill,
484
+ priority: learnedPriorityFor(entry.skill),
485
+ guidance: `Activate SKC ${entry.skill} workflow (learned from ${entry.positives} routed prompts)`,
486
+ pattern,
487
+ learned: true,
488
+ });
489
+ }
490
+ return definitions;
491
+ }
492
+
493
+ /** Stems of `candidate` are all extensions of `existing`'s — same rule, less general. */
494
+ function isRedundantWith(existing: StoredPattern, candidate: StoredPattern): boolean {
495
+ if (existing.skill !== candidate.skill) return false;
496
+ const [ea, eb] = existing.stems.split(STEM_SEPARATOR);
497
+ const [ca, cb] = candidate.stems.split(STEM_SEPARATOR);
498
+ if (!ea || !eb || !ca || !cb) return false;
499
+ return ca.startsWith(ea) && cb.startsWith(eb);
500
+ }
501
+
502
+ export interface RoutingObservation {
503
+ text: string;
504
+ /** Null means the router deliberately chose no workflow — a negative example. */
505
+ skill: CanonicalSkcWorkflowSkill | null;
506
+ confidence?: number | undefined;
507
+ calibrated?: boolean | undefined;
508
+ }
509
+
510
+ export interface LearningOutcome {
511
+ promoted: SkillKeywordDefinition[];
512
+ retracted: number;
513
+ }
514
+
515
+ /**
516
+ * Feed one routing answer into the table.
517
+ *
518
+ * Best-effort by construction: it runs after the turn's routing decision is
519
+ * already made, so a failure here changes nothing the user can observe.
520
+ */
521
+ export async function observeRouting(observation: RoutingObservation): Promise<LearningOutcome> {
522
+ const outcome: LearningOutcome = { promoted: [], retracted: 0 };
523
+ const shingles = mineShingles(observation.text);
524
+ if (shingles.length === 0) return outcome;
525
+
526
+ const store = await readStore();
527
+ const now = Date.now();
528
+ const print = fingerprint(observation.text);
529
+ const blocked = new Set(store.blocked);
530
+ const mined = new Set(shingles);
531
+ let dirty = false;
532
+
533
+ // A promoted pattern that appears in a prompt routed elsewhere was never
534
+ // discriminative. Retract it — the cost of leaving it in is a workflow the
535
+ // user did not ask for, every time they use that phrasing.
536
+ const survivingEntries: StoredPattern[] = [];
537
+ for (const entry of store.entries) {
538
+ if (!mined.has(entry.stems)) {
539
+ survivingEntries.push(entry);
540
+ continue;
541
+ }
542
+ if (entry.skill === observation.skill) {
543
+ if (!entry.fingerprints.includes(print)) {
544
+ entry.fingerprints = [print, ...entry.fingerprints].slice(0, MAX_FINGERPRINTS);
545
+ entry.positives++;
546
+ }
547
+ entry.lastSeen = now;
548
+ dirty = true;
549
+ survivingEntries.push(entry);
550
+ continue;
551
+ }
552
+ logger.debug("decisions/keyword-learning: retracting contradicted pattern", {
553
+ stems: displayKeyword(entry.stems),
554
+ had: entry.skill,
555
+ saw: observation.skill ?? "none",
556
+ });
557
+ blocked.add(entry.stems);
558
+ outcome.retracted++;
559
+ dirty = true;
560
+ }
561
+ store.entries = survivingEntries;
562
+
563
+ const required =
564
+ observation.calibrated && (observation.confidence ?? 0) >= MIN_PROMOTION_CONFIDENCE
565
+ ? PROMOTION_OBSERVATIONS_CALIBRATED
566
+ : PROMOTION_OBSERVATIONS_ORDINAL;
567
+
568
+ const survivingCandidates: StoredPattern[] = [];
569
+ for (const candidate of store.candidates) {
570
+ if (blocked.has(candidate.stems)) {
571
+ dirty = true;
572
+ continue;
573
+ }
574
+ if (!mined.has(candidate.stems)) {
575
+ if (now - candidate.lastSeen > CANDIDATE_TTL_MS) {
576
+ dirty = true;
577
+ continue;
578
+ }
579
+ survivingCandidates.push(candidate);
580
+ continue;
581
+ }
582
+ // Seen again, but with a different answer — or with none at all. Either way
583
+ // the stems do not separate the classes, so stop tracking them for good.
584
+ if (candidate.skill !== observation.skill) {
585
+ blocked.add(candidate.stems);
586
+ dirty = true;
587
+ continue;
588
+ }
589
+ dirty = true;
590
+ if (!candidate.fingerprints.includes(print)) {
591
+ candidate.fingerprints = [print, ...candidate.fingerprints].slice(0, MAX_FINGERPRINTS);
592
+ candidate.positives++;
593
+ }
594
+ candidate.lastSeen = now;
595
+ if (candidate.positives < required) {
596
+ survivingCandidates.push(candidate);
597
+ continue;
598
+ }
599
+ // Promotion. A more general pattern already covering this one makes it dead
600
+ // weight; a less general one it covers gets replaced, because the general
601
+ // rule is the one worth keeping.
602
+ if (store.entries.some(entry => isRedundantWith(entry, candidate))) continue;
603
+ store.entries = store.entries.filter(entry => !isRedundantWith(candidate, entry));
604
+ store.entries.unshift(candidate);
605
+ const pattern = compilePattern(candidate.stems);
606
+ if (pattern) {
607
+ outcome.promoted.push({
608
+ keyword: displayKeyword(candidate.stems),
609
+ skill: candidate.skill,
610
+ priority: learnedPriorityFor(candidate.skill),
611
+ guidance: `Activate SKC ${candidate.skill} workflow (learned)`,
612
+ pattern,
613
+ learned: true,
614
+ });
615
+ }
616
+ logger.debug("decisions/keyword-learning: promoted pattern", {
617
+ stems: displayKeyword(candidate.stems),
618
+ skill: candidate.skill,
619
+ positives: candidate.positives,
620
+ });
621
+ }
622
+ store.candidates = survivingCandidates;
623
+
624
+ // A "none" answer only ever removes. Opening candidates for it would mine the
625
+ // vocabulary of ordinary coding requests, which is every prompt SKC gets.
626
+ if (observation.skill) {
627
+ const known = new Set([...store.entries, ...store.candidates].map(item => item.stems));
628
+ for (const stems of shingles) {
629
+ if (known.has(stems) || blocked.has(stems)) continue;
630
+ store.candidates.push({
631
+ stems,
632
+ skill: observation.skill,
633
+ positives: 1,
634
+ fingerprints: [print],
635
+ firstSeen: now,
636
+ lastSeen: now,
637
+ });
638
+ dirty = true;
639
+ }
640
+ }
641
+
642
+ if (!dirty) return outcome;
643
+
644
+ store.blocked = [...blocked].slice(-MAX_CANDIDATES);
645
+ if (store.candidates.length > MAX_CANDIDATES) {
646
+ // Evict the least reinforced first; a high-positive candidate is the one
647
+ // closest to being worth something.
648
+ store.candidates.sort((a, b) => b.positives - a.positives || b.lastSeen - a.lastSeen);
649
+ store.candidates.length = MAX_CANDIDATES;
650
+ }
651
+ if (store.entries.length > MAX_ENTRIES) {
652
+ store.entries.sort((a, b) => b.positives - a.positives || b.lastSeen - a.lastSeen);
653
+ store.entries.length = MAX_ENTRIES;
654
+ }
655
+ await writeStore(store);
656
+ return outcome;
657
+ }
658
+
659
+ export interface LearnedKeywordSummary {
660
+ entries: { keyword: string; skill: CanonicalSkcWorkflowSkill; positives: number; lastSeen: number }[];
661
+ candidates: number;
662
+ blocked: number;
663
+ }
664
+
665
+ /** What the table has actually learned, for `skc` surfaces and tests. */
666
+ export async function summarizeLearnedKeywords(): Promise<LearnedKeywordSummary> {
667
+ const store = await readStore();
668
+ return {
669
+ entries: store.entries.map(entry => ({
670
+ keyword: displayKeyword(entry.stems),
671
+ skill: entry.skill,
672
+ positives: entry.positives,
673
+ lastSeen: entry.lastSeen,
674
+ })),
675
+ candidates: store.candidates.length,
676
+ blocked: store.blocked.length,
677
+ };
678
+ }