@sayknow-cli/coding-agent 0.5.25 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/CHANGELOG.md +36 -1
  2. package/dist/types/config/settings-schema.d.ts +51 -5
  3. package/dist/types/config/task-model-specialties.d.ts +55 -0
  4. package/dist/types/decisions/keyword-learning.d.ts +61 -0
  5. package/dist/types/decisions/llm-backend.d.ts +13 -1
  6. package/dist/types/decisions/prompt-triage.d.ts +42 -0
  7. package/dist/types/decisions/skill-routing.d.ts +41 -6
  8. package/dist/types/decisions/task-routing.d.ts +96 -11
  9. package/dist/types/hooks/native-prompt-routing.d.ts +21 -0
  10. package/dist/types/hooks/native-skill-hook.d.ts +3 -0
  11. package/dist/types/hooks/skill-keywords.d.ts +9 -0
  12. package/dist/types/hooks/skill-state.d.ts +20 -3
  13. package/dist/types/hooks/ui-skill-keywords.d.ts +15 -0
  14. package/dist/types/i18n/messages/en.d.ts +15 -0
  15. package/dist/types/lsp/index.d.ts +1 -1
  16. package/dist/types/lsp/types.d.ts +1 -1
  17. package/dist/types/modes/components/model-selector.d.ts +11 -0
  18. package/dist/types/sdk/session.d.ts +3 -13
  19. package/dist/types/session/agent-session.d.ts +8 -0
  20. package/dist/types/session/auth-storage-discovery.d.ts +13 -0
  21. package/dist/types/task/index.d.ts +1 -1
  22. package/dist/types/task/receipt.d.ts +2 -0
  23. package/dist/types/task/types.d.ts +114 -18
  24. package/dist/types/tools/browser.d.ts +2 -2
  25. package/dist/types/tools/subagent.d.ts +2 -2
  26. package/package.json +7 -7
  27. package/scripts/eval-skill-routing.ts +37 -12
  28. package/src/config/settings-schema.ts +64 -12
  29. package/src/config/task-model-specialties.ts +131 -0
  30. package/src/decisions/index.ts +8 -2
  31. package/src/decisions/keyword-learning.ts +678 -0
  32. package/src/decisions/llm-backend.ts +213 -67
  33. package/src/decisions/prompt-triage.ts +163 -0
  34. package/src/decisions/skill-routing.ts +39 -56
  35. package/src/decisions/task-routing.ts +382 -66
  36. package/src/decisions/typesafe-backend.ts +3 -0
  37. package/src/hooks/native-prompt-routing.ts +190 -0
  38. package/src/hooks/native-skill-hook.ts +21 -12
  39. package/src/hooks/skill-keywords.ts +9 -0
  40. package/src/hooks/skill-state.ts +41 -10
  41. package/src/hooks/ui-skill-keywords.ts +67 -10
  42. package/src/i18n/messages/de.ts +16 -0
  43. package/src/i18n/messages/en.ts +16 -0
  44. package/src/i18n/messages/es.ts +16 -0
  45. package/src/i18n/messages/fr.ts +16 -0
  46. package/src/i18n/messages/ja.ts +16 -0
  47. package/src/i18n/messages/ko.ts +16 -0
  48. package/src/i18n/messages/zh.ts +16 -0
  49. package/src/internal-urls/docs-index.generated.ts +1 -1
  50. package/src/main.ts +1 -1
  51. package/src/modes/components/model-selector.ts +275 -34
  52. package/src/modes/controllers/selector-controller.ts +50 -2
  53. package/src/modes/shared/agent-wire/command-dispatch.ts +1 -1
  54. package/src/prompts/tools/task.md +1 -0
  55. package/src/sdk/session.ts +5 -82
  56. package/src/session/agent-session.ts +137 -37
  57. package/src/session/auth-storage-discovery.ts +83 -0
  58. package/src/slash-commands/builtin-registry.ts +11 -9
  59. package/src/task/index.ts +98 -40
  60. package/src/task/receipt.ts +3 -0
  61. package/src/task/types.ts +44 -0
@@ -154,11 +154,21 @@ export interface LlmBackendDeps {
154
154
  maxInputCostPerMTok?: number;
155
155
  /** Injected in tests to make the local-runtime probe deterministic. */
156
156
  fetchImpl?: typeof fetch;
157
+ /** Injected in tests to script provider answers without a network. */
158
+ completeImpl?: typeof completeSimple;
159
+ /** Per-candidate deadline; injected in tests so a hung provider does not cost real seconds. */
160
+ attemptTimeoutMs?: number;
157
161
  registry: ModelRegistry;
158
162
  settings: Settings;
159
163
  sessionId?: string;
160
164
  /** Overrides role resolution; used by callers that already picked a model. */
161
165
  model?: Model<Api>;
166
+ /**
167
+ * Provider of the model the caller is already talking to. Its small model is tried
168
+ * first when nothing was configured; see `rankSmallModels` for why. A thunk is
169
+ * accepted so a long-lived backend follows the session when the user switches model.
170
+ */
171
+ preferredProvider?: string | (() => string | undefined);
162
172
  }
163
173
 
164
174
  /**
@@ -188,6 +198,23 @@ function isTextOnly(model: Model<Api>): boolean {
188
198
  return (model.input ?? ["text"]).includes("text") && !(model.output ?? ["text"]).includes("image");
189
199
  }
190
200
 
201
+ /**
202
+ * Local runtimes list embedding models next to chat models with identical metadata.
203
+ * Measured: `ollama/nomic-embed-text:latest` was ranked as a candidate and answered
204
+ * HTTP 400 "does not support chat". Nothing in `Model` says so; the id does.
205
+ */
206
+ const EMBEDDING_MODEL_ID = /embed/i;
207
+
208
+ /**
209
+ * Parameter count from a local model id (`qwen3:1.7b`, `gemma4:e4b`, `lfm2-24b-a2b`),
210
+ * used only to order local candidates: smaller loads faster and answers faster, and a
211
+ * five-way choice does not need a 30b model. Unparseable ids sort last.
212
+ */
213
+ function localModelSize(id: string): number {
214
+ const match = /(\d+(?:\.\d+)?)b(?![a-z])/i.exec(id);
215
+ return match ? Number(match[1]) : Number.POSITIVE_INFINITY;
216
+ }
217
+
191
218
  /**
192
219
  * Locally hosted runtimes. A decision answered here costs no tokens at all and the
193
220
  * state never leaves the machine, which is the strongest possible fit for this feature.
@@ -236,7 +263,50 @@ async function isLocalRuntimeAlive(baseUrl: string, fetchImpl: typeof fetch = fe
236
263
  }
237
264
 
238
265
  /**
239
- * Pick a small, fast text model.
266
+ * A model that answered a decision with an error or a timeout, skipped for a while.
267
+ *
268
+ * The catalog lists models the provider no longer serves — `claude-3-haiku-20240307`
269
+ * was the cheapest Anthropic entry and answered 404 on every call — and it lists
270
+ * "free" reasoning tiers that ignore `disableReasoning` and take ~9s to answer. Either
271
+ * one, chosen blindly, turned the fallback into a fixed ~1-9s stall that answered
272
+ * nothing, on every turn, forever. Remembering the failure means one bad turn per
273
+ * model per TTL, not one per prompt.
274
+ */
275
+ const DEAD_MODEL_TTL_MS = 10 * 60_000;
276
+ const deadModels = new Map<string, { at: number; reason: string }>();
277
+
278
+ /** Reset between tests. */
279
+ export function clearDeadModelCache(): void {
280
+ deadModels.clear();
281
+ }
282
+
283
+ function modelKey(model: Model<Api>): string {
284
+ return `${model.provider}/${model.id}`;
285
+ }
286
+
287
+ function isDead(model: Model<Api>): boolean {
288
+ const entry = deadModels.get(modelKey(model));
289
+ if (!entry) return false;
290
+ if (Date.now() - entry.at < DEAD_MODEL_TTL_MS) return true;
291
+ deadModels.delete(modelKey(model));
292
+ return false;
293
+ }
294
+
295
+ /**
296
+ * Candidates tried per decision when nothing was configured. Two attempts of
297
+ * `ATTEMPT_TIMEOUT_MS` fit inside the service's 8s deadline with room for the
298
+ * liveness probe; a third only runs when the earlier ones failed fast (404, auth).
299
+ */
300
+ const MAX_AUTO_ATTEMPTS = 3;
301
+ /**
302
+ * Per-attempt budget. The hosted small models measured here answer in 0.3-1.3s; the
303
+ * ones that blow this are reasoning tiers that think despite being told not to, and
304
+ * the right response to those is the next candidate, not a longer wait.
305
+ */
306
+ const ATTEMPT_TIMEOUT_MS = 3_500;
307
+
308
+ /**
309
+ * Rank small, fast text models, best first.
240
310
  *
241
311
  * Sorting by price alone is a trap, and it was measured: the cheapest qualifying model
242
312
  * on this registry is free but took **4.8s** per routing decision — three times slower
@@ -244,29 +314,44 @@ async function isLocalRuntimeAlive(baseUrl: string, fetchImpl: typeof fetch = fe
244
314
  * are dominated by reasoning models. A decision service that is cheap and slow has
245
315
  * missed the point twice over.
246
316
  *
247
- * So non-reasoning wins first, price second. Ties break by id so the choice is stable
248
- * across runs; a backend that silently changed model between turns would make routing
249
- * non-reproducible, which is most of what this feature is for.
317
+ * The provider the user is already chatting with goes first. Measured on a registry
318
+ * with Anthropic, Codex and Z.ai keys: price order put four `zai/glm-*` tiers ahead of
319
+ * `claude-haiku-4-5`, and the first of them took 8.9s — past the service deadline, so
320
+ * the answer was nothing. The user's own provider is the credential known to work, the
321
+ * bill they expect to see, and (Haiku next to Opus, mini next to GPT) the small model
322
+ * they would have picked by hand. Within a provider, non-reasoning wins, then price.
323
+ * Ties break by id so the choice is stable across runs; a backend that silently changed
324
+ * model between turns would make routing non-reproducible, which is most of what this
325
+ * feature is for.
250
326
  */
251
- async function pickSmallModel(
327
+ async function rankSmallModels(
252
328
  available: Model<Api>[],
253
329
  costCeiling: number,
330
+ preferredProvider: string | undefined,
254
331
  fetchImpl?: typeof fetch,
255
- ): Promise<Model<Api> | undefined> {
256
- // A local runtime that is actually up wins outright: zero tokens, zero egress. Its
257
- // size is not screened the way hosted models are — if the user loaded it, they chose
258
- // it, and trying costs nothing.
332
+ ): Promise<Model<Api>[]> {
333
+ // A live local runtime goes first: zero tokens, zero egress. One candidate only —
334
+ // measured with Ollama on CPU, every loaded model blew the attempt budget on a cold
335
+ // start, and with five of them ranked ahead of every hosted model the hosted
336
+ // fallback was never reached before the service deadline. One local miss now costs
337
+ // one attempt, after which the dead-model cache sends the next ten minutes of
338
+ // decisions straight to the hosted candidate.
339
+ const ranked: Model<Api>[] = [];
259
340
  const local = available
260
- .filter(model => LOCAL_PROVIDERS.has(model.provider) && isTextOnly(model))
261
- .sort((a, b) => a.id.localeCompare(b.id));
341
+ .filter(model => LOCAL_PROVIDERS.has(model.provider) && isTextOnly(model) && !EMBEDDING_MODEL_ID.test(model.id))
342
+ .sort((a, b) => localModelSize(a.id) - localModelSize(b.id) || a.id.localeCompare(b.id));
262
343
  for (const model of local) {
344
+ // The smallest one missing parks the whole runtime for the TTL; trying the next
345
+ // size up would only repeat the cold-start stall on the next turn.
346
+ if (isDead(model)) break;
263
347
  if (await isLocalRuntimeAlive(model.baseUrl, fetchImpl)) {
264
- logger.debug("decisions/llm: using local runtime", { id: `${model.provider}/${model.id}` });
265
- return model;
348
+ logger.debug("decisions/llm: using local runtime", { id: modelKey(model) });
349
+ ranked.push(model);
350
+ break;
266
351
  }
267
352
  }
268
353
 
269
- return available
354
+ const hosted = available
270
355
  .filter(
271
356
  model =>
272
357
  isTextOnly(model) &&
@@ -276,12 +361,18 @@ async function pickSmallModel(
276
361
  )
277
362
  .sort(
278
363
  (a, b) =>
279
- Number(!!a.reasoning) - Number(!!b.reasoning) || a.cost.input - b.cost.input || a.id.localeCompare(b.id),
280
- )[0];
364
+ Number(b.provider === preferredProvider) - Number(a.provider === preferredProvider) ||
365
+ Number(!!a.reasoning) - Number(!!b.reasoning) ||
366
+ a.cost.input - b.cost.input ||
367
+ a.id.localeCompare(b.id),
368
+ );
369
+ return ranked.concat(hosted);
281
370
  }
282
371
 
283
372
  export function createLlmDecisionBackend(deps: LlmBackendDeps): DecisionBackend {
284
373
  const costCeiling = deps.maxInputCostPerMTok ?? DEFAULT_MAX_INPUT_COST_PER_MTOK;
374
+ const complete = deps.completeImpl ?? completeSimple;
375
+ const attemptTimeoutMs = deps.attemptTimeoutMs ?? ATTEMPT_TIMEOUT_MS;
285
376
  return {
286
377
  name: "llm",
287
378
  async decide(request: DecisionRequest): Promise<DecisionResult | null> {
@@ -290,67 +381,122 @@ export function createLlmDecisionBackend(deps: LlmBackendDeps): DecisionBackend
290
381
  // Resolution order, cheapest intent first:
291
382
  // 1. an explicit override — the caller already decided
292
383
  // 2. the `smol` role — the user already decided
293
- // 3. the cheapest small model on hand — nobody decided, so decide safely
384
+ // 3. the ranked small models on hand — nobody decided, so decide safely,
385
+ // and move on when one is dead or slow
294
386
  // `default` is deliberately absent: it is whatever the user chats with, which is
295
387
  // exactly the frontier model this feature exists to avoid spending on.
296
- const chosen =
297
- deps.model ??
298
- resolveRoleSelection(["smol"], deps.settings, available, deps.registry)?.model ??
299
- (await pickSmallModel(available, costCeiling, deps.fetchImpl));
300
- if (!chosen) {
388
+ const configured =
389
+ deps.model ?? resolveRoleSelection(["smol"], deps.settings, available, deps.registry)?.model;
390
+ const preferredProvider =
391
+ typeof deps.preferredProvider === "function" ? deps.preferredProvider() : deps.preferredProvider;
392
+ const candidates = configured
393
+ ? [configured]
394
+ : (await rankSmallModels(available, costCeiling, preferredProvider, deps.fetchImpl))
395
+ .filter(model => !isDead(model))
396
+ .slice(0, MAX_AUTO_ATTEMPTS);
397
+ if (candidates.length === 0) {
301
398
  logger.debug("decisions/llm: no small model available; leaving the decision to existing behaviour");
302
399
  return null;
303
400
  }
304
- const model = chosen;
305
- // The ceiling still applies to an explicitly configured `smol` role — a role can
306
- // point anywhere, including at a frontier model.
307
- if (!deps.model && model.cost.input > costCeiling) {
308
- logger.debug("decisions/llm: declining, model too expensive for a decision", {
309
- id: `${model.provider}/${model.id}`,
310
- inputCostPerMTok: model.cost.input,
311
- ceiling: costCeiling,
312
- });
313
- return null;
314
- }
315
- const apiKey = await deps.registry.getApiKey(model, deps.sessionId);
316
- if (!apiKey) {
317
- logger.debug("decisions/llm: no credential", { provider: model.provider, id: model.id });
318
- return null;
319
- }
320
401
 
321
402
  const text = stateToText(request.state);
322
403
  const state = text.length > MAX_STATE_CHARS ? `${text.slice(0, MAX_STATE_CHARS)}…` : text;
323
- const started = Date.now();
324
- const response = await completeSimple(
325
- model,
326
- {
327
- systemPrompt: [SYSTEM_PROMPT],
328
- messages: [{ role: "user", content: `<state>\n${state}\n</state>`, timestamp: Date.now() }],
329
- tools: [buildTool(request.questions)],
330
- },
331
- {
332
- apiKey,
333
- maxTokens: model.reasoning ? Math.max(MAX_TOKENS, REASONING_SAFE_MAX_TOKENS) : MAX_TOKENS,
334
- disableReasoning: true,
335
- toolChoice: { type: "tool", name: TOOL_NAME },
336
- signal: request.signal,
337
- },
338
- );
339
404
 
340
- const args = readToolArguments(response.content);
341
- if (!args) {
342
- logger.debug("decisions/llm: model did not emit the forced tool call");
343
- return null;
405
+ for (const model of candidates) {
406
+ if (request.signal?.aborted) return null;
407
+ // The ceiling still applies to an explicitly configured `smol` role — a role can
408
+ // point anywhere, including at a frontier model.
409
+ if (!deps.model && model.cost.input > costCeiling) {
410
+ logger.debug("decisions/llm: declining, model too expensive for a decision", {
411
+ id: modelKey(model),
412
+ inputCostPerMTok: model.cost.input,
413
+ ceiling: costCeiling,
414
+ });
415
+ return null;
416
+ }
417
+ const apiKey = await deps.registry.getApiKey(model, deps.sessionId);
418
+ if (!apiKey) {
419
+ logger.debug("decisions/llm: no credential", { provider: model.provider, id: model.id });
420
+ return null;
421
+ }
422
+ // The credential lookup awaited; an abort in that window would be missed by
423
+ // the listener registered below.
424
+ if (request.signal?.aborted) return null;
425
+
426
+ const controller = new AbortController();
427
+ const abortOnCaller = () => controller.abort();
428
+ request.signal?.addEventListener("abort", abortOnCaller, { once: true });
429
+ let timedOut = false;
430
+ const timer = setTimeout(() => {
431
+ timedOut = true;
432
+ controller.abort();
433
+ }, attemptTimeoutMs);
434
+ const started = Date.now();
435
+ let response: AssistantMessage;
436
+ try {
437
+ response = await complete(
438
+ model,
439
+ {
440
+ systemPrompt: [SYSTEM_PROMPT],
441
+ messages: [{ role: "user", content: `<state>\n${state}\n</state>`, timestamp: Date.now() }],
442
+ tools: [buildTool(request.questions)],
443
+ },
444
+ {
445
+ apiKey,
446
+ maxTokens: model.reasoning ? Math.max(MAX_TOKENS, REASONING_SAFE_MAX_TOKENS) : MAX_TOKENS,
447
+ disableReasoning: true,
448
+ toolChoice: { type: "tool", name: TOOL_NAME },
449
+ signal: controller.signal,
450
+ },
451
+ );
452
+ } catch (error) {
453
+ // A thrown transport error is as dead as a 404 for our purposes.
454
+ response = {
455
+ role: "assistant",
456
+ content: [],
457
+ stopReason: "error",
458
+ errorMessage: String(error),
459
+ } as unknown as AssistantMessage;
460
+ } finally {
461
+ clearTimeout(timer);
462
+ request.signal?.removeEventListener("abort", abortOnCaller);
463
+ }
464
+
465
+ if (request.signal?.aborted) return null;
466
+ if (timedOut || response.stopReason === "error" || response.stopReason === "aborted") {
467
+ const reason = timedOut
468
+ ? `timeout after ${attemptTimeoutMs}ms`
469
+ : `${response.errorStatus ?? response.stopReason}: ${(response.errorMessage ?? "").slice(0, 200)}`;
470
+ deadModels.set(modelKey(model), { at: Date.now(), reason });
471
+ logger.debug("decisions/llm: model failed, trying the next candidate", {
472
+ id: modelKey(model),
473
+ reason,
474
+ durationMs: Date.now() - started,
475
+ });
476
+ continue;
477
+ }
478
+
479
+ const args = readToolArguments(response.content);
480
+ if (!args) {
481
+ // The provider answered but ignored the forced tool: a model answer, not an
482
+ // availability problem, so it is neither retried nor remembered as dead.
483
+ logger.debug("decisions/llm: model did not emit the forced tool call", {
484
+ id: modelKey(model),
485
+ stopReason: response.stopReason,
486
+ });
487
+ return null;
488
+ }
489
+ const answers = toAnswers(request.questions, args);
490
+ if (Object.keys(answers).length === 0) return null;
491
+ return {
492
+ answers,
493
+ backend: "llm",
494
+ model: modelKey(model),
495
+ calibrated: false,
496
+ durationMs: Date.now() - started,
497
+ };
344
498
  }
345
- const answers = toAnswers(request.questions, args);
346
- if (Object.keys(answers).length === 0) return null;
347
- return {
348
- answers,
349
- backend: "llm",
350
- model: `${model.provider}/${model.id}`,
351
- calibrated: false,
352
- durationMs: Date.now() - started,
353
- };
499
+ return null;
354
500
  },
355
501
  };
356
502
  }
@@ -0,0 +1,163 @@
1
+ /**
2
+ * One typed decision per user turn, answering every routing question SKC has.
3
+ *
4
+ * Typed decisions started out wired to exactly one question — a five-way
5
+ * workflow enum — while the thirteen bundled UI skills were still selected by
6
+ * twenty hand-written regexes whose own comment measured them at 5 of 8 real
7
+ * frontend prompts and described the result as "not activation, it is hope".
8
+ *
9
+ * Both questions are about the same sentence, so they belong in the same call.
10
+ * The decision API takes a map of questions and returns a map of answers, which
11
+ * means adding the UI question costs one extra criteria block in the prompt and
12
+ * **zero** extra round trips: same latency budget, same deadline, same backend.
13
+ *
14
+ * Each question is skipped when a free deterministic stage already answered it,
15
+ * so the call shrinks to whatever is genuinely unknown — and vanishes entirely
16
+ * when nothing is.
17
+ */
18
+ import { logger } from "@sayknow-cli/utils";
19
+ import { BUNDLED_SKC_UI_SKILL_NAMES, type BundledSkcUiSkillName } from "../defaults/skc-ui-skills";
20
+ import { BUNDLED_UI_SKILL_MEANINGS } from "../hooks/ui-skill-keywords";
21
+ import { CANONICAL_SKC_WORKFLOW_SKILLS, type CanonicalSkcWorkflowSkill } from "../skill-state/active-state";
22
+ import type { DecisionService } from "./index";
23
+ import {
24
+ buildRoutingCriteria,
25
+ MAX_PROMPT_CHARS,
26
+ MIN_CALIBRATED_CONFIDENCE,
27
+ MIN_PROMPT_CHARS,
28
+ NONE_CHOICE,
29
+ promptSignalLength,
30
+ } from "./skill-routing";
31
+ import type { Question } from "./types";
32
+
33
+ const UI_INSTRUCTIONS =
34
+ "Which bundled UI craft skill should be loaded for this request? Choose none unless the request is about building, reviewing, or polishing a user-visible interface.";
35
+
36
+ /** Exported so tests can assert the contract the model is actually given. */
37
+ export function buildUiSkillCriteria(): Record<string, string> {
38
+ const criteria: Record<string, string> = {};
39
+ for (const skill of BUNDLED_SKC_UI_SKILL_NAMES) criteria[skill] = BUNDLED_UI_SKILL_MEANINGS[skill];
40
+ criteria[NONE_CHOICE] =
41
+ "Not interface work: backend, data, infrastructure, tooling, SKC's own terminal UI, or a question with no surface to build.";
42
+ return criteria;
43
+ }
44
+
45
+ export interface PromptTriage {
46
+ /** Null means the router deliberately chose no workflow, not that it failed. */
47
+ workflow: CanonicalSkcWorkflowSkill | null;
48
+ uiSkill: BundledSkcUiSkillName | null;
49
+ /** Only meaningful when {@link calibrated} is true. */
50
+ workflowConfidence: number | undefined;
51
+ calibrated: boolean;
52
+ }
53
+
54
+ export interface PromptTriageRequest {
55
+ text: string;
56
+ /** Skip the workflow question — the keyword table already answered it. */
57
+ skipWorkflow?: boolean;
58
+ /** Skip the UI question — the regex table already matched. */
59
+ skipUiSkill?: boolean;
60
+ signal?: AbortSignal | undefined;
61
+ }
62
+
63
+ export type PromptTriager = (request: PromptTriageRequest) => Promise<PromptTriage | null>;
64
+
65
+ function resolveChoice<T extends string>(
66
+ answer: unknown,
67
+ allowed: readonly T[],
68
+ calibrated: boolean,
69
+ label: string,
70
+ ): { value: T | null; confidence: number | undefined } {
71
+ const choice = answer as { type?: string; choice?: string; confidence?: number } | undefined;
72
+ if (choice?.type !== "choice") return { value: null, confidence: undefined };
73
+ if (choice.choice === NONE_CHOICE) return { value: null, confidence: choice.confidence };
74
+ const match = allowed.find(candidate => candidate === choice.choice);
75
+ if (!match) return { value: null, confidence: choice.confidence };
76
+ if (calibrated && (choice.confidence ?? 0) < MIN_CALIBRATED_CONFIDENCE) {
77
+ // A floor is only meaningful against a calibrated probability; against an
78
+ // ordinal score it would reject answers that are simply scaled differently.
79
+ logger.debug("decisions/prompt-triage: below confidence floor", {
80
+ question: label,
81
+ choice: choice.choice,
82
+ confidence: choice.confidence,
83
+ floor: MIN_CALIBRATED_CONFIDENCE,
84
+ });
85
+ return { value: null, confidence: choice.confidence };
86
+ }
87
+ return { value: match, confidence: choice.confidence };
88
+ }
89
+
90
+ /**
91
+ * Build the per-turn triager.
92
+ *
93
+ * Returns null when the service is disabled, the prompt is too short to carry
94
+ * intent, every question was already answered for free, or the backend did not
95
+ * produce a usable result. Null means "no information", which is different from
96
+ * a result whose fields are all null — that one is the router saying "none", and
97
+ * the keyword learner treats it as a negative example.
98
+ */
99
+ export function createPromptTriage(service: DecisionService): PromptTriager {
100
+ const workflowCriteria = buildRoutingCriteria();
101
+ const uiCriteria = buildUiSkillCriteria();
102
+ return async (request: PromptTriageRequest): Promise<PromptTriage | null> => {
103
+ if (!service.enabled) return null;
104
+ const trimmed = request.text.trim();
105
+ if (promptSignalLength(trimmed) < MIN_PROMPT_CHARS) return null;
106
+ const state = trimmed.length > MAX_PROMPT_CHARS ? trimmed.slice(0, MAX_PROMPT_CHARS) : trimmed;
107
+
108
+ const questions: Record<string, Question> = {};
109
+ if (!request.skipWorkflow) {
110
+ questions.workflow = {
111
+ type: "choice",
112
+ instructions:
113
+ "Which workflow should handle this user request? Choose none unless the request clearly calls for one of the workflows.",
114
+ criteria: workflowCriteria,
115
+ };
116
+ }
117
+ if (!request.skipUiSkill) {
118
+ questions.uiSkill = { type: "choice", instructions: UI_INSTRUCTIONS, criteria: uiCriteria };
119
+ }
120
+ if (Object.keys(questions).length === 0) return null;
121
+
122
+ const result = await service.decide({ state, questions, signal: request.signal });
123
+ if (!result) return null;
124
+
125
+ const workflow = request.skipWorkflow
126
+ ? { value: null, confidence: undefined }
127
+ : resolveChoice(result.answers.workflow, CANONICAL_SKC_WORKFLOW_SKILLS, result.calibrated, "workflow");
128
+ const uiSkill = request.skipUiSkill
129
+ ? { value: null, confidence: undefined }
130
+ : resolveChoice(result.answers.uiSkill, BUNDLED_SKC_UI_SKILL_NAMES, result.calibrated, "uiSkill");
131
+
132
+ logger.debug("decisions/prompt-triage: answered", {
133
+ workflow: workflow.value,
134
+ uiSkill: uiSkill.value,
135
+ backend: result.backend,
136
+ model: result.model,
137
+ calibrated: result.calibrated,
138
+ durationMs: result.durationMs,
139
+ });
140
+ return {
141
+ workflow: workflow.value,
142
+ uiSkill: uiSkill.value,
143
+ workflowConfidence: workflow.confidence,
144
+ calibrated: result.calibrated,
145
+ };
146
+ };
147
+ }
148
+
149
+ export type SkillRouter = (text: string, signal?: AbortSignal) => Promise<CanonicalSkcWorkflowSkill | null>;
150
+
151
+ /**
152
+ * Workflow-only view of the triager.
153
+ *
154
+ * A few lines over the same implementation rather than a second one: the eval
155
+ * harness in `scripts/eval-skill-routing.ts` and the routing tests want
156
+ * text-in/skill-out and have no UI question to ask, and a parallel router would
157
+ * be free to drift away from the thresholds this one enforces.
158
+ */
159
+ export function createSemanticSkillRouter(service: DecisionService): SkillRouter {
160
+ const triage = createPromptTriage(service);
161
+ return async (text: string, signal?: AbortSignal): Promise<CanonicalSkcWorkflowSkill | null> =>
162
+ (await triage({ text, skipUiSkill: true, signal }))?.workflow ?? null;
163
+ }
@@ -1,25 +1,29 @@
1
1
  /**
2
- * Semantic fallback for workflow-skill routing.
2
+ * The workflow-routing question: what the model is asked, and the thresholds the
3
+ * answer is held to.
3
4
  *
4
- * The keyword table in `hooks/skill-keywords.ts` is thirteen literal strings. It is
5
+ * The keyword table in `hooks/skill-keywords.ts` is a list of literal strings. It is
5
6
  * exact and free, and it is the right first stage — but measured against realistic
6
7
  * paraphrases it recalls 4/17, and **0/9 in Korean**, which is most of our users. A
7
8
  * miss is not fatal (the model still sees the routing rules in the system prompt), but
8
9
  * it means the deterministic gate simply does not exist for those prompts.
9
10
  *
10
- * This module fills that gap only where the keyword stage produced nothing:
11
+ * The staging that closes the gap lives in `prompt-triage.ts`:
11
12
  *
12
- * keyword (exact, free) -> semantic (this, one cheap call) -> system prompt (as today)
13
+ * keyword + learned (exact, free) -> one typed decision -> system prompt (as today)
13
14
  *
14
- * The two stages fail in opposite directions, which is why both are kept. Measured on
15
- * the same 22 prompts, the literal stage is the one that catches `ultragoal this` and
16
- * `consensus plan`; the semantic stage is the one that catches everything Korean.
15
+ * The stages fail in opposite directions, which is why all of them are kept. Measured
16
+ * on the same 22 prompts, the literal stage is the one that catches `ultragoal this`
17
+ * and `consensus plan`; the semantic stage is the one that catches everything Korean.
18
+ *
19
+ * This file holds only the contract and the numbers, so the criteria cannot drift away
20
+ * from the thresholds that judge answers against them.
17
21
  */
18
- import { logger } from "@sayknow-cli/utils";
19
22
  import { CANONICAL_SKC_WORKFLOW_SKILLS, type CanonicalSkcWorkflowSkill } from "../skill-state/active-state";
20
- import type { DecisionService } from "./index";
21
23
 
22
- const NONE = "none";
24
+ /** Shared across every routing question so one answer shape covers them all. */
25
+ export const NONE_CHOICE = "none";
26
+ const NONE = NONE_CHOICE;
23
27
 
24
28
  /**
25
29
  * What each workflow is *for*, in the words a user would recognise. These descriptions
@@ -40,7 +44,7 @@ const WORKFLOW_MEANINGS: Record<CanonicalSkcWorkflowSkill, string> = {
40
44
  team: "The work is large enough to split across several coordinated workers running in parallel.",
41
45
  };
42
46
 
43
- const ROUTING_INSTRUCTIONS =
47
+ export const ROUTING_INSTRUCTIONS =
44
48
  "Which workflow should handle this user request? Choose none unless the request clearly calls for one of the workflows.";
45
49
 
46
50
  /** Exported so tests can assert the contract the model is actually given. */
@@ -52,10 +56,30 @@ export function buildRoutingCriteria(): Record<string, string> {
52
56
  return criteria;
53
57
  }
54
58
 
55
- /** Prompts below this length never carry enough signal to justify a model round-trip. */
56
- const MIN_PROMPT_CHARS = 12;
59
+ /**
60
+ * Prompts below this length never carry enough signal to justify a model round-trip.
61
+ * Measured with {@link promptSignalLength}, not `String.length`: the floor was fitted
62
+ * to English and Korean, and in code units a complete Chinese request is shorter than
63
+ * "ok thanks".
64
+ */
65
+ export const MIN_PROMPT_CHARS = 12;
57
66
  /** Only the opening of a prompt decides its workflow; the rest is payload. */
58
- const MAX_PROMPT_CHARS = 4_000;
67
+ export const MAX_PROMPT_CHARS = 4_000;
68
+
69
+ const DENSE_SCRIPT_PATTERN = /[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Hangul}]/gu;
70
+
71
+ /**
72
+ * Prompt length in Latin-letter equivalents.
73
+ *
74
+ * One Han character, kana or Hangul syllable carries what two or three Latin letters
75
+ * do, so each counts double. "先做架构设计" (6 code units) is a whole request; "谢谢" and
76
+ * "고마워요" still fall under the floor, which is the point of having one.
77
+ */
78
+ export function promptSignalLength(text: string): number {
79
+ let dense = 0;
80
+ for (const _ of text.matchAll(DENSE_SCRIPT_PATTERN)) dense++;
81
+ return text.length + dense;
82
+ }
59
83
 
60
84
  /**
61
85
  * Minimum calibrated confidence required to activate a workflow.
@@ -79,45 +103,4 @@ const MAX_PROMPT_CHARS = 4_000;
79
103
  * through a forced enum has no meaningful confidence to compare against, so gating on a
80
104
  * number it did not really produce would just be superstition.
81
105
  */
82
- const MIN_CALIBRATED_CONFIDENCE = 0.75;
83
-
84
- export type SkillRouter = (text: string) => Promise<CanonicalSkcWorkflowSkill | null>;
85
-
86
- /**
87
- * Build the semantic router. Returns null-resolving function when the service is
88
- * disabled so the caller keeps its existing behaviour with no branching.
89
- */
90
- export function createSemanticSkillRouter(service: DecisionService): SkillRouter {
91
- const criteria = buildRoutingCriteria();
92
- return async (text: string): Promise<CanonicalSkcWorkflowSkill | null> => {
93
- if (!service.enabled) return null;
94
- const trimmed = text.trim();
95
- if (trimmed.length < MIN_PROMPT_CHARS) return null;
96
- const state = trimmed.length > MAX_PROMPT_CHARS ? trimmed.slice(0, MAX_PROMPT_CHARS) : trimmed;
97
-
98
- const result = await service.decide({
99
- state,
100
- questions: { workflow: { type: "choice", instructions: ROUTING_INSTRUCTIONS, criteria } },
101
- });
102
- const answer = result?.answers.workflow;
103
- if (!result || answer?.type !== "choice" || answer.choice === NONE) return null;
104
- const skill = CANONICAL_SKC_WORKFLOW_SKILLS.find(candidate => candidate === answer.choice);
105
- if (!skill) return null;
106
- if (result.calibrated && (answer.confidence ?? 0) < MIN_CALIBRATED_CONFIDENCE) {
107
- logger.debug("decisions/skill-routing: below confidence floor, leaving routing alone", {
108
- skill,
109
- confidence: answer.confidence,
110
- floor: MIN_CALIBRATED_CONFIDENCE,
111
- });
112
- return null;
113
- }
114
- logger.debug("decisions/skill-routing: semantic match", {
115
- skill,
116
- backend: result.backend,
117
- confidence: answer.confidence,
118
- calibrated: result.calibrated,
119
- durationMs: result.durationMs,
120
- });
121
- return skill;
122
- };
123
- }
106
+ export const MIN_CALIBRATED_CONFIDENCE = 0.75;