@sayknow-cli/coding-agent 0.5.25 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/CHANGELOG.md +36 -1
  2. package/dist/types/config/settings-schema.d.ts +51 -5
  3. package/dist/types/config/task-model-specialties.d.ts +55 -0
  4. package/dist/types/decisions/keyword-learning.d.ts +61 -0
  5. package/dist/types/decisions/llm-backend.d.ts +13 -1
  6. package/dist/types/decisions/prompt-triage.d.ts +42 -0
  7. package/dist/types/decisions/skill-routing.d.ts +41 -6
  8. package/dist/types/decisions/task-routing.d.ts +96 -11
  9. package/dist/types/hooks/native-prompt-routing.d.ts +21 -0
  10. package/dist/types/hooks/native-skill-hook.d.ts +3 -0
  11. package/dist/types/hooks/skill-keywords.d.ts +9 -0
  12. package/dist/types/hooks/skill-state.d.ts +20 -3
  13. package/dist/types/hooks/ui-skill-keywords.d.ts +15 -0
  14. package/dist/types/i18n/messages/en.d.ts +15 -0
  15. package/dist/types/lsp/index.d.ts +1 -1
  16. package/dist/types/lsp/types.d.ts +1 -1
  17. package/dist/types/modes/components/model-selector.d.ts +11 -0
  18. package/dist/types/sdk/session.d.ts +3 -13
  19. package/dist/types/session/agent-session.d.ts +8 -0
  20. package/dist/types/session/auth-storage-discovery.d.ts +13 -0
  21. package/dist/types/task/index.d.ts +1 -1
  22. package/dist/types/task/receipt.d.ts +2 -0
  23. package/dist/types/task/types.d.ts +114 -18
  24. package/dist/types/tools/browser.d.ts +2 -2
  25. package/dist/types/tools/subagent.d.ts +2 -2
  26. package/package.json +7 -7
  27. package/scripts/eval-skill-routing.ts +37 -12
  28. package/src/config/settings-schema.ts +64 -12
  29. package/src/config/task-model-specialties.ts +131 -0
  30. package/src/decisions/index.ts +8 -2
  31. package/src/decisions/keyword-learning.ts +678 -0
  32. package/src/decisions/llm-backend.ts +213 -67
  33. package/src/decisions/prompt-triage.ts +163 -0
  34. package/src/decisions/skill-routing.ts +39 -56
  35. package/src/decisions/task-routing.ts +382 -66
  36. package/src/decisions/typesafe-backend.ts +3 -0
  37. package/src/hooks/native-prompt-routing.ts +190 -0
  38. package/src/hooks/native-skill-hook.ts +21 -12
  39. package/src/hooks/skill-keywords.ts +9 -0
  40. package/src/hooks/skill-state.ts +41 -10
  41. package/src/hooks/ui-skill-keywords.ts +67 -10
  42. package/src/i18n/messages/de.ts +16 -0
  43. package/src/i18n/messages/en.ts +16 -0
  44. package/src/i18n/messages/es.ts +16 -0
  45. package/src/i18n/messages/fr.ts +16 -0
  46. package/src/i18n/messages/ja.ts +16 -0
  47. package/src/i18n/messages/ko.ts +16 -0
  48. package/src/i18n/messages/zh.ts +16 -0
  49. package/src/internal-urls/docs-index.generated.ts +1 -1
  50. package/src/main.ts +1 -1
  51. package/src/modes/components/model-selector.ts +275 -34
  52. package/src/modes/controllers/selector-controller.ts +50 -2
  53. package/src/modes/shared/agent-wire/command-dispatch.ts +1 -1
  54. package/src/prompts/tools/task.md +1 -0
  55. package/src/sdk/session.ts +5 -82
  56. package/src/session/agent-session.ts +137 -37
  57. package/src/session/auth-storage-discovery.ts +83 -0
  58. package/src/slash-commands/builtin-registry.ts +11 -9
  59. package/src/task/index.ts +98 -40
  60. package/src/task/receipt.ts +3 -0
  61. package/src/task/types.ts +44 -0
package/CHANGELOG.md CHANGED
@@ -2,7 +2,42 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
- ## [0.5.25] - 2026-09-21
5
+ ## [0.6.0] - 2026-09-24
6
+
7
+ ### Added
8
+
9
+ - The 13 bundled UI skills are now part of the typed decision instead of 20 hand-written regexes. Measured recall of the pattern table on real Korean prompts was 5/8; a prompt like "이 부분 손보기 전에 어떻게 갈지부터 같이 정하고 넘어가자" matched nothing and the skill never activated. The picked skill still arrives as the same hidden `developer` reminder, now labelled `semantic match` when the model answered and `matched` when a pattern did.
10
+ - Workflow routing and UI skill selection are **one** decision call, not two. The triage asks both questions in a single request (`questions: ["workflow", "uiSkill"]`), and a prompt the pattern table already answered drops that question from the request instead of paying for it — so adding UI routing cost zero extra round trips.
11
+ - The skill keyword table now fills itself. Every answered routing decision mines ordered two-stem shingles from the prompt into `~/.skc/agent/learned-skill-keywords.json`. A shingle is promoted to a real keyword after 2 independent prompts when the backend reports calibrated confidence ≥ 0.9, or 3 when it only ranks; a single contradicting answer blocks that shingle permanently, so a coincidence cannot become a rule. Learned entries carry strictly negative priority — a hand-written keyword always wins — and a promoted pattern answers for free on the next matching prompt. `decisions.keywordLearning: false` turns the store off; `SKC_LEARNED_KEYWORDS_PATH` relocates it.
12
+ - Routing and learning work in whatever language the prompt is written in. Mining uses the platform Unicode segmenter, so Chinese, Japanese and Thai prompts with no spaces come apart into words (Han runs the ICU dictionary misses fall back to character bigrams; hiragana-only tokens are dropped as grammar); non-ASCII alphabets are kept instead of discarded, and a learned stem gets a Unicode word boundary in spaced scripts and a bare substring match in unspaced ones. The length floor that skips the model on "ok thanks" is measured in signal rather than code units, so a six-character Chinese request is no longer thrown away. Measured against the hosted `jev` backend: 40/40 across ko, en, zh, ja, es and ru, and a ralplan phrasing repeated twice in Chinese, Japanese or Russian promotes a keyword that fires on an unseen sentence in that language.
13
+ - The Codex host gets the same router. `skc codex-native-hook` ran the hand-written keyword table and nothing else: the semantic stage had a parameter waiting for it since it shipped, and learned keywords were never loaded there, so a phrasing learned in a CLI session did nothing under Codex. The hook now runs all three stages — hand-written keywords, learned keywords, then one typed decision for whatever those missed — on the same backend chain (TypeSafe key, else a running local runtime, else the small model of the provider you are logged into) and feeds the answer back into the learned table. Because the hook is a fresh process per prompt, credentials and the model registry are opened only on a prompt the tables did not fully answer and closed before it exits; a prompt they did answer still costs the same ~0.3s it did before. Measured through the real hook subprocess: a Chinese ralplan phrasing routed by the model on the first prompt fired from the learned table on the next one.
14
+
15
+ ### Changed
16
+
17
+ - Typed decisions (`decisions.enabled`) are now **on by default**. The keyword stage of workflow routing still runs first and free; the model stage only spends a small-model call on turns the table did not answer. Set `decisions.enabled: false` to route on the keyword table alone.
18
+ - The small-model fallback now ranks the **provider you are chatting with** first. Measured on a registry with Anthropic, Codex and Z.ai keys: price order put four free `zai/glm-*` reasoning tiers ahead of `claude-haiku-4-5`, and the first took 8.9s — past the 8s service deadline, so the answer was nothing. Haiku next to Opus, mini next to GPT is also the model you would have picked by hand.
19
+ - The routing decision now goes through the session's own `streamFn` instead of a bare `completeSimple`, so it gets the same proxy, credential-invalidation and retry wrapping as the turn, and a host that injects a transport intercepts it. Hosts opt sessions in with `AgentSessionConfig.semanticWorkflowRouting` (`createAgentSession` does); a bare `AgentSession` never routes, so scripted transports are not consumed by a call they did not script. While the decision is in flight the session reports `isStreaming`, so a second `prompt()` queues or rejects as it would mid-turn, and `abort()` cancels the decision.
20
+ - `task.modelRouting.enabled` now defaults to **true**. The tier lists it reads are empty until you set them, so nothing changes for anyone who has not configured them — but shipping it off meant a user who *had* configured tiers still got no routing, which made the whole feature dead code.
21
+
22
+ ### Fixed
23
+
24
+ - The small-model fallback answered **nothing** on any registry with an Anthropic key: the cheapest qualifying entry, `claude-3-haiku-20240307`, is retired and returns 404, and the backend logged it as "model did not emit the forced tool call" and gave up. The three retired Claude 3.x ids are now dropped from the catalog *and* from stale on-disk cache rows, the backend logs the real status, and a candidate that errors or blows a 3.5s attempt budget is skipped for the next one and remembered for ten minutes so the stall is paid once, not per prompt.
25
+ - A local runtime contributes **one** candidate — the smallest chat model — and embedding models (`nomic-embed-text`) are no longer offered a chat call. Measured with Ollama: five local models ranked ahead of every hosted one, each timed out on cold start, and the hosted model was never reached before the deadline.
26
+ - A decision request whose caller had already aborted — measured: `abort()` landing during the previous backend's credential lookup — ran a full attempt budget of silence, because an abort listener added to a signal that has already fired never runs. Every backend now checks the signal after each await.
27
+
28
+ ## [0.5.26] - 2026-09-23
29
+ ### Added
30
+
31
+ - `/model` role rows now open a second level so a model can be assigned to a *detailed use* rather than only to a whole role. `planner`/`architect` offer backend architecture and frontend design, `executor` offers implementation and testing, `critic` offers review; `default` has none and still assigns in one keystroke. The two bulk rows deliberately stay canonical-only — an action that also rewrote five detailed-use overrides could not be undone from the same menu. Assignments persist to the new `task.modelRouting.specialtyModels` record, whose keys are allowlisted to those five ids. Saving one while `task.modelRouting.enabled` is off persists it and says so instead of silently enabling a feature you did not ask for. First-level rows are now stable descriptors rather than positional index arithmetic, which is what previously made every added row a chance to assign a model to the wrong role.
32
+ - Subagent model routing now runs **one classification per child task** instead of one per `task` call. A batch shares an agent but not a workload, so an implementation slice and a test slice in the same call no longer have to share a single verdict. Each child dispatches its own ordered chain — detailed-use model, then tier model, then the role's configured chain — so a detailed-use model that cannot authenticate still falls through to something the role can actually run. A high-risk assignment skips the detailed-use axis entirely, because a lateral swap can move sideways into something weaker.
33
+ - Receipts and progress now carry what the router **requested** separately from what the spawn ran on. Because the dispatched value is a fallback chain, conflating the two would let a receipt claim a detailed-use model was used when the chain actually fell through to the baseline. When the decision backend is an ordinary LLM it returns no probabilities at all, so the record carries an ordinal clarity score and `calibrated: false` rather than a fabricated confidence.
34
+ - The `task` tool accepts `.specialty` per task. Declaring the kind of work is the deterministic path to a detailed-use model: no classifier runs, `task.modelRouting.enabled` is not consulted, and no confidence bar applies — if the user assigned a model to that specialty under `/model`, the child runs on it and leaves it only on a transport error (429, 5xx, auth, quota), when the child session's fallback chain advances to the role baseline composed behind it. A declared specialty ignores the menu's role grouping on purpose: the setting is one flat map, and a frontend build delegated to `executor` should get the frontend model. Receipts mark these `declared: true` so they cannot be mistaken for a classifier verdict. Auto-detection from assignment text is unchanged and still opt-in.
35
+
36
+ ### Fixed
37
+
38
+ - Model-selector suites that assert UI copy now pin the locale — and restore it. They previously inherited whatever `language` the developer had configured, because a sibling suite restores the real agent directory and reloads real settings, and one suite pinned Korean without putting it back — so whether they passed depended on test file ordering.
39
+ - The first-level `/model` role rows (`Set as EXECUTOR (Executor)`) were hardcoded English inside an otherwise localized menu. They now go through i18n in all seven locales; the redundant `(Name)` suffix is dropped where the locale does not need it, and the role tag stays verbatim because it is the same identifier receipts and `/model` arguments use.
40
+ - Login, `--list-models`, `/model`, and `/model provider/id` now refresh provider catalogs online so newly published Kimi/Grok (and other OpenAI-compatible) ids appear without waiting for a bundled `models.json` regen or a 2h cache TTL.
6
41
 
7
42
  ## [0.5.22] - 2026-09-19
8
43
 
@@ -104,6 +104,11 @@ interface RecordDef<T> {
104
104
  type: "record";
105
105
  default: Record<string, T>;
106
106
  valueSchema?: RecordValueDef;
107
+ /**
108
+ * Closed key set. When present, reconciliation rejects any other key instead of
109
+ * silently carrying a typo that no consumer will ever read.
110
+ */
111
+ keys?: readonly string[];
107
112
  ui?: UiBase;
108
113
  }
109
114
  type SettingDef = BooleanDef | StringDef | NumberDef | EnumDef<readonly string[]> | ArrayDef<unknown> | RecordDef<unknown>;
@@ -2414,11 +2419,29 @@ export declare const SETTINGS_SCHEMA: {
2414
2419
  };
2415
2420
  readonly "decisions.enabled": {
2416
2421
  readonly type: "boolean";
2417
- readonly default: false;
2422
+ readonly default: true;
2418
2423
  readonly ui: {
2419
2424
  readonly tab: "context";
2420
2425
  readonly label: "Typed decisions";
2421
- readonly description: "Add a model-backed second stage to workflow routing. The keyword table already runs on every turn and costs nothing; this handles the phrasings it cannot express, which is most wording that is not a literal match. Costs one small model call, and only on turns the keyword table did not already answer. Any failure falls back to keyword-only behaviour, but a successful answer can also select a different workflow than the deep-interview ambiguity detector would have.";
2426
+ readonly description: "Model-backed second stage for workflow routing. The keyword table runs first on every turn and costs nothing; this handles the phrasings it cannot express, which is most wording that is not a literal match. Costs one small model call, only on turns the keyword table did not already answer, on the cheapest backend available: TypeSafe when a key is stored, else a local runtime that is running, else the small model of the provider you are chatting with. Any failure falls back to keyword-only behaviour, but a successful answer can also select a different workflow than the deep-interview ambiguity detector would have. Turn off to route on the keyword table alone.";
2427
+ };
2428
+ };
2429
+ /**
2430
+ * Feed routing answers back into the deterministic keyword table.
2431
+ *
2432
+ * The hand-written table cannot be grown by hand for Korean — measured recall
2433
+ * was 0/9 — so it grows itself instead: a two-stem pattern that produced the
2434
+ * same routing answer on two distinct prompts, and was never contradicted, is
2435
+ * promoted and thereafter fires for free. One contradiction retracts it
2436
+ * permanently. Stems and hashes are stored; prompt text never is.
2437
+ */
2438
+ readonly "decisions.keywordLearning": {
2439
+ readonly type: "boolean";
2440
+ readonly default: true;
2441
+ readonly ui: {
2442
+ readonly tab: "context";
2443
+ readonly label: "Learn routing keywords";
2444
+ readonly description: "Remember the phrasings the routing model resolves, so repeating one stops costing a model call. A pattern must give the same answer on two different prompts before it fires on its own, and a single disagreement removes it for good. Only word stems and hashes are written to disk, never your prompts. Turn off to keep the keyword table frozen at its built-in entries.";
2422
2445
  };
2423
2446
  };
2424
2447
  /**
@@ -2430,12 +2453,14 @@ export declare const SETTINGS_SCHEMA: {
2430
2453
  * mid-session invalidates the prompt cache, which on a long context costs
2431
2454
  * more than the cheaper tier saves.
2432
2455
  *
2433
- * Needs `decisions.enabled` and at least two tiers configured. Without both
2434
- * it never fires and the configured role models are used unchanged.
2456
+ * Needs `decisions.enabled` and at least two tiers configured. The tiers are
2457
+ * empty by default, so this is a no-op until the user sets them — which is why
2458
+ * it defaults on: shipping it off meant the routing code existed and never ran
2459
+ * even for users who had configured the tiers it needs.
2435
2460
  */
2436
2461
  readonly "task.modelRouting.enabled": {
2437
2462
  readonly type: "boolean";
2438
- readonly default: false;
2463
+ readonly default: true;
2439
2464
  readonly ui: {
2440
2465
  readonly tab: "tasks";
2441
2466
  readonly label: "Route subagent models per task";
@@ -2471,6 +2496,27 @@ export declare const SETTINGS_SCHEMA: {
2471
2496
  readonly type: "string";
2472
2497
  readonly default: "";
2473
2498
  };
2499
+ /**
2500
+ * Per-specialty models — the **work-kind** axis.
2501
+ *
2502
+ * Keys are the bounded ids in `config/task-model-specialties.ts`; values use the
2503
+ * same selector grammar as any other model setting, so a chain is allowed. A
2504
+ * specialty with no entry inherits the role's resolved model unchanged, which is
2505
+ * why the default is empty rather than pre-populated.
2506
+ *
2507
+ * These are routing hints, never agents: nothing here widens the canonical role
2508
+ * roster, model-profile role keys, tool grants, or spawn permissions. Saving one
2509
+ * does not enable `task.modelRouting.enabled`; the assignment surface reports the
2510
+ * disabled state instead of silently turning routing on.
2511
+ */
2512
+ readonly "task.modelRouting.specialtyModels": {
2513
+ readonly type: "record";
2514
+ readonly default: Record<string, ModelSelectorValue>;
2515
+ readonly valueSchema: {
2516
+ readonly type: "model-selector-value";
2517
+ };
2518
+ readonly keys: readonly ["backendArchitecture", "frontendDesign", "implementation", "testing", "review"];
2519
+ };
2474
2520
  readonly "ttsr.enabled": {
2475
2521
  readonly type: "boolean";
2476
2522
  readonly default: true;
@@ -0,0 +1,55 @@
1
+ /** Stable configuration/routing identifiers. Display strings are localized separately. */
2
+ export declare const TASK_MODEL_SPECIALTY_IDS: readonly ["backendArchitecture", "frontendDesign", "implementation", "testing", "review"];
3
+ export type TaskModelSpecialty = (typeof TASK_MODEL_SPECIALTY_IDS)[number];
4
+ /**
5
+ * Which canonical role agents may receive each specialty.
6
+ *
7
+ * Planning roles share the two design specialties on purpose: the user's intent is
8
+ * "a backend-strong model plans backend work", not "planner and architect each get
9
+ * their own private backend model". Implementation roles never take a design
10
+ * specialty, mirroring the existing frontend-domain rule.
11
+ */
12
+ export declare const TASK_MODEL_SPECIALTY_ROLES: Readonly<Record<TaskModelSpecialty, readonly string[]>>;
13
+ /** Neutral classifier outcome: the work does not clearly belong to one specialty. */
14
+ export declare const TASK_MODEL_SPECIALTY_NONE: "none";
15
+ /** Bounded-key predicate for the `task.modelRouting.specialtyModels` record. */
16
+ export declare function isTaskModelSpecialty(value: unknown): value is TaskModelSpecialty;
17
+ /** Specialties a given canonical role agent is allowed to receive, in declaration order. */
18
+ export declare function specialtiesForRole(agentName: string): TaskModelSpecialty[];
19
+ /** Whether a specialty may be applied to work routed to this canonical role agent. */
20
+ export declare function specialtySupportsRole(specialty: TaskModelSpecialty, agentName: string): boolean;
21
+ /** Where a composed candidate came from, so a receipt never overstates what was selected. */
22
+ export type TaskRoutingSource = "specialty" | "legacy-frontend" | "tier" | "baseline";
23
+ export interface TaskRoutingCandidate {
24
+ /** The configured selector, normalized but otherwise untouched. */
25
+ selector: string;
26
+ source: TaskRoutingSource;
27
+ /** Set when `source` is `specialty` or `legacy-frontend`. */
28
+ specialty?: TaskModelSpecialty;
29
+ /** Set when `source` is `tier`. */
30
+ tier?: string;
31
+ }
32
+ /**
33
+ * Identity used for deduplication.
34
+ *
35
+ * An explicit thinking suffix is part of the identity: `provider/model:high` and
36
+ * `provider/model:low` are different configured intents, and collapsing them would
37
+ * silently drop the user's effort choice from a chain.
38
+ */
39
+ export declare function specialtySelectorIdentity(selector: string): string;
40
+ /** The `provider/model` part, ignoring any thinking suffix. Used for "same model" checks. */
41
+ export declare function specialtySelectorHead(selector: string): string;
42
+ /** Expand a configured selector value into normalized candidates tagged with their origin. */
43
+ export declare function toRoutingCandidates(value: string | readonly string[] | undefined, source: TaskRoutingSource, extra?: {
44
+ specialty?: TaskModelSpecialty;
45
+ tier?: string;
46
+ }): TaskRoutingCandidate[];
47
+ /**
48
+ * Concatenate candidate segments, keeping the first occurrence of each identity.
49
+ *
50
+ * A selector that also exists in the baseline segment is attributed to `baseline`
51
+ * even when an earlier specialty segment introduced it. Resolving that entry proves
52
+ * only that the role's own configured model was usable — reporting it as a specialty
53
+ * hit would claim a routing decision that never happened.
54
+ */
55
+ export declare function dedupeRoutingCandidates(segments: readonly TaskRoutingCandidate[][]): TaskRoutingCandidate[];
@@ -0,0 +1,61 @@
1
+ import type { SkillKeywordDefinition } from "../hooks/skill-keywords";
2
+ import type { CanonicalSkcWorkflowSkill } from "../skill-state/active-state";
3
+ export declare const LEARNED_KEYWORD_STORE_VERSION = 1;
4
+ /** Test seam. Also lets a host keep the table out of the user's home. */
5
+ export declare function setLearnedKeywordStorePath(value: string | undefined): void;
6
+ export declare function resetLearnedKeywordCache(): void;
7
+ /**
8
+ * Where the learned table lives.
9
+ *
10
+ * User-global on purpose: a phrasing you use is a phrasing you use, and making it
11
+ * per-repository would mean relearning the same rule in every checkout.
12
+ * `SKC_LEARNED_KEYWORDS_PATH` redirects it, so a test driving a real session under
13
+ * a temp agent dir cannot write into the developer's actual home.
14
+ */
15
+ export declare function getLearnedKeywordStorePath(): string;
16
+ /**
17
+ * Ordered stem pairs within a three-token window.
18
+ *
19
+ * The gap is what makes a learned pattern generalize where the literal table
20
+ * cannot: a prompt that says "make me a plan document" mines `plan … document`,
21
+ * which then also matches "make the plan into a document" — the phrasing that
22
+ * made the hand-written entry miss.
23
+ */
24
+ export declare function mineShingles(text: string): string[];
25
+ /**
26
+ * Promoted patterns, in the shape the deterministic stage consumes.
27
+ *
28
+ * Priority sits one below the hand-written entry for the same workflow: when a
29
+ * learned pattern and an enumerated keyword disagree, the human wins.
30
+ */
31
+ export declare function loadLearnedKeywordDefinitions(): Promise<SkillKeywordDefinition[]>;
32
+ export interface RoutingObservation {
33
+ text: string;
34
+ /** Null means the router deliberately chose no workflow — a negative example. */
35
+ skill: CanonicalSkcWorkflowSkill | null;
36
+ confidence?: number | undefined;
37
+ calibrated?: boolean | undefined;
38
+ }
39
+ export interface LearningOutcome {
40
+ promoted: SkillKeywordDefinition[];
41
+ retracted: number;
42
+ }
43
+ /**
44
+ * Feed one routing answer into the table.
45
+ *
46
+ * Best-effort by construction: it runs after the turn's routing decision is
47
+ * already made, so a failure here changes nothing the user can observe.
48
+ */
49
+ export declare function observeRouting(observation: RoutingObservation): Promise<LearningOutcome>;
50
+ export interface LearnedKeywordSummary {
51
+ entries: {
52
+ keyword: string;
53
+ skill: CanonicalSkcWorkflowSkill;
54
+ positives: number;
55
+ lastSeen: number;
56
+ }[];
57
+ candidates: number;
58
+ blocked: number;
59
+ }
60
+ /** What the table has actually learned, for `skc` surfaces and tests. */
61
+ export declare function summarizeLearnedKeywords(): Promise<LearnedKeywordSummary>;
@@ -16,7 +16,7 @@
16
16
  * 2. Every question goes in **one** call. Splitting them multiplies cost and latency
17
17
  * while the enum constraint already keeps each field independent.
18
18
  */
19
- import { type Api, type Model } from "@sayknow-cli/ai";
19
+ import { type Api, completeSimple, type Model } from "@sayknow-cli/ai";
20
20
  import type { ModelRegistry } from "../config/model-registry";
21
21
  import type { Settings } from "../config/settings";
22
22
  import { type DecisionBackend } from "./types";
@@ -40,12 +40,24 @@ export interface LlmBackendDeps {
40
40
  maxInputCostPerMTok?: number;
41
41
  /** Injected in tests to make the local-runtime probe deterministic. */
42
42
  fetchImpl?: typeof fetch;
43
+ /** Injected in tests to script provider answers without a network. */
44
+ completeImpl?: typeof completeSimple;
45
+ /** Per-candidate deadline; injected in tests so a hung provider does not cost real seconds. */
46
+ attemptTimeoutMs?: number;
43
47
  registry: ModelRegistry;
44
48
  settings: Settings;
45
49
  sessionId?: string;
46
50
  /** Overrides role resolution; used by callers that already picked a model. */
47
51
  model?: Model<Api>;
52
+ /**
53
+ * Provider of the model the caller is already talking to. Its small model is tried
54
+ * first when nothing was configured; see `rankSmallModels` for why. A thunk is
55
+ * accepted so a long-lived backend follows the session when the user switches model.
56
+ */
57
+ preferredProvider?: string | (() => string | undefined);
48
58
  }
49
59
  /** Reset between tests; also lets a caller force a fresh probe after starting a runtime. */
50
60
  export declare function clearLocalRuntimeLivenessCache(): void;
61
+ /** Reset between tests. */
62
+ export declare function clearDeadModelCache(): void;
51
63
  export declare function createLlmDecisionBackend(deps: LlmBackendDeps): DecisionBackend;
@@ -0,0 +1,42 @@
1
+ import { type BundledSkcUiSkillName } from "../defaults/skc-ui-skills";
2
+ import { type CanonicalSkcWorkflowSkill } from "../skill-state/active-state";
3
+ import type { DecisionService } from "./index";
4
+ /** Exported so tests can assert the contract the model is actually given. */
5
+ export declare function buildUiSkillCriteria(): Record<string, string>;
6
+ export interface PromptTriage {
7
+ /** Null means the router deliberately chose no workflow, not that it failed. */
8
+ workflow: CanonicalSkcWorkflowSkill | null;
9
+ uiSkill: BundledSkcUiSkillName | null;
10
+ /** Only meaningful when {@link calibrated} is true. */
11
+ workflowConfidence: number | undefined;
12
+ calibrated: boolean;
13
+ }
14
+ export interface PromptTriageRequest {
15
+ text: string;
16
+ /** Skip the workflow question — the keyword table already answered it. */
17
+ skipWorkflow?: boolean;
18
+ /** Skip the UI question — the regex table already matched. */
19
+ skipUiSkill?: boolean;
20
+ signal?: AbortSignal | undefined;
21
+ }
22
+ export type PromptTriager = (request: PromptTriageRequest) => Promise<PromptTriage | null>;
23
+ /**
24
+ * Build the per-turn triager.
25
+ *
26
+ * Returns null when the service is disabled, the prompt is too short to carry
27
+ * intent, every question was already answered for free, or the backend did not
28
+ * produce a usable result. Null means "no information", which is different from
29
+ * a result whose fields are all null — that one is the router saying "none", and
30
+ * the keyword learner treats it as a negative example.
31
+ */
32
+ export declare function createPromptTriage(service: DecisionService): PromptTriager;
33
+ export type SkillRouter = (text: string, signal?: AbortSignal) => Promise<CanonicalSkcWorkflowSkill | null>;
34
+ /**
35
+ * Workflow-only view of the triager.
36
+ *
37
+ * A few lines over the same implementation rather than a second one: the eval
38
+ * harness in `scripts/eval-skill-routing.ts` and the routing tests want
39
+ * text-in/skill-out and have no UI question to ask, and a parallel router would
40
+ * be free to drift away from the thresholds this one enforces.
41
+ */
42
+ export declare function createSemanticSkillRouter(service: DecisionService): SkillRouter;
@@ -1,10 +1,45 @@
1
- import { type CanonicalSkcWorkflowSkill } from "../skill-state/active-state";
2
- import type { DecisionService } from "./index";
1
+ /** Shared across every routing question so one answer shape covers them all. */
2
+ export declare const NONE_CHOICE = "none";
3
+ export declare const ROUTING_INSTRUCTIONS = "Which workflow should handle this user request? Choose none unless the request clearly calls for one of the workflows.";
3
4
  /** Exported so tests can assert the contract the model is actually given. */
4
5
  export declare function buildRoutingCriteria(): Record<string, string>;
5
- export type SkillRouter = (text: string) => Promise<CanonicalSkcWorkflowSkill | null>;
6
6
  /**
7
- * Build the semantic router. Returns null-resolving function when the service is
8
- * disabled so the caller keeps its existing behaviour with no branching.
7
+ * Prompts below this length never carry enough signal to justify a model round-trip.
8
+ * Measured with {@link promptSignalLength}, not `String.length`: the floor was fitted
9
+ * to English and Korean, and in code units a complete Chinese request is shorter than
10
+ * "ok thanks".
9
11
  */
10
- export declare function createSemanticSkillRouter(service: DecisionService): SkillRouter;
12
+ export declare const MIN_PROMPT_CHARS = 12;
13
+ /** Only the opening of a prompt decides its workflow; the rest is payload. */
14
+ export declare const MAX_PROMPT_CHARS = 4000;
15
+ /**
16
+ * Prompt length in Latin-letter equivalents.
17
+ *
18
+ * One Han character, kana or Hangul syllable carries what two or three Latin letters
19
+ * do, so each counts double. "先做架构设计" (6 code units) is a whole request; "谢谢" and
20
+ * "고마워요" still fall under the floor, which is the point of having one.
21
+ */
22
+ export declare function promptSignalLength(text: string): number;
23
+ /**
24
+ * Minimum calibrated confidence required to activate a workflow.
25
+ *
26
+ * Activation is a strong move: it switches on the mutation guard, the Stop hook and the
27
+ * ask tool. Getting it wrong is worse than missing, because the user did not ask for any
28
+ * of that and has no obvious way to see why it appeared.
29
+ *
30
+ * Measured over ten routing prompts against the hosted model: every answer it reported
31
+ * at 1.00 was correct, and its single wrong answer reported 0.71. The lowest *correct*
32
+ * confidence was 0.67 — and that case was "none", so gating it out costs nothing. A
33
+ * floor here therefore removes the observed error without removing a real activation.
34
+ *
35
+ * One prompt sits close to this line. "추측하지 말고 모르는 건 다 물어봐" resolves to
36
+ * deep-interview in 8/8 samples but at 0.76-0.83, so the floor has roughly 0.01 of
37
+ * headroom on it. Raising the floor would drop a correct activation; lowering it would
38
+ * re-admit the 0.71 error. Treat 0.75 as fitted to a small sample and re-derive it from
39
+ * real usage rather than nudging it on a hunch.
40
+ *
41
+ * Only applied when the backend reports `calibrated: true`. An ordinary LLM answering
42
+ * through a forced enum has no meaningful confidence to compare against, so gating on a
43
+ * number it did not really produce would just be superstition.
44
+ */
45
+ export declare const MIN_CALIBRATED_CONFIDENCE = 0.75;
@@ -1,3 +1,6 @@
1
+ import type { ModelSelectorValue } from "../config/model-selector-value";
2
+ import type { Settings } from "../config/settings";
3
+ import { type TaskModelSpecialty, type TaskRoutingCandidate, type TaskRoutingSource } from "../config/task-model-specialties";
1
4
  import type { DecisionService } from "./index";
2
5
  /** Ordered cheapest to most capable. The order *is* the policy's direction. */
3
6
  export declare const TASK_TIERS: readonly ["fast", "balanced", "deep"];
@@ -22,26 +25,57 @@ export interface TaskRoutingPolicy {
22
25
  /**
23
26
  * Model for frontend planning, when the assignment reads as frontend work.
24
27
  *
25
- * This is the **domain** axis, not a rung on the ladder: a design-strong model
26
- * is not "better" than a code-strong one, it is a different specialty. Only
27
- * planning roles ever take it, and only laterally — the implementation roles
28
- * stay on the difficulty ladder.
28
+ * Superseded by `specialtyModels.frontendDesign`; kept as the fallback source
29
+ * so an existing configuration keeps working untouched until it is migrated.
29
30
  */
30
31
  frontendModel?: string;
31
32
  /**
32
- * Bar for the lateral swap above. Directional bars do not apply here because
33
- * neither direction is "spending more": being wrong either way costs quality,
34
- * symmetrically, so one bar is the whole story.
33
+ * Per-specialty models — the work-kind axis.
34
+ *
35
+ * Absent entries inherit the role's own chain, which is why an unset specialty
36
+ * is not an error and does not suppress the difficulty ladder.
37
+ */
38
+ specialtyModels?: Partial<Record<TaskModelSpecialty, ModelSelectorValue>>;
39
+ /**
40
+ * Bar for a lateral swap on a **calibrated** backend. Directional bars do not
41
+ * apply here because neither direction is "spending more": being wrong either
42
+ * way costs quality, symmetrically, so one bar is the whole story.
35
43
  */
36
44
  minDomainConfidence: number;
45
+ /**
46
+ * Bar for a lateral swap on an **uncalibrated** backend.
47
+ *
48
+ * The ordinary logged-in model cannot report a probability, and asking it for
49
+ * one measurably degrades the answer, so its choice carries no `confidence`.
50
+ * What it *can* report is an ordinal strength. Requiring a high ordinal is not
51
+ * the same guarantee as a calibrated threshold, and the result is recorded as
52
+ * uncalibrated — but refusing to route at all would make a user's explicit
53
+ * specialty selection silently inert on the default backend.
54
+ */
55
+ minSpecialtyOrdinal: number;
37
56
  }
38
57
  export declare const DEFAULT_TASK_ROUTING_POLICY: Omit<TaskRoutingPolicy, "tiers">;
58
+ /**
59
+ * The settings surface this module reads.
60
+ *
61
+ * Narrowed to `get` so the router cannot quietly start writing settings, and so
62
+ * a caller only has to supply a reader rather than a whole `Settings` instance.
63
+ */
64
+ export type TaskRoutingSettingsReader = Pick<Settings, "get">;
65
+ /**
66
+ * Build the routing policy from settings, or null when routing must not run.
67
+ *
68
+ * Null is returned for two distinct reasons that both mean "leave the configured
69
+ * model alone": the feature is off, or it is on but nothing is configured to
70
+ * route *to*. A lone tier is not an axis — there is nowhere to move from it —
71
+ * so two tiers is the floor unless a specialty or the legacy frontend model
72
+ * supplies a lateral target instead.
73
+ */
74
+ export declare function buildTaskRoutingPolicyFromSettings(settings: TaskRoutingSettingsReader): TaskRoutingPolicy | null;
39
75
  /**
40
76
  * Roles whose output is a plan or a design review.
41
77
  *
42
- * These are the only roles the domain swap applies to — the user's intent is
43
- * "a design-strong model *plans* the frontend; implementation stays where it
44
- * is". Executor keeps the difficulty ladder regardless of domain.
78
+ * Derived from the specialty compatibility map so the two never drift apart.
45
79
  */
46
80
  export declare const PLANNING_ROLES: ReadonlySet<string>;
47
81
  export interface TaskRoutingRequest {
@@ -50,14 +84,65 @@ export interface TaskRoutingRequest {
50
84
  assignment: string;
51
85
  /** Whatever the role is configured to use today, used as the direction baseline. */
52
86
  currentModel: string | undefined;
87
+ /**
88
+ * The role's fully resolved chain, in order. The composed candidate list ends
89
+ * with this, so a specialty or tier that cannot be authenticated falls through
90
+ * to the model the role would have used anyway.
91
+ */
92
+ baselineChain?: readonly string[];
53
93
  signal?: AbortSignal;
54
94
  }
55
95
  export interface TaskRoutingResult {
96
+ /** Head of the composed chain — what the spawn runs on if it authenticates. */
56
97
  model: string;
57
- /** Null when the move was a domain swap — that axis has no ladder. */
98
+ /** Null when the move was a specialty swap — that axis has no ladder. */
58
99
  tier: TaskTier | null;
59
100
  reason: string;
101
+ /** Ordered, provenance-tagged chain for the existing auth-aware resolver. */
102
+ candidates: TaskRoutingCandidate[];
103
+ /** What the classifier asked for. The *effective* source is only known after resolution. */
104
+ requestedSource: TaskRoutingSource;
105
+ requestedSpecialty?: TaskModelSpecialty;
106
+ requestedTier?: TaskTier;
107
+ /**
108
+ * True when the caller named the specialty on the spawn itself. No classifier
109
+ * ran, so `calibrated`/`confidence`/`ordinalStrength` describe nothing here.
110
+ */
111
+ declared: boolean;
112
+ /** False means `ordinalStrength` ranks, and no probability was available. */
113
+ calibrated: boolean;
114
+ confidence?: number;
115
+ ordinalStrength?: number;
116
+ }
117
+ export interface DeclaredSpecialtyRequest {
118
+ agentName: string;
119
+ specialty: TaskModelSpecialty;
120
+ /** Whatever the role is configured to use today. */
121
+ currentModel: string | undefined;
122
+ /** The role's fully resolved chain; always the tail so a dead specialty model falls through. */
123
+ baselineChain?: readonly string[];
60
124
  }
125
+ /**
126
+ * Route a spawn whose caller *declared* the kind of work.
127
+ *
128
+ * This is the deterministic half of the specialty axis. Nothing here asks a
129
+ * classifier, reads `task.modelRouting.enabled`, or applies a confidence bar:
130
+ * the user put a model on this specialty in `/model`, the caller says this is
131
+ * that work, and the only remaining reason not to run on it is that it fails —
132
+ * which the child session's fallback chain handles at the transport boundary
133
+ * (429, 5xx, auth, quota) by advancing to the role's baseline behind it.
134
+ *
135
+ * Role eligibility is deliberately not checked. The menu groups specialties
136
+ * under the roles that usually do that work, but the setting is one flat map:
137
+ * a frontend model the user chose for design is the same frontend model they
138
+ * expect when the *implementation* of that frontend is delegated. Refusing
139
+ * here would make "frontend uses a different model" false for exactly the
140
+ * spawns where it matters most.
141
+ *
142
+ * Returns null only when nothing is configured for the specialty, or when the
143
+ * configured model is already what the role would run on anyway.
144
+ */
145
+ export declare function resolveDeclaredSpecialtyRouting(settings: TaskRoutingSettingsReader, request: DeclaredSpecialtyRequest): TaskRoutingResult | null;
61
146
  /**
62
147
  * Decide the model for one subagent spawn, or null to leave the configured one alone.
63
148
  *
@@ -0,0 +1,21 @@
1
+ import type { PromptTriager } from "../decisions/prompt-triage";
2
+ import { type SkillActiveState } from "./skill-state";
3
+ export interface NativePromptRoutingInput {
4
+ cwd: string;
5
+ text: string;
6
+ sessionId?: string;
7
+ threadId?: string;
8
+ turnId?: string;
9
+ stateDir?: string;
10
+ /** `config.yml` files in precedence order, lowest first; resolved by the hook. */
11
+ configPaths: readonly string[];
12
+ agentDir?: string;
13
+ /** Injected in tests so routing is exercised without credentials or a network. */
14
+ triager?: PromptTriager;
15
+ }
16
+ export interface NativePromptRoutingResult {
17
+ skillState: SkillActiveState | null;
18
+ /** Directive for a UI skill the model picked because the pattern table missed. */
19
+ uiSkillContext: string | null;
20
+ }
21
+ export declare function routeNativePrompt(input: NativePromptRoutingInput): Promise<NativePromptRoutingResult>;
@@ -1,3 +1,4 @@
1
+ import type { PromptTriager } from "../decisions/prompt-triage";
1
2
  import { type EffectiveSkillConfigInput } from "./skill-state";
2
3
  export type SkcNativeHookEventName = "UserPromptSubmit" | "Stop";
3
4
  export interface SkcNativeHookDispatchResult {
@@ -11,6 +12,8 @@ interface SkcNativeHookDispatchOptions {
11
12
  stateDir?: string;
12
13
  effectiveSkillConfig?: EffectiveSkillConfigInput;
13
14
  configPaths?: string[];
15
+ /** Injected in tests so the model stage runs without credentials or a network. */
16
+ triager?: PromptTriager;
14
17
  }
15
18
  export declare function clearSkcNativeSkillHookCachesForTesting(): void;
16
19
  export declare function getSkcNativeSkillHookCacheStatsForTesting(): {
@@ -4,6 +4,15 @@ export interface SkillKeywordDefinition {
4
4
  skill: SkcWorkflowSkill;
5
5
  priority: number;
6
6
  guidance: string;
7
+ /**
8
+ * Pre-compiled matcher, replacing the literal-substring compilation of
9
+ * `keyword`. Only learned entries set it: they are two stems with a bounded
10
+ * gap, which a literal string cannot express. `keyword` stays human-readable
11
+ * for logs and for the guidance line.
12
+ */
13
+ pattern?: RegExp;
14
+ /** Mined from routing answers rather than written by hand. */
15
+ learned?: boolean;
7
16
  }
8
17
  export declare const SKC_WORKFLOW_SKILLS: readonly ["deep-interview", "ralplan", "ultragoal", "team"];
9
18
  export type SkcWorkflowSkill = CanonicalSkcWorkflowSkill;
@@ -2,7 +2,7 @@ import type { SkillDiscoverySettings } from "../config/skill-settings-defaults";
2
2
  import { type SkillActiveState } from "../skill-state/active-state";
3
3
  import { initialPhaseForSkill } from "../skill-state/initial-phase";
4
4
  export { initialPhaseForSkill };
5
- import { type SkcWorkflowSkill } from "./skill-keywords";
5
+ import { type SkcWorkflowSkill, type SkillKeywordDefinition } from "./skill-keywords";
6
6
  export declare const SKC_STATE_DIR = ".skc";
7
7
  export declare const SKILL_ACTIVE_STATE_FILE = "skill-active-state.json";
8
8
  export interface EffectiveSkillConfigInput {
@@ -15,6 +15,8 @@ export interface SkillKeywordMatch {
15
15
  keyword: string;
16
16
  skill: SkcWorkflowSkill;
17
17
  priority: number;
18
+ /** Matched a pattern mined from routing answers, not a hand-written keyword. */
19
+ learned?: boolean;
18
20
  }
19
21
  export type { SkillActiveEntry, SkillActiveState } from "../skill-state/active-state";
20
22
  export interface ModeState {
@@ -38,6 +40,12 @@ export interface RecordSkillActivationInput {
38
40
  turnId?: string;
39
41
  nowIso?: string;
40
42
  stateDir?: string;
43
+ /**
44
+ * Patterns promoted by the keyword learner, appended after the hand-written
45
+ * table. The caller loads them because it knows whether learning is on; see
46
+ * `detectSkillKeywords`.
47
+ */
48
+ learned?: readonly SkillKeywordDefinition[];
41
49
  /**
42
50
  * Semantic fallback, consulted only when no keyword matched. Supplying it turns the
43
51
  * literal keyword table into a two-stage router; omitting it keeps the historical
@@ -60,8 +68,17 @@ export interface UserPromptSubmitStateInput {
60
68
  prompt?: string;
61
69
  sessionFile?: string;
62
70
  }
63
- export declare function detectSkillKeywords(text: string): SkillKeywordMatch[];
64
- export declare function detectPrimarySkillKeyword(text: string): SkillKeywordMatch | null;
71
+ /**
72
+ * Match a prompt against the keyword table.
73
+ *
74
+ * `learned` carries patterns mined from semantic routing answers (see
75
+ * `decisions/keyword-learning.ts`). It is a parameter rather than a module-level
76
+ * load because this file is imported by the hook process, where a synchronous
77
+ * disk read on every prompt is not acceptable and the caller already knows
78
+ * whether learning is enabled.
79
+ */
80
+ export declare function detectSkillKeywords(text: string, learned?: readonly SkillKeywordDefinition[]): SkillKeywordMatch[];
81
+ export declare function detectPrimarySkillKeyword(text: string, learned?: readonly SkillKeywordDefinition[]): SkillKeywordMatch | null;
65
82
  export declare function resolveSkcStateDir(cwd: string, stateDir?: string): string;
66
83
  export interface StateRecoveryDiagnostic {
67
84
  kind: "skill-active-state" | "mode-state";