@sayknow-cli/coding-agent 0.5.22 → 0.5.24

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,8 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [0.5.24] - 2026-09-21
6
+
5
7
  ## [0.5.22] - 2026-09-19
6
8
 
7
9
  ### Added
@@ -2418,9 +2418,45 @@ export declare const SETTINGS_SCHEMA: {
2418
2418
  readonly ui: {
2419
2419
  readonly tab: "context";
2420
2420
  readonly label: "Typed decisions";
2421
- readonly description: "Let a cheap model answer typed questions the deterministic rules cannot. Currently routes workflow skills when the keyword table finds nothing — which is every non-English phrasing. Costs one small model call on those prompts; every failure falls back to today's behaviour.";
2421
+ readonly description: "Add a model-backed second stage to workflow routing. The keyword table already runs on every turn and costs nothing; this handles the phrasings it cannot express, which is most wording that is not a literal match. Costs one small model call, and only on turns the keyword table did not already answer. Any failure falls back to keyword-only behaviour, but a successful answer can also select a different workflow than the deep-interview ambiguity detector would have.";
2422
2422
  };
2423
2423
  };
2424
+ /**
2425
+ * Per-spawn model routing for subagents.
2426
+ *
2427
+ * A role's configured model is a standing guess about the average task that
2428
+ * role gets; this moves it when a particular assignment is clearly harder or
2429
+ * clearly more mechanical. Subagents only — routing the main loop's model
2430
+ * mid-session invalidates the prompt cache, which on a long context costs
2431
+ * more than the cheaper tier saves.
2432
+ *
2433
+ * Needs `decisions.enabled` and at least two tiers configured. Without both
2434
+ * it never fires and the configured role models are used unchanged.
2435
+ */
2436
+ readonly "task.modelRouting.enabled": {
2437
+ readonly type: "boolean";
2438
+ readonly default: false;
2439
+ readonly ui: {
2440
+ readonly tab: "tasks";
2441
+ readonly label: "Route subagent models per task";
2442
+ readonly description: "Ask a cheap model how demanding each subagent assignment is, and move that spawn to a cheaper or stronger model. Needs typed decisions on and the tier models below set. Moving to a cheaper model requires more confidence than moving to a stronger one, because being wrong about it costs a retry.";
2443
+ };
2444
+ };
2445
+ /** Cheapest tier. Mechanical, local, single-file work. */
2446
+ readonly "task.modelRouting.fastModel": {
2447
+ readonly type: "string";
2448
+ readonly default: "";
2449
+ };
2450
+ /** Middle tier. Ordinary engineering against an existing pattern. */
2451
+ readonly "task.modelRouting.balancedModel": {
2452
+ readonly type: "string";
2453
+ readonly default: "";
2454
+ };
2455
+ /** Most capable tier. Unclear cause, cross-cutting design, hard to undo. */
2456
+ readonly "task.modelRouting.deepModel": {
2457
+ readonly type: "string";
2458
+ readonly default: "";
2459
+ };
2424
2460
  readonly "ttsr.enabled": {
2425
2461
  readonly type: "boolean";
2426
2462
  readonly default: true;
@@ -1,5 +1,7 @@
1
1
  import { type CanonicalSkcWorkflowSkill } from "../skill-state/active-state";
2
2
  import type { DecisionService } from "./index";
3
+ /** Exported so tests can assert the contract the model is actually given. */
4
+ export declare function buildRoutingCriteria(): Record<string, string>;
3
5
  export type SkillRouter = (text: string) => Promise<CanonicalSkcWorkflowSkill | null>;
4
6
  /**
5
7
  * Build the semantic router. Returns null-resolving function when the service is
@@ -0,0 +1,44 @@
1
+ import type { DecisionService } from "./index";
2
+ /** Ordered cheapest to most capable. The order *is* the policy's direction. */
3
+ export declare const TASK_TIERS: readonly ["fast", "balanced", "deep"];
4
+ export type TaskTier = (typeof TASK_TIERS)[number];
5
+ export interface TaskTierModels {
6
+ fast?: string;
7
+ balanced?: string;
8
+ deep?: string;
9
+ }
10
+ export interface TaskRoutingPolicy {
11
+ tiers: TaskTierModels;
12
+ /**
13
+ * Bar to move to a more capable model. Being wrong costs money.
14
+ */
15
+ minUpgradeConfidence: number;
16
+ /**
17
+ * Bar to move to a cheaper model. Being wrong means real work handled by a
18
+ * model too small for it, which is discovered late and costs a retry — so
19
+ * this bar sits higher than the upgrade bar on purpose.
20
+ */
21
+ minDowngradeConfidence: number;
22
+ }
23
+ export declare const DEFAULT_TASK_ROUTING_POLICY: Omit<TaskRoutingPolicy, "tiers">;
24
+ export interface TaskRoutingRequest {
25
+ agentName: string;
26
+ /** The assignment text the subagent will act on. */
27
+ assignment: string;
28
+ /** Whatever the role is configured to use today, used as the direction baseline. */
29
+ currentModel: string | undefined;
30
+ signal?: AbortSignal;
31
+ }
32
+ export interface TaskRoutingResult {
33
+ model: string;
34
+ tier: TaskTier;
35
+ reason: string;
36
+ }
37
+ /**
38
+ * Decide the model for one subagent spawn, or null to leave the configured one alone.
39
+ *
40
+ * Every failure path returns null: no tiers configured, decisions disabled, no
41
+ * backend, a timeout, an answer outside the enum. A subagent that runs on its
42
+ * configured model is the status quo, and the status quo is always acceptable.
43
+ */
44
+ export declare function routeTaskModel(service: DecisionService, policy: TaskRoutingPolicy, request: TaskRoutingRequest): Promise<TaskRoutingResult | null>;
@@ -81,6 +81,15 @@ export declare class SelectorController {
81
81
  }): void;
82
82
  showCommandPalette(commands: SlashCommand[], actions: CommandPaletteAction[], executeSlashCommand: (name: string) => Promise<void>): void;
83
83
  showProviderOnboarding(): void;
84
+ /**
85
+ * Take a TypeSafe key and verify it before storing.
86
+ *
87
+ * Verification is not optional here. The decision service fails open by design, so an
88
+ * unverified bad key produces no error anywhere: decisions silently keep coming from
89
+ * the user's own model while the UI claims TypeSafe is on. Better to keep the prompt
90
+ * open and say the key was rejected.
91
+ */
92
+ showTypeSafeKeyPrompt(): void;
84
93
  showCustomModelPresetWizard(snapshot: ModelProfileConfig): void;
85
94
  showCustomProviderWizard(): void;
86
95
  showEffortSelector(): void;
@@ -264,6 +264,7 @@ export declare class InteractiveMode implements InteractiveModeContext {
264
264
  temporaryOnly?: boolean;
265
265
  }): void;
266
266
  showEffortSelector(): void;
267
+ showTypeSafeKeyPrompt(): void;
267
268
  showProviderOnboarding(): void;
268
269
  showPluginSelector(mode?: "install" | "uninstall"): void;
269
270
  showUserMessageSelector(): void;
@@ -280,6 +280,8 @@ export interface InteractiveModeContext {
280
280
  }): void;
281
281
  showEffortSelector(): void;
282
282
  showProviderOnboarding(): void;
283
+ /** Open the TypeSafe key prompt (typed decisions; not a chat model). */
284
+ showTypeSafeKeyPrompt(): void;
283
285
  showPluginSelector(mode?: "install" | "uninstall"): void;
284
286
  showUserMessageSelector(): void;
285
287
  showTreeSelector(): void;
@@ -2,6 +2,11 @@ export declare const MODEL_ONBOARDING_API_PROVIDER_COMMAND = "/provider add --co
2
2
  export declare const MODEL_ONBOARDING_PROVIDER_PRESET_COMMAND = "/provider add --preset <minimax|minimax-cn|glm>";
3
3
  export declare const MODEL_ONBOARDING_SETUP_COMMAND = "skc setup provider";
4
4
  export declare const MODEL_ONBOARDING_OAUTH_COMMAND = "/provider login [provider-id] or /login [provider-id]";
5
+ /**
6
+ * TypeSafe is not a chat model and never appears in the model list, so the only way a
7
+ * user learns it exists is from the surfaces where they go to add credentials.
8
+ */
9
+ export declare const MODEL_ONBOARDING_TYPESAFE_COMMAND = "/provider typesafe";
5
10
  export declare function formatModelOnboardingGuidance(): string;
6
11
  export declare function formatModelOnboardingInlineHint(): string;
7
12
  export declare function formatNoModelOnboardingError(): string;
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@sayknow-cli/coding-agent",
4
- "version": "0.5.22",
4
+ "version": "0.5.24",
5
5
  "description": "Sayknow-CLI CLI with read, bash, edit, write tools and session management",
6
6
  "homepage": "https://sayknow-cli.com",
7
7
  "author": "jaybeyond",
@@ -54,12 +54,12 @@
54
54
  "@agentclientprotocol/sdk": "1.3.0",
55
55
  "@babel/parser": "^7.29.3",
56
56
  "@mozilla/readability": "^0.6.0",
57
- "@sayknow-cli/stats": "0.5.22",
58
- "@sayknow-cli/agent-core": "0.5.22",
59
- "@sayknow-cli/ai": "0.5.22",
60
- "@sayknow-cli/natives": "0.5.22",
61
- "@sayknow-cli/tui": "0.5.22",
62
- "@sayknow-cli/utils": "0.5.22",
57
+ "@sayknow-cli/stats": "0.5.24",
58
+ "@sayknow-cli/agent-core": "0.5.24",
59
+ "@sayknow-cli/ai": "0.5.24",
60
+ "@sayknow-cli/natives": "0.5.24",
61
+ "@sayknow-cli/tui": "0.5.24",
62
+ "@sayknow-cli/utils": "0.5.24",
63
63
  "@puppeteer/browsers": "^2.13.0",
64
64
  "@types/turndown": "5.0.6",
65
65
  "@xterm/headless": "^6.0.0",
@@ -126,12 +126,12 @@ async function main(): Promise<void> {
126
126
 
127
127
  console.log("\n=== stage comparison ===");
128
128
  for (const [label, pick] of [
129
- ["keyword only (today)", (row: (typeof rows)[number]) => row.keyword],
130
- ["semantic only", (row: (typeof rows)[number]) => row.semantic],
131
- ["hybrid (shipped)", hybrid],
129
+ ["keyword only (Codex hook)", (row: (typeof rows)[number]) => row.keyword],
130
+ ["semantic only (SHIPPED)", (row: (typeof rows)[number]) => row.semantic],
131
+ ["keyword+semantic (upper bound)", hybrid],
132
132
  ] as const) {
133
133
  console.log(
134
- `${label.padEnd(22)} all ${pct(...score(pick, () => true))} ko ${pct(...score(pick, ko))} en ${pct(...score(pick, en))} clean-negatives ${pct(...score(pick, negatives))}`,
134
+ `${label.padEnd(32)} all ${pct(...score(pick, () => true))} ko ${pct(...score(pick, ko))} en ${pct(...score(pick, en))} clean-negatives ${pct(...score(pick, negatives))}`,
135
135
  );
136
136
  }
137
137
 
@@ -140,12 +140,13 @@ async function main(): Promise<void> {
140
140
  `\nlatency p50 ${latencies[Math.floor(latencies.length / 2)]}ms p95 ${latencies[Math.max(0, Math.ceil(latencies.length * 0.95) - 1)]}ms max ${latencies.at(-1)}ms`,
141
141
  );
142
142
 
143
- const misses = rows.filter(row => hybrid(row) !== row.case.expect);
143
+ // Report against what actually ships in this host, not against the upper bound.
144
+ const misses = rows.filter(row => row.semantic !== row.case.expect);
144
145
  if (misses.length > 0) {
145
- console.log("\n=== hybrid misses ===");
146
+ console.log("\n=== misses in shipped configuration (semantic only) ===");
146
147
  for (const row of misses)
147
148
  console.log(
148
- ` [${row.case.lang}] want=${row.case.expect ?? "none"} got=${hybrid(row) ?? "none"} :: ${row.case.prompt}`,
149
+ ` [${row.case.lang}] want=${row.case.expect ?? "none"} got=${row.semantic ?? "none"} :: ${row.case.prompt}`,
149
150
  );
150
151
  }
151
152
 
@@ -2007,10 +2007,39 @@ export const SETTINGS_SCHEMA = {
2007
2007
  tab: "context",
2008
2008
  label: "Typed decisions",
2009
2009
  description:
2010
- "Let a cheap model answer typed questions the deterministic rules cannot. Currently routes workflow skills when the keyword table finds nothing — which is every non-English phrasing. Costs one small model call on those prompts; every failure falls back to today's behaviour.",
2010
+ "Add a model-backed second stage to workflow routing. The keyword table already runs on every turn and costs nothing; this handles the phrasings it cannot express, which is most wording that is not a literal match. Costs one small model call, and only on turns the keyword table did not already answer. Any failure falls back to keyword-only behaviour, but a successful answer can also select a different workflow than the deep-interview ambiguity detector would have.",
2011
2011
  },
2012
2012
  },
2013
2013
 
2014
+ /**
2015
+ * Per-spawn model routing for subagents.
2016
+ *
2017
+ * A role's configured model is a standing guess about the average task that
2018
+ * role gets; this moves it when a particular assignment is clearly harder or
2019
+ * clearly more mechanical. Subagents only — routing the main loop's model
2020
+ * mid-session invalidates the prompt cache, which on a long context costs
2021
+ * more than the cheaper tier saves.
2022
+ *
2023
+ * Needs `decisions.enabled` and at least two tiers configured. Without both
2024
+ * it never fires and the configured role models are used unchanged.
2025
+ */
2026
+ "task.modelRouting.enabled": {
2027
+ type: "boolean",
2028
+ default: false,
2029
+ ui: {
2030
+ tab: "tasks",
2031
+ label: "Route subagent models per task",
2032
+ description:
2033
+ "Ask a cheap model how demanding each subagent assignment is, and move that spawn to a cheaper or stronger model. Needs typed decisions on and the tier models below set. Moving to a cheaper model requires more confidence than moving to a stronger one, because being wrong about it costs a retry.",
2034
+ },
2035
+ },
2036
+ /** Cheapest tier. Mechanical, local, single-file work. */
2037
+ "task.modelRouting.fastModel": { type: "string", default: "" },
2038
+ /** Middle tier. Ordinary engineering against an existing pattern. */
2039
+ "task.modelRouting.balancedModel": { type: "string", default: "" },
2040
+ /** Most capable tier. Unclear cause, cross-cutting design, hard to undo. */
2041
+ "task.modelRouting.deepModel": { type: "string", default: "" },
2042
+
2014
2043
  // TTSR
2015
2044
  "ttsr.enabled": {
2016
2045
  type: "boolean",
@@ -26,8 +26,13 @@ const NONE = "none";
26
26
  * are the whole contract with the model — the enum ids alone carry almost no signal.
27
27
  */
28
28
  const WORKFLOW_MEANINGS: Record<CanonicalSkcWorkflowSkill, string> = {
29
+ // Scoped to the *behaviour* the user is asking for, not only to the state of the
30
+ // request. The earlier wording ("vague about what to build") described a property of
31
+ // the spec, so a direct instruction to ask rather than assume — the request is not
32
+ // vague, it is an order about how to proceed — landed on `none`. Measured: that one
33
+ // prompt was the sole miss in the 23-case set.
29
34
  "deep-interview":
30
- "The request is vague about what to build. The user wants to be interviewed and have requirements elicited before anything is designed or written.",
35
+ "The user wants requirements drawn out of them by questioning before anything is built. Includes explicit instructions to ask rather than assume.",
31
36
  ralplan:
32
37
  "The user wants a deliberate plan, design comparison, or approval before any code is touched. Architecture or sequencing risk is involved.",
33
38
  ultragoal:
@@ -38,7 +43,8 @@ const WORKFLOW_MEANINGS: Record<CanonicalSkcWorkflowSkill, string> = {
38
43
  const ROUTING_INSTRUCTIONS =
39
44
  "Which workflow should handle this user request? Choose none unless the request clearly calls for one of the workflows.";
40
45
 
41
- function buildCriteria(): Record<string, string> {
46
+ /** Exported so tests can assert the contract the model is actually given. */
47
+ export function buildRoutingCriteria(): Record<string, string> {
42
48
  const criteria: Record<string, string> = {};
43
49
  for (const skill of CANONICAL_SKC_WORKFLOW_SKILLS) criteria[skill] = WORKFLOW_MEANINGS[skill];
44
50
  criteria[NONE] =
@@ -51,6 +57,30 @@ const MIN_PROMPT_CHARS = 12;
51
57
  /** Only the opening of a prompt decides its workflow; the rest is payload. */
52
58
  const MAX_PROMPT_CHARS = 4_000;
53
59
 
60
+ /**
61
+ * Minimum calibrated confidence required to activate a workflow.
62
+ *
63
+ * Activation is a strong move: it switches on the mutation guard, the Stop hook and the
64
+ * ask tool. Getting it wrong is worse than missing, because the user did not ask for any
65
+ * of that and has no obvious way to see why it appeared.
66
+ *
67
+ * Measured over ten routing prompts against the hosted model: every answer it reported
68
+ * at 1.00 was correct, and its single wrong answer reported 0.71. The lowest *correct*
69
+ * confidence was 0.67 — and that case was "none", so gating it out costs nothing. A
70
+ * floor here therefore removes the observed error without removing a real activation.
71
+ *
72
+ * One prompt sits close to this line. "추측하지 말고 모르는 건 다 물어봐" resolves to
73
+ * deep-interview in 8/8 samples but at 0.76-0.83, so the floor has roughly 0.01 of
74
+ * headroom on it. Raising the floor would drop a correct activation; lowering it would
75
+ * re-admit the 0.71 error. Treat 0.75 as fitted to a small sample and re-derive it from
76
+ * real usage rather than nudging it on a hunch.
77
+ *
78
+ * Only applied when the backend reports `calibrated: true`. An ordinary LLM answering
79
+ * through a forced enum has no meaningful confidence to compare against, so gating on a
80
+ * number it did not really produce would just be superstition.
81
+ */
82
+ const MIN_CALIBRATED_CONFIDENCE = 0.75;
83
+
54
84
  export type SkillRouter = (text: string) => Promise<CanonicalSkcWorkflowSkill | null>;
55
85
 
56
86
  /**
@@ -58,7 +88,7 @@ export type SkillRouter = (text: string) => Promise<CanonicalSkcWorkflowSkill |
58
88
  * disabled so the caller keeps its existing behaviour with no branching.
59
89
  */
60
90
  export function createSemanticSkillRouter(service: DecisionService): SkillRouter {
61
- const criteria = buildCriteria();
91
+ const criteria = buildRoutingCriteria();
62
92
  return async (text: string): Promise<CanonicalSkcWorkflowSkill | null> => {
63
93
  if (!service.enabled) return null;
64
94
  const trimmed = text.trim();
@@ -73,9 +103,19 @@ export function createSemanticSkillRouter(service: DecisionService): SkillRouter
73
103
  if (!result || answer?.type !== "choice" || answer.choice === NONE) return null;
74
104
  const skill = CANONICAL_SKC_WORKFLOW_SKILLS.find(candidate => candidate === answer.choice);
75
105
  if (!skill) return null;
106
+ if (result.calibrated && (answer.confidence ?? 0) < MIN_CALIBRATED_CONFIDENCE) {
107
+ logger.debug("decisions/skill-routing: below confidence floor, leaving routing alone", {
108
+ skill,
109
+ confidence: answer.confidence,
110
+ floor: MIN_CALIBRATED_CONFIDENCE,
111
+ });
112
+ return null;
113
+ }
76
114
  logger.debug("decisions/skill-routing: semantic match", {
77
115
  skill,
78
116
  backend: result.backend,
117
+ confidence: answer.confidence,
118
+ calibrated: result.calibrated,
79
119
  durationMs: result.durationMs,
80
120
  });
81
121
  return skill;
@@ -0,0 +1,196 @@
1
+ /**
2
+ * Pick the model a subagent runs on from the work it was handed.
3
+ *
4
+ * A role's configured model is a standing guess about the *average* task that
5
+ * role gets. It cannot be right for every one: the same executor is handed both
6
+ * a one-line rename and a migration across twelve files. This asks about the
7
+ * actual assignment and moves the model when the answer is clear enough.
8
+ *
9
+ * Only subagents. The main loop's model is deliberately out of scope — changing
10
+ * it mid-session invalidates the prompt cache, and on a long context re-caching
11
+ * routinely costs more than the cheaper tier saves. A subagent starts with its
12
+ * own context, so there is nothing to invalidate.
13
+ */
14
+ import { logger } from "@sayknow-cli/utils";
15
+ import type { DecisionService } from "./index";
16
+ import type { Question } from "./types";
17
+
18
+ /** Ordered cheapest to most capable. The order *is* the policy's direction. */
19
+ export const TASK_TIERS = ["fast", "balanced", "deep"] as const;
20
+ export type TaskTier = (typeof TASK_TIERS)[number];
21
+
22
+ export interface TaskTierModels {
23
+ fast?: string;
24
+ balanced?: string;
25
+ deep?: string;
26
+ }
27
+
28
+ export interface TaskRoutingPolicy {
29
+ tiers: TaskTierModels;
30
+ /**
31
+ * Bar to move to a more capable model. Being wrong costs money.
32
+ */
33
+ minUpgradeConfidence: number;
34
+ /**
35
+ * Bar to move to a cheaper model. Being wrong means real work handled by a
36
+ * model too small for it, which is discovered late and costs a retry — so
37
+ * this bar sits higher than the upgrade bar on purpose.
38
+ */
39
+ minDowngradeConfidence: number;
40
+ }
41
+
42
+ export const DEFAULT_TASK_ROUTING_POLICY: Omit<TaskRoutingPolicy, "tiers"> = {
43
+ // Deliberately higher than the reference implementation's 0.3/0.6. That one
44
+ // assumes a frontier default with a cheap tier to fall to, so "up" is the
45
+ // rare move. Here the configured role models are already chosen per role, so
46
+ // overriding one needs a stronger signal in either direction.
47
+ minUpgradeConfidence: 0.5,
48
+ minDowngradeConfidence: 0.75,
49
+ };
50
+
51
+ /**
52
+ * The questions describe the *work*, never a model name.
53
+ *
54
+ * Naming models in the criteria would bind the classifier to one lineup and
55
+ * make every model swap a prompt change. It also invites the model to reason
56
+ * about price, which is not what it is good at.
57
+ */
58
+ function buildQuestions(): Record<string, Question> {
59
+ return {
60
+ tier: {
61
+ type: "choice",
62
+ instructions: "How demanding is this assignment?",
63
+ criteria: {
64
+ fast: "Mechanical and local. A rename, a typo, a one-file edit, running a command and reporting what it printed.",
65
+ balanced: "Ordinary engineering. Several files, an existing pattern to follow, normal debugging.",
66
+ deep: "Hard or high-stakes. Unclear cause, cross-cutting design, subtle correctness, or work that is hard to undo.",
67
+ },
68
+ },
69
+ risky: {
70
+ type: "noul",
71
+ instructions:
72
+ "Does this assignment touch production, money, credentials, published releases, or state that cannot be undone?",
73
+ },
74
+ };
75
+ }
76
+
77
+ export interface TaskRoutingRequest {
78
+ agentName: string;
79
+ /** The assignment text the subagent will act on. */
80
+ assignment: string;
81
+ /** Whatever the role is configured to use today, used as the direction baseline. */
82
+ currentModel: string | undefined;
83
+ signal?: AbortSignal;
84
+ }
85
+
86
+ export interface TaskRoutingResult {
87
+ model: string;
88
+ tier: TaskTier;
89
+ reason: string;
90
+ }
91
+
92
+ /** Where a concrete model id sits in the ladder, or null when it is not one of ours. */
93
+ function rankOf(model: string | undefined, tiers: TaskTierModels): number | null {
94
+ if (!model) return null;
95
+ const index = TASK_TIERS.findIndex(tier => tiers[tier] && matchesModel(tiers[tier] as string, model));
96
+ return index === -1 ? null : index;
97
+ }
98
+
99
+ /**
100
+ * Compare a configured tier model against the role's current selector.
101
+ *
102
+ * Selectors carry a thinking suffix (`provider/id:high`) that the tier table
103
+ * may or may not repeat, so compare the part before it.
104
+ */
105
+ function matchesModel(a: string, b: string): boolean {
106
+ const base = (value: string) => value.split(":")[0]?.trim().toLowerCase() ?? "";
107
+ return base(a) === base(b);
108
+ }
109
+
110
+ /**
111
+ * Is a move from `current` to `wanted` allowed at this confidence?
112
+ *
113
+ * A backend that cannot report a calibrated confidence may only move a request
114
+ * **up**. Spending less on an unmeasured hunch is the bad trade: the upgrade's
115
+ * worst case is an overpriced answer, the downgrade's is a wrong one.
116
+ */
117
+ function allowed(
118
+ wanted: number,
119
+ current: number | null,
120
+ confidence: number | undefined,
121
+ calibrated: boolean,
122
+ policy: TaskRoutingPolicy,
123
+ ): boolean {
124
+ if (current !== null && wanted === current) return false;
125
+ const isDowngrade = current !== null && wanted < current;
126
+ if (!calibrated || confidence === undefined) return !isDowngrade;
127
+ return confidence >= (isDowngrade ? policy.minDowngradeConfidence : policy.minUpgradeConfidence);
128
+ }
129
+
130
+ /**
131
+ * Decide the model for one subagent spawn, or null to leave the configured one alone.
132
+ *
133
+ * Every failure path returns null: no tiers configured, decisions disabled, no
134
+ * backend, a timeout, an answer outside the enum. A subagent that runs on its
135
+ * configured model is the status quo, and the status quo is always acceptable.
136
+ */
137
+ export async function routeTaskModel(
138
+ service: DecisionService,
139
+ policy: TaskRoutingPolicy,
140
+ request: TaskRoutingRequest,
141
+ ): Promise<TaskRoutingResult | null> {
142
+ const configured = TASK_TIERS.filter(tier => policy.tiers[tier]);
143
+ // One tier is not a ladder; with nothing to move between there is no decision
144
+ // worth paying a model call for.
145
+ if (configured.length < 2) return null;
146
+
147
+ const assignment = request.assignment.trim();
148
+ if (assignment.length < 24) return null;
149
+
150
+ const result = await service.decide({
151
+ state: `Agent: ${request.agentName}\n\nAssignment:\n${assignment.slice(0, 4_000)}`,
152
+ questions: buildQuestions(),
153
+ signal: request.signal,
154
+ });
155
+ if (!result) return null;
156
+
157
+ const answer = result.answers.tier;
158
+ if (answer?.type !== "choice") return null;
159
+ let tier = TASK_TIERS.find(candidate => candidate === answer.choice);
160
+ if (!tier) return null;
161
+
162
+ // Work that cannot be undone takes the most capable tier available and skips
163
+ // the confidence bars — this one is not a confidence question. It may only
164
+ // ever raise the tier, never lower it, or "this is risky" would end up
165
+ // *downgrading* an assignment already running deep.
166
+ const riskAnswer = result.answers.risky;
167
+ const forcedByRisk = riskAnswer?.type === "noul" && riskAnswer.noul > 0.7;
168
+ if (forcedByRisk) {
169
+ const deepest = configured[configured.length - 1] as TaskTier;
170
+ const currentRank = rankOf(request.currentModel, policy.tiers);
171
+ const wantedRank = Math.max(TASK_TIERS.indexOf(deepest), currentRank ?? 0);
172
+ tier = TASK_TIERS[wantedRank] as TaskTier;
173
+ }
174
+
175
+ const model = policy.tiers[tier];
176
+ if (!model || (request.currentModel && matchesModel(model, request.currentModel))) return null;
177
+
178
+ const currentRank = rankOf(request.currentModel, policy.tiers);
179
+ const wantedRank = TASK_TIERS.indexOf(tier);
180
+ if (!forcedByRisk && !allowed(wantedRank, currentRank, answer.confidence, result.calibrated, policy)) {
181
+ logger.debug("decisions/task-routing: below the bar, keeping the configured model", {
182
+ agent: request.agentName,
183
+ wanted: tier,
184
+ current: request.currentModel,
185
+ confidence: answer.confidence,
186
+ calibrated: result.calibrated,
187
+ });
188
+ return null;
189
+ }
190
+
191
+ const reason = forcedByRisk
192
+ ? `${tier}, forced by risk`
193
+ : `${tier} (confidence ${answer.confidence?.toFixed(2) ?? "n/d"})`;
194
+ logger.debug("decisions/task-routing: routed", { agent: request.agentName, model, reason });
195
+ return { model, tier, reason };
196
+ }
@@ -881,7 +881,7 @@ export class SelectorController {
881
881
  } else if (action === "import-credentials") {
882
882
  void this.#handleCredentialImport();
883
883
  } else if (action === "typesafe-key") {
884
- this.#showTypeSafeKeyPrompt();
884
+ this.showTypeSafeKeyPrompt();
885
885
  } else {
886
886
  this.ctx.showStatus(formatProviderOnboardingCommandGuide());
887
887
  }
@@ -903,7 +903,7 @@ export class SelectorController {
903
903
  * the user's own model while the UI claims TypeSafe is on. Better to keep the prompt
904
904
  * open and say the key was rejected.
905
905
  */
906
- #showTypeSafeKeyPrompt(): void {
906
+ showTypeSafeKeyPrompt(): void {
907
907
  this.showSelector(done => {
908
908
  let prompt: TypeSafeKeyPromptComponent | undefined;
909
909
  prompt = new TypeSafeKeyPromptComponent(
@@ -2133,6 +2133,10 @@ export class InteractiveMode implements InteractiveModeContext {
2133
2133
  this.#selectorController.showEffortSelector();
2134
2134
  }
2135
2135
 
2136
+ showTypeSafeKeyPrompt(): void {
2137
+ this.#selectorController.showTypeSafeKeyPrompt();
2138
+ }
2139
+
2136
2140
  showProviderOnboarding(): void {
2137
2141
  this.#selectorController.showProviderOnboarding();
2138
2142
  }
@@ -321,6 +321,8 @@ export interface InteractiveModeContext {
321
321
  showModelSelector(options?: { temporaryOnly?: boolean }): void;
322
322
  showEffortSelector(): void;
323
323
  showProviderOnboarding(): void;
324
+ /** Open the TypeSafe key prompt (typed decisions; not a chat model). */
325
+ showTypeSafeKeyPrompt(): void;
324
326
  showPluginSelector(mode?: "install" | "uninstall"): void;
325
327
  showUserMessageSelector(): void;
326
328
  showTreeSelector(): void;
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  name: architect
3
3
  description: Read-only architecture and code-review agent with severity-rated findings and status verdicts
4
- tools: read, search, find, lsp, ast_grep, web_search, bash, report_finding, irc
4
+ tools: read, search, find, lsp, ast_grep, web_search, bash, report_finding, skill, irc
5
5
  thinking-level: high
6
6
  blocking: true
7
7
  forkContext: allowed
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  name: critic
3
3
  description: Read-only plan critic that approves only actionable, verifiable execution plans
4
- tools: read, search, find, lsp, ast_grep, web_search, bash, irc
4
+ tools: read, search, find, lsp, ast_grep, web_search, bash, skill, irc
5
5
  thinking-level: high
6
6
  bashAllowedPrefixes:
7
7
  - skc ralplan --write
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  name: planner
3
3
  description: Read-only planning agent for sequencing, acceptance criteria, risks, and handoff shape
4
- tools: read, search, find, lsp, ast_grep, web_search, bash, irc
4
+ tools: read, search, find, lsp, ast_grep, web_search, bash, skill, irc
5
5
  thinking-level: medium
6
6
  bashAllowedPrefixes:
7
7
  - skc ralplan --write
@@ -267,6 +267,7 @@ import {
267
267
  detectPrimarySkillKeyword,
268
268
  ensureWorkflowSkillActivationState,
269
269
  } from "../hooks/skill-state";
270
+ import { buildUiSkillActivationContext } from "../hooks/ui-skill-keywords";
270
271
  import { initializeLocalRoot, type LocalProtocolOptions, resolveLocalUrlToPath } from "../internal-urls";
271
272
  import { shutdownAll as shutdownAllLspClients } from "../lsp/client";
272
273
  import { resolveMemoryBackend } from "../memory-backend";
@@ -313,6 +314,7 @@ import {
313
314
  } from "../skc-runtime/session-state-sidecar";
314
315
  import { requestSkcWorkerIntegrationAttempt } from "../skc-runtime/team-runtime";
315
316
  import {
317
+ type CanonicalSkcWorkflowSkill,
316
318
  isCanonicalSkcWorkflowSkill,
317
319
  readVisibleSkillActiveState,
318
320
  syncSkillActiveState,
@@ -7558,18 +7560,41 @@ export class AgentSession {
7558
7560
  * @throws Error if no model selected or no API key available (when not streaming)
7559
7561
  */
7560
7562
  /**
7561
- * Second-stage workflow routing for prompts the keyword table cannot see.
7563
+ * Semantic workflow routing for prompts the keyword table cannot express.
7562
7564
  *
7563
7565
  * Runs in this process, not the hook process: the hook only receives paths and
7564
7566
  * config, so it has no model registry and no credentials to call anything with.
7565
7567
  *
7566
- * Deliberately best-effort — a disabled setting, a missing credential, a timeout or
7567
- * a nonsense answer all resolve to "no activation", which is precisely the
7568
- * behaviour before this stage existed.
7568
+ * **The keyword table is not consulted here, and that is deliberate.** An earlier
7569
+ * version returned early on a keyword hit, on the assumption that the deterministic
7570
+ * stage had already activated the workflow. That assumption holds only under the
7571
+ * Codex host, where `skc codex-native-hook` runs on `UserPromptSubmit`. This session
7572
+ * never fires that hook, so the early return meant a prompt containing an enumerated
7573
+ * keyword activated *nothing at all* — strictly worse than before the keywords
7574
+ * existed, because the semantic stage had been handling those phrasings.
7575
+ *
7576
+ * Keywords remain advisory in this host, as they always were: the routing rules in
7577
+ * the system prompt describe them to the model. Only this stage activates, and only
7578
+ * when it is confident enough to be worth the mutation guard and Stop hook that
7579
+ * activation switches on.
7580
+ *
7581
+ * Deliberately best-effort — a disabled setting, a missing credential, a timeout, a
7582
+ * nonsense answer or low confidence all resolve to "no activation", which is
7583
+ * precisely the behaviour before this stage existed.
7569
7584
  */
7570
7585
  async #routeWorkflowSemantically(text: string): Promise<void> {
7586
+ // Stage one: the keyword table. Free, deterministic, and measured at zero false
7587
+ // positives, so it is not gated behind the opt-in setting — gating it was why an
7588
+ // enumerated phrase activated nothing in this host while the Codex hook activated
7589
+ // it fine. Activating here makes the two hosts agree.
7590
+ const keyword = detectPrimarySkillKeyword(text);
7591
+ if (keyword) {
7592
+ await this.#activateWorkflowSkill(keyword.skill);
7593
+ return;
7594
+ }
7595
+
7596
+ // Stage two costs a model call, so it stays opt-in.
7571
7597
  if (!this.settings.get("decisions.enabled")) return;
7572
- if (detectPrimarySkillKeyword(text)) return; // deterministic stage already decided
7573
7598
  try {
7574
7599
  this.#semanticSkillRouter ??= createSemanticSkillRouter(
7575
7600
  createDecisionService({
@@ -7581,6 +7606,20 @@ export class AgentSession {
7581
7606
  );
7582
7607
  const skill = await this.#semanticSkillRouter(text);
7583
7608
  if (!skill) return;
7609
+ await this.#activateWorkflowSkill(skill);
7610
+ } catch (error) {
7611
+ logger.debug("agent-session: semantic workflow routing failed", { error: String(error) });
7612
+ }
7613
+ }
7614
+
7615
+ /**
7616
+ * Seed workflow state and attach the ask tool.
7617
+ *
7618
+ * Both routing stages funnel through here, and both are best-effort: a failure to
7619
+ * write state must never take down the user's turn, so it is logged and swallowed.
7620
+ */
7621
+ async #activateWorkflowSkill(skill: CanonicalSkcWorkflowSkill): Promise<void> {
7622
+ try {
7584
7623
  await ensureWorkflowSkillActivationState({
7585
7624
  cwd: this.sessionManager.getCwd(),
7586
7625
  skill,
@@ -7588,7 +7627,7 @@ export class AgentSession {
7588
7627
  });
7589
7628
  this.#attachAskTool();
7590
7629
  } catch (error) {
7591
- logger.debug("agent-session: semantic workflow routing failed", { error: String(error) });
7630
+ logger.debug("agent-session: workflow activation failed", { skill, error: String(error) });
7592
7631
  }
7593
7632
  }
7594
7633
 
@@ -7650,10 +7689,18 @@ export class AgentSession {
7650
7689
  const deepInterviewUserIntentEpoch =
7651
7690
  claimsGenuineUserIntent && !this.isStreaming ? this.#claimDeepInterviewUserIntent() : undefined;
7652
7691
 
7653
- // The keyword table in `hooks/skill-keywords.ts` is thirteen literal strings, so a
7692
+ // The keyword table in `hooks/skill-keywords.ts` is a list of literal strings, so a
7654
7693
  // Korean phrasing of "plan this before you touch code" activates nothing. Ask a
7655
- // cheap model only when the deterministic stage found nothing, and only for real
7656
- // user turns. Any failure leaves routing to the system prompt, exactly as before.
7694
+ // cheap model instead, and only for real user turns. Any failure leaves routing to
7695
+ // the system prompt, exactly as before.
7696
+ //
7697
+ // Ordering matters: this runs *after* the deep-interview intent claim above but
7698
+ // before streaming, so a workflow it activates is in place before the ambiguity
7699
+ // detector in `skc-runtime/deep-interview-ambiguity.ts` would otherwise seed one.
7700
+ // Verified end to end: "설계가 위험해 보여 … 승인받을 문서부터 만들자" seeds
7701
+ // deep-interview with the setting off and ralplan with it on. Both are plausible
7702
+ // readings and ralplan is the better one here, but the point is that enabling this
7703
+ // can *change* an activation rather than only add one where there was none.
7657
7704
  if (claimsGenuineUserIntent && !this.isStreaming) await this.#routeWorkflowSemantically(expandedText);
7658
7705
 
7659
7706
  // If streaming, queue via steer() or followUp() based on option
@@ -7689,6 +7736,7 @@ export class AgentSession {
7689
7736
  const hasPendingUserDirective = this.#toolChoiceQueue.inspect().includes("user-force");
7690
7737
  const eagerTodoPrelude =
7691
7738
  !options?.synthetic && !hasPendingUserDirective ? this.#createEagerTodoPrelude(expandedText) : undefined;
7739
+ const uiSkillPrelude = options?.synthetic ? undefined : this.#createUiSkillPrelude(expandedText);
7692
7740
 
7693
7741
  const userContent: (TextContent | ImageContent)[] = [{ type: "text", text: expandedText }];
7694
7742
  if (options?.images) {
@@ -7717,7 +7765,13 @@ export class AgentSession {
7717
7765
  try {
7718
7766
  await this.#promptWithMessage(message, expandedText, {
7719
7767
  ...options,
7720
- prependMessages: eagerTodoPrelude ? [eagerTodoPrelude.message] : undefined,
7768
+ prependMessages:
7769
+ eagerTodoPrelude || uiSkillPrelude
7770
+ ? [
7771
+ ...(uiSkillPrelude ? [uiSkillPrelude] : []),
7772
+ ...(eagerTodoPrelude ? [eagerTodoPrelude.message] : []),
7773
+ ]
7774
+ : undefined,
7721
7775
  admissionLease: admission,
7722
7776
  resetRetryReplaySafety: true,
7723
7777
  });
@@ -11964,6 +12018,35 @@ export class AgentSession {
11964
12018
  });
11965
12019
  }
11966
12020
 
12021
+ /**
12022
+ * Deterministic activation for the bundled frontend UI/UX skills.
12023
+ *
12024
+ * `hooks/ui-skill-keywords.ts` already knows how to match a prompt against all
12025
+ * thirteen bundled skills in Korean and English, but the only caller was the Codex
12026
+ * `UserPromptSubmit` hook — which this host never fires. In an SKC session the skills
12027
+ * were therefore advertised solely by a sentence in the system prompt, leaving it to
12028
+ * the model to notice and obey. That is not activation, it is hope.
12029
+ *
12030
+ * The matcher is deliberately conservative and measured that way: on sixteen real
12031
+ * prompts it caught 5 of 8 frontend requests and produced **zero** false positives on
12032
+ * the 8 backend ones. Missing a match costs nothing — the system-prompt sentence is
12033
+ * still there — while a wrong match would load a design skill onto a database task.
12034
+ * That asymmetry is why a reminder is the right shape here and a forced tool call is
12035
+ * not.
12036
+ */
12037
+ #createUiSkillPrelude(promptText: string): AgentMessage | undefined {
12038
+ if (this.#planModeState?.enabled) return undefined;
12039
+ const directive = buildUiSkillActivationContext(promptText);
12040
+ if (!directive) return undefined;
12041
+ logger.debug("agent-session: bundled UI skill matched", { promptChars: promptText.length });
12042
+ return {
12043
+ role: "developer",
12044
+ content: [{ type: "text", text: `<system-reminder>\n${directive}\n</system-reminder>` }],
12045
+ attribution: "agent",
12046
+ timestamp: Date.now(),
12047
+ };
12048
+ }
12049
+
11967
12050
  #createEagerTodoPrelude(promptText: string): { message: AgentMessage; toolChoice?: ToolChoice } | undefined {
11968
12051
  const eagerTodosEnabled = this.settings.get("todo.eager");
11969
12052
  const todosEnabled = this.settings.get("todo.enabled");
@@ -6,6 +6,11 @@ export const MODEL_ONBOARDING_PROVIDER_PRESET_COMMAND = "/provider add --preset
6
6
 
7
7
  export const MODEL_ONBOARDING_SETUP_COMMAND = "skc setup provider";
8
8
  export const MODEL_ONBOARDING_OAUTH_COMMAND = "/provider login [provider-id] or /login [provider-id]";
9
+ /**
10
+ * TypeSafe is not a chat model and never appears in the model list, so the only way a
11
+ * user learns it exists is from the surfaces where they go to add credentials.
12
+ */
13
+ export const MODEL_ONBOARDING_TYPESAFE_COMMAND = "/provider typesafe";
9
14
 
10
15
  export function formatModelOnboardingGuidance(): string {
11
16
  return [
@@ -15,12 +20,13 @@ export function formatModelOnboardingGuidance(): string {
15
20
  `Provider presets: ${MODEL_ONBOARDING_PROVIDER_PRESET_COMMAND} (or ${MODEL_ONBOARDING_SETUP_COMMAND} --preset <preset>).`,
16
21
  `API-compatible custom providers: ${MODEL_ONBOARDING_API_PROVIDER_COMMAND}.`,
17
22
  `OAuth/subscription providers: ${MODEL_ONBOARDING_OAUTH_COMMAND}.`,
23
+ `Typed decisions (not a chat model): ${MODEL_ONBOARDING_TYPESAFE_COMMAND} adds a TypeSafe key.`,
18
24
  "Then run /model to select a configured model or assign it to a target.",
19
25
  ].join("\n");
20
26
  }
21
27
 
22
28
  export function formatModelOnboardingInlineHint(): string {
23
- return `Add MiniMax/GLM presets with ${MODEL_ONBOARDING_PROVIDER_PRESET_COMMAND}; custom API providers with ${MODEL_ONBOARDING_API_PROVIDER_COMMAND} (or ${MODEL_ONBOARDING_SETUP_COMMAND}); OAuth/subscription with ${MODEL_ONBOARDING_OAUTH_COMMAND}; then run /model for DEFAULT, EXECUTOR, ARCHITECT, PLANNER, and CRITIC.`;
29
+ return `Add MiniMax/GLM presets with ${MODEL_ONBOARDING_PROVIDER_PRESET_COMMAND}; custom API providers with ${MODEL_ONBOARDING_API_PROVIDER_COMMAND} (or ${MODEL_ONBOARDING_SETUP_COMMAND}); OAuth/subscription with ${MODEL_ONBOARDING_OAUTH_COMMAND}; TypeSafe typed decisions with ${MODEL_ONBOARDING_TYPESAFE_COMMAND}; then run /model for DEFAULT, EXECUTOR, ARCHITECT, PLANNER, and CRITIC.`;
24
30
  }
25
31
 
26
32
  export function formatNoModelOnboardingError(): string {
@@ -1288,7 +1288,7 @@ const BUILTIN_SLASH_COMMAND_REGISTRY: ReadonlyArray<SlashCommandSpec> = [
1288
1288
  {
1289
1289
  name: "provider",
1290
1290
  description: "Set up API-compatible providers or login providers",
1291
- inlineHint: "add|login",
1291
+ inlineHint: "add|login|typesafe",
1292
1292
  allowArgs: true,
1293
1293
  handle: async (command, runtime) => {
1294
1294
  const args = command.args.trim();
@@ -1296,6 +1296,15 @@ const BUILTIN_SLASH_COMMAND_REGISTRY: ReadonlyArray<SlashCommandSpec> = [
1296
1296
  await runtime.output(providerSetupUsage());
1297
1297
  return commandConsumed();
1298
1298
  }
1299
+ if (args === "typesafe") {
1300
+ await runtime.output(
1301
+ "TypeSafe key entry needs an interactive terminal.\n" +
1302
+ "Run it in the TUI (/provider typesafe) or from a shell:\n" +
1303
+ " TYPESAFE_API_KEY=<key> skc setup typesafe\n" +
1304
+ " skc setup typesafe --remove",
1305
+ );
1306
+ return commandConsumed();
1307
+ }
1299
1308
  if (args === "login" || args.startsWith("login ")) {
1300
1309
  const providerId = args.slice("login".length).trim();
1301
1310
  const loginCommand = providerId ? `/login ${providerId}` : "/login [provider-id]";
@@ -1361,6 +1370,14 @@ const BUILTIN_SLASH_COMMAND_REGISTRY: ReadonlyArray<SlashCommandSpec> = [
1361
1370
  runtime.ctx.editor.setText("");
1362
1371
  return;
1363
1372
  }
1373
+ // TypeSafe is not a chat model, so it cannot live in the model list. A direct
1374
+ // subcommand keeps it one step away from `/model`, where users actually look
1375
+ // for "add a key", instead of buried in the onboarding menu.
1376
+ if (args === "typesafe") {
1377
+ runtime.ctx.showTypeSafeKeyPrompt();
1378
+ runtime.ctx.editor.setText("");
1379
+ return;
1380
+ }
1364
1381
  if (args.startsWith("add ")) {
1365
1382
  const parsed = parseProviderSetupSlashArgs(args.slice(4));
1366
1383
  try {
package/src/task/index.ts CHANGED
@@ -17,7 +17,7 @@ import * as os from "node:os";
17
17
  import path from "node:path";
18
18
  import type { AgentTool, AgentToolResult, AgentToolUpdateCallback } from "@sayknow-cli/agent-core";
19
19
  import type { Model, Usage } from "@sayknow-cli/ai";
20
- import { $pickenv, prompt, Snowflake } from "@sayknow-cli/utils";
20
+ import { $pickenv, logger, prompt, Snowflake } from "@sayknow-cli/utils";
21
21
  import type { ToolSession } from "..";
22
22
  import { AsyncJobManager, OwnerSubagentShutdownError, type ResumeRunner } from "../async";
23
23
  import { resolveAgentModelPatterns } from "../config/model-resolver";
@@ -461,6 +461,61 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
461
461
  return this.session.settings.get("task.simple");
462
462
  }
463
463
 
464
+ /**
465
+ * Pick a model for this spawn from the assignment, or null to keep the configured one.
466
+ *
467
+ * Opt-in twice over: `decisions.enabled` must be on *and* at least two tier
468
+ * models configured. That double gate is deliberate — a user who set explicit
469
+ * per-role models chose them on purpose, and silently overriding those from a
470
+ * classifier would be a worse default than doing nothing.
471
+ *
472
+ * One decision per spawn, not per task: every task in a call runs on the same
473
+ * agent and the same model, so asking per task would pay N times for a value
474
+ * that can only be set once.
475
+ */
476
+ async #routeSpawnModel(
477
+ agentName: string,
478
+ tasks: ReadonlyArray<{ description?: string; assignment?: string }> | undefined,
479
+ currentModel: string | readonly string[] | undefined,
480
+ ): Promise<string | undefined> {
481
+ if (!this.session.settings.get("task.modelRouting.enabled")) return undefined;
482
+ const tiers = {
483
+ fast: this.session.settings.get("task.modelRouting.fastModel") || undefined,
484
+ balanced: this.session.settings.get("task.modelRouting.balancedModel") || undefined,
485
+ deep: this.session.settings.get("task.modelRouting.deepModel") || undefined,
486
+ };
487
+ if (Object.values(tiers).filter(Boolean).length < 2) return undefined;
488
+
489
+ const assignment = (tasks ?? [])
490
+ .map(task => [task.description, task.assignment].filter(Boolean).join("\n"))
491
+ .filter(Boolean)
492
+ .join("\n\n");
493
+ if (!assignment) return undefined;
494
+
495
+ try {
496
+ const { createDecisionService } = await import("../decisions");
497
+ const { DEFAULT_TASK_ROUTING_POLICY, routeTaskModel } = await import("../decisions/task-routing");
498
+ const registry = this.session.modelRegistry;
499
+ if (!registry) return undefined;
500
+ const routed = await routeTaskModel(
501
+ createDecisionService({ registry, settings: this.session.settings, enabled: true }),
502
+ { ...DEFAULT_TASK_ROUTING_POLICY, tiers },
503
+ // A role may be configured with a fallback chain; the first entry is what it
504
+ // actually runs on, so that is the baseline the direction is measured from.
505
+ {
506
+ agentName,
507
+ assignment,
508
+ currentModel: Array.isArray(currentModel) ? currentModel[0] : currentModel,
509
+ },
510
+ );
511
+ return routed?.model;
512
+ } catch (error) {
513
+ // Routing is an optimisation. A failure here must never stop a spawn.
514
+ logger.debug("task: spawn model routing failed", { agent: agentName, error: String(error) });
515
+ return undefined;
516
+ }
517
+ }
518
+
464
519
  /**
465
520
  * Create a TaskTool instance.
466
521
  * Repository authority is captured from session cwd *before* agent discovery so
@@ -1114,9 +1169,14 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1114
1169
  // Apply per-agent model override from settings (highest priority)
1115
1170
  const agentModelOverrides = this.session.settings.get("task.agentModelOverrides");
1116
1171
  const settingsModelOverride = agentModelOverrides[agentName];
1172
+ // Per-spawn routing sits *above* the configured role model but uses it as the
1173
+ // baseline: the decision is "is this particular assignment heavier or lighter
1174
+ // than what this role normally gets", not "pick a model from scratch". Declining
1175
+ // leaves the configured value exactly as it was.
1176
+ const routedModelOverride = await this.#routeSpawnModel(agentName, boundParams.tasks, settingsModelOverride);
1117
1177
  const parentActiveModelPattern = this.session.getActiveModelString?.();
1118
1178
  const modelOverride = resolveAgentModelPatterns({
1119
- settingsOverride: settingsModelOverride,
1179
+ settingsOverride: routedModelOverride ?? settingsModelOverride,
1120
1180
  agentModel: effectiveAgent.model,
1121
1181
  settings: this.session.settings,
1122
1182
  activeModelPattern: parentActiveModelPattern,