@sayknow-cli/coding-agent 0.5.25 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/CHANGELOG.md +36 -1
  2. package/dist/types/config/settings-schema.d.ts +51 -5
  3. package/dist/types/config/task-model-specialties.d.ts +55 -0
  4. package/dist/types/decisions/keyword-learning.d.ts +61 -0
  5. package/dist/types/decisions/llm-backend.d.ts +13 -1
  6. package/dist/types/decisions/prompt-triage.d.ts +42 -0
  7. package/dist/types/decisions/skill-routing.d.ts +41 -6
  8. package/dist/types/decisions/task-routing.d.ts +96 -11
  9. package/dist/types/hooks/native-prompt-routing.d.ts +21 -0
  10. package/dist/types/hooks/native-skill-hook.d.ts +3 -0
  11. package/dist/types/hooks/skill-keywords.d.ts +9 -0
  12. package/dist/types/hooks/skill-state.d.ts +20 -3
  13. package/dist/types/hooks/ui-skill-keywords.d.ts +15 -0
  14. package/dist/types/i18n/messages/en.d.ts +15 -0
  15. package/dist/types/lsp/index.d.ts +1 -1
  16. package/dist/types/lsp/types.d.ts +1 -1
  17. package/dist/types/modes/components/model-selector.d.ts +11 -0
  18. package/dist/types/sdk/session.d.ts +3 -13
  19. package/dist/types/session/agent-session.d.ts +8 -0
  20. package/dist/types/session/auth-storage-discovery.d.ts +13 -0
  21. package/dist/types/task/index.d.ts +1 -1
  22. package/dist/types/task/receipt.d.ts +2 -0
  23. package/dist/types/task/types.d.ts +114 -18
  24. package/dist/types/tools/browser.d.ts +2 -2
  25. package/dist/types/tools/subagent.d.ts +2 -2
  26. package/package.json +7 -7
  27. package/scripts/eval-skill-routing.ts +37 -12
  28. package/src/config/settings-schema.ts +64 -12
  29. package/src/config/task-model-specialties.ts +131 -0
  30. package/src/decisions/index.ts +8 -2
  31. package/src/decisions/keyword-learning.ts +678 -0
  32. package/src/decisions/llm-backend.ts +213 -67
  33. package/src/decisions/prompt-triage.ts +163 -0
  34. package/src/decisions/skill-routing.ts +39 -56
  35. package/src/decisions/task-routing.ts +382 -66
  36. package/src/decisions/typesafe-backend.ts +3 -0
  37. package/src/hooks/native-prompt-routing.ts +190 -0
  38. package/src/hooks/native-skill-hook.ts +21 -12
  39. package/src/hooks/skill-keywords.ts +9 -0
  40. package/src/hooks/skill-state.ts +41 -10
  41. package/src/hooks/ui-skill-keywords.ts +67 -10
  42. package/src/i18n/messages/de.ts +16 -0
  43. package/src/i18n/messages/en.ts +16 -0
  44. package/src/i18n/messages/es.ts +16 -0
  45. package/src/i18n/messages/fr.ts +16 -0
  46. package/src/i18n/messages/ja.ts +16 -0
  47. package/src/i18n/messages/ko.ts +16 -0
  48. package/src/i18n/messages/zh.ts +16 -0
  49. package/src/internal-urls/docs-index.generated.ts +1 -1
  50. package/src/main.ts +1 -1
  51. package/src/modes/components/model-selector.ts +275 -34
  52. package/src/modes/controllers/selector-controller.ts +50 -2
  53. package/src/modes/shared/agent-wire/command-dispatch.ts +1 -1
  54. package/src/prompts/tools/task.md +1 -0
  55. package/src/sdk/session.ts +5 -82
  56. package/src/session/agent-session.ts +137 -37
  57. package/src/session/auth-storage-discovery.ts +83 -0
  58. package/src/slash-commands/builtin-registry.ts +11 -9
  59. package/src/task/index.ts +98 -40
  60. package/src/task/receipt.ts +3 -0
  61. package/src/task/types.ts +44 -0
@@ -203,7 +203,9 @@ import type { SettingPath } from "../config/settings-schema";
203
203
  import { getDefault } from "../config/settings-schema";
204
204
  import { RawSseDebugBuffer } from "../debug/raw-sse-buffer";
205
205
  import { createDecisionService } from "../decisions";
206
- import { createSemanticSkillRouter, type SkillRouter } from "../decisions/skill-routing";
206
+ import { loadLearnedKeywordDefinitions, observeRouting } from "../decisions/keyword-learning";
207
+ import { createPromptTriage, type PromptTriager } from "../decisions/prompt-triage";
208
+ import type { BundledSkcUiSkillName } from "../defaults/skc-ui-skills";
207
209
  import { loadCapability } from "../discovery";
208
210
  import { expandApplyPatchToEntries, normalizeDiff, normalizeToLF, ParseError, previewPatch, stripBom } from "../edit";
209
211
  import { MAX_EDIT_FILE_BYTES } from "../edit/read-file";
@@ -267,7 +269,11 @@ import {
267
269
  detectPrimarySkillKeyword,
268
270
  ensureWorkflowSkillActivationState,
269
271
  } from "../hooks/skill-state";
270
- import { buildUiSkillActivationContext } from "../hooks/ui-skill-keywords";
272
+ import {
273
+ buildUiSkillActivationContext,
274
+ buildUiSkillDirectiveForSkill,
275
+ detectUiSkillKeywords,
276
+ } from "../hooks/ui-skill-keywords";
271
277
  import { initializeLocalRoot, type LocalProtocolOptions, resolveLocalUrlToPath } from "../internal-urls";
272
278
  import { shutdownAll as shutdownAllLspClients } from "../lsp/client";
273
279
  import { resolveMemoryBackend } from "../memory-backend";
@@ -533,6 +539,14 @@ export interface AgentSessionConfig {
533
539
  toolRegistry?: Map<string, AgentTool>;
534
540
  /** Tool-session factory context used to lazily attach workflow-gate-only tools. */
535
541
  workflowGateToolSession?: ToolSession;
542
+ /**
543
+ * Stage two of workflow routing: before a genuine user turn, ask a small model
544
+ * through the session's own transport which SKC workflow the prompt calls for.
545
+ * Hosts that own an interactive user (`createAgentSession`) turn this on; a bare
546
+ * session leaves it off so a scripted or injected transport is never consumed by a
547
+ * call the host did not script. `decisions.enabled` still governs the user side.
548
+ */
549
+ semanticWorkflowRouting?: boolean;
536
550
  /** Current session pre-LLM message transform pipeline */
537
551
  transformContext?: (messages: AgentMessage[], signal?: AbortSignal) => AgentMessage[] | Promise<AgentMessage[]>;
538
552
  /** Provider payload hook used by the active session request path */
@@ -1851,11 +1865,18 @@ export class AgentSession {
1851
1865
  #modelRegistry: ModelRegistry;
1852
1866
 
1853
1867
  /** Built on first use; the decision service resolves model and credential lazily. */
1854
- #semanticSkillRouter?: SkillRouter;
1868
+ #promptTriager?: PromptTriager;
1869
+
1870
+ /**
1871
+ * UI skill the semantic stage picked for this turn, when the regex table did
1872
+ * not. Cleared per prompt: it describes one sentence, not the session.
1873
+ */
1874
+ #semanticUiSkill?: BundledSkcUiSkillName;
1855
1875
 
1856
1876
  // Tool registry and prompt builder for extensions
1857
1877
  #toolRegistry: Map<string, AgentTool>;
1858
1878
  #workflowGateToolSession: ToolSession | undefined;
1879
+ #semanticWorkflowRouting: boolean;
1859
1880
  #transformContext: (messages: AgentMessage[], signal?: AbortSignal) => AgentMessage[] | Promise<AgentMessage[]>;
1860
1881
  #onPayload: SimpleStreamOptions["onPayload"] | undefined;
1861
1882
  #onResponse: SimpleStreamOptions["onResponse"] | undefined;
@@ -1979,6 +2000,12 @@ export class AgentSession {
1979
2000
  nonEditDeterminations: 0,
1980
2001
  };
1981
2002
  #promptInFlightCount = 0;
2003
+ /**
2004
+ * A genuine user prompt is being routed to a workflow before its turn starts. Counts
2005
+ * as streaming so a second `prompt()` queues or rejects exactly as it would mid-turn,
2006
+ * and `abort()` cancels the routing through the preflight signal.
2007
+ */
2008
+ #workflowRoutingInFlight = false;
1982
2009
  #agentEventHandlersInFlight = 0;
1983
2010
  #queuedExtensionEventCount = 0;
1984
2011
  #extensionTurnGeneration = 0;
@@ -2405,6 +2432,7 @@ export class AgentSession {
2405
2432
  }
2406
2433
  this.#toolRegistry = config.toolRegistry ?? new Map();
2407
2434
  this.#workflowGateToolSession = config.workflowGateToolSession;
2435
+ this.#semanticWorkflowRouting = config.semanticWorkflowRouting ?? false;
2408
2436
  this.#requestedToolNames = config.requestedToolNames;
2409
2437
  this.#explicitlyDisabledTools = config.explicitlyDisabledTools === true;
2410
2438
  this.#transformContext = config.transformContext ?? (messages => messages);
@@ -5497,7 +5525,7 @@ export class AgentSession {
5497
5525
 
5498
5526
  /** Whether agent is currently streaming a response */
5499
5527
  get isStreaming(): boolean {
5500
- return this.agent.state.isStreaming || this.#promptInFlightCount > 0;
5528
+ return this.agent.state.isStreaming || this.#promptInFlightCount > 0 || this.#workflowRoutingInFlight;
5501
5529
  }
5502
5530
 
5503
5531
  /** Wait until streaming and session settlement work are fully settled. */
@@ -7560,55 +7588,112 @@ export class AgentSession {
7560
7588
  * @throws Error if no model selected or no API key available (when not streaming)
7561
7589
  */
7562
7590
  /**
7563
- * Semantic workflow routing for prompts the keyword table cannot express.
7591
+ * Per-turn prompt triage: every routing question SKC asks about a user's
7592
+ * sentence, answered in one place.
7564
7593
  *
7565
7594
  * Runs in this process, not the hook process: the hook only receives paths and
7566
7595
  * config, so it has no model registry and no credentials to call anything with.
7567
7596
  *
7568
- * **The keyword table is not consulted here, and that is deliberate.** An earlier
7569
- * version returned early on a keyword hit, on the assumption that the deterministic
7570
- * stage had already activated the workflow. That assumption holds only under the
7571
- * Codex host, where `skc codex-native-hook` runs on `UserPromptSubmit`. This session
7572
- * never fires that hook, so the early return meant a prompt containing an enumerated
7573
- * keyword activated *nothing at all* — strictly worse than before the keywords
7574
- * existed, because the semantic stage had been handling those phrasings.
7597
+ * Three stages, cheapest first, and each one shrinks the next:
7598
+ *
7599
+ * 1. **Hand-written tables.** The eighteen workflow keywords and the twenty UI
7600
+ * regexes. Free, deterministic, measured at zero false positives, so they
7601
+ * are not gated behind the opt-in setting.
7602
+ * 2. **Learned patterns.** Two-stem rules mined from previous semantic answers
7603
+ * (`decisions/keyword-learning.ts`). Also free, also deterministic, and the
7604
+ * reason a phrasing the user repeats stops costing a model call.
7605
+ * 3. **One typed decision** for whatever stages 1 and 2 left unanswered. Both
7606
+ * questions ride the same call, so adding UI routing cost no round trip.
7575
7607
  *
7576
- * Keywords remain advisory in this host, as they always were: the routing rules in
7577
- * the system prompt describe them to the model. Only this stage activates, and only
7578
- * when it is confident enough to be worth the mutation guard and Stop hook that
7579
- * activation switches on.
7608
+ * Whatever stage three answers is fed back into stage two, which is what makes
7609
+ * the keyword table self-populating rather than a list somebody has to grow by
7610
+ * hand — the original list recalled 0/9 on Korean prompts for exactly that
7611
+ * reason.
7580
7612
  *
7581
- * Deliberately best-effort — a disabled setting, a missing credential, a timeout, a
7582
- * nonsense answer or low confidence all resolve to "no activation", which is
7583
- * precisely the behaviour before this stage existed.
7613
+ * Keyword hits activate here rather than returning early. An earlier version
7614
+ * returned early on the assumption that the deterministic stage had already
7615
+ * activated the workflow, which holds only under the Codex host where `skc
7616
+ * codex-native-hook` runs on `UserPromptSubmit`. This session never fires that
7617
+ * hook, so the early return meant an enumerated keyword activated *nothing*.
7618
+ *
7619
+ * Deliberately best-effort — a disabled setting, a missing credential, a
7620
+ * timeout, a nonsense answer or low confidence all resolve to "no activation",
7621
+ * which is precisely the behaviour before this stage existed.
7584
7622
  */
7585
- async #routeWorkflowSemantically(text: string): Promise<void> {
7586
- // Stage one: the keyword table. Free, deterministic, and measured at zero false
7587
- // positives, so it is not gated behind the opt-in setting — gating it was why an
7588
- // enumerated phrase activated nothing in this host while the Codex hook activated
7589
- // it fine. Activating here makes the two hosts agree.
7590
- const keyword = detectPrimarySkillKeyword(text);
7623
+ async #routeWorkflowSemantically(text: string, signal: AbortSignal): Promise<void> {
7624
+ this.#semanticUiSkill = undefined;
7625
+ const learningEnabled = this.settings.get("decisions.keywordLearning");
7626
+ // Loaded per turn rather than cached on the session: a second session in the
7627
+ // same repo promotes patterns too, and a user who watched one get learned
7628
+ // expects the next prompt to use it, not the next restart.
7629
+ const learned = learningEnabled ? await loadLearnedKeywordDefinitions() : [];
7630
+ const keyword = detectPrimarySkillKeyword(text, learned);
7591
7631
  if (keyword) {
7632
+ logger.debug("agent-session: workflow keyword match", {
7633
+ skill: keyword.skill,
7634
+ learned: keyword.learned === true,
7635
+ });
7592
7636
  await this.#activateWorkflowSkill(keyword.skill);
7593
- return;
7594
7637
  }
7595
-
7596
- // Stage two costs a model call, so it stays opt-in.
7597
- if (!this.settings.get("decisions.enabled")) return;
7638
+ // The regex table answers the UI question for free when it matches; asking
7639
+ // the model as well would pay for an answer already in hand.
7640
+ const uiMatched = detectUiSkillKeywords(text).length > 0;
7641
+ if (keyword && uiMatched) return;
7642
+
7643
+ // Stage three costs a model call: the host must have opted the session in and
7644
+ // the user must not have turned the setting (on by default) off.
7645
+ if (!this.#semanticWorkflowRouting || !this.settings.get("decisions.enabled")) return;
7646
+ // Only the model call counts as streaming. Flagging the keyword stage too made
7647
+ // `isStreaming` flicker true for one microtask on every prompt, which a poller
7648
+ // mistook for the turn having started.
7649
+ this.#workflowRoutingInFlight = true;
7598
7650
  try {
7599
- this.#semanticSkillRouter ??= createSemanticSkillRouter(
7651
+ this.#promptTriager ??= createPromptTriage(
7600
7652
  createDecisionService({
7601
7653
  registry: this.#modelRegistry,
7602
7654
  settings: this.settings,
7603
7655
  sessionId: this.sessionManager.getSessionId(),
7656
+ preferredProvider: () => this.model?.provider,
7657
+ // The turn's own transport, not a bare `completeSimple`: the same proxy,
7658
+ // credential-invalidation and retry wrapping the main request gets, and
7659
+ // a host that injects a stream function (tests, embedders) intercepts the
7660
+ // decision too instead of watching it go to the network behind its back.
7661
+ completeImpl: async (model, context, options) =>
7662
+ (await this.agent.streamFn(model, context, options)).result(),
7604
7663
  enabled: true,
7605
7664
  }),
7606
7665
  );
7607
- const skill = await this.#semanticSkillRouter(text);
7608
- if (!skill) return;
7609
- await this.#activateWorkflowSkill(skill);
7666
+ const started = Date.now();
7667
+ const triage = await this.#promptTriager({
7668
+ text,
7669
+ skipWorkflow: Boolean(keyword),
7670
+ skipUiSkill: uiMatched,
7671
+ signal,
7672
+ });
7673
+ logger.debug("agent-session: prompt triage", {
7674
+ workflow: triage?.workflow ?? null,
7675
+ uiSkill: triage?.uiSkill ?? null,
7676
+ durationMs: Date.now() - started,
7677
+ });
7678
+ if (!triage || signal.aborted) return;
7679
+ if (triage.uiSkill) this.#semanticUiSkill = triage.uiSkill;
7680
+ if (keyword) return;
7681
+ if (triage.workflow) await this.#activateWorkflowSkill(triage.workflow);
7682
+ if (learningEnabled) {
7683
+ // Awaited, not fired and forgotten: an unawaited write racing the next
7684
+ // turn's read is how a learned pattern gets lost on the prompt that was
7685
+ // supposed to promote it. The store is a few KB and already in memory.
7686
+ await observeRouting({
7687
+ text,
7688
+ skill: triage.workflow,
7689
+ confidence: triage.workflowConfidence,
7690
+ calibrated: triage.calibrated,
7691
+ });
7692
+ }
7610
7693
  } catch (error) {
7611
- logger.debug("agent-session: semantic workflow routing failed", { error: String(error) });
7694
+ logger.debug("agent-session: prompt triage failed", { error: String(error) });
7695
+ } finally {
7696
+ this.#workflowRoutingInFlight = false;
7612
7697
  }
7613
7698
  }
7614
7699
 
@@ -7701,7 +7786,12 @@ export class AgentSession {
7701
7786
  // deep-interview with the setting off and ralplan with it on. Both are plausible
7702
7787
  // readings and ralplan is the better one here, but the point is that enabling this
7703
7788
  // can *change* an activation rather than only add one where there was none.
7704
- if (claimsGenuineUserIntent && !this.isStreaming) await this.#routeWorkflowSemantically(expandedText);
7789
+ if (claimsGenuineUserIntent && !this.isStreaming) {
7790
+ const routingGeneration = this.#promptGeneration;
7791
+ const routingSignal = this.#promptPreflightAbortController.signal;
7792
+ await this.#routeWorkflowSemantically(expandedText, routingSignal);
7793
+ this.#throwIfPromptPreflightCancelled(routingGeneration, routingSignal);
7794
+ }
7705
7795
 
7706
7796
  // If streaming, queue via steer() or followUp() based on option
7707
7797
  if (this.isStreaming) {
@@ -12033,12 +12123,22 @@ export class AgentSession {
12033
12123
  * still there — while a wrong match would load a design skill onto a database task.
12034
12124
  * That asymmetry is why a reminder is the right shape here and a forced tool call is
12035
12125
  * not.
12126
+ *
12127
+ * The three frontend prompts the patterns missed are now covered by the typed
12128
+ * decision in `#routeWorkflowSemantically`, which rides the same call as workflow
12129
+ * routing and therefore costs nothing extra. The patterns stay in front of it: when
12130
+ * they match, the model is never asked.
12036
12131
  */
12037
12132
  #createUiSkillPrelude(promptText: string): AgentMessage | undefined {
12038
12133
  if (this.#planModeState?.enabled) return undefined;
12039
- const directive = buildUiSkillActivationContext(promptText);
12134
+ const matched = buildUiSkillActivationContext(promptText);
12135
+ const directive =
12136
+ matched ?? (this.#semanticUiSkill ? buildUiSkillDirectiveForSkill(this.#semanticUiSkill) : null);
12040
12137
  if (!directive) return undefined;
12041
- logger.debug("agent-session: bundled UI skill matched", { promptChars: promptText.length });
12138
+ logger.debug("agent-session: bundled UI skill matched", {
12139
+ promptChars: promptText.length,
12140
+ source: matched ? "pattern" : "semantic",
12141
+ });
12042
12142
  return {
12043
12143
  role: "developer",
12044
12144
  content: [{ type: "text", text: `<system-reminder>\n${directive}\n</system-reminder>` }],
@@ -0,0 +1,83 @@
1
+ /**
2
+ * Credential store discovery, on its own so a short-lived process can open the
3
+ * store without importing the SDK.
4
+ *
5
+ * `sdk/session.ts` re-exports `discoverAuthStorage` and remains the public entry
6
+ * point. Measured: importing `sdk/session` costs ~400ms of module graph; the
7
+ * Codex prompt hook runs once per prompt and already spends 260ms, so it opens
8
+ * credentials through this file instead.
9
+ */
10
+ import { getAgentDbPath, getAgentDir } from "@sayknow-cli/utils";
11
+ import { resolveConfigValue } from "../config/resolve-config-value";
12
+ import { resolveAuthBrokerConfig } from "./auth-broker-config";
13
+ import { AuthBrokerClient, AuthStorage, RemoteAuthCredentialStore } from "./auth-storage";
14
+
15
+ /**
16
+ * Create an AuthStorage instance.
17
+ *
18
+ * Default: local SQLite store at `<agentDir>/agent.db`.
19
+ *
20
+ * Broker mode: when `SKC_AUTH_BROKER_URL` is set, credentials are pulled from
21
+ * a remote auth-broker over the wire. Refresh tokens never leave the broker;
22
+ * the client receives access tokens with `refresh = "__remote__"` and calls
23
+ * back into the broker through the {@link AuthStorageOptions.refreshOAuthCredential}
24
+ * override to re-mint access tokens when needed.
25
+ */
26
+ export async function discoverAuthStorage(agentDir: string = getAgentDir()): Promise<AuthStorage> {
27
+ const brokerConfig = await resolveAuthBrokerConfig();
28
+ const credentialRankingMode = resolveCredentialRankingMode();
29
+ if (brokerConfig) {
30
+ const client = new AuthBrokerClient({ url: brokerConfig.url, token: brokerConfig.token });
31
+ const initialResult = await client.fetchSnapshot();
32
+ if (initialResult.status !== 200) throw new Error("Auth broker returned no initial snapshot");
33
+ const store = new RemoteAuthCredentialStore({ client, initialSnapshot: initialResult.snapshot });
34
+ // Refresh + usage hooks live on RemoteAuthCredentialStore; AuthStorage
35
+ // discovers them automatically when no explicit option overrides them.
36
+ const storage = new AuthStorage(store, {
37
+ configValueResolver: resolveConfigValue,
38
+ sourceLabel: `broker ${brokerConfig.url}`,
39
+ credentialRankingMode,
40
+ });
41
+ try {
42
+ await storage.reload();
43
+ } catch (error) {
44
+ try {
45
+ storage.close();
46
+ } catch {
47
+ // Preserve the initial reload failure.
48
+ }
49
+ throw error;
50
+ }
51
+ return storage;
52
+ }
53
+ const dbPath = getAgentDbPath(agentDir);
54
+ const storage = await AuthStorage.create(dbPath, {
55
+ configValueResolver: resolveConfigValue,
56
+ sourceLabel: `local ${dbPath}`,
57
+ credentialRankingMode,
58
+ });
59
+ try {
60
+ await storage.reload();
61
+ } catch (error) {
62
+ try {
63
+ storage.close();
64
+ } catch {
65
+ // Preserve the initial reload failure.
66
+ }
67
+ throw error;
68
+ }
69
+ return storage;
70
+ }
71
+
72
+ /**
73
+ * Opt-in multi-account credential ranking mode, read from the
74
+ * `SKC_CREDENTIAL_RANKING_MODE` env var. Unset/unknown → `undefined`, leaving
75
+ * {@link AuthStorage}'s default (`balanced`) untouched. `earliest-reset`
76
+ * switches to earliest-expiry-first selection so soon-to-reset tumbling-window
77
+ * quota is drained before it is lost.
78
+ */
79
+ function resolveCredentialRankingMode(): "balanced" | "earliest-reset" | undefined {
80
+ const raw = process.env.SKC_CREDENTIAL_RANKING_MODE?.trim();
81
+ if (raw === "balanced" || raw === "earliest-reset") return raw;
82
+ return undefined;
83
+ }
@@ -360,9 +360,8 @@ async function resolveModelCommandSelection(
360
360
  }
361
361
 
362
362
  const providerRef = parseProviderQualifiedSelector(selector);
363
- const discoverableProviders = runtime.session.modelRegistry?.getDiscoverableProviders?.() ?? [];
364
- if (providerRef && discoverableProviders.includes(providerRef.provider)) {
365
- await runtime.session.modelRegistry.refreshProvider?.(providerRef.provider, "online");
363
+ if (providerRef && runtime.session.modelRegistry?.refreshProvider) {
364
+ await runtime.session.modelRegistry.refreshProvider(providerRef.provider, "online");
366
365
  availableModels = runtime.session.getAvailableModels?.() ?? [];
367
366
  const refreshedSelection = resolveModelCommandSelectionFromAvailable(
368
367
  runtime,
@@ -372,12 +371,15 @@ async function resolveModelCommandSelection(
372
371
  if (refreshedSelection) {
373
372
  return { ok: true, selection: refreshedSelection };
374
373
  }
375
- return {
376
- ok: false,
377
- failure: {
378
- message: formatDiscoverableProviderFailure(selector, providerRef.provider, providerRef.modelId, runtime),
379
- },
380
- };
374
+ const discoverableProviders = runtime.session.modelRegistry.getDiscoverableProviders?.() ?? [];
375
+ if (discoverableProviders.includes(providerRef.provider)) {
376
+ return {
377
+ ok: false,
378
+ failure: {
379
+ message: formatDiscoverableProviderFailure(selector, providerRef.provider, providerRef.modelId, runtime),
380
+ },
381
+ };
382
+ }
381
383
  }
382
384
 
383
385
  return {
package/src/task/index.ts CHANGED
@@ -21,6 +21,9 @@ import { $pickenv, logger, prompt, Snowflake } from "@sayknow-cli/utils";
21
21
  import type { ToolSession } from "..";
22
22
  import { AsyncJobManager, OwnerSubagentShutdownError, type ResumeRunner } from "../async";
23
23
  import { resolveAgentModelPatterns } from "../config/model-resolver";
24
+ import { normalizeModelSelectorValue } from "../config/model-selector-value";
25
+ import type { TaskModelSpecialty } from "../config/task-model-specialties";
26
+ import type { TaskRoutingResult } from "../decisions/task-routing";
24
27
  import type { Theme } from "../modes/theme/theme";
25
28
  import planModeSubagentPrompt from "../prompts/system/plan-mode-subagent.md" with { type: "text" };
26
29
  import taskDescriptionTemplate from "../prompts/tools/task.md" with { type: "text" };
@@ -37,6 +40,7 @@ import {
37
40
  type SingleResult,
38
41
  type TaskItem,
39
42
  type TaskParams,
43
+ type TaskRoutingAttribution,
40
44
  type TaskToolDetails,
41
45
  type TaskToolSchemaInstance,
42
46
  } from "./types";
@@ -212,6 +216,7 @@ export type {
212
216
  SubagentLifecyclePayload,
213
217
  SubagentProgressPayload,
214
218
  TaskParams,
219
+ TaskRoutingAttribution,
215
220
  TaskToolDetails,
216
221
  } from "./types";
217
222
  export {
@@ -462,54 +467,80 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
462
467
  }
463
468
 
464
469
  /**
465
- * Pick a model for this spawn from the assignment, or null to keep the configured one.
470
+ * Route a child whose caller declared the kind of work.
466
471
  *
467
- * Opt-in twice over: `decisions.enabled` must be on *and* at least two tier
468
- * models configured. That double gate is deliberate — a user who set explicit
469
- * per-role models chose them on purpose, and silently overriding those from a
470
- * classifier would be a worse default than doing nothing.
472
+ * Deterministic by design: no classifier, no `task.modelRouting.enabled`
473
+ * gate. The user assigned a model to this specialty in `/model`; if the caller
474
+ * says the work is that specialty, the child runs on that model and leaves it
475
+ * only when it errors — the child session's fallback chain advances to the
476
+ * role baseline composed behind it.
477
+ */
478
+ async #routeDeclaredSpecialty(
479
+ agentName: string,
480
+ specialty: TaskModelSpecialty,
481
+ baselineChain: readonly string[],
482
+ ): Promise<TaskRoutingResult | undefined> {
483
+ try {
484
+ const { resolveDeclaredSpecialtyRouting } = await import("../decisions/task-routing");
485
+ return (
486
+ resolveDeclaredSpecialtyRouting(this.session.settings, {
487
+ agentName,
488
+ specialty,
489
+ currentModel: baselineChain[0],
490
+ baselineChain,
491
+ }) ?? undefined
492
+ );
493
+ } catch (error) {
494
+ // A declared specialty with nothing behind it still spawns on the role model.
495
+ logger.debug("task: declared specialty routing failed", { agent: agentName, specialty, error: String(error) });
496
+ return undefined;
497
+ }
498
+ }
499
+
500
+ /**
501
+ * Pick a model for one child assignment, or undefined to keep the configured one.
471
502
  *
472
- * One decision per spawn, not per task: every task in a call runs on the same
473
- * agent and the same model, so asking per task would pay N times for a value
474
- * that can only be set once.
503
+ * This is the *guessing* half: a classifier reads the assignment and decides.
504
+ * Opt-in twice over: `task.modelRouting.enabled` must be on *and* an axis must
505
+ * have something to move on. That double gate is deliberate — a user who set
506
+ * explicit per-role models chose them on purpose, and silently overriding those
507
+ * from a classifier would be a worse default than doing nothing. A caller that
508
+ * *knows* the kind of work declares it instead (`#routeDeclaredSpecialty`).
509
+ *
510
+ * One decision per *child*, not per call. Every task in a call shares an agent,
511
+ * but not a workload: a batch can hold an implementation slice and a test slice,
512
+ * and a single joined classification would have to answer for both at once.
475
513
  */
476
514
  async #routeSpawnModel(
477
515
  agentName: string,
478
- tasks: ReadonlyArray<{ description?: string; assignment?: string }> | undefined,
479
- currentModel: string | readonly string[] | undefined,
480
- ): Promise<string | undefined> {
516
+ assignment: string | undefined,
517
+ baselineChain: readonly string[],
518
+ ): Promise<TaskRoutingResult | undefined> {
519
+ // Cheap guard before the dynamic import so a disabled feature costs nothing.
481
520
  if (!this.session.settings.get("task.modelRouting.enabled")) return undefined;
482
- const tiers = {
483
- fast: this.session.settings.get("task.modelRouting.fastModel") || undefined,
484
- balanced: this.session.settings.get("task.modelRouting.balancedModel") || undefined,
485
- deep: this.session.settings.get("task.modelRouting.deepModel") || undefined,
486
- };
487
- const frontendModel = this.session.settings.get("task.modelRouting.frontendModel") || undefined;
488
- if (Object.values(tiers).filter(Boolean).length < 2 && !frontendModel) return undefined;
489
-
490
- const assignment = (tasks ?? [])
491
- .map(task => [task.description, task.assignment].filter(Boolean).join("\n"))
492
- .filter(Boolean)
493
- .join("\n\n");
494
- if (!assignment) return undefined;
521
+ const trimmed = assignment?.trim();
522
+ if (!trimmed) return undefined;
495
523
 
496
524
  try {
497
525
  const { createDecisionService } = await import("../decisions");
498
- const { DEFAULT_TASK_ROUTING_POLICY, routeTaskModel } = await import("../decisions/task-routing");
526
+ const { buildTaskRoutingPolicyFromSettings, routeTaskModel } = await import("../decisions/task-routing");
527
+ const policy = buildTaskRoutingPolicyFromSettings(this.session.settings);
528
+ if (!policy) return undefined;
499
529
  const registry = this.session.modelRegistry;
500
530
  if (!registry) return undefined;
501
531
  const routed = await routeTaskModel(
502
532
  createDecisionService({ registry, settings: this.session.settings, enabled: true }),
503
- { ...DEFAULT_TASK_ROUTING_POLICY, tiers, frontendModel },
504
- // A role may be configured with a fallback chain; the first entry is what it
505
- // actually runs on, so that is the baseline the direction is measured from.
533
+ policy,
506
534
  {
507
535
  agentName,
508
- assignment,
509
- currentModel: Array.isArray(currentModel) ? currentModel[0] : currentModel,
536
+ assignment: trimmed,
537
+ // A role may be configured with a fallback chain; the first entry is what it
538
+ // actually runs on, so that is the baseline the direction is measured from.
539
+ currentModel: baselineChain[0],
540
+ baselineChain,
510
541
  },
511
542
  );
512
- return routed?.model;
543
+ return routed ?? undefined;
513
544
  } catch (error) {
514
545
  // Routing is an optimisation. A failure here must never stop a spawn.
515
546
  logger.debug("task: spawn model routing failed", { agent: agentName, error: String(error) });
@@ -1170,19 +1201,18 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1170
1201
  // Apply per-agent model override from settings (highest priority)
1171
1202
  const agentModelOverrides = this.session.settings.get("task.agentModelOverrides");
1172
1203
  const settingsModelOverride = agentModelOverrides[agentName];
1173
- // Per-spawn routing sits *above* the configured role model but uses it as the
1174
- // baseline: the decision is "is this particular assignment heavier or lighter
1175
- // than what this role normally gets", not "pick a model from scratch". Declining
1176
- // leaves the configured value exactly as it was.
1177
- const routedModelOverride = await this.#routeSpawnModel(agentName, boundParams.tasks, settingsModelOverride);
1204
+ // The role's own resolved chain. Per-child routing composes *in front of* this
1205
+ // and never replaces it: a specialty or tier that cannot be authenticated must
1206
+ // still fall through to the model the role would have used anyway.
1178
1207
  const parentActiveModelPattern = this.session.getActiveModelString?.();
1179
1208
  const modelOverride = resolveAgentModelPatterns({
1180
- settingsOverride: routedModelOverride ?? settingsModelOverride,
1209
+ settingsOverride: settingsModelOverride,
1181
1210
  agentModel: effectiveAgent.model,
1182
1211
  settings: this.session.settings,
1183
1212
  activeModelPattern: parentActiveModelPattern,
1184
1213
  fallbackModelPattern: this.session.getModelString?.(),
1185
1214
  });
1215
+ const baselineChain = normalizeModelSelectorValue(modelOverride);
1186
1216
  const thinkingLevelOverride = effectiveAgent.thinkingLevel;
1187
1217
 
1188
1218
  // Output schema priority: task call > agent frontmatter > inherited parent session.
@@ -1474,6 +1504,28 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1474
1504
  sessionFile?: string | null;
1475
1505
  },
1476
1506
  ) => {
1507
+ // Route THIS child. A declared specialty is deterministic; otherwise a
1508
+ // batch shares an agent but not a workload, so a joined classification
1509
+ // would have to answer for an implementation slice and a test slice at
1510
+ // the same time.
1511
+ const routed = task.specialty
1512
+ ? await this.#routeDeclaredSpecialty(agentName, task.specialty, baselineChain)
1513
+ : await this.#routeSpawnModel(agentName, task.assignment, baselineChain);
1514
+ const taskModelOverride = routed ? routed.candidates.map(candidate => candidate.selector) : modelOverride;
1515
+ // Requested, not effective: the chain above can still fall through to a
1516
+ // later candidate, so this records intent and nothing more.
1517
+ const taskRouting: TaskRoutingAttribution | undefined = routed
1518
+ ? {
1519
+ source: routed.requestedSource,
1520
+ specialty: routed.requestedSpecialty,
1521
+ tier: routed.requestedTier,
1522
+ declared: routed.declared,
1523
+ calibrated: routed.calibrated,
1524
+ confidence: routed.confidence,
1525
+ ordinalStrength: routed.ordinalStrength,
1526
+ reason: routed.reason,
1527
+ }
1528
+ : undefined;
1477
1529
  const forkContextSeed = prebuiltForkContextSeeds?.get(task.id) ?? (await buildForkContextSeed(task));
1478
1530
  const forkContext = requestsForkContext(task)
1479
1531
  ? { mode: task.inheritContext, clonedTokens: forkContextSeed?.metadata.approximateTokens ?? 0 }
@@ -1516,7 +1568,7 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1516
1568
  resumeMessage: overrides?.resumeMessage ?? executionOverrides?.resumeMessage,
1517
1569
  subagentId: task.id,
1518
1570
  taskDepth,
1519
- modelOverride,
1571
+ modelOverride: taskModelOverride,
1520
1572
  parentActiveModelPattern,
1521
1573
  parentSessionId: this.session.getSessionId?.() ?? undefined,
1522
1574
  thinkingLevel: thinkingLevelOverride,
@@ -1533,6 +1585,8 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1533
1585
  onProgress: progress => {
1534
1586
  progressMap.set(index, {
1535
1587
  ...structuredClone(progress),
1588
+ modelOverride: taskModelOverride,
1589
+ ...(taskRouting ? { routing: taskRouting } : {}),
1536
1590
  });
1537
1591
  AsyncJobManager.instance()?.recordSubagentProgress(task.id, progress);
1538
1592
  emitProgress();
@@ -1589,7 +1643,7 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1589
1643
  resumeMessage: overrides?.resumeMessage ?? executionOverrides?.resumeMessage,
1590
1644
  subagentId: task.id,
1591
1645
  taskDepth,
1592
- modelOverride,
1646
+ modelOverride: taskModelOverride,
1593
1647
  parentActiveModelPattern,
1594
1648
  parentSessionId: this.session.getSessionId?.() ?? undefined,
1595
1649
  thinkingLevel: thinkingLevelOverride,
@@ -1606,6 +1660,8 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1606
1660
  onProgress: progress => {
1607
1661
  progressMap.set(index, {
1608
1662
  ...structuredClone(progress),
1663
+ modelOverride: taskModelOverride,
1664
+ ...(taskRouting ? { routing: taskRouting } : {}),
1609
1665
  });
1610
1666
  AsyncJobManager.instance()?.recordSubagentProgress(task.id, progress);
1611
1667
  emitProgress();
@@ -1629,6 +1685,7 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1629
1685
  const resultWithForkContext = {
1630
1686
  ...result,
1631
1687
  ...(forkContext ? { forkContext } : {}),
1688
+ ...(taskRouting ? { routing: taskRouting } : {}),
1632
1689
  forkContextAdvisory,
1633
1690
  repositoryBinding: publicRepositoryBinding(taskRepositoryBinding),
1634
1691
  };
@@ -1709,7 +1766,8 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1709
1766
  truncated: false,
1710
1767
  durationMs: Date.now() - taskStart,
1711
1768
  tokens: 0,
1712
- modelOverride,
1769
+ modelOverride: taskModelOverride,
1770
+ ...(taskRouting ? { routing: taskRouting } : {}),
1713
1771
  forkContext,
1714
1772
  error: message,
1715
1773
  };
@@ -28,6 +28,8 @@ export interface TaskResultReceipt {
28
28
  contextTokens?: number;
29
29
  contextWindow?: number;
30
30
  modelOverride?: string | string[];
31
+ /** What the router asked for, kept separate from what the spawn ran on. */
32
+ routing?: SingleResult["routing"];
31
33
  modelSubstitutionWarning?: SingleResult["modelSubstitutionWarning"];
32
34
  usage?: SingleResult["usage"];
33
35
  cost?: number;
@@ -245,6 +247,7 @@ export function buildTaskReceipt(raw: SingleResult): TaskResultReceipt {
245
247
  contextTokens: raw.contextTokens,
246
248
  contextWindow: raw.contextWindow,
247
249
  modelOverride: raw.modelOverride,
250
+ routing: raw.routing,
248
251
  modelSubstitutionWarning: raw.modelSubstitutionWarning,
249
252
  usage: raw.usage,
250
253
  cost: raw.usage?.cost.total,