@sayknow-cli/coding-agent 0.5.25 → 0.5.26

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/task/index.ts CHANGED
@@ -21,6 +21,9 @@ import { $pickenv, logger, prompt, Snowflake } from "@sayknow-cli/utils";
21
21
  import type { ToolSession } from "..";
22
22
  import { AsyncJobManager, OwnerSubagentShutdownError, type ResumeRunner } from "../async";
23
23
  import { resolveAgentModelPatterns } from "../config/model-resolver";
24
+ import { normalizeModelSelectorValue } from "../config/model-selector-value";
25
+ import type { TaskModelSpecialty } from "../config/task-model-specialties";
26
+ import type { TaskRoutingResult } from "../decisions/task-routing";
24
27
  import type { Theme } from "../modes/theme/theme";
25
28
  import planModeSubagentPrompt from "../prompts/system/plan-mode-subagent.md" with { type: "text" };
26
29
  import taskDescriptionTemplate from "../prompts/tools/task.md" with { type: "text" };
@@ -37,6 +40,7 @@ import {
37
40
  type SingleResult,
38
41
  type TaskItem,
39
42
  type TaskParams,
43
+ type TaskRoutingAttribution,
40
44
  type TaskToolDetails,
41
45
  type TaskToolSchemaInstance,
42
46
  } from "./types";
@@ -212,6 +216,7 @@ export type {
212
216
  SubagentLifecyclePayload,
213
217
  SubagentProgressPayload,
214
218
  TaskParams,
219
+ TaskRoutingAttribution,
215
220
  TaskToolDetails,
216
221
  } from "./types";
217
222
  export {
@@ -462,54 +467,80 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
462
467
  }
463
468
 
464
469
  /**
465
- * Pick a model for this spawn from the assignment, or null to keep the configured one.
470
+ * Route a child whose caller declared the kind of work.
466
471
  *
467
- * Opt-in twice over: `decisions.enabled` must be on *and* at least two tier
468
- * models configured. That double gate is deliberate — a user who set explicit
469
- * per-role models chose them on purpose, and silently overriding those from a
470
- * classifier would be a worse default than doing nothing.
472
+ * Deterministic by design: no classifier, no `task.modelRouting.enabled`
473
+ * gate. The user assigned a model to this specialty in `/model`; if the caller
474
+ * says the work is that specialty, the child runs on that model and leaves it
475
+ * only when it errors — the child session's fallback chain advances to the
476
+ * role baseline composed behind it.
477
+ */
478
+ async #routeDeclaredSpecialty(
479
+ agentName: string,
480
+ specialty: TaskModelSpecialty,
481
+ baselineChain: readonly string[],
482
+ ): Promise<TaskRoutingResult | undefined> {
483
+ try {
484
+ const { resolveDeclaredSpecialtyRouting } = await import("../decisions/task-routing");
485
+ return (
486
+ resolveDeclaredSpecialtyRouting(this.session.settings, {
487
+ agentName,
488
+ specialty,
489
+ currentModel: baselineChain[0],
490
+ baselineChain,
491
+ }) ?? undefined
492
+ );
493
+ } catch (error) {
494
+ // A declared specialty with nothing behind it still spawns on the role model.
495
+ logger.debug("task: declared specialty routing failed", { agent: agentName, specialty, error: String(error) });
496
+ return undefined;
497
+ }
498
+ }
499
+
500
+ /**
501
+ * Pick a model for one child assignment, or undefined to keep the configured one.
471
502
  *
472
- * One decision per spawn, not per task: every task in a call runs on the same
473
- * agent and the same model, so asking per task would pay N times for a value
474
- * that can only be set once.
503
+ * This is the *guessing* half: a classifier reads the assignment and decides.
504
+ * Opt-in twice over: `task.modelRouting.enabled` must be on *and* an axis must
505
+ * have something to move on. That double gate is deliberate — a user who set
506
+ * explicit per-role models chose them on purpose, and silently overriding those
507
+ * from a classifier would be a worse default than doing nothing. A caller that
508
+ * *knows* the kind of work declares it instead (`#routeDeclaredSpecialty`).
509
+ *
510
+ * One decision per *child*, not per call. Every task in a call shares an agent,
511
+ * but not a workload: a batch can hold an implementation slice and a test slice,
512
+ * and a single joined classification would have to answer for both at once.
475
513
  */
476
514
  async #routeSpawnModel(
477
515
  agentName: string,
478
- tasks: ReadonlyArray<{ description?: string; assignment?: string }> | undefined,
479
- currentModel: string | readonly string[] | undefined,
480
- ): Promise<string | undefined> {
516
+ assignment: string | undefined,
517
+ baselineChain: readonly string[],
518
+ ): Promise<TaskRoutingResult | undefined> {
519
+ // Cheap guard before the dynamic import so a disabled feature costs nothing.
481
520
  if (!this.session.settings.get("task.modelRouting.enabled")) return undefined;
482
- const tiers = {
483
- fast: this.session.settings.get("task.modelRouting.fastModel") || undefined,
484
- balanced: this.session.settings.get("task.modelRouting.balancedModel") || undefined,
485
- deep: this.session.settings.get("task.modelRouting.deepModel") || undefined,
486
- };
487
- const frontendModel = this.session.settings.get("task.modelRouting.frontendModel") || undefined;
488
- if (Object.values(tiers).filter(Boolean).length < 2 && !frontendModel) return undefined;
489
-
490
- const assignment = (tasks ?? [])
491
- .map(task => [task.description, task.assignment].filter(Boolean).join("\n"))
492
- .filter(Boolean)
493
- .join("\n\n");
494
- if (!assignment) return undefined;
521
+ const trimmed = assignment?.trim();
522
+ if (!trimmed) return undefined;
495
523
 
496
524
  try {
497
525
  const { createDecisionService } = await import("../decisions");
498
- const { DEFAULT_TASK_ROUTING_POLICY, routeTaskModel } = await import("../decisions/task-routing");
526
+ const { buildTaskRoutingPolicyFromSettings, routeTaskModel } = await import("../decisions/task-routing");
527
+ const policy = buildTaskRoutingPolicyFromSettings(this.session.settings);
528
+ if (!policy) return undefined;
499
529
  const registry = this.session.modelRegistry;
500
530
  if (!registry) return undefined;
501
531
  const routed = await routeTaskModel(
502
532
  createDecisionService({ registry, settings: this.session.settings, enabled: true }),
503
- { ...DEFAULT_TASK_ROUTING_POLICY, tiers, frontendModel },
504
- // A role may be configured with a fallback chain; the first entry is what it
505
- // actually runs on, so that is the baseline the direction is measured from.
533
+ policy,
506
534
  {
507
535
  agentName,
508
- assignment,
509
- currentModel: Array.isArray(currentModel) ? currentModel[0] : currentModel,
536
+ assignment: trimmed,
537
+ // A role may be configured with a fallback chain; the first entry is what it
538
+ // actually runs on, so that is the baseline the direction is measured from.
539
+ currentModel: baselineChain[0],
540
+ baselineChain,
510
541
  },
511
542
  );
512
- return routed?.model;
543
+ return routed ?? undefined;
513
544
  } catch (error) {
514
545
  // Routing is an optimisation. A failure here must never stop a spawn.
515
546
  logger.debug("task: spawn model routing failed", { agent: agentName, error: String(error) });
@@ -1170,19 +1201,18 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1170
1201
  // Apply per-agent model override from settings (highest priority)
1171
1202
  const agentModelOverrides = this.session.settings.get("task.agentModelOverrides");
1172
1203
  const settingsModelOverride = agentModelOverrides[agentName];
1173
- // Per-spawn routing sits *above* the configured role model but uses it as the
1174
- // baseline: the decision is "is this particular assignment heavier or lighter
1175
- // than what this role normally gets", not "pick a model from scratch". Declining
1176
- // leaves the configured value exactly as it was.
1177
- const routedModelOverride = await this.#routeSpawnModel(agentName, boundParams.tasks, settingsModelOverride);
1204
+ // The role's own resolved chain. Per-child routing composes *in front of* this
1205
+ // and never replaces it: a specialty or tier that cannot be authenticated must
1206
+ // still fall through to the model the role would have used anyway.
1178
1207
  const parentActiveModelPattern = this.session.getActiveModelString?.();
1179
1208
  const modelOverride = resolveAgentModelPatterns({
1180
- settingsOverride: routedModelOverride ?? settingsModelOverride,
1209
+ settingsOverride: settingsModelOverride,
1181
1210
  agentModel: effectiveAgent.model,
1182
1211
  settings: this.session.settings,
1183
1212
  activeModelPattern: parentActiveModelPattern,
1184
1213
  fallbackModelPattern: this.session.getModelString?.(),
1185
1214
  });
1215
+ const baselineChain = normalizeModelSelectorValue(modelOverride);
1186
1216
  const thinkingLevelOverride = effectiveAgent.thinkingLevel;
1187
1217
 
1188
1218
  // Output schema priority: task call > agent frontmatter > inherited parent session.
@@ -1474,6 +1504,28 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1474
1504
  sessionFile?: string | null;
1475
1505
  },
1476
1506
  ) => {
1507
+ // Route THIS child. A declared specialty is deterministic; otherwise a
1508
+ // batch shares an agent but not a workload, so a joined classification
1509
+ // would have to answer for an implementation slice and a test slice at
1510
+ // the same time.
1511
+ const routed = task.specialty
1512
+ ? await this.#routeDeclaredSpecialty(agentName, task.specialty, baselineChain)
1513
+ : await this.#routeSpawnModel(agentName, task.assignment, baselineChain);
1514
+ const taskModelOverride = routed ? routed.candidates.map(candidate => candidate.selector) : modelOverride;
1515
+ // Requested, not effective: the chain above can still fall through to a
1516
+ // later candidate, so this records intent and nothing more.
1517
+ const taskRouting: TaskRoutingAttribution | undefined = routed
1518
+ ? {
1519
+ source: routed.requestedSource,
1520
+ specialty: routed.requestedSpecialty,
1521
+ tier: routed.requestedTier,
1522
+ declared: routed.declared,
1523
+ calibrated: routed.calibrated,
1524
+ confidence: routed.confidence,
1525
+ ordinalStrength: routed.ordinalStrength,
1526
+ reason: routed.reason,
1527
+ }
1528
+ : undefined;
1477
1529
  const forkContextSeed = prebuiltForkContextSeeds?.get(task.id) ?? (await buildForkContextSeed(task));
1478
1530
  const forkContext = requestsForkContext(task)
1479
1531
  ? { mode: task.inheritContext, clonedTokens: forkContextSeed?.metadata.approximateTokens ?? 0 }
@@ -1516,7 +1568,7 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1516
1568
  resumeMessage: overrides?.resumeMessage ?? executionOverrides?.resumeMessage,
1517
1569
  subagentId: task.id,
1518
1570
  taskDepth,
1519
- modelOverride,
1571
+ modelOverride: taskModelOverride,
1520
1572
  parentActiveModelPattern,
1521
1573
  parentSessionId: this.session.getSessionId?.() ?? undefined,
1522
1574
  thinkingLevel: thinkingLevelOverride,
@@ -1533,6 +1585,8 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1533
1585
  onProgress: progress => {
1534
1586
  progressMap.set(index, {
1535
1587
  ...structuredClone(progress),
1588
+ modelOverride: taskModelOverride,
1589
+ ...(taskRouting ? { routing: taskRouting } : {}),
1536
1590
  });
1537
1591
  AsyncJobManager.instance()?.recordSubagentProgress(task.id, progress);
1538
1592
  emitProgress();
@@ -1589,7 +1643,7 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1589
1643
  resumeMessage: overrides?.resumeMessage ?? executionOverrides?.resumeMessage,
1590
1644
  subagentId: task.id,
1591
1645
  taskDepth,
1592
- modelOverride,
1646
+ modelOverride: taskModelOverride,
1593
1647
  parentActiveModelPattern,
1594
1648
  parentSessionId: this.session.getSessionId?.() ?? undefined,
1595
1649
  thinkingLevel: thinkingLevelOverride,
@@ -1606,6 +1660,8 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1606
1660
  onProgress: progress => {
1607
1661
  progressMap.set(index, {
1608
1662
  ...structuredClone(progress),
1663
+ modelOverride: taskModelOverride,
1664
+ ...(taskRouting ? { routing: taskRouting } : {}),
1609
1665
  });
1610
1666
  AsyncJobManager.instance()?.recordSubagentProgress(task.id, progress);
1611
1667
  emitProgress();
@@ -1629,6 +1685,7 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1629
1685
  const resultWithForkContext = {
1630
1686
  ...result,
1631
1687
  ...(forkContext ? { forkContext } : {}),
1688
+ ...(taskRouting ? { routing: taskRouting } : {}),
1632
1689
  forkContextAdvisory,
1633
1690
  repositoryBinding: publicRepositoryBinding(taskRepositoryBinding),
1634
1691
  };
@@ -1709,7 +1766,8 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1709
1766
  truncated: false,
1710
1767
  durationMs: Date.now() - taskStart,
1711
1768
  tokens: 0,
1712
- modelOverride,
1769
+ modelOverride: taskModelOverride,
1770
+ ...(taskRouting ? { routing: taskRouting } : {}),
1713
1771
  forkContext,
1714
1772
  error: message,
1715
1773
  };
@@ -28,6 +28,8 @@ export interface TaskResultReceipt {
28
28
  contextTokens?: number;
29
29
  contextWindow?: number;
30
30
  modelOverride?: string | string[];
31
+ /** What the router asked for, kept separate from what the spawn ran on. */
32
+ routing?: SingleResult["routing"];
31
33
  modelSubstitutionWarning?: SingleResult["modelSubstitutionWarning"];
32
34
  usage?: SingleResult["usage"];
33
35
  cost?: number;
@@ -245,6 +247,7 @@ export function buildTaskReceipt(raw: SingleResult): TaskResultReceipt {
245
247
  contextTokens: raw.contextTokens,
246
248
  contextWindow: raw.contextWindow,
247
249
  modelOverride: raw.modelOverride,
250
+ routing: raw.routing,
248
251
  modelSubstitutionWarning: raw.modelSubstitutionWarning,
249
252
  usage: raw.usage,
250
253
  cost: raw.usage?.cost.total,
package/src/task/types.ts CHANGED
@@ -2,6 +2,12 @@ import type { ThinkingLevel } from "@sayknow-cli/agent-core";
2
2
  import type { Usage } from "@sayknow-cli/ai";
3
3
  import { $env } from "@sayknow-cli/utils";
4
4
  import * as z from "zod/v4";
5
+ import {
6
+ TASK_MODEL_SPECIALTY_IDS,
7
+ type TaskModelSpecialty,
8
+ type TaskRoutingSource,
9
+ } from "../config/task-model-specialties";
10
+ import type { TaskTier } from "../decisions/task-routing";
5
11
  import { isValidTaskId, TASK_ID_DESCRIPTION } from "./id";
6
12
  import type { TaskResultReceipt } from "./receipt";
7
13
  import type { SpawnRoiReconciliation } from "./roi-reconciliation";
@@ -103,6 +109,12 @@ const createTaskItemSchema = (_contextEnabled: boolean) =>
103
109
  .describe(
104
110
  "typed executor mode: default keeps ordinary executor behavior; ultragoal-red-team injects the Ultragoal QA/red-team prompt fragment. Prefer this over free-form assignment text (#2698).",
105
111
  ),
112
+ specialty: z
113
+ .enum(TASK_MODEL_SPECIALTY_IDS)
114
+ .optional()
115
+ .describe(
116
+ "kind of work, so the model the user assigned to it under /model runs this child: backendArchitecture, frontendDesign, implementation, testing, or review. Declaring it is deterministic — no classifier, no routing switch; the child only leaves that model if it errors (429, 5xx, auth, quota). Omit to keep the agent's role model, or to let auto-detection decide when task.modelRouting.enabled is on.",
117
+ ),
106
118
  inheritContext: z
107
119
  .enum(["none", "receipt", "last-turn", "bounded", "full"])
108
120
  .optional()
@@ -250,6 +262,34 @@ export interface ModelSubstitutionWarning {
250
262
  reason: "auth_unavailable" | "assistant_model_mismatch";
251
263
  }
252
264
 
265
+ /**
266
+ * What the model router *asked for* on one child — deliberately not what it ran on.
267
+ *
268
+ * The dispatched value is a fallback chain, so the head can lose to a later
269
+ * candidate when it fails to authenticate. Recording the request separately is
270
+ * what keeps a receipt from claiming a specialty model was used when the spawn
271
+ * actually fell through to the role's baseline. `ModelSubstitutionWarning`
272
+ * covers the disagreement; this covers the intent.
273
+ */
274
+ export interface TaskRoutingAttribution {
275
+ /** Axis the head came from. `baseline` means the router declined to move. */
276
+ source: TaskRoutingSource;
277
+ /** Bounded specialty id, present only when the specialty axis won. */
278
+ specialty?: TaskModelSpecialty;
279
+ /** Tier the classifier settled on. Null for a specialty swap, which has no ladder. */
280
+ tier?: TaskTier;
281
+ /** True when the caller declared the specialty on the spawn; no classifier ran. */
282
+ declared: boolean;
283
+ /** False for ordinary LLM backends, which return no probabilities at all. */
284
+ calibrated: boolean;
285
+ /** Probability. Present only when `calibrated` is true — never synthesised. */
286
+ confidence?: number;
287
+ /** Ordinal clarity in [0,1], recorded in place of a probability when uncalibrated. */
288
+ ordinalStrength?: number;
289
+ /** Router's own explanation, surfaced verbatim on the receipt. */
290
+ reason: string;
291
+ }
292
+
253
293
  /** Progress tracking for a single agent */
254
294
  export interface AgentProgress {
255
295
  index: number;
@@ -283,6 +323,8 @@ export interface AgentProgress {
283
323
  durationMs: number;
284
324
  modelOverride?: string | string[];
285
325
  modelSubstitutionWarning?: ModelSubstitutionWarning;
326
+ /** What the router asked for on this child. See {@link TaskRoutingAttribution}. */
327
+ routing?: TaskRoutingAttribution;
286
328
  /** Data extracted by registered subprocess tool handlers (keyed by tool name) */
287
329
  extractedToolData?: Record<string, unknown[]>;
288
330
  /**
@@ -345,6 +387,8 @@ export interface SingleResult {
345
387
  /** Model's context window in tokens, when known. */
346
388
  contextWindow?: number;
347
389
  modelOverride?: string | string[];
390
+ /** What the router asked for on this child. See {@link TaskRoutingAttribution}. */
391
+ routing?: TaskRoutingAttribution;
348
392
  modelSubstitutionWarning?: ModelSubstitutionWarning;
349
393
  error?: string;
350
394
  aborted?: boolean;