@sayknow-cli/coding-agent 0.5.24 → 0.5.26

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/task/index.ts CHANGED
@@ -21,6 +21,9 @@ import { $pickenv, logger, prompt, Snowflake } from "@sayknow-cli/utils";
21
21
  import type { ToolSession } from "..";
22
22
  import { AsyncJobManager, OwnerSubagentShutdownError, type ResumeRunner } from "../async";
23
23
  import { resolveAgentModelPatterns } from "../config/model-resolver";
24
+ import { normalizeModelSelectorValue } from "../config/model-selector-value";
25
+ import type { TaskModelSpecialty } from "../config/task-model-specialties";
26
+ import type { TaskRoutingResult } from "../decisions/task-routing";
24
27
  import type { Theme } from "../modes/theme/theme";
25
28
  import planModeSubagentPrompt from "../prompts/system/plan-mode-subagent.md" with { type: "text" };
26
29
  import taskDescriptionTemplate from "../prompts/tools/task.md" with { type: "text" };
@@ -37,6 +40,7 @@ import {
37
40
  type SingleResult,
38
41
  type TaskItem,
39
42
  type TaskParams,
43
+ type TaskRoutingAttribution,
40
44
  type TaskToolDetails,
41
45
  type TaskToolSchemaInstance,
42
46
  } from "./types";
@@ -212,6 +216,7 @@ export type {
212
216
  SubagentLifecyclePayload,
213
217
  SubagentProgressPayload,
214
218
  TaskParams,
219
+ TaskRoutingAttribution,
215
220
  TaskToolDetails,
216
221
  } from "./types";
217
222
  export {
@@ -462,53 +467,80 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
462
467
  }
463
468
 
464
469
  /**
465
- * Pick a model for this spawn from the assignment, or null to keep the configured one.
470
+ * Route a child whose caller declared the kind of work.
466
471
  *
467
- * Opt-in twice over: `decisions.enabled` must be on *and* at least two tier
468
- * models configured. That double gate is deliberate — a user who set explicit
469
- * per-role models chose them on purpose, and silently overriding those from a
470
- * classifier would be a worse default than doing nothing.
472
+ * Deterministic by design: no classifier, no `task.modelRouting.enabled`
473
+ * gate. The user assigned a model to this specialty in `/model`; if the caller
474
+ * says the work is that specialty, the child runs on that model and leaves it
475
+ * only when it errors — the child session's fallback chain advances to the
476
+ * role baseline composed behind it.
477
+ */
478
+ async #routeDeclaredSpecialty(
479
+ agentName: string,
480
+ specialty: TaskModelSpecialty,
481
+ baselineChain: readonly string[],
482
+ ): Promise<TaskRoutingResult | undefined> {
483
+ try {
484
+ const { resolveDeclaredSpecialtyRouting } = await import("../decisions/task-routing");
485
+ return (
486
+ resolveDeclaredSpecialtyRouting(this.session.settings, {
487
+ agentName,
488
+ specialty,
489
+ currentModel: baselineChain[0],
490
+ baselineChain,
491
+ }) ?? undefined
492
+ );
493
+ } catch (error) {
494
+ // A declared specialty with nothing behind it still spawns on the role model.
495
+ logger.debug("task: declared specialty routing failed", { agent: agentName, specialty, error: String(error) });
496
+ return undefined;
497
+ }
498
+ }
499
+
500
+ /**
501
+ * Pick a model for one child assignment, or undefined to keep the configured one.
471
502
  *
472
- * One decision per spawn, not per task: every task in a call runs on the same
473
- * agent and the same model, so asking per task would pay N times for a value
474
- * that can only be set once.
503
+ * This is the *guessing* half: a classifier reads the assignment and decides.
504
+ * Opt-in twice over: `task.modelRouting.enabled` must be on *and* an axis must
505
+ * have something to move on. That double gate is deliberate — a user who set
506
+ * explicit per-role models chose them on purpose, and silently overriding those
507
+ * from a classifier would be a worse default than doing nothing. A caller that
508
+ * *knows* the kind of work declares it instead (`#routeDeclaredSpecialty`).
509
+ *
510
+ * One decision per *child*, not per call. Every task in a call shares an agent,
511
+ * but not a workload: a batch can hold an implementation slice and a test slice,
512
+ * and a single joined classification would have to answer for both at once.
475
513
  */
476
514
  async #routeSpawnModel(
477
515
  agentName: string,
478
- tasks: ReadonlyArray<{ description?: string; assignment?: string }> | undefined,
479
- currentModel: string | readonly string[] | undefined,
480
- ): Promise<string | undefined> {
516
+ assignment: string | undefined,
517
+ baselineChain: readonly string[],
518
+ ): Promise<TaskRoutingResult | undefined> {
519
+ // Cheap guard before the dynamic import so a disabled feature costs nothing.
481
520
  if (!this.session.settings.get("task.modelRouting.enabled")) return undefined;
482
- const tiers = {
483
- fast: this.session.settings.get("task.modelRouting.fastModel") || undefined,
484
- balanced: this.session.settings.get("task.modelRouting.balancedModel") || undefined,
485
- deep: this.session.settings.get("task.modelRouting.deepModel") || undefined,
486
- };
487
- if (Object.values(tiers).filter(Boolean).length < 2) return undefined;
488
-
489
- const assignment = (tasks ?? [])
490
- .map(task => [task.description, task.assignment].filter(Boolean).join("\n"))
491
- .filter(Boolean)
492
- .join("\n\n");
493
- if (!assignment) return undefined;
521
+ const trimmed = assignment?.trim();
522
+ if (!trimmed) return undefined;
494
523
 
495
524
  try {
496
525
  const { createDecisionService } = await import("../decisions");
497
- const { DEFAULT_TASK_ROUTING_POLICY, routeTaskModel } = await import("../decisions/task-routing");
526
+ const { buildTaskRoutingPolicyFromSettings, routeTaskModel } = await import("../decisions/task-routing");
527
+ const policy = buildTaskRoutingPolicyFromSettings(this.session.settings);
528
+ if (!policy) return undefined;
498
529
  const registry = this.session.modelRegistry;
499
530
  if (!registry) return undefined;
500
531
  const routed = await routeTaskModel(
501
532
  createDecisionService({ registry, settings: this.session.settings, enabled: true }),
502
- { ...DEFAULT_TASK_ROUTING_POLICY, tiers },
503
- // A role may be configured with a fallback chain; the first entry is what it
504
- // actually runs on, so that is the baseline the direction is measured from.
533
+ policy,
505
534
  {
506
535
  agentName,
507
- assignment,
508
- currentModel: Array.isArray(currentModel) ? currentModel[0] : currentModel,
536
+ assignment: trimmed,
537
+ // A role may be configured with a fallback chain; the first entry is what it
538
+ // actually runs on, so that is the baseline the direction is measured from.
539
+ currentModel: baselineChain[0],
540
+ baselineChain,
509
541
  },
510
542
  );
511
- return routed?.model;
543
+ return routed ?? undefined;
512
544
  } catch (error) {
513
545
  // Routing is an optimisation. A failure here must never stop a spawn.
514
546
  logger.debug("task: spawn model routing failed", { agent: agentName, error: String(error) });
@@ -1169,19 +1201,18 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1169
1201
  // Apply per-agent model override from settings (highest priority)
1170
1202
  const agentModelOverrides = this.session.settings.get("task.agentModelOverrides");
1171
1203
  const settingsModelOverride = agentModelOverrides[agentName];
1172
- // Per-spawn routing sits *above* the configured role model but uses it as the
1173
- // baseline: the decision is "is this particular assignment heavier or lighter
1174
- // than what this role normally gets", not "pick a model from scratch". Declining
1175
- // leaves the configured value exactly as it was.
1176
- const routedModelOverride = await this.#routeSpawnModel(agentName, boundParams.tasks, settingsModelOverride);
1204
+ // The role's own resolved chain. Per-child routing composes *in front of* this
1205
+ // and never replaces it: a specialty or tier that cannot be authenticated must
1206
+ // still fall through to the model the role would have used anyway.
1177
1207
  const parentActiveModelPattern = this.session.getActiveModelString?.();
1178
1208
  const modelOverride = resolveAgentModelPatterns({
1179
- settingsOverride: routedModelOverride ?? settingsModelOverride,
1209
+ settingsOverride: settingsModelOverride,
1180
1210
  agentModel: effectiveAgent.model,
1181
1211
  settings: this.session.settings,
1182
1212
  activeModelPattern: parentActiveModelPattern,
1183
1213
  fallbackModelPattern: this.session.getModelString?.(),
1184
1214
  });
1215
+ const baselineChain = normalizeModelSelectorValue(modelOverride);
1185
1216
  const thinkingLevelOverride = effectiveAgent.thinkingLevel;
1186
1217
 
1187
1218
  // Output schema priority: task call > agent frontmatter > inherited parent session.
@@ -1473,6 +1504,28 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1473
1504
  sessionFile?: string | null;
1474
1505
  },
1475
1506
  ) => {
1507
+ // Route THIS child. A declared specialty is deterministic; otherwise a
1508
+ // batch shares an agent but not a workload, so a joined classification
1509
+ // would have to answer for an implementation slice and a test slice at
1510
+ // the same time.
1511
+ const routed = task.specialty
1512
+ ? await this.#routeDeclaredSpecialty(agentName, task.specialty, baselineChain)
1513
+ : await this.#routeSpawnModel(agentName, task.assignment, baselineChain);
1514
+ const taskModelOverride = routed ? routed.candidates.map(candidate => candidate.selector) : modelOverride;
1515
+ // Requested, not effective: the chain above can still fall through to a
1516
+ // later candidate, so this records intent and nothing more.
1517
+ const taskRouting: TaskRoutingAttribution | undefined = routed
1518
+ ? {
1519
+ source: routed.requestedSource,
1520
+ specialty: routed.requestedSpecialty,
1521
+ tier: routed.requestedTier,
1522
+ declared: routed.declared,
1523
+ calibrated: routed.calibrated,
1524
+ confidence: routed.confidence,
1525
+ ordinalStrength: routed.ordinalStrength,
1526
+ reason: routed.reason,
1527
+ }
1528
+ : undefined;
1476
1529
  const forkContextSeed = prebuiltForkContextSeeds?.get(task.id) ?? (await buildForkContextSeed(task));
1477
1530
  const forkContext = requestsForkContext(task)
1478
1531
  ? { mode: task.inheritContext, clonedTokens: forkContextSeed?.metadata.approximateTokens ?? 0 }
@@ -1515,7 +1568,7 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1515
1568
  resumeMessage: overrides?.resumeMessage ?? executionOverrides?.resumeMessage,
1516
1569
  subagentId: task.id,
1517
1570
  taskDepth,
1518
- modelOverride,
1571
+ modelOverride: taskModelOverride,
1519
1572
  parentActiveModelPattern,
1520
1573
  parentSessionId: this.session.getSessionId?.() ?? undefined,
1521
1574
  thinkingLevel: thinkingLevelOverride,
@@ -1532,6 +1585,8 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1532
1585
  onProgress: progress => {
1533
1586
  progressMap.set(index, {
1534
1587
  ...structuredClone(progress),
1588
+ modelOverride: taskModelOverride,
1589
+ ...(taskRouting ? { routing: taskRouting } : {}),
1535
1590
  });
1536
1591
  AsyncJobManager.instance()?.recordSubagentProgress(task.id, progress);
1537
1592
  emitProgress();
@@ -1588,7 +1643,7 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1588
1643
  resumeMessage: overrides?.resumeMessage ?? executionOverrides?.resumeMessage,
1589
1644
  subagentId: task.id,
1590
1645
  taskDepth,
1591
- modelOverride,
1646
+ modelOverride: taskModelOverride,
1592
1647
  parentActiveModelPattern,
1593
1648
  parentSessionId: this.session.getSessionId?.() ?? undefined,
1594
1649
  thinkingLevel: thinkingLevelOverride,
@@ -1605,6 +1660,8 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1605
1660
  onProgress: progress => {
1606
1661
  progressMap.set(index, {
1607
1662
  ...structuredClone(progress),
1663
+ modelOverride: taskModelOverride,
1664
+ ...(taskRouting ? { routing: taskRouting } : {}),
1608
1665
  });
1609
1666
  AsyncJobManager.instance()?.recordSubagentProgress(task.id, progress);
1610
1667
  emitProgress();
@@ -1628,6 +1685,7 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1628
1685
  const resultWithForkContext = {
1629
1686
  ...result,
1630
1687
  ...(forkContext ? { forkContext } : {}),
1688
+ ...(taskRouting ? { routing: taskRouting } : {}),
1631
1689
  forkContextAdvisory,
1632
1690
  repositoryBinding: publicRepositoryBinding(taskRepositoryBinding),
1633
1691
  };
@@ -1708,7 +1766,8 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
1708
1766
  truncated: false,
1709
1767
  durationMs: Date.now() - taskStart,
1710
1768
  tokens: 0,
1711
- modelOverride,
1769
+ modelOverride: taskModelOverride,
1770
+ ...(taskRouting ? { routing: taskRouting } : {}),
1712
1771
  forkContext,
1713
1772
  error: message,
1714
1773
  };
@@ -28,6 +28,8 @@ export interface TaskResultReceipt {
28
28
  contextTokens?: number;
29
29
  contextWindow?: number;
30
30
  modelOverride?: string | string[];
31
+ /** What the router asked for, kept separate from what the spawn ran on. */
32
+ routing?: SingleResult["routing"];
31
33
  modelSubstitutionWarning?: SingleResult["modelSubstitutionWarning"];
32
34
  usage?: SingleResult["usage"];
33
35
  cost?: number;
@@ -245,6 +247,7 @@ export function buildTaskReceipt(raw: SingleResult): TaskResultReceipt {
245
247
  contextTokens: raw.contextTokens,
246
248
  contextWindow: raw.contextWindow,
247
249
  modelOverride: raw.modelOverride,
250
+ routing: raw.routing,
248
251
  modelSubstitutionWarning: raw.modelSubstitutionWarning,
249
252
  usage: raw.usage,
250
253
  cost: raw.usage?.cost.total,
package/src/task/types.ts CHANGED
@@ -2,6 +2,12 @@ import type { ThinkingLevel } from "@sayknow-cli/agent-core";
2
2
  import type { Usage } from "@sayknow-cli/ai";
3
3
  import { $env } from "@sayknow-cli/utils";
4
4
  import * as z from "zod/v4";
5
+ import {
6
+ TASK_MODEL_SPECIALTY_IDS,
7
+ type TaskModelSpecialty,
8
+ type TaskRoutingSource,
9
+ } from "../config/task-model-specialties";
10
+ import type { TaskTier } from "../decisions/task-routing";
5
11
  import { isValidTaskId, TASK_ID_DESCRIPTION } from "./id";
6
12
  import type { TaskResultReceipt } from "./receipt";
7
13
  import type { SpawnRoiReconciliation } from "./roi-reconciliation";
@@ -103,6 +109,12 @@ const createTaskItemSchema = (_contextEnabled: boolean) =>
103
109
  .describe(
104
110
  "typed executor mode: default keeps ordinary executor behavior; ultragoal-red-team injects the Ultragoal QA/red-team prompt fragment. Prefer this over free-form assignment text (#2698).",
105
111
  ),
112
+ specialty: z
113
+ .enum(TASK_MODEL_SPECIALTY_IDS)
114
+ .optional()
115
+ .describe(
116
+ "kind of work, so the model the user assigned to it under /model runs this child: backendArchitecture, frontendDesign, implementation, testing, or review. Declaring it is deterministic — no classifier, no routing switch; the child only leaves that model if it errors (429, 5xx, auth, quota). Omit to keep the agent's role model, or to let auto-detection decide when task.modelRouting.enabled is on.",
117
+ ),
106
118
  inheritContext: z
107
119
  .enum(["none", "receipt", "last-turn", "bounded", "full"])
108
120
  .optional()
@@ -250,6 +262,34 @@ export interface ModelSubstitutionWarning {
250
262
  reason: "auth_unavailable" | "assistant_model_mismatch";
251
263
  }
252
264
 
265
+ /**
266
+ * What the model router *asked for* on one child — deliberately not what it ran on.
267
+ *
268
+ * The dispatched value is a fallback chain, so the head can lose to a later
269
+ * candidate when it fails to authenticate. Recording the request separately is
270
+ * what keeps a receipt from claiming a specialty model was used when the spawn
271
+ * actually fell through to the role's baseline. `ModelSubstitutionWarning`
272
+ * covers the disagreement; this covers the intent.
273
+ */
274
+ export interface TaskRoutingAttribution {
275
+ /** Axis the head came from. `baseline` means the router declined to move. */
276
+ source: TaskRoutingSource;
277
+ /** Bounded specialty id, present only when the specialty axis won. */
278
+ specialty?: TaskModelSpecialty;
279
+ /** Tier the classifier settled on. Null for a specialty swap, which has no ladder. */
280
+ tier?: TaskTier;
281
+ /** True when the caller declared the specialty on the spawn; no classifier ran. */
282
+ declared: boolean;
283
+ /** False for ordinary LLM backends, which return no probabilities at all. */
284
+ calibrated: boolean;
285
+ /** Probability. Present only when `calibrated` is true — never synthesised. */
286
+ confidence?: number;
287
+ /** Ordinal clarity in [0,1], recorded in place of a probability when uncalibrated. */
288
+ ordinalStrength?: number;
289
+ /** Router's own explanation, surfaced verbatim on the receipt. */
290
+ reason: string;
291
+ }
292
+
253
293
  /** Progress tracking for a single agent */
254
294
  export interface AgentProgress {
255
295
  index: number;
@@ -283,6 +323,8 @@ export interface AgentProgress {
283
323
  durationMs: number;
284
324
  modelOverride?: string | string[];
285
325
  modelSubstitutionWarning?: ModelSubstitutionWarning;
326
+ /** What the router asked for on this child. See {@link TaskRoutingAttribution}. */
327
+ routing?: TaskRoutingAttribution;
286
328
  /** Data extracted by registered subprocess tool handlers (keyed by tool name) */
287
329
  extractedToolData?: Record<string, unknown[]>;
288
330
  /**
@@ -345,6 +387,8 @@ export interface SingleResult {
345
387
  /** Model's context window in tokens, when known. */
346
388
  contextWindow?: number;
347
389
  modelOverride?: string | string[];
390
+ /** What the router asked for on this child. See {@link TaskRoutingAttribution}. */
391
+ routing?: TaskRoutingAttribution;
348
392
  modelSubstitutionWarning?: ModelSubstitutionWarning;
349
393
  error?: string;
350
394
  aborted?: boolean;