@sayknow-cli/coding-agent 0.5.23 → 0.5.25
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md
CHANGED
|
@@ -2421,6 +2421,56 @@ export declare const SETTINGS_SCHEMA: {
|
|
|
2421
2421
|
readonly description: "Add a model-backed second stage to workflow routing. The keyword table already runs on every turn and costs nothing; this handles the phrasings it cannot express, which is most wording that is not a literal match. Costs one small model call, and only on turns the keyword table did not already answer. Any failure falls back to keyword-only behaviour, but a successful answer can also select a different workflow than the deep-interview ambiguity detector would have.";
|
|
2422
2422
|
};
|
|
2423
2423
|
};
|
|
2424
|
+
/**
|
|
2425
|
+
* Per-spawn model routing for subagents.
|
|
2426
|
+
*
|
|
2427
|
+
* A role's configured model is a standing guess about the average task that
|
|
2428
|
+
* role gets; this moves it when a particular assignment is clearly harder or
|
|
2429
|
+
* clearly more mechanical. Subagents only — routing the main loop's model
|
|
2430
|
+
* mid-session invalidates the prompt cache, which on a long context costs
|
|
2431
|
+
* more than the cheaper tier saves.
|
|
2432
|
+
*
|
|
2433
|
+
* Needs `decisions.enabled` and at least two tiers configured. Without both
|
|
2434
|
+
* it never fires and the configured role models are used unchanged.
|
|
2435
|
+
*/
|
|
2436
|
+
readonly "task.modelRouting.enabled": {
|
|
2437
|
+
readonly type: "boolean";
|
|
2438
|
+
readonly default: false;
|
|
2439
|
+
readonly ui: {
|
|
2440
|
+
readonly tab: "tasks";
|
|
2441
|
+
readonly label: "Route subagent models per task";
|
|
2442
|
+
readonly description: "Ask a cheap model how demanding each subagent assignment is, and move that spawn to a cheaper or stronger model. Needs typed decisions on and the tier models below set. Moving to a cheaper model requires more confidence than moving to a stronger one, because being wrong about it costs a retry.";
|
|
2443
|
+
};
|
|
2444
|
+
};
|
|
2445
|
+
/** Cheapest tier. Mechanical, local, single-file work. */
|
|
2446
|
+
readonly "task.modelRouting.fastModel": {
|
|
2447
|
+
readonly type: "string";
|
|
2448
|
+
readonly default: "";
|
|
2449
|
+
};
|
|
2450
|
+
/** Middle tier. Ordinary engineering against an existing pattern. */
|
|
2451
|
+
readonly "task.modelRouting.balancedModel": {
|
|
2452
|
+
readonly type: "string";
|
|
2453
|
+
readonly default: "";
|
|
2454
|
+
};
|
|
2455
|
+
/** Most capable tier. Unclear cause, cross-cutting design, hard to undo. */
|
|
2456
|
+
readonly "task.modelRouting.deepModel": {
|
|
2457
|
+
readonly type: "string";
|
|
2458
|
+
readonly default: "";
|
|
2459
|
+
};
|
|
2460
|
+
/**
|
|
2461
|
+
* Frontend planning model — the **domain** axis.
|
|
2462
|
+
*
|
|
2463
|
+
* When a planning role (planner/architect) is handed work that clearly reads
|
|
2464
|
+
* as frontend/UI, it takes this model instead of the ladder's answer. A
|
|
2465
|
+
* design-strong model is not "more capable" than a code-strong one; this is a
|
|
2466
|
+
* lateral swap. Unset disables it entirely, and implementation roles are
|
|
2467
|
+
* never swapped by domain — the user asked for the design model to plan the
|
|
2468
|
+
* frontend, not to write it.
|
|
2469
|
+
*/
|
|
2470
|
+
readonly "task.modelRouting.frontendModel": {
|
|
2471
|
+
readonly type: "string";
|
|
2472
|
+
readonly default: "";
|
|
2473
|
+
};
|
|
2424
2474
|
readonly "ttsr.enabled": {
|
|
2425
2475
|
readonly type: "boolean";
|
|
2426
2476
|
readonly default: true;
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
import type { DecisionService } from "./index";
|
|
2
|
+
/** Ordered cheapest to most capable. The order *is* the policy's direction. */
|
|
3
|
+
export declare const TASK_TIERS: readonly ["fast", "balanced", "deep"];
|
|
4
|
+
export type TaskTier = (typeof TASK_TIERS)[number];
|
|
5
|
+
export interface TaskTierModels {
|
|
6
|
+
fast?: string;
|
|
7
|
+
balanced?: string;
|
|
8
|
+
deep?: string;
|
|
9
|
+
}
|
|
10
|
+
export interface TaskRoutingPolicy {
|
|
11
|
+
tiers: TaskTierModels;
|
|
12
|
+
/**
|
|
13
|
+
* Bar to move to a more capable model. Being wrong costs money.
|
|
14
|
+
*/
|
|
15
|
+
minUpgradeConfidence: number;
|
|
16
|
+
/**
|
|
17
|
+
* Bar to move to a cheaper model. Being wrong means real work handled by a
|
|
18
|
+
* model too small for it, which is discovered late and costs a retry — so
|
|
19
|
+
* this bar sits higher than the upgrade bar on purpose.
|
|
20
|
+
*/
|
|
21
|
+
minDowngradeConfidence: number;
|
|
22
|
+
/**
|
|
23
|
+
* Model for frontend planning, when the assignment reads as frontend work.
|
|
24
|
+
*
|
|
25
|
+
* This is the **domain** axis, not a rung on the ladder: a design-strong model
|
|
26
|
+
* is not "better" than a code-strong one, it is a different specialty. Only
|
|
27
|
+
* planning roles ever take it, and only laterally — the implementation roles
|
|
28
|
+
* stay on the difficulty ladder.
|
|
29
|
+
*/
|
|
30
|
+
frontendModel?: string;
|
|
31
|
+
/**
|
|
32
|
+
* Bar for the lateral swap above. Directional bars do not apply here because
|
|
33
|
+
* neither direction is "spending more": being wrong either way costs quality,
|
|
34
|
+
* symmetrically, so one bar is the whole story.
|
|
35
|
+
*/
|
|
36
|
+
minDomainConfidence: number;
|
|
37
|
+
}
|
|
38
|
+
export declare const DEFAULT_TASK_ROUTING_POLICY: Omit<TaskRoutingPolicy, "tiers">;
|
|
39
|
+
/**
|
|
40
|
+
* Roles whose output is a plan or a design review.
|
|
41
|
+
*
|
|
42
|
+
* These are the only roles the domain swap applies to — the user's intent is
|
|
43
|
+
* "a design-strong model *plans* the frontend; implementation stays where it
|
|
44
|
+
* is". Executor keeps the difficulty ladder regardless of domain.
|
|
45
|
+
*/
|
|
46
|
+
export declare const PLANNING_ROLES: ReadonlySet<string>;
|
|
47
|
+
export interface TaskRoutingRequest {
|
|
48
|
+
agentName: string;
|
|
49
|
+
/** The assignment text the subagent will act on. */
|
|
50
|
+
assignment: string;
|
|
51
|
+
/** Whatever the role is configured to use today, used as the direction baseline. */
|
|
52
|
+
currentModel: string | undefined;
|
|
53
|
+
signal?: AbortSignal;
|
|
54
|
+
}
|
|
55
|
+
export interface TaskRoutingResult {
|
|
56
|
+
model: string;
|
|
57
|
+
/** Null when the move was a domain swap — that axis has no ladder. */
|
|
58
|
+
tier: TaskTier | null;
|
|
59
|
+
reason: string;
|
|
60
|
+
}
|
|
61
|
+
/**
|
|
62
|
+
* Decide the model for one subagent spawn, or null to leave the configured one alone.
|
|
63
|
+
*
|
|
64
|
+
* Every failure path returns null: no tiers configured, decisions disabled, no
|
|
65
|
+
* backend, a timeout, an answer outside the enum. A subagent that runs on its
|
|
66
|
+
* configured model is the status quo, and the status quo is always acceptable.
|
|
67
|
+
*/
|
|
68
|
+
export declare function routeTaskModel(service: DecisionService, policy: TaskRoutingPolicy, request: TaskRoutingRequest): Promise<TaskRoutingResult | null>;
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@sayknow-cli/coding-agent",
|
|
4
|
-
"version": "0.5.
|
|
4
|
+
"version": "0.5.25",
|
|
5
5
|
"description": "Sayknow-CLI CLI with read, bash, edit, write tools and session management",
|
|
6
6
|
"homepage": "https://sayknow-cli.com",
|
|
7
7
|
"author": "jaybeyond",
|
|
@@ -54,12 +54,12 @@
|
|
|
54
54
|
"@agentclientprotocol/sdk": "1.3.0",
|
|
55
55
|
"@babel/parser": "^7.29.3",
|
|
56
56
|
"@mozilla/readability": "^0.6.0",
|
|
57
|
-
"@sayknow-cli/stats": "0.5.
|
|
58
|
-
"@sayknow-cli/agent-core": "0.5.
|
|
59
|
-
"@sayknow-cli/ai": "0.5.
|
|
60
|
-
"@sayknow-cli/natives": "0.5.
|
|
61
|
-
"@sayknow-cli/tui": "0.5.
|
|
62
|
-
"@sayknow-cli/utils": "0.5.
|
|
57
|
+
"@sayknow-cli/stats": "0.5.25",
|
|
58
|
+
"@sayknow-cli/agent-core": "0.5.25",
|
|
59
|
+
"@sayknow-cli/ai": "0.5.25",
|
|
60
|
+
"@sayknow-cli/natives": "0.5.25",
|
|
61
|
+
"@sayknow-cli/tui": "0.5.25",
|
|
62
|
+
"@sayknow-cli/utils": "0.5.25",
|
|
63
63
|
"@puppeteer/browsers": "^2.13.0",
|
|
64
64
|
"@types/turndown": "5.0.6",
|
|
65
65
|
"@xterm/headless": "^6.0.0",
|
|
@@ -2011,6 +2011,46 @@ export const SETTINGS_SCHEMA = {
|
|
|
2011
2011
|
},
|
|
2012
2012
|
},
|
|
2013
2013
|
|
|
2014
|
+
/**
|
|
2015
|
+
* Per-spawn model routing for subagents.
|
|
2016
|
+
*
|
|
2017
|
+
* A role's configured model is a standing guess about the average task that
|
|
2018
|
+
* role gets; this moves it when a particular assignment is clearly harder or
|
|
2019
|
+
* clearly more mechanical. Subagents only — routing the main loop's model
|
|
2020
|
+
* mid-session invalidates the prompt cache, which on a long context costs
|
|
2021
|
+
* more than the cheaper tier saves.
|
|
2022
|
+
*
|
|
2023
|
+
* Needs `decisions.enabled` and at least two tiers configured. Without both
|
|
2024
|
+
* it never fires and the configured role models are used unchanged.
|
|
2025
|
+
*/
|
|
2026
|
+
"task.modelRouting.enabled": {
|
|
2027
|
+
type: "boolean",
|
|
2028
|
+
default: false,
|
|
2029
|
+
ui: {
|
|
2030
|
+
tab: "tasks",
|
|
2031
|
+
label: "Route subagent models per task",
|
|
2032
|
+
description:
|
|
2033
|
+
"Ask a cheap model how demanding each subagent assignment is, and move that spawn to a cheaper or stronger model. Needs typed decisions on and the tier models below set. Moving to a cheaper model requires more confidence than moving to a stronger one, because being wrong about it costs a retry.",
|
|
2034
|
+
},
|
|
2035
|
+
},
|
|
2036
|
+
/** Cheapest tier. Mechanical, local, single-file work. */
|
|
2037
|
+
"task.modelRouting.fastModel": { type: "string", default: "" },
|
|
2038
|
+
/** Middle tier. Ordinary engineering against an existing pattern. */
|
|
2039
|
+
"task.modelRouting.balancedModel": { type: "string", default: "" },
|
|
2040
|
+
/** Most capable tier. Unclear cause, cross-cutting design, hard to undo. */
|
|
2041
|
+
"task.modelRouting.deepModel": { type: "string", default: "" },
|
|
2042
|
+
/**
|
|
2043
|
+
* Frontend planning model — the **domain** axis.
|
|
2044
|
+
*
|
|
2045
|
+
* When a planning role (planner/architect) is handed work that clearly reads
|
|
2046
|
+
* as frontend/UI, it takes this model instead of the ladder's answer. A
|
|
2047
|
+
* design-strong model is not "more capable" than a code-strong one; this is a
|
|
2048
|
+
* lateral swap. Unset disables it entirely, and implementation roles are
|
|
2049
|
+
* never swapped by domain — the user asked for the design model to plan the
|
|
2050
|
+
* frontend, not to write it.
|
|
2051
|
+
*/
|
|
2052
|
+
"task.modelRouting.frontendModel": { type: "string", default: "" },
|
|
2053
|
+
|
|
2014
2054
|
// TTSR
|
|
2015
2055
|
"ttsr.enabled": {
|
|
2016
2056
|
type: "boolean",
|
|
@@ -0,0 +1,258 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Pick the model a subagent runs on from the work it was handed.
|
|
3
|
+
*
|
|
4
|
+
* A role's configured model is a standing guess about the *average* task that
|
|
5
|
+
* role gets. It cannot be right for every one: the same executor is handed both
|
|
6
|
+
* a one-line rename and a migration across twelve files. This asks about the
|
|
7
|
+
* actual assignment and moves the model when the answer is clear enough.
|
|
8
|
+
*
|
|
9
|
+
* Only subagents. The main loop's model is deliberately out of scope — changing
|
|
10
|
+
* it mid-session invalidates the prompt cache, and on a long context re-caching
|
|
11
|
+
* routinely costs more than the cheaper tier saves. A subagent starts with its
|
|
12
|
+
* own context, so there is nothing to invalidate.
|
|
13
|
+
*/
|
|
14
|
+
import { logger } from "@sayknow-cli/utils";
|
|
15
|
+
import type { DecisionService } from "./index";
|
|
16
|
+
import type { Question } from "./types";
|
|
17
|
+
|
|
18
|
+
/** Ordered cheapest to most capable. The order *is* the policy's direction. */
|
|
19
|
+
export const TASK_TIERS = ["fast", "balanced", "deep"] as const;
|
|
20
|
+
export type TaskTier = (typeof TASK_TIERS)[number];
|
|
21
|
+
|
|
22
|
+
export interface TaskTierModels {
|
|
23
|
+
fast?: string;
|
|
24
|
+
balanced?: string;
|
|
25
|
+
deep?: string;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
export interface TaskRoutingPolicy {
|
|
29
|
+
tiers: TaskTierModels;
|
|
30
|
+
/**
|
|
31
|
+
* Bar to move to a more capable model. Being wrong costs money.
|
|
32
|
+
*/
|
|
33
|
+
minUpgradeConfidence: number;
|
|
34
|
+
/**
|
|
35
|
+
* Bar to move to a cheaper model. Being wrong means real work handled by a
|
|
36
|
+
* model too small for it, which is discovered late and costs a retry — so
|
|
37
|
+
* this bar sits higher than the upgrade bar on purpose.
|
|
38
|
+
*/
|
|
39
|
+
minDowngradeConfidence: number;
|
|
40
|
+
/**
|
|
41
|
+
* Model for frontend planning, when the assignment reads as frontend work.
|
|
42
|
+
*
|
|
43
|
+
* This is the **domain** axis, not a rung on the ladder: a design-strong model
|
|
44
|
+
* is not "better" than a code-strong one, it is a different specialty. Only
|
|
45
|
+
* planning roles ever take it, and only laterally — the implementation roles
|
|
46
|
+
* stay on the difficulty ladder.
|
|
47
|
+
*/
|
|
48
|
+
frontendModel?: string;
|
|
49
|
+
/**
|
|
50
|
+
* Bar for the lateral swap above. Directional bars do not apply here because
|
|
51
|
+
* neither direction is "spending more": being wrong either way costs quality,
|
|
52
|
+
* symmetrically, so one bar is the whole story.
|
|
53
|
+
*/
|
|
54
|
+
minDomainConfidence: number;
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
export const DEFAULT_TASK_ROUTING_POLICY: Omit<TaskRoutingPolicy, "tiers"> = {
|
|
58
|
+
// Deliberately higher than the reference implementation's 0.3/0.6. That one
|
|
59
|
+
// assumes a frontier default with a cheap tier to fall to, so "up" is the
|
|
60
|
+
// rare move. Here the configured role models are already chosen per role, so
|
|
61
|
+
// overriding one needs a stronger signal in either direction.
|
|
62
|
+
minUpgradeConfidence: 0.5,
|
|
63
|
+
minDowngradeConfidence: 0.75,
|
|
64
|
+
minDomainConfidence: 0.6,
|
|
65
|
+
};
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* Roles whose output is a plan or a design review.
|
|
69
|
+
*
|
|
70
|
+
* These are the only roles the domain swap applies to — the user's intent is
|
|
71
|
+
* "a design-strong model *plans* the frontend; implementation stays where it
|
|
72
|
+
* is". Executor keeps the difficulty ladder regardless of domain.
|
|
73
|
+
*/
|
|
74
|
+
export const PLANNING_ROLES: ReadonlySet<string> = new Set(["planner", "architect"]);
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* The questions describe the *work*, never a model name.
|
|
78
|
+
*
|
|
79
|
+
* Naming models in the criteria would bind the classifier to one lineup and
|
|
80
|
+
* make every model swap a prompt change. It also invites the model to reason
|
|
81
|
+
* about price, which is not what it is good at.
|
|
82
|
+
*/
|
|
83
|
+
function buildQuestions(): Record<string, Question> {
|
|
84
|
+
return {
|
|
85
|
+
tier: {
|
|
86
|
+
type: "choice",
|
|
87
|
+
instructions: "How demanding is this assignment?",
|
|
88
|
+
criteria: {
|
|
89
|
+
fast: "Mechanical and local. A rename, a typo, a one-file edit, running a command and reporting what it printed.",
|
|
90
|
+
balanced: "Ordinary engineering. Several files, an existing pattern to follow, normal debugging.",
|
|
91
|
+
deep: "Hard or high-stakes. Unclear cause, cross-cutting design, subtle correctness, or work that is hard to undo.",
|
|
92
|
+
},
|
|
93
|
+
},
|
|
94
|
+
risky: {
|
|
95
|
+
type: "noul",
|
|
96
|
+
instructions:
|
|
97
|
+
"Does this assignment touch production, money, credentials, published releases, or state that cannot be undone?",
|
|
98
|
+
},
|
|
99
|
+
domain: {
|
|
100
|
+
type: "noul",
|
|
101
|
+
instructions:
|
|
102
|
+
"Is this assignment frontend/UI work — interfaces, components, visual design, styling or interaction — rather than data, APIs, infrastructure or business logic?",
|
|
103
|
+
},
|
|
104
|
+
};
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
export interface TaskRoutingRequest {
|
|
108
|
+
agentName: string;
|
|
109
|
+
/** The assignment text the subagent will act on. */
|
|
110
|
+
assignment: string;
|
|
111
|
+
/** Whatever the role is configured to use today, used as the direction baseline. */
|
|
112
|
+
currentModel: string | undefined;
|
|
113
|
+
signal?: AbortSignal;
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
export interface TaskRoutingResult {
|
|
117
|
+
model: string;
|
|
118
|
+
/** Null when the move was a domain swap — that axis has no ladder. */
|
|
119
|
+
tier: TaskTier | null;
|
|
120
|
+
reason: string;
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/** Where a concrete model id sits in the ladder, or null when it is not one of ours. */
|
|
124
|
+
function rankOf(model: string | undefined, tiers: TaskTierModels): number | null {
|
|
125
|
+
if (!model) return null;
|
|
126
|
+
const index = TASK_TIERS.findIndex(tier => tiers[tier] && matchesModel(tiers[tier] as string, model));
|
|
127
|
+
return index === -1 ? null : index;
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
/**
|
|
131
|
+
* Compare a configured tier model against the role's current selector.
|
|
132
|
+
*
|
|
133
|
+
* Selectors carry a thinking suffix (`provider/id:high`) that the tier table
|
|
134
|
+
* may or may not repeat, so compare the part before it.
|
|
135
|
+
*/
|
|
136
|
+
function matchesModel(a: string, b: string): boolean {
|
|
137
|
+
const base = (value: string) => value.split(":")[0]?.trim().toLowerCase() ?? "";
|
|
138
|
+
return base(a) === base(b);
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
/**
|
|
142
|
+
* Is a move from `current` to `wanted` allowed at this confidence?
|
|
143
|
+
*
|
|
144
|
+
* A backend that cannot report a calibrated confidence may only move a request
|
|
145
|
+
* **up**. Spending less on an unmeasured hunch is the bad trade: the upgrade's
|
|
146
|
+
* worst case is an overpriced answer, the downgrade's is a wrong one.
|
|
147
|
+
*/
|
|
148
|
+
function allowed(
|
|
149
|
+
wanted: number,
|
|
150
|
+
current: number | null,
|
|
151
|
+
confidence: number | undefined,
|
|
152
|
+
calibrated: boolean,
|
|
153
|
+
policy: TaskRoutingPolicy,
|
|
154
|
+
): boolean {
|
|
155
|
+
if (current !== null && wanted === current) return false;
|
|
156
|
+
const isDowngrade = current !== null && wanted < current;
|
|
157
|
+
if (!calibrated || confidence === undefined) return !isDowngrade;
|
|
158
|
+
return confidence >= (isDowngrade ? policy.minDowngradeConfidence : policy.minUpgradeConfidence);
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
/**
|
|
162
|
+
* Decide the model for one subagent spawn, or null to leave the configured one alone.
|
|
163
|
+
*
|
|
164
|
+
* Every failure path returns null: no tiers configured, decisions disabled, no
|
|
165
|
+
* backend, a timeout, an answer outside the enum. A subagent that runs on its
|
|
166
|
+
* configured model is the status quo, and the status quo is always acceptable.
|
|
167
|
+
*/
|
|
168
|
+
export async function routeTaskModel(
|
|
169
|
+
service: DecisionService,
|
|
170
|
+
policy: TaskRoutingPolicy,
|
|
171
|
+
request: TaskRoutingRequest,
|
|
172
|
+
): Promise<TaskRoutingResult | null> {
|
|
173
|
+
const configured = TASK_TIERS.filter(tier => policy.tiers[tier]);
|
|
174
|
+
const frontendModel = policy.frontendModel?.trim() || undefined;
|
|
175
|
+
// Neither axis has anything to move on: no ladder and no domain model.
|
|
176
|
+
if (configured.length < 2 && !frontendModel) return null;
|
|
177
|
+
|
|
178
|
+
const assignment = request.assignment.trim();
|
|
179
|
+
if (assignment.length < 24) return null;
|
|
180
|
+
|
|
181
|
+
const result = await service.decide({
|
|
182
|
+
state: `Agent: ${request.agentName}\n\nAssignment:\n${assignment.slice(0, 4_000)}`,
|
|
183
|
+
questions: buildQuestions(),
|
|
184
|
+
signal: request.signal,
|
|
185
|
+
});
|
|
186
|
+
if (!result) return null;
|
|
187
|
+
|
|
188
|
+
// --- Domain axis: lateral swap for planning roles ---
|
|
189
|
+
//
|
|
190
|
+
// A design-strong model is not "more capable" than a code-strong one, so this
|
|
191
|
+
// is not a rung on the ladder and the directional bars do not apply. When the
|
|
192
|
+
// assignment clearly reads as frontend work and a frontend model is
|
|
193
|
+
// configured, planning roles take it — that is the whole of the user's
|
|
194
|
+
// intent: the design model *plans* the frontend, implementation stays put.
|
|
195
|
+
// When the swap fires, the difficulty ladder is skipped entirely for this
|
|
196
|
+
// spawn; for planning, design judgment is the point, not raw capability.
|
|
197
|
+
const domainAnswer = result.answers.domain;
|
|
198
|
+
if (
|
|
199
|
+
frontendModel &&
|
|
200
|
+
PLANNING_ROLES.has(request.agentName) &&
|
|
201
|
+
domainAnswer?.type === "noul" &&
|
|
202
|
+
domainAnswer.noul >= policy.minDomainConfidence
|
|
203
|
+
) {
|
|
204
|
+
if (request.currentModel && matchesModel(frontendModel, request.currentModel)) return null;
|
|
205
|
+
logger.debug("decisions/task-routing: routed", {
|
|
206
|
+
agent: request.agentName,
|
|
207
|
+
model: frontendModel,
|
|
208
|
+
reason: `frontend (domain ${domainAnswer.noul.toFixed(2)})`,
|
|
209
|
+
});
|
|
210
|
+
return {
|
|
211
|
+
model: frontendModel,
|
|
212
|
+
tier: null,
|
|
213
|
+
reason: `frontend (domain ${domainAnswer.noul.toFixed(2)})`,
|
|
214
|
+
};
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
// --- Difficulty axis: the ladder ---
|
|
218
|
+
if (configured.length < 2) return null;
|
|
219
|
+
const answer = result.answers.tier;
|
|
220
|
+
if (answer?.type !== "choice") return null;
|
|
221
|
+
let tier = TASK_TIERS.find(candidate => candidate === answer.choice);
|
|
222
|
+
if (!tier) return null;
|
|
223
|
+
|
|
224
|
+
// Work that cannot be undone takes the most capable tier available and skips
|
|
225
|
+
// the confidence bars — this one is not a confidence question. It may only
|
|
226
|
+
// ever raise the tier, never lower it, or "this is risky" would end up
|
|
227
|
+
// *downgrading* an assignment already running deep.
|
|
228
|
+
const riskAnswer = result.answers.risky;
|
|
229
|
+
const forcedByRisk = riskAnswer?.type === "noul" && riskAnswer.noul > 0.7;
|
|
230
|
+
if (forcedByRisk) {
|
|
231
|
+
const deepest = configured[configured.length - 1] as TaskTier;
|
|
232
|
+
const currentRank = rankOf(request.currentModel, policy.tiers);
|
|
233
|
+
const wantedRank = Math.max(TASK_TIERS.indexOf(deepest), currentRank ?? 0);
|
|
234
|
+
tier = TASK_TIERS[wantedRank] as TaskTier;
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
const model = policy.tiers[tier];
|
|
238
|
+
if (!model || (request.currentModel && matchesModel(model, request.currentModel))) return null;
|
|
239
|
+
|
|
240
|
+
const currentRank = rankOf(request.currentModel, policy.tiers);
|
|
241
|
+
const wantedRank = TASK_TIERS.indexOf(tier);
|
|
242
|
+
if (!forcedByRisk && !allowed(wantedRank, currentRank, answer.confidence, result.calibrated, policy)) {
|
|
243
|
+
logger.debug("decisions/task-routing: below the bar, keeping the configured model", {
|
|
244
|
+
agent: request.agentName,
|
|
245
|
+
wanted: tier,
|
|
246
|
+
current: request.currentModel,
|
|
247
|
+
confidence: answer.confidence,
|
|
248
|
+
calibrated: result.calibrated,
|
|
249
|
+
});
|
|
250
|
+
return null;
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
const reason = forcedByRisk
|
|
254
|
+
? `${tier}, forced by risk`
|
|
255
|
+
: `${tier} (confidence ${answer.confidence?.toFixed(2) ?? "n/d"})`;
|
|
256
|
+
logger.debug("decisions/task-routing: routed", { agent: request.agentName, model, reason });
|
|
257
|
+
return { model, tier, reason };
|
|
258
|
+
}
|
package/src/task/index.ts
CHANGED
|
@@ -17,7 +17,7 @@ import * as os from "node:os";
|
|
|
17
17
|
import path from "node:path";
|
|
18
18
|
import type { AgentTool, AgentToolResult, AgentToolUpdateCallback } from "@sayknow-cli/agent-core";
|
|
19
19
|
import type { Model, Usage } from "@sayknow-cli/ai";
|
|
20
|
-
import { $pickenv, prompt, Snowflake } from "@sayknow-cli/utils";
|
|
20
|
+
import { $pickenv, logger, prompt, Snowflake } from "@sayknow-cli/utils";
|
|
21
21
|
import type { ToolSession } from "..";
|
|
22
22
|
import { AsyncJobManager, OwnerSubagentShutdownError, type ResumeRunner } from "../async";
|
|
23
23
|
import { resolveAgentModelPatterns } from "../config/model-resolver";
|
|
@@ -461,6 +461,62 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
|
|
|
461
461
|
return this.session.settings.get("task.simple");
|
|
462
462
|
}
|
|
463
463
|
|
|
464
|
+
/**
|
|
465
|
+
* Pick a model for this spawn from the assignment, or null to keep the configured one.
|
|
466
|
+
*
|
|
467
|
+
* Opt-in twice over: `decisions.enabled` must be on *and* at least two tier
|
|
468
|
+
* models configured. That double gate is deliberate — a user who set explicit
|
|
469
|
+
* per-role models chose them on purpose, and silently overriding those from a
|
|
470
|
+
* classifier would be a worse default than doing nothing.
|
|
471
|
+
*
|
|
472
|
+
* One decision per spawn, not per task: every task in a call runs on the same
|
|
473
|
+
* agent and the same model, so asking per task would pay N times for a value
|
|
474
|
+
* that can only be set once.
|
|
475
|
+
*/
|
|
476
|
+
async #routeSpawnModel(
|
|
477
|
+
agentName: string,
|
|
478
|
+
tasks: ReadonlyArray<{ description?: string; assignment?: string }> | undefined,
|
|
479
|
+
currentModel: string | readonly string[] | undefined,
|
|
480
|
+
): Promise<string | undefined> {
|
|
481
|
+
if (!this.session.settings.get("task.modelRouting.enabled")) return undefined;
|
|
482
|
+
const tiers = {
|
|
483
|
+
fast: this.session.settings.get("task.modelRouting.fastModel") || undefined,
|
|
484
|
+
balanced: this.session.settings.get("task.modelRouting.balancedModel") || undefined,
|
|
485
|
+
deep: this.session.settings.get("task.modelRouting.deepModel") || undefined,
|
|
486
|
+
};
|
|
487
|
+
const frontendModel = this.session.settings.get("task.modelRouting.frontendModel") || undefined;
|
|
488
|
+
if (Object.values(tiers).filter(Boolean).length < 2 && !frontendModel) return undefined;
|
|
489
|
+
|
|
490
|
+
const assignment = (tasks ?? [])
|
|
491
|
+
.map(task => [task.description, task.assignment].filter(Boolean).join("\n"))
|
|
492
|
+
.filter(Boolean)
|
|
493
|
+
.join("\n\n");
|
|
494
|
+
if (!assignment) return undefined;
|
|
495
|
+
|
|
496
|
+
try {
|
|
497
|
+
const { createDecisionService } = await import("../decisions");
|
|
498
|
+
const { DEFAULT_TASK_ROUTING_POLICY, routeTaskModel } = await import("../decisions/task-routing");
|
|
499
|
+
const registry = this.session.modelRegistry;
|
|
500
|
+
if (!registry) return undefined;
|
|
501
|
+
const routed = await routeTaskModel(
|
|
502
|
+
createDecisionService({ registry, settings: this.session.settings, enabled: true }),
|
|
503
|
+
{ ...DEFAULT_TASK_ROUTING_POLICY, tiers, frontendModel },
|
|
504
|
+
// A role may be configured with a fallback chain; the first entry is what it
|
|
505
|
+
// actually runs on, so that is the baseline the direction is measured from.
|
|
506
|
+
{
|
|
507
|
+
agentName,
|
|
508
|
+
assignment,
|
|
509
|
+
currentModel: Array.isArray(currentModel) ? currentModel[0] : currentModel,
|
|
510
|
+
},
|
|
511
|
+
);
|
|
512
|
+
return routed?.model;
|
|
513
|
+
} catch (error) {
|
|
514
|
+
// Routing is an optimisation. A failure here must never stop a spawn.
|
|
515
|
+
logger.debug("task: spawn model routing failed", { agent: agentName, error: String(error) });
|
|
516
|
+
return undefined;
|
|
517
|
+
}
|
|
518
|
+
}
|
|
519
|
+
|
|
464
520
|
/**
|
|
465
521
|
* Create a TaskTool instance.
|
|
466
522
|
* Repository authority is captured from session cwd *before* agent discovery so
|
|
@@ -1114,9 +1170,14 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
|
|
|
1114
1170
|
// Apply per-agent model override from settings (highest priority)
|
|
1115
1171
|
const agentModelOverrides = this.session.settings.get("task.agentModelOverrides");
|
|
1116
1172
|
const settingsModelOverride = agentModelOverrides[agentName];
|
|
1173
|
+
// Per-spawn routing sits *above* the configured role model but uses it as the
|
|
1174
|
+
// baseline: the decision is "is this particular assignment heavier or lighter
|
|
1175
|
+
// than what this role normally gets", not "pick a model from scratch". Declining
|
|
1176
|
+
// leaves the configured value exactly as it was.
|
|
1177
|
+
const routedModelOverride = await this.#routeSpawnModel(agentName, boundParams.tasks, settingsModelOverride);
|
|
1117
1178
|
const parentActiveModelPattern = this.session.getActiveModelString?.();
|
|
1118
1179
|
const modelOverride = resolveAgentModelPatterns({
|
|
1119
|
-
settingsOverride: settingsModelOverride,
|
|
1180
|
+
settingsOverride: routedModelOverride ?? settingsModelOverride,
|
|
1120
1181
|
agentModel: effectiveAgent.model,
|
|
1121
1182
|
settings: this.session.settings,
|
|
1122
1183
|
activeModelPattern: parentActiveModelPattern,
|