@sayknow-cli/coding-agent 0.5.25 → 0.5.26
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +13 -1
- package/dist/types/config/settings-schema.d.ts +26 -0
- package/dist/types/config/task-model-specialties.d.ts +55 -0
- package/dist/types/decisions/task-routing.d.ts +96 -11
- package/dist/types/i18n/messages/en.d.ts +15 -0
- package/dist/types/lsp/index.d.ts +1 -1
- package/dist/types/lsp/types.d.ts +1 -1
- package/dist/types/modes/components/model-selector.d.ts +11 -0
- package/dist/types/task/index.d.ts +1 -1
- package/dist/types/task/receipt.d.ts +2 -0
- package/dist/types/task/types.d.ts +114 -18
- package/dist/types/tools/subagent.d.ts +2 -2
- package/package.json +7 -7
- package/src/config/settings-schema.ts +37 -7
- package/src/config/task-model-specialties.ts +131 -0
- package/src/decisions/task-routing.ts +382 -66
- package/src/i18n/messages/de.ts +16 -0
- package/src/i18n/messages/en.ts +16 -0
- package/src/i18n/messages/es.ts +16 -0
- package/src/i18n/messages/fr.ts +16 -0
- package/src/i18n/messages/ja.ts +16 -0
- package/src/i18n/messages/ko.ts +16 -0
- package/src/i18n/messages/zh.ts +16 -0
- package/src/internal-urls/docs-index.generated.ts +1 -1
- package/src/main.ts +1 -1
- package/src/modes/components/model-selector.ts +275 -34
- package/src/modes/controllers/selector-controller.ts +50 -2
- package/src/modes/shared/agent-wire/command-dispatch.ts +1 -1
- package/src/prompts/tools/task.md +1 -0
- package/src/slash-commands/builtin-registry.ts +11 -9
- package/src/task/index.ts +98 -40
- package/src/task/receipt.ts +3 -0
- package/src/task/types.ts +44 -0
|
@@ -6,14 +6,36 @@
|
|
|
6
6
|
* a one-line rename and a migration across twelve files. This asks about the
|
|
7
7
|
* actual assignment and moves the model when the answer is clear enough.
|
|
8
8
|
*
|
|
9
|
+
* Two axes, asked in one call:
|
|
10
|
+
*
|
|
11
|
+
* - **Difficulty** — the fast/balanced/deep ladder. Directional, so moving down
|
|
12
|
+
* costs more confidence than moving up.
|
|
13
|
+
* - **Specialty** — the kind of work (backend architecture, frontend design,
|
|
14
|
+
* implementation, test work, review). Lateral, so a single bar applies.
|
|
15
|
+
*
|
|
9
16
|
* Only subagents. The main loop's model is deliberately out of scope — changing
|
|
10
17
|
* it mid-session invalidates the prompt cache, and on a long context re-caching
|
|
11
18
|
* routinely costs more than the cheaper tier saves. A subagent starts with its
|
|
12
19
|
* own context, so there is nothing to invalidate.
|
|
13
20
|
*/
|
|
14
21
|
import { logger } from "@sayknow-cli/utils";
|
|
22
|
+
import type { ModelSelectorValue } from "../config/model-selector-value";
|
|
23
|
+
import type { Settings } from "../config/settings";
|
|
24
|
+
import {
|
|
25
|
+
dedupeRoutingCandidates,
|
|
26
|
+
isTaskModelSpecialty,
|
|
27
|
+
specialtySelectorHead,
|
|
28
|
+
specialtySupportsRole,
|
|
29
|
+
TASK_MODEL_SPECIALTY_IDS,
|
|
30
|
+
TASK_MODEL_SPECIALTY_NONE,
|
|
31
|
+
TASK_MODEL_SPECIALTY_ROLES,
|
|
32
|
+
type TaskModelSpecialty,
|
|
33
|
+
type TaskRoutingCandidate,
|
|
34
|
+
type TaskRoutingSource,
|
|
35
|
+
toRoutingCandidates,
|
|
36
|
+
} from "../config/task-model-specialties";
|
|
15
37
|
import type { DecisionService } from "./index";
|
|
16
|
-
import type { Question } from "./types";
|
|
38
|
+
import type { Answer, Question } from "./types";
|
|
17
39
|
|
|
18
40
|
/** Ordered cheapest to most capable. The order *is* the policy's direction. */
|
|
19
41
|
export const TASK_TIERS = ["fast", "balanced", "deep"] as const;
|
|
@@ -40,18 +62,34 @@ export interface TaskRoutingPolicy {
|
|
|
40
62
|
/**
|
|
41
63
|
* Model for frontend planning, when the assignment reads as frontend work.
|
|
42
64
|
*
|
|
43
|
-
*
|
|
44
|
-
*
|
|
45
|
-
* planning roles ever take it, and only laterally — the implementation roles
|
|
46
|
-
* stay on the difficulty ladder.
|
|
65
|
+
* Superseded by `specialtyModels.frontendDesign`; kept as the fallback source
|
|
66
|
+
* so an existing configuration keeps working untouched until it is migrated.
|
|
47
67
|
*/
|
|
48
68
|
frontendModel?: string;
|
|
49
69
|
/**
|
|
50
|
-
*
|
|
51
|
-
*
|
|
52
|
-
*
|
|
70
|
+
* Per-specialty models — the work-kind axis.
|
|
71
|
+
*
|
|
72
|
+
* Absent entries inherit the role's own chain, which is why an unset specialty
|
|
73
|
+
* is not an error and does not suppress the difficulty ladder.
|
|
74
|
+
*/
|
|
75
|
+
specialtyModels?: Partial<Record<TaskModelSpecialty, ModelSelectorValue>>;
|
|
76
|
+
/**
|
|
77
|
+
* Bar for a lateral swap on a **calibrated** backend. Directional bars do not
|
|
78
|
+
* apply here because neither direction is "spending more": being wrong either
|
|
79
|
+
* way costs quality, symmetrically, so one bar is the whole story.
|
|
53
80
|
*/
|
|
54
81
|
minDomainConfidence: number;
|
|
82
|
+
/**
|
|
83
|
+
* Bar for a lateral swap on an **uncalibrated** backend.
|
|
84
|
+
*
|
|
85
|
+
* The ordinary logged-in model cannot report a probability, and asking it for
|
|
86
|
+
* one measurably degrades the answer, so its choice carries no `confidence`.
|
|
87
|
+
* What it *can* report is an ordinal strength. Requiring a high ordinal is not
|
|
88
|
+
* the same guarantee as a calibrated threshold, and the result is recorded as
|
|
89
|
+
* uncalibrated — but refusing to route at all would make a user's explicit
|
|
90
|
+
* specialty selection silently inert on the default backend.
|
|
91
|
+
*/
|
|
92
|
+
minSpecialtyOrdinal: number;
|
|
55
93
|
}
|
|
56
94
|
|
|
57
95
|
export const DEFAULT_TASK_ROUTING_POLICY: Omit<TaskRoutingPolicy, "tiers"> = {
|
|
@@ -62,16 +100,55 @@ export const DEFAULT_TASK_ROUTING_POLICY: Omit<TaskRoutingPolicy, "tiers"> = {
|
|
|
62
100
|
minUpgradeConfidence: 0.5,
|
|
63
101
|
minDowngradeConfidence: 0.75,
|
|
64
102
|
minDomainConfidence: 0.6,
|
|
103
|
+
// One step above "probably yes" on the ordinal ladder the uncalibrated backend
|
|
104
|
+
// emits, so "unclear" and "probably yes" both decline.
|
|
105
|
+
minSpecialtyOrdinal: 0.75,
|
|
65
106
|
};
|
|
66
107
|
|
|
108
|
+
/**
|
|
109
|
+
* The settings surface this module reads.
|
|
110
|
+
*
|
|
111
|
+
* Narrowed to `get` so the router cannot quietly start writing settings, and so
|
|
112
|
+
* a caller only has to supply a reader rather than a whole `Settings` instance.
|
|
113
|
+
*/
|
|
114
|
+
export type TaskRoutingSettingsReader = Pick<Settings, "get">;
|
|
115
|
+
|
|
116
|
+
/** True when the value names at least one model rather than being blank. */
|
|
117
|
+
function hasConfiguredModel(value: ModelSelectorValue | undefined): boolean {
|
|
118
|
+
if (Array.isArray(value)) return value.some(entry => entry.trim().length > 0);
|
|
119
|
+
return typeof value === "string" && value.trim().length > 0;
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
/**
|
|
123
|
+
* Build the routing policy from settings, or null when routing must not run.
|
|
124
|
+
*
|
|
125
|
+
* Null is returned for two distinct reasons that both mean "leave the configured
|
|
126
|
+
* model alone": the feature is off, or it is on but nothing is configured to
|
|
127
|
+
* route *to*. A lone tier is not an axis — there is nowhere to move from it —
|
|
128
|
+
* so two tiers is the floor unless a specialty or the legacy frontend model
|
|
129
|
+
* supplies a lateral target instead.
|
|
130
|
+
*/
|
|
131
|
+
export function buildTaskRoutingPolicyFromSettings(settings: TaskRoutingSettingsReader): TaskRoutingPolicy | null {
|
|
132
|
+
if (!settings.get("task.modelRouting.enabled")) return null;
|
|
133
|
+
const tiers: TaskTierModels = {
|
|
134
|
+
fast: settings.get("task.modelRouting.fastModel") || undefined,
|
|
135
|
+
balanced: settings.get("task.modelRouting.balancedModel") || undefined,
|
|
136
|
+
deep: settings.get("task.modelRouting.deepModel") || undefined,
|
|
137
|
+
};
|
|
138
|
+
const frontendModel = settings.get("task.modelRouting.frontendModel") || undefined;
|
|
139
|
+
const specialtyModels = settings.get("task.modelRouting.specialtyModels") ?? {};
|
|
140
|
+
const hasSpecialty = TASK_MODEL_SPECIALTY_IDS.some(id => hasConfiguredModel(specialtyModels[id]));
|
|
141
|
+
const tierCount = Object.values(tiers).filter(Boolean).length;
|
|
142
|
+
if (tierCount < 2 && !frontendModel && !hasSpecialty) return null;
|
|
143
|
+
return { ...DEFAULT_TASK_ROUTING_POLICY, tiers, frontendModel, specialtyModels };
|
|
144
|
+
}
|
|
145
|
+
|
|
67
146
|
/**
|
|
68
147
|
* Roles whose output is a plan or a design review.
|
|
69
148
|
*
|
|
70
|
-
*
|
|
71
|
-
* "a design-strong model *plans* the frontend; implementation stays where it
|
|
72
|
-
* is". Executor keeps the difficulty ladder regardless of domain.
|
|
149
|
+
* Derived from the specialty compatibility map so the two never drift apart.
|
|
73
150
|
*/
|
|
74
|
-
export const PLANNING_ROLES: ReadonlySet<string> = new Set(
|
|
151
|
+
export const PLANNING_ROLES: ReadonlySet<string> = new Set(TASK_MODEL_SPECIALTY_ROLES.frontendDesign);
|
|
75
152
|
|
|
76
153
|
/**
|
|
77
154
|
* The questions describe the *work*, never a model name.
|
|
@@ -96,10 +173,26 @@ function buildQuestions(): Record<string, Question> {
|
|
|
96
173
|
instructions:
|
|
97
174
|
"Does this assignment touch production, money, credentials, published releases, or state that cannot be undone?",
|
|
98
175
|
},
|
|
99
|
-
|
|
176
|
+
specialty: {
|
|
177
|
+
type: "choice",
|
|
178
|
+
instructions: "Which kind of work is this assignment?",
|
|
179
|
+
criteria: {
|
|
180
|
+
backendArchitecture:
|
|
181
|
+
"Designing or reviewing backend structure: APIs, data models, services, storage, infrastructure.",
|
|
182
|
+
frontendDesign:
|
|
183
|
+
"Designing or reviewing an interface: layout, visual design, interaction, components, styling.",
|
|
184
|
+
implementation:
|
|
185
|
+
"Writing or changing code against an established pattern, where the approach is already settled.",
|
|
186
|
+
testing: "Designing, writing, debugging or running tests and verification.",
|
|
187
|
+
review: "Judging existing work for correctness, regressions, or maintainability.",
|
|
188
|
+
[TASK_MODEL_SPECIALTY_NONE]:
|
|
189
|
+
"General, mixed, or unclear work that does not sit in exactly one of the categories above.",
|
|
190
|
+
},
|
|
191
|
+
},
|
|
192
|
+
specialtyClear: {
|
|
100
193
|
type: "noul",
|
|
101
194
|
instructions:
|
|
102
|
-
"
|
|
195
|
+
"Does this assignment clearly belong to exactly one of those kinds of work, rather than spanning several or being unclear?",
|
|
103
196
|
},
|
|
104
197
|
};
|
|
105
198
|
}
|
|
@@ -110,14 +203,111 @@ export interface TaskRoutingRequest {
|
|
|
110
203
|
assignment: string;
|
|
111
204
|
/** Whatever the role is configured to use today, used as the direction baseline. */
|
|
112
205
|
currentModel: string | undefined;
|
|
206
|
+
/**
|
|
207
|
+
* The role's fully resolved chain, in order. The composed candidate list ends
|
|
208
|
+
* with this, so a specialty or tier that cannot be authenticated falls through
|
|
209
|
+
* to the model the role would have used anyway.
|
|
210
|
+
*/
|
|
211
|
+
baselineChain?: readonly string[];
|
|
113
212
|
signal?: AbortSignal;
|
|
114
213
|
}
|
|
115
214
|
|
|
116
215
|
export interface TaskRoutingResult {
|
|
216
|
+
/** Head of the composed chain — what the spawn runs on if it authenticates. */
|
|
117
217
|
model: string;
|
|
118
|
-
/** Null when the move was a
|
|
218
|
+
/** Null when the move was a specialty swap — that axis has no ladder. */
|
|
119
219
|
tier: TaskTier | null;
|
|
120
220
|
reason: string;
|
|
221
|
+
/** Ordered, provenance-tagged chain for the existing auth-aware resolver. */
|
|
222
|
+
candidates: TaskRoutingCandidate[];
|
|
223
|
+
/** What the classifier asked for. The *effective* source is only known after resolution. */
|
|
224
|
+
requestedSource: TaskRoutingSource;
|
|
225
|
+
requestedSpecialty?: TaskModelSpecialty;
|
|
226
|
+
requestedTier?: TaskTier;
|
|
227
|
+
/**
|
|
228
|
+
* True when the caller named the specialty on the spawn itself. No classifier
|
|
229
|
+
* ran, so `calibrated`/`confidence`/`ordinalStrength` describe nothing here.
|
|
230
|
+
*/
|
|
231
|
+
declared: boolean;
|
|
232
|
+
/** False means `ordinalStrength` ranks, and no probability was available. */
|
|
233
|
+
calibrated: boolean;
|
|
234
|
+
confidence?: number;
|
|
235
|
+
ordinalStrength?: number;
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
export interface DeclaredSpecialtyRequest {
|
|
239
|
+
agentName: string;
|
|
240
|
+
specialty: TaskModelSpecialty;
|
|
241
|
+
/** Whatever the role is configured to use today. */
|
|
242
|
+
currentModel: string | undefined;
|
|
243
|
+
/** The role's fully resolved chain; always the tail so a dead specialty model falls through. */
|
|
244
|
+
baselineChain?: readonly string[];
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
/**
|
|
248
|
+
* Route a spawn whose caller *declared* the kind of work.
|
|
249
|
+
*
|
|
250
|
+
* This is the deterministic half of the specialty axis. Nothing here asks a
|
|
251
|
+
* classifier, reads `task.modelRouting.enabled`, or applies a confidence bar:
|
|
252
|
+
* the user put a model on this specialty in `/model`, the caller says this is
|
|
253
|
+
* that work, and the only remaining reason not to run on it is that it fails —
|
|
254
|
+
* which the child session's fallback chain handles at the transport boundary
|
|
255
|
+
* (429, 5xx, auth, quota) by advancing to the role's baseline behind it.
|
|
256
|
+
*
|
|
257
|
+
* Role eligibility is deliberately not checked. The menu groups specialties
|
|
258
|
+
* under the roles that usually do that work, but the setting is one flat map:
|
|
259
|
+
* a frontend model the user chose for design is the same frontend model they
|
|
260
|
+
* expect when the *implementation* of that frontend is delegated. Refusing
|
|
261
|
+
* here would make "frontend uses a different model" false for exactly the
|
|
262
|
+
* spawns where it matters most.
|
|
263
|
+
*
|
|
264
|
+
* Returns null only when nothing is configured for the specialty, or when the
|
|
265
|
+
* configured model is already what the role would run on anyway.
|
|
266
|
+
*/
|
|
267
|
+
export function resolveDeclaredSpecialtyRouting(
|
|
268
|
+
settings: TaskRoutingSettingsReader,
|
|
269
|
+
request: DeclaredSpecialtyRequest,
|
|
270
|
+
): TaskRoutingResult | null {
|
|
271
|
+
const specialtyModels = settings.get("task.modelRouting.specialtyModels") ?? {};
|
|
272
|
+
const configured = specialtyModels[request.specialty];
|
|
273
|
+
let value: ModelSelectorValue | undefined;
|
|
274
|
+
let source: Extract<TaskRoutingSource, "specialty" | "legacy-frontend"> = "specialty";
|
|
275
|
+
if (hasConfiguredModel(configured)) {
|
|
276
|
+
value = configured;
|
|
277
|
+
} else if (request.specialty === "frontendDesign") {
|
|
278
|
+
// Legacy compatibility: the old single frontend selector still answers for
|
|
279
|
+
// frontend design when no explicit entry has replaced it.
|
|
280
|
+
const legacy = settings.get("task.modelRouting.frontendModel")?.trim();
|
|
281
|
+
if (legacy) {
|
|
282
|
+
value = legacy;
|
|
283
|
+
source = "legacy-frontend";
|
|
284
|
+
}
|
|
285
|
+
}
|
|
286
|
+
if (value === undefined) return null;
|
|
287
|
+
|
|
288
|
+
const specialtyCandidates = toRoutingCandidates(value, source, { specialty: request.specialty });
|
|
289
|
+
const head = specialtyCandidates[0];
|
|
290
|
+
if (!head) return null;
|
|
291
|
+
// Already there: the declared model is the role's own. Reporting a route would
|
|
292
|
+
// claim a swap that never happened.
|
|
293
|
+
if (request.currentModel && matchesModel(head.selector, request.currentModel)) return null;
|
|
294
|
+
|
|
295
|
+
const candidates = dedupeRoutingCandidates([specialtyCandidates, baselineCandidates(request)]);
|
|
296
|
+
const effectiveHead = candidates[0];
|
|
297
|
+
if (!effectiveHead || effectiveHead.source === "baseline") return null;
|
|
298
|
+
const label = source === "legacy-frontend" ? "frontendDesign (legacy selector)" : request.specialty;
|
|
299
|
+
const reason = `${label}, declared by caller`;
|
|
300
|
+
logger.debug("decisions/task-routing: routed", { agent: request.agentName, model: effectiveHead.selector, reason });
|
|
301
|
+
return {
|
|
302
|
+
model: effectiveHead.selector,
|
|
303
|
+
tier: null,
|
|
304
|
+
reason,
|
|
305
|
+
candidates,
|
|
306
|
+
requestedSource: source,
|
|
307
|
+
requestedSpecialty: request.specialty,
|
|
308
|
+
declared: true,
|
|
309
|
+
calibrated: false,
|
|
310
|
+
};
|
|
121
311
|
}
|
|
122
312
|
|
|
123
313
|
/** Where a concrete model id sits in the ladder, or null when it is not one of ours. */
|
|
@@ -134,8 +324,7 @@ function rankOf(model: string | undefined, tiers: TaskTierModels): number | null
|
|
|
134
324
|
* may or may not repeat, so compare the part before it.
|
|
135
325
|
*/
|
|
136
326
|
function matchesModel(a: string, b: string): boolean {
|
|
137
|
-
|
|
138
|
-
return base(a) === base(b);
|
|
327
|
+
return specialtySelectorHead(a) === specialtySelectorHead(b);
|
|
139
328
|
}
|
|
140
329
|
|
|
141
330
|
/**
|
|
@@ -158,6 +347,68 @@ function allowed(
|
|
|
158
347
|
return confidence >= (isDowngrade ? policy.minDowngradeConfidence : policy.minUpgradeConfidence);
|
|
159
348
|
}
|
|
160
349
|
|
|
350
|
+
/** The role's own chain, used as the tail of every composed candidate list. */
|
|
351
|
+
function baselineCandidates(
|
|
352
|
+
request: Pick<TaskRoutingRequest, "baselineChain" | "currentModel">,
|
|
353
|
+
): TaskRoutingCandidate[] {
|
|
354
|
+
const chain = request.baselineChain?.length ? request.baselineChain : [request.currentModel ?? ""];
|
|
355
|
+
return toRoutingCandidates([...chain], "baseline");
|
|
356
|
+
}
|
|
357
|
+
|
|
358
|
+
interface SpecialtySelection {
|
|
359
|
+
specialty: TaskModelSpecialty;
|
|
360
|
+
source: Extract<TaskRoutingSource, "specialty" | "legacy-frontend">;
|
|
361
|
+
value: ModelSelectorValue;
|
|
362
|
+
ordinalStrength?: number;
|
|
363
|
+
confidence?: number;
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
/**
|
|
367
|
+
* Resolve the specialty axis, or null to leave it to the difficulty ladder.
|
|
368
|
+
*
|
|
369
|
+
* Declines are deliberately quiet and numerous: an unrecognized option, `none`,
|
|
370
|
+
* an incompatible role, no configured model, a below-bar answer, or a missing
|
|
371
|
+
* clarity signal all mean "the user did not ask for this, carry on".
|
|
372
|
+
*/
|
|
373
|
+
function resolveSpecialty(
|
|
374
|
+
policy: TaskRoutingPolicy,
|
|
375
|
+
request: TaskRoutingRequest,
|
|
376
|
+
answers: Record<string, Answer>,
|
|
377
|
+
calibrated: boolean,
|
|
378
|
+
): SpecialtySelection | null {
|
|
379
|
+
const answer = answers.specialty;
|
|
380
|
+
if (answer?.type !== "choice") return null;
|
|
381
|
+
const choice = answer.choice;
|
|
382
|
+
if (typeof choice !== "string" || choice === TASK_MODEL_SPECIALTY_NONE) return null;
|
|
383
|
+
if (!isTaskModelSpecialty(choice)) return null;
|
|
384
|
+
if (!specialtySupportsRole(choice, request.agentName)) return null;
|
|
385
|
+
|
|
386
|
+
const clarity = answers.specialtyClear;
|
|
387
|
+
const ordinalStrength = clarity?.type === "noul" && typeof clarity.noul === "number" ? clarity.noul : undefined;
|
|
388
|
+
const confidence = typeof answer.confidence === "number" ? answer.confidence : undefined;
|
|
389
|
+
|
|
390
|
+
if (calibrated) {
|
|
391
|
+
// A calibrated backend reports a real probability; threshold on it directly.
|
|
392
|
+
if (confidence === undefined || confidence < policy.minDomainConfidence) return null;
|
|
393
|
+
} else {
|
|
394
|
+
// No probability is available. Require an explicit high ordinal instead, and
|
|
395
|
+
// never dress that number up as confidence downstream.
|
|
396
|
+
if (ordinalStrength === undefined || ordinalStrength < policy.minSpecialtyOrdinal) return null;
|
|
397
|
+
}
|
|
398
|
+
|
|
399
|
+
const configured = policy.specialtyModels?.[choice];
|
|
400
|
+
if (configured !== undefined && (typeof configured !== "string" || configured.trim().length > 0)) {
|
|
401
|
+
return { specialty: choice, source: "specialty", value: configured, ordinalStrength, confidence };
|
|
402
|
+
}
|
|
403
|
+
// Legacy compatibility: the old single frontend selector still answers for
|
|
404
|
+
// frontend design when no explicit entry has replaced it.
|
|
405
|
+
const legacy = policy.frontendModel?.trim();
|
|
406
|
+
if (choice === "frontendDesign" && legacy) {
|
|
407
|
+
return { specialty: choice, source: "legacy-frontend", value: legacy, ordinalStrength, confidence };
|
|
408
|
+
}
|
|
409
|
+
return null;
|
|
410
|
+
}
|
|
411
|
+
|
|
161
412
|
/**
|
|
162
413
|
* Decide the model for one subagent spawn, or null to leave the configured one alone.
|
|
163
414
|
*
|
|
@@ -172,8 +423,12 @@ export async function routeTaskModel(
|
|
|
172
423
|
): Promise<TaskRoutingResult | null> {
|
|
173
424
|
const configured = TASK_TIERS.filter(tier => policy.tiers[tier]);
|
|
174
425
|
const frontendModel = policy.frontendModel?.trim() || undefined;
|
|
175
|
-
|
|
176
|
-
|
|
426
|
+
const hasSpecialtyModels = TASK_MODEL_SPECIALTY_IDS.some(id => {
|
|
427
|
+
const value = policy.specialtyModels?.[id];
|
|
428
|
+
return Array.isArray(value) ? value.length > 0 : typeof value === "string" && value.trim().length > 0;
|
|
429
|
+
});
|
|
430
|
+
// No axis has anything to move on: no ladder, no specialty model, no legacy one.
|
|
431
|
+
if (configured.length < 2 && !frontendModel && !hasSpecialtyModels) return null;
|
|
177
432
|
|
|
178
433
|
const assignment = request.assignment.trim();
|
|
179
434
|
if (assignment.length < 24) return null;
|
|
@@ -185,74 +440,135 @@ export async function routeTaskModel(
|
|
|
185
440
|
});
|
|
186
441
|
if (!result) return null;
|
|
187
442
|
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
// A design-strong model is not "more capable" than a code-strong one, so this
|
|
191
|
-
// is not a rung on the ladder and the directional bars do not apply. When the
|
|
192
|
-
// assignment clearly reads as frontend work and a frontend model is
|
|
193
|
-
// configured, planning roles take it — that is the whole of the user's
|
|
194
|
-
// intent: the design model *plans* the frontend, implementation stays put.
|
|
195
|
-
// When the swap fires, the difficulty ladder is skipped entirely for this
|
|
196
|
-
// spawn; for planning, design judgment is the point, not raw capability.
|
|
197
|
-
const domainAnswer = result.answers.domain;
|
|
198
|
-
if (
|
|
199
|
-
frontendModel &&
|
|
200
|
-
PLANNING_ROLES.has(request.agentName) &&
|
|
201
|
-
domainAnswer?.type === "noul" &&
|
|
202
|
-
domainAnswer.noul >= policy.minDomainConfidence
|
|
203
|
-
) {
|
|
204
|
-
if (request.currentModel && matchesModel(frontendModel, request.currentModel)) return null;
|
|
205
|
-
logger.debug("decisions/task-routing: routed", {
|
|
206
|
-
agent: request.agentName,
|
|
207
|
-
model: frontendModel,
|
|
208
|
-
reason: `frontend (domain ${domainAnswer.noul.toFixed(2)})`,
|
|
209
|
-
});
|
|
210
|
-
return {
|
|
211
|
-
model: frontendModel,
|
|
212
|
-
tier: null,
|
|
213
|
-
reason: `frontend (domain ${domainAnswer.noul.toFixed(2)})`,
|
|
214
|
-
};
|
|
215
|
-
}
|
|
216
|
-
|
|
217
|
-
// --- Difficulty axis: the ladder ---
|
|
218
|
-
if (configured.length < 2) return null;
|
|
219
|
-
const answer = result.answers.tier;
|
|
220
|
-
if (answer?.type !== "choice") return null;
|
|
221
|
-
let tier = TASK_TIERS.find(candidate => candidate === answer.choice);
|
|
222
|
-
if (!tier) return null;
|
|
443
|
+
const answers = result.answers;
|
|
444
|
+
const baseline = baselineCandidates(request);
|
|
223
445
|
|
|
446
|
+
// --- Risk floor ---
|
|
447
|
+
//
|
|
224
448
|
// Work that cannot be undone takes the most capable tier available and skips
|
|
225
449
|
// the confidence bars — this one is not a confidence question. It may only
|
|
226
450
|
// ever raise the tier, never lower it, or "this is risky" would end up
|
|
227
|
-
// *downgrading* an assignment already running deep.
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
451
|
+
// *downgrading* an assignment already running deep. It also outranks the
|
|
452
|
+
// specialty axis: a design-strong model is not the safety property being
|
|
453
|
+
// asked for here.
|
|
454
|
+
const riskAnswer = answers.risky;
|
|
455
|
+
const forcedByRisk = riskAnswer?.type === "noul" && typeof riskAnswer.noul === "number" && riskAnswer.noul > 0.7;
|
|
456
|
+
|
|
457
|
+
const tierAnswer = answers.tier;
|
|
458
|
+
const declaredTier =
|
|
459
|
+
tierAnswer?.type === "choice" ? TASK_TIERS.find(candidate => candidate === tierAnswer.choice) : undefined;
|
|
460
|
+
const tierConfidence =
|
|
461
|
+
tierAnswer?.type === "choice" && typeof tierAnswer.confidence === "number" ? tierAnswer.confidence : undefined;
|
|
462
|
+
|
|
463
|
+
if (forcedByRisk && configured.length > 0) {
|
|
231
464
|
const deepest = configured[configured.length - 1] as TaskTier;
|
|
232
465
|
const currentRank = rankOf(request.currentModel, policy.tiers);
|
|
233
466
|
const wantedRank = Math.max(TASK_TIERS.indexOf(deepest), currentRank ?? 0);
|
|
234
|
-
tier = TASK_TIERS[wantedRank] as TaskTier;
|
|
467
|
+
const tier = TASK_TIERS[wantedRank] as TaskTier;
|
|
468
|
+
const model = policy.tiers[tier];
|
|
469
|
+
if (!model || (request.currentModel && matchesModel(model, request.currentModel))) return null;
|
|
470
|
+
const candidates = dedupeRoutingCandidates([toRoutingCandidates(model, "tier", { tier }), baseline]);
|
|
471
|
+
const head = candidates[0];
|
|
472
|
+
if (!head) return null;
|
|
473
|
+
const reason = `${tier}, forced by risk`;
|
|
474
|
+
logger.debug("decisions/task-routing: routed", { agent: request.agentName, model: head.selector, reason });
|
|
475
|
+
return {
|
|
476
|
+
model: head.selector,
|
|
477
|
+
tier,
|
|
478
|
+
reason,
|
|
479
|
+
candidates,
|
|
480
|
+
requestedSource: "tier",
|
|
481
|
+
requestedTier: tier,
|
|
482
|
+
declared: false,
|
|
483
|
+
calibrated: result.calibrated,
|
|
484
|
+
confidence: tierConfidence,
|
|
485
|
+
};
|
|
486
|
+
}
|
|
487
|
+
|
|
488
|
+
// --- Specialty axis: lateral swap for the kind of work ---
|
|
489
|
+
const specialty = resolveSpecialty(policy, request, answers, result.calibrated);
|
|
490
|
+
if (specialty) {
|
|
491
|
+
const specialtyCandidates = toRoutingCandidates(specialty.value, specialty.source, {
|
|
492
|
+
specialty: specialty.specialty,
|
|
493
|
+
});
|
|
494
|
+
const head = specialtyCandidates[0];
|
|
495
|
+
// Already there: nothing to move, and the ladder should not fire either —
|
|
496
|
+
// the user's specialty choice is the standing answer for this work.
|
|
497
|
+
if (head && request.currentModel && matchesModel(head.selector, request.currentModel)) return null;
|
|
498
|
+
if (head) {
|
|
499
|
+
// The tier is still worth composing *behind* the specialty: if the specialty
|
|
500
|
+
// model cannot be authenticated, the difficulty answer is the next best guess.
|
|
501
|
+
const tierSegment =
|
|
502
|
+
declaredTier && policy.tiers[declaredTier]
|
|
503
|
+
? toRoutingCandidates(policy.tiers[declaredTier] as string, "tier", { tier: declaredTier })
|
|
504
|
+
: [];
|
|
505
|
+
const candidates = dedupeRoutingCandidates([specialtyCandidates, tierSegment, baseline]);
|
|
506
|
+
const effectiveHead = candidates[0];
|
|
507
|
+
if (effectiveHead && effectiveHead.source !== "baseline") {
|
|
508
|
+
const strength = result.calibrated
|
|
509
|
+
? `confidence ${specialty.confidence?.toFixed(2) ?? "n/d"}`
|
|
510
|
+
: `clarity ${specialty.ordinalStrength?.toFixed(2) ?? "n/d"}, uncalibrated`;
|
|
511
|
+
const label =
|
|
512
|
+
specialty.source === "legacy-frontend" ? "frontendDesign (legacy selector)" : specialty.specialty;
|
|
513
|
+
const reason = `${label} (${strength})`;
|
|
514
|
+
logger.debug("decisions/task-routing: routed", {
|
|
515
|
+
agent: request.agentName,
|
|
516
|
+
model: effectiveHead.selector,
|
|
517
|
+
reason,
|
|
518
|
+
});
|
|
519
|
+
return {
|
|
520
|
+
model: effectiveHead.selector,
|
|
521
|
+
tier: null,
|
|
522
|
+
reason,
|
|
523
|
+
candidates,
|
|
524
|
+
requestedSource: specialty.source,
|
|
525
|
+
requestedSpecialty: specialty.specialty,
|
|
526
|
+
requestedTier: declaredTier,
|
|
527
|
+
declared: false,
|
|
528
|
+
calibrated: result.calibrated,
|
|
529
|
+
confidence: result.calibrated ? specialty.confidence : undefined,
|
|
530
|
+
ordinalStrength: result.calibrated ? undefined : specialty.ordinalStrength,
|
|
531
|
+
};
|
|
532
|
+
}
|
|
533
|
+
}
|
|
235
534
|
}
|
|
236
535
|
|
|
536
|
+
// --- Difficulty axis: the ladder ---
|
|
537
|
+
if (configured.length < 2) return null;
|
|
538
|
+
if (!declaredTier) return null;
|
|
539
|
+
const tier = declaredTier;
|
|
540
|
+
|
|
237
541
|
const model = policy.tiers[tier];
|
|
238
542
|
if (!model || (request.currentModel && matchesModel(model, request.currentModel))) return null;
|
|
239
543
|
|
|
240
544
|
const currentRank = rankOf(request.currentModel, policy.tiers);
|
|
241
545
|
const wantedRank = TASK_TIERS.indexOf(tier);
|
|
242
|
-
if (!
|
|
546
|
+
if (!allowed(wantedRank, currentRank, tierConfidence, result.calibrated, policy)) {
|
|
243
547
|
logger.debug("decisions/task-routing: below the bar, keeping the configured model", {
|
|
244
548
|
agent: request.agentName,
|
|
245
549
|
wanted: tier,
|
|
246
550
|
current: request.currentModel,
|
|
247
|
-
confidence:
|
|
551
|
+
confidence: tierConfidence,
|
|
552
|
+
declared: false,
|
|
248
553
|
calibrated: result.calibrated,
|
|
249
554
|
});
|
|
250
555
|
return null;
|
|
251
556
|
}
|
|
252
557
|
|
|
253
|
-
const
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
558
|
+
const candidates = dedupeRoutingCandidates([toRoutingCandidates(model, "tier", { tier }), baseline]);
|
|
559
|
+
const head = candidates[0];
|
|
560
|
+
if (!head) return null;
|
|
561
|
+
const reason = `${tier} (confidence ${tierConfidence?.toFixed(2) ?? "n/d"})`;
|
|
562
|
+
logger.debug("decisions/task-routing: routed", { agent: request.agentName, model: head.selector, reason });
|
|
563
|
+
return {
|
|
564
|
+
model: head.selector,
|
|
565
|
+
tier,
|
|
566
|
+
reason,
|
|
567
|
+
candidates,
|
|
568
|
+
requestedSource: "tier",
|
|
569
|
+
requestedTier: tier,
|
|
570
|
+
declared: false,
|
|
571
|
+
calibrated: result.calibrated,
|
|
572
|
+
confidence: tierConfidence,
|
|
573
|
+
};
|
|
258
574
|
}
|
package/src/i18n/messages/de.ts
CHANGED
|
@@ -75,6 +75,22 @@ export const de: Partial<Record<MsgKey, string>> = {
|
|
|
75
75
|
"modelSelector.noMatching": "Keine passenden Modelle.",
|
|
76
76
|
"modelSelector.modelName": "Modellname: {value}",
|
|
77
77
|
"modelSelector.actionFor": "Aktion für: {id}",
|
|
78
|
+
"modelSelector.setAsTarget": "Als {tag} ({name}) festlegen",
|
|
79
|
+
"modelSelector.setForAllRoleAgents": "Fur alle Rollen-Agenten setzen",
|
|
80
|
+
"modelSelector.setForAllTargets": "Fur alle Ziele setzen",
|
|
81
|
+
"modelSelector.hasDetailedUses": "(detaillierte Aufgaben)",
|
|
82
|
+
"modelSelector.detailedUseFor": "Detaillierte Aufgabe fur {target}: {id}",
|
|
83
|
+
"modelSelector.generalRole": "Allgemein (ganze Rolle)",
|
|
84
|
+
"modelSelector.resetSpecialties": "Detaillierte Aufgaben zurucksetzen",
|
|
85
|
+
"modelSelector.specialty.backendArchitecture": "Backend-Architektur",
|
|
86
|
+
"modelSelector.specialty.frontendDesign": "Frontend-Design",
|
|
87
|
+
"modelSelector.specialty.implementation": "Implementierung",
|
|
88
|
+
"modelSelector.specialty.testing": "Tests",
|
|
89
|
+
"modelSelector.specialty.review": "Review",
|
|
90
|
+
"modelSelector.specialtySaved": "{specialty} verwendet {value}.",
|
|
91
|
+
"modelSelector.specialtySavedRoutingOff":
|
|
92
|
+
"{specialty} verwendet {value}, sobald eine Aufgabe diese Arbeit deklariert. Automatische Erkennung aus dem Auftragstext bleibt aus, bis task.modelRouting.enabled aktiv ist.",
|
|
93
|
+
"modelSelector.specialtyCleared": "Detaillierte Aufgaben fur {target} zuruckgesetzt.",
|
|
78
94
|
"modelSelector.reasoningFor": "Reasoning für {target}: {id}",
|
|
79
95
|
"modelSelector.temporaryModel": "temporäres Modell",
|
|
80
96
|
};
|
package/src/i18n/messages/en.ts
CHANGED
|
@@ -95,6 +95,22 @@ export const en = {
|
|
|
95
95
|
"modelSelector.noMatching": "No matching models.",
|
|
96
96
|
"modelSelector.modelName": "Model Name: {value}",
|
|
97
97
|
"modelSelector.actionFor": "Action for: {id}",
|
|
98
|
+
"modelSelector.setAsTarget": "Set as {tag} ({name})",
|
|
99
|
+
"modelSelector.setForAllRoleAgents": "Set for all role agents",
|
|
100
|
+
"modelSelector.setForAllTargets": "Set for all targets",
|
|
101
|
+
"modelSelector.hasDetailedUses": "(detailed uses)",
|
|
102
|
+
"modelSelector.detailedUseFor": "Detailed use for {target}: {id}",
|
|
103
|
+
"modelSelector.generalRole": "General (whole role)",
|
|
104
|
+
"modelSelector.resetSpecialties": "Clear detailed-use overrides",
|
|
105
|
+
"modelSelector.specialty.backendArchitecture": "Backend architecture",
|
|
106
|
+
"modelSelector.specialty.frontendDesign": "Frontend design",
|
|
107
|
+
"modelSelector.specialty.implementation": "Implementation",
|
|
108
|
+
"modelSelector.specialty.testing": "Testing",
|
|
109
|
+
"modelSelector.specialty.review": "Review",
|
|
110
|
+
"modelSelector.specialtySaved": "{specialty} will use {value}.",
|
|
111
|
+
"modelSelector.specialtySavedRoutingOff":
|
|
112
|
+
"{specialty} will use {value} whenever a task declares that work. Auto-detection from assignment text stays off until you enable task.modelRouting.enabled.",
|
|
113
|
+
"modelSelector.specialtyCleared": "Cleared detailed-use overrides for {target}.",
|
|
98
114
|
"modelSelector.reasoningFor": "Reasoning for {target}: {id}",
|
|
99
115
|
"modelSelector.temporaryModel": "temporary model",
|
|
100
116
|
} as const;
|
package/src/i18n/messages/es.ts
CHANGED
|
@@ -75,6 +75,22 @@ export const es: Partial<Record<MsgKey, string>> = {
|
|
|
75
75
|
"modelSelector.noMatching": "No hay modelos coincidentes.",
|
|
76
76
|
"modelSelector.modelName": "Nombre del modelo: {value}",
|
|
77
77
|
"modelSelector.actionFor": "Acción para: {id}",
|
|
78
|
+
"modelSelector.setAsTarget": "Asignar como {tag} ({name})",
|
|
79
|
+
"modelSelector.setForAllRoleAgents": "Asignar a todos los agentes de rol",
|
|
80
|
+
"modelSelector.setForAllTargets": "Asignar a todos los destinos",
|
|
81
|
+
"modelSelector.hasDetailedUses": "(usos detallados)",
|
|
82
|
+
"modelSelector.detailedUseFor": "Uso detallado de {target}: {id}",
|
|
83
|
+
"modelSelector.generalRole": "General (todo el rol)",
|
|
84
|
+
"modelSelector.resetSpecialties": "Borrar los usos detallados",
|
|
85
|
+
"modelSelector.specialty.backendArchitecture": "Arquitectura de backend",
|
|
86
|
+
"modelSelector.specialty.frontendDesign": "Diseño de frontend",
|
|
87
|
+
"modelSelector.specialty.implementation": "Implementación",
|
|
88
|
+
"modelSelector.specialty.testing": "Pruebas",
|
|
89
|
+
"modelSelector.specialty.review": "Revisión",
|
|
90
|
+
"modelSelector.specialtySaved": "{specialty} usará {value}.",
|
|
91
|
+
"modelSelector.specialtySavedRoutingOff":
|
|
92
|
+
"{specialty} usará {value} cuando una tarea declare ese trabajo. La detección automática desde el texto de la tarea sigue desactivada hasta que actives task.modelRouting.enabled.",
|
|
93
|
+
"modelSelector.specialtyCleared": "Se borraron los usos detallados de {target}.",
|
|
78
94
|
"modelSelector.reasoningFor": "Razonamiento para {target}: {id}",
|
|
79
95
|
"modelSelector.temporaryModel": "modelo temporal",
|
|
80
96
|
};
|
package/src/i18n/messages/fr.ts
CHANGED
|
@@ -75,6 +75,22 @@ export const fr: Partial<Record<MsgKey, string>> = {
|
|
|
75
75
|
"modelSelector.noMatching": "Aucun modèle correspondant.",
|
|
76
76
|
"modelSelector.modelName": "Nom du modèle : {value}",
|
|
77
77
|
"modelSelector.actionFor": "Action pour : {id}",
|
|
78
|
+
"modelSelector.setAsTarget": "Définir comme {tag} ({name})",
|
|
79
|
+
"modelSelector.setForAllRoleAgents": "Definir pour tous les agents de role",
|
|
80
|
+
"modelSelector.setForAllTargets": "Definir pour toutes les cibles",
|
|
81
|
+
"modelSelector.hasDetailedUses": "(usages detailles)",
|
|
82
|
+
"modelSelector.detailedUseFor": "Usage detaille pour {target} : {id}",
|
|
83
|
+
"modelSelector.generalRole": "General (tout le role)",
|
|
84
|
+
"modelSelector.resetSpecialties": "Effacer les usages detailles",
|
|
85
|
+
"modelSelector.specialty.backendArchitecture": "Architecture backend",
|
|
86
|
+
"modelSelector.specialty.frontendDesign": "Design frontend",
|
|
87
|
+
"modelSelector.specialty.implementation": "Implementation",
|
|
88
|
+
"modelSelector.specialty.testing": "Tests",
|
|
89
|
+
"modelSelector.specialty.review": "Revue",
|
|
90
|
+
"modelSelector.specialtySaved": "{specialty} utilisera {value}.",
|
|
91
|
+
"modelSelector.specialtySavedRoutingOff":
|
|
92
|
+
"{specialty} utilisera {value} des qu'une tache declare ce travail. La detection automatique depuis le texte reste inactive sans task.modelRouting.enabled.",
|
|
93
|
+
"modelSelector.specialtyCleared": "Usages detailles effaces pour {target}.",
|
|
78
94
|
"modelSelector.reasoningFor": "Raisonnement pour {target} : {id}",
|
|
79
95
|
"modelSelector.temporaryModel": "modèle temporaire",
|
|
80
96
|
};
|