@sayknow-cli/coding-agent 0.5.24 → 0.5.26
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +13 -1
- package/dist/types/config/settings-schema.d.ts +40 -0
- package/dist/types/config/task-model-specialties.d.ts +55 -0
- package/dist/types/decisions/task-routing.d.ts +110 -1
- package/dist/types/i18n/messages/en.d.ts +15 -0
- package/dist/types/lsp/index.d.ts +1 -1
- package/dist/types/lsp/types.d.ts +1 -1
- package/dist/types/modes/components/model-selector.d.ts +11 -0
- package/dist/types/task/index.d.ts +1 -1
- package/dist/types/task/receipt.d.ts +2 -0
- package/dist/types/task/types.d.ts +114 -18
- package/dist/types/tools/subagent.d.ts +2 -2
- package/package.json +7 -7
- package/src/config/settings-schema.ts +48 -7
- package/src/config/task-model-specialties.ts +131 -0
- package/src/decisions/task-routing.ts +401 -23
- package/src/i18n/messages/de.ts +16 -0
- package/src/i18n/messages/en.ts +16 -0
- package/src/i18n/messages/es.ts +16 -0
- package/src/i18n/messages/fr.ts +16 -0
- package/src/i18n/messages/ja.ts +16 -0
- package/src/i18n/messages/ko.ts +16 -0
- package/src/i18n/messages/zh.ts +16 -0
- package/src/internal-urls/docs-index.generated.ts +1 -1
- package/src/main.ts +1 -1
- package/src/modes/components/model-selector.ts +275 -34
- package/src/modes/controllers/selector-controller.ts +50 -2
- package/src/modes/shared/agent-wire/command-dispatch.ts +1 -1
- package/src/prompts/tools/task.md +1 -0
- package/src/slash-commands/builtin-registry.ts +11 -9
- package/src/task/index.ts +98 -39
- package/src/task/receipt.ts +3 -0
- package/src/task/types.ts +44 -0
|
@@ -6,14 +6,36 @@
|
|
|
6
6
|
* a one-line rename and a migration across twelve files. This asks about the
|
|
7
7
|
* actual assignment and moves the model when the answer is clear enough.
|
|
8
8
|
*
|
|
9
|
+
* Two axes, asked in one call:
|
|
10
|
+
*
|
|
11
|
+
* - **Difficulty** — the fast/balanced/deep ladder. Directional, so moving down
|
|
12
|
+
* costs more confidence than moving up.
|
|
13
|
+
* - **Specialty** — the kind of work (backend architecture, frontend design,
|
|
14
|
+
* implementation, test work, review). Lateral, so a single bar applies.
|
|
15
|
+
*
|
|
9
16
|
* Only subagents. The main loop's model is deliberately out of scope — changing
|
|
10
17
|
* it mid-session invalidates the prompt cache, and on a long context re-caching
|
|
11
18
|
* routinely costs more than the cheaper tier saves. A subagent starts with its
|
|
12
19
|
* own context, so there is nothing to invalidate.
|
|
13
20
|
*/
|
|
14
21
|
import { logger } from "@sayknow-cli/utils";
|
|
22
|
+
import type { ModelSelectorValue } from "../config/model-selector-value";
|
|
23
|
+
import type { Settings } from "../config/settings";
|
|
24
|
+
import {
|
|
25
|
+
dedupeRoutingCandidates,
|
|
26
|
+
isTaskModelSpecialty,
|
|
27
|
+
specialtySelectorHead,
|
|
28
|
+
specialtySupportsRole,
|
|
29
|
+
TASK_MODEL_SPECIALTY_IDS,
|
|
30
|
+
TASK_MODEL_SPECIALTY_NONE,
|
|
31
|
+
TASK_MODEL_SPECIALTY_ROLES,
|
|
32
|
+
type TaskModelSpecialty,
|
|
33
|
+
type TaskRoutingCandidate,
|
|
34
|
+
type TaskRoutingSource,
|
|
35
|
+
toRoutingCandidates,
|
|
36
|
+
} from "../config/task-model-specialties";
|
|
15
37
|
import type { DecisionService } from "./index";
|
|
16
|
-
import type { Question } from "./types";
|
|
38
|
+
import type { Answer, Question } from "./types";
|
|
17
39
|
|
|
18
40
|
/** Ordered cheapest to most capable. The order *is* the policy's direction. */
|
|
19
41
|
export const TASK_TIERS = ["fast", "balanced", "deep"] as const;
|
|
@@ -37,6 +59,37 @@ export interface TaskRoutingPolicy {
|
|
|
37
59
|
* this bar sits higher than the upgrade bar on purpose.
|
|
38
60
|
*/
|
|
39
61
|
minDowngradeConfidence: number;
|
|
62
|
+
/**
|
|
63
|
+
* Model for frontend planning, when the assignment reads as frontend work.
|
|
64
|
+
*
|
|
65
|
+
* Superseded by `specialtyModels.frontendDesign`; kept as the fallback source
|
|
66
|
+
* so an existing configuration keeps working untouched until it is migrated.
|
|
67
|
+
*/
|
|
68
|
+
frontendModel?: string;
|
|
69
|
+
/**
|
|
70
|
+
* Per-specialty models — the work-kind axis.
|
|
71
|
+
*
|
|
72
|
+
* Absent entries inherit the role's own chain, which is why an unset specialty
|
|
73
|
+
* is not an error and does not suppress the difficulty ladder.
|
|
74
|
+
*/
|
|
75
|
+
specialtyModels?: Partial<Record<TaskModelSpecialty, ModelSelectorValue>>;
|
|
76
|
+
/**
|
|
77
|
+
* Bar for a lateral swap on a **calibrated** backend. Directional bars do not
|
|
78
|
+
* apply here because neither direction is "spending more": being wrong either
|
|
79
|
+
* way costs quality, symmetrically, so one bar is the whole story.
|
|
80
|
+
*/
|
|
81
|
+
minDomainConfidence: number;
|
|
82
|
+
/**
|
|
83
|
+
* Bar for a lateral swap on an **uncalibrated** backend.
|
|
84
|
+
*
|
|
85
|
+
* The ordinary logged-in model cannot report a probability, and asking it for
|
|
86
|
+
* one measurably degrades the answer, so its choice carries no `confidence`.
|
|
87
|
+
* What it *can* report is an ordinal strength. Requiring a high ordinal is not
|
|
88
|
+
* the same guarantee as a calibrated threshold, and the result is recorded as
|
|
89
|
+
* uncalibrated — but refusing to route at all would make a user's explicit
|
|
90
|
+
* specialty selection silently inert on the default backend.
|
|
91
|
+
*/
|
|
92
|
+
minSpecialtyOrdinal: number;
|
|
40
93
|
}
|
|
41
94
|
|
|
42
95
|
export const DEFAULT_TASK_ROUTING_POLICY: Omit<TaskRoutingPolicy, "tiers"> = {
|
|
@@ -46,8 +99,57 @@ export const DEFAULT_TASK_ROUTING_POLICY: Omit<TaskRoutingPolicy, "tiers"> = {
|
|
|
46
99
|
// overriding one needs a stronger signal in either direction.
|
|
47
100
|
minUpgradeConfidence: 0.5,
|
|
48
101
|
minDowngradeConfidence: 0.75,
|
|
102
|
+
minDomainConfidence: 0.6,
|
|
103
|
+
// One step above "probably yes" on the ordinal ladder the uncalibrated backend
|
|
104
|
+
// emits, so "unclear" and "probably yes" both decline.
|
|
105
|
+
minSpecialtyOrdinal: 0.75,
|
|
49
106
|
};
|
|
50
107
|
|
|
108
|
+
/**
|
|
109
|
+
* The settings surface this module reads.
|
|
110
|
+
*
|
|
111
|
+
* Narrowed to `get` so the router cannot quietly start writing settings, and so
|
|
112
|
+
* a caller only has to supply a reader rather than a whole `Settings` instance.
|
|
113
|
+
*/
|
|
114
|
+
export type TaskRoutingSettingsReader = Pick<Settings, "get">;
|
|
115
|
+
|
|
116
|
+
/** True when the value names at least one model rather than being blank. */
|
|
117
|
+
function hasConfiguredModel(value: ModelSelectorValue | undefined): boolean {
|
|
118
|
+
if (Array.isArray(value)) return value.some(entry => entry.trim().length > 0);
|
|
119
|
+
return typeof value === "string" && value.trim().length > 0;
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
/**
|
|
123
|
+
* Build the routing policy from settings, or null when routing must not run.
|
|
124
|
+
*
|
|
125
|
+
* Null is returned for two distinct reasons that both mean "leave the configured
|
|
126
|
+
* model alone": the feature is off, or it is on but nothing is configured to
|
|
127
|
+
* route *to*. A lone tier is not an axis — there is nowhere to move from it —
|
|
128
|
+
* so two tiers is the floor unless a specialty or the legacy frontend model
|
|
129
|
+
* supplies a lateral target instead.
|
|
130
|
+
*/
|
|
131
|
+
export function buildTaskRoutingPolicyFromSettings(settings: TaskRoutingSettingsReader): TaskRoutingPolicy | null {
|
|
132
|
+
if (!settings.get("task.modelRouting.enabled")) return null;
|
|
133
|
+
const tiers: TaskTierModels = {
|
|
134
|
+
fast: settings.get("task.modelRouting.fastModel") || undefined,
|
|
135
|
+
balanced: settings.get("task.modelRouting.balancedModel") || undefined,
|
|
136
|
+
deep: settings.get("task.modelRouting.deepModel") || undefined,
|
|
137
|
+
};
|
|
138
|
+
const frontendModel = settings.get("task.modelRouting.frontendModel") || undefined;
|
|
139
|
+
const specialtyModels = settings.get("task.modelRouting.specialtyModels") ?? {};
|
|
140
|
+
const hasSpecialty = TASK_MODEL_SPECIALTY_IDS.some(id => hasConfiguredModel(specialtyModels[id]));
|
|
141
|
+
const tierCount = Object.values(tiers).filter(Boolean).length;
|
|
142
|
+
if (tierCount < 2 && !frontendModel && !hasSpecialty) return null;
|
|
143
|
+
return { ...DEFAULT_TASK_ROUTING_POLICY, tiers, frontendModel, specialtyModels };
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/**
|
|
147
|
+
* Roles whose output is a plan or a design review.
|
|
148
|
+
*
|
|
149
|
+
* Derived from the specialty compatibility map so the two never drift apart.
|
|
150
|
+
*/
|
|
151
|
+
export const PLANNING_ROLES: ReadonlySet<string> = new Set(TASK_MODEL_SPECIALTY_ROLES.frontendDesign);
|
|
152
|
+
|
|
51
153
|
/**
|
|
52
154
|
* The questions describe the *work*, never a model name.
|
|
53
155
|
*
|
|
@@ -71,6 +173,27 @@ function buildQuestions(): Record<string, Question> {
|
|
|
71
173
|
instructions:
|
|
72
174
|
"Does this assignment touch production, money, credentials, published releases, or state that cannot be undone?",
|
|
73
175
|
},
|
|
176
|
+
specialty: {
|
|
177
|
+
type: "choice",
|
|
178
|
+
instructions: "Which kind of work is this assignment?",
|
|
179
|
+
criteria: {
|
|
180
|
+
backendArchitecture:
|
|
181
|
+
"Designing or reviewing backend structure: APIs, data models, services, storage, infrastructure.",
|
|
182
|
+
frontendDesign:
|
|
183
|
+
"Designing or reviewing an interface: layout, visual design, interaction, components, styling.",
|
|
184
|
+
implementation:
|
|
185
|
+
"Writing or changing code against an established pattern, where the approach is already settled.",
|
|
186
|
+
testing: "Designing, writing, debugging or running tests and verification.",
|
|
187
|
+
review: "Judging existing work for correctness, regressions, or maintainability.",
|
|
188
|
+
[TASK_MODEL_SPECIALTY_NONE]:
|
|
189
|
+
"General, mixed, or unclear work that does not sit in exactly one of the categories above.",
|
|
190
|
+
},
|
|
191
|
+
},
|
|
192
|
+
specialtyClear: {
|
|
193
|
+
type: "noul",
|
|
194
|
+
instructions:
|
|
195
|
+
"Does this assignment clearly belong to exactly one of those kinds of work, rather than spanning several or being unclear?",
|
|
196
|
+
},
|
|
74
197
|
};
|
|
75
198
|
}
|
|
76
199
|
|
|
@@ -80,13 +203,111 @@ export interface TaskRoutingRequest {
|
|
|
80
203
|
assignment: string;
|
|
81
204
|
/** Whatever the role is configured to use today, used as the direction baseline. */
|
|
82
205
|
currentModel: string | undefined;
|
|
206
|
+
/**
|
|
207
|
+
* The role's fully resolved chain, in order. The composed candidate list ends
|
|
208
|
+
* with this, so a specialty or tier that cannot be authenticated falls through
|
|
209
|
+
* to the model the role would have used anyway.
|
|
210
|
+
*/
|
|
211
|
+
baselineChain?: readonly string[];
|
|
83
212
|
signal?: AbortSignal;
|
|
84
213
|
}
|
|
85
214
|
|
|
86
215
|
export interface TaskRoutingResult {
|
|
216
|
+
/** Head of the composed chain — what the spawn runs on if it authenticates. */
|
|
87
217
|
model: string;
|
|
88
|
-
|
|
218
|
+
/** Null when the move was a specialty swap — that axis has no ladder. */
|
|
219
|
+
tier: TaskTier | null;
|
|
89
220
|
reason: string;
|
|
221
|
+
/** Ordered, provenance-tagged chain for the existing auth-aware resolver. */
|
|
222
|
+
candidates: TaskRoutingCandidate[];
|
|
223
|
+
/** What the classifier asked for. The *effective* source is only known after resolution. */
|
|
224
|
+
requestedSource: TaskRoutingSource;
|
|
225
|
+
requestedSpecialty?: TaskModelSpecialty;
|
|
226
|
+
requestedTier?: TaskTier;
|
|
227
|
+
/**
|
|
228
|
+
* True when the caller named the specialty on the spawn itself. No classifier
|
|
229
|
+
* ran, so `calibrated`/`confidence`/`ordinalStrength` describe nothing here.
|
|
230
|
+
*/
|
|
231
|
+
declared: boolean;
|
|
232
|
+
/** False means `ordinalStrength` ranks, and no probability was available. */
|
|
233
|
+
calibrated: boolean;
|
|
234
|
+
confidence?: number;
|
|
235
|
+
ordinalStrength?: number;
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
export interface DeclaredSpecialtyRequest {
|
|
239
|
+
agentName: string;
|
|
240
|
+
specialty: TaskModelSpecialty;
|
|
241
|
+
/** Whatever the role is configured to use today. */
|
|
242
|
+
currentModel: string | undefined;
|
|
243
|
+
/** The role's fully resolved chain; always the tail so a dead specialty model falls through. */
|
|
244
|
+
baselineChain?: readonly string[];
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
/**
|
|
248
|
+
* Route a spawn whose caller *declared* the kind of work.
|
|
249
|
+
*
|
|
250
|
+
* This is the deterministic half of the specialty axis. Nothing here asks a
|
|
251
|
+
* classifier, reads `task.modelRouting.enabled`, or applies a confidence bar:
|
|
252
|
+
* the user put a model on this specialty in `/model`, the caller says this is
|
|
253
|
+
* that work, and the only remaining reason not to run on it is that it fails —
|
|
254
|
+
* which the child session's fallback chain handles at the transport boundary
|
|
255
|
+
* (429, 5xx, auth, quota) by advancing to the role's baseline behind it.
|
|
256
|
+
*
|
|
257
|
+
* Role eligibility is deliberately not checked. The menu groups specialties
|
|
258
|
+
* under the roles that usually do that work, but the setting is one flat map:
|
|
259
|
+
* a frontend model the user chose for design is the same frontend model they
|
|
260
|
+
* expect when the *implementation* of that frontend is delegated. Refusing
|
|
261
|
+
* here would make "frontend uses a different model" false for exactly the
|
|
262
|
+
* spawns where it matters most.
|
|
263
|
+
*
|
|
264
|
+
* Returns null only when nothing is configured for the specialty, or when the
|
|
265
|
+
* configured model is already what the role would run on anyway.
|
|
266
|
+
*/
|
|
267
|
+
export function resolveDeclaredSpecialtyRouting(
|
|
268
|
+
settings: TaskRoutingSettingsReader,
|
|
269
|
+
request: DeclaredSpecialtyRequest,
|
|
270
|
+
): TaskRoutingResult | null {
|
|
271
|
+
const specialtyModels = settings.get("task.modelRouting.specialtyModels") ?? {};
|
|
272
|
+
const configured = specialtyModels[request.specialty];
|
|
273
|
+
let value: ModelSelectorValue | undefined;
|
|
274
|
+
let source: Extract<TaskRoutingSource, "specialty" | "legacy-frontend"> = "specialty";
|
|
275
|
+
if (hasConfiguredModel(configured)) {
|
|
276
|
+
value = configured;
|
|
277
|
+
} else if (request.specialty === "frontendDesign") {
|
|
278
|
+
// Legacy compatibility: the old single frontend selector still answers for
|
|
279
|
+
// frontend design when no explicit entry has replaced it.
|
|
280
|
+
const legacy = settings.get("task.modelRouting.frontendModel")?.trim();
|
|
281
|
+
if (legacy) {
|
|
282
|
+
value = legacy;
|
|
283
|
+
source = "legacy-frontend";
|
|
284
|
+
}
|
|
285
|
+
}
|
|
286
|
+
if (value === undefined) return null;
|
|
287
|
+
|
|
288
|
+
const specialtyCandidates = toRoutingCandidates(value, source, { specialty: request.specialty });
|
|
289
|
+
const head = specialtyCandidates[0];
|
|
290
|
+
if (!head) return null;
|
|
291
|
+
// Already there: the declared model is the role's own. Reporting a route would
|
|
292
|
+
// claim a swap that never happened.
|
|
293
|
+
if (request.currentModel && matchesModel(head.selector, request.currentModel)) return null;
|
|
294
|
+
|
|
295
|
+
const candidates = dedupeRoutingCandidates([specialtyCandidates, baselineCandidates(request)]);
|
|
296
|
+
const effectiveHead = candidates[0];
|
|
297
|
+
if (!effectiveHead || effectiveHead.source === "baseline") return null;
|
|
298
|
+
const label = source === "legacy-frontend" ? "frontendDesign (legacy selector)" : request.specialty;
|
|
299
|
+
const reason = `${label}, declared by caller`;
|
|
300
|
+
logger.debug("decisions/task-routing: routed", { agent: request.agentName, model: effectiveHead.selector, reason });
|
|
301
|
+
return {
|
|
302
|
+
model: effectiveHead.selector,
|
|
303
|
+
tier: null,
|
|
304
|
+
reason,
|
|
305
|
+
candidates,
|
|
306
|
+
requestedSource: source,
|
|
307
|
+
requestedSpecialty: request.specialty,
|
|
308
|
+
declared: true,
|
|
309
|
+
calibrated: false,
|
|
310
|
+
};
|
|
90
311
|
}
|
|
91
312
|
|
|
92
313
|
/** Where a concrete model id sits in the ladder, or null when it is not one of ours. */
|
|
@@ -103,8 +324,7 @@ function rankOf(model: string | undefined, tiers: TaskTierModels): number | null
|
|
|
103
324
|
* may or may not repeat, so compare the part before it.
|
|
104
325
|
*/
|
|
105
326
|
function matchesModel(a: string, b: string): boolean {
|
|
106
|
-
|
|
107
|
-
return base(a) === base(b);
|
|
327
|
+
return specialtySelectorHead(a) === specialtySelectorHead(b);
|
|
108
328
|
}
|
|
109
329
|
|
|
110
330
|
/**
|
|
@@ -127,6 +347,68 @@ function allowed(
|
|
|
127
347
|
return confidence >= (isDowngrade ? policy.minDowngradeConfidence : policy.minUpgradeConfidence);
|
|
128
348
|
}
|
|
129
349
|
|
|
350
|
+
/** The role's own chain, used as the tail of every composed candidate list. */
|
|
351
|
+
function baselineCandidates(
|
|
352
|
+
request: Pick<TaskRoutingRequest, "baselineChain" | "currentModel">,
|
|
353
|
+
): TaskRoutingCandidate[] {
|
|
354
|
+
const chain = request.baselineChain?.length ? request.baselineChain : [request.currentModel ?? ""];
|
|
355
|
+
return toRoutingCandidates([...chain], "baseline");
|
|
356
|
+
}
|
|
357
|
+
|
|
358
|
+
interface SpecialtySelection {
|
|
359
|
+
specialty: TaskModelSpecialty;
|
|
360
|
+
source: Extract<TaskRoutingSource, "specialty" | "legacy-frontend">;
|
|
361
|
+
value: ModelSelectorValue;
|
|
362
|
+
ordinalStrength?: number;
|
|
363
|
+
confidence?: number;
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
/**
|
|
367
|
+
* Resolve the specialty axis, or null to leave it to the difficulty ladder.
|
|
368
|
+
*
|
|
369
|
+
* Declines are deliberately quiet and numerous: an unrecognized option, `none`,
|
|
370
|
+
* an incompatible role, no configured model, a below-bar answer, or a missing
|
|
371
|
+
* clarity signal all mean "the user did not ask for this, carry on".
|
|
372
|
+
*/
|
|
373
|
+
function resolveSpecialty(
|
|
374
|
+
policy: TaskRoutingPolicy,
|
|
375
|
+
request: TaskRoutingRequest,
|
|
376
|
+
answers: Record<string, Answer>,
|
|
377
|
+
calibrated: boolean,
|
|
378
|
+
): SpecialtySelection | null {
|
|
379
|
+
const answer = answers.specialty;
|
|
380
|
+
if (answer?.type !== "choice") return null;
|
|
381
|
+
const choice = answer.choice;
|
|
382
|
+
if (typeof choice !== "string" || choice === TASK_MODEL_SPECIALTY_NONE) return null;
|
|
383
|
+
if (!isTaskModelSpecialty(choice)) return null;
|
|
384
|
+
if (!specialtySupportsRole(choice, request.agentName)) return null;
|
|
385
|
+
|
|
386
|
+
const clarity = answers.specialtyClear;
|
|
387
|
+
const ordinalStrength = clarity?.type === "noul" && typeof clarity.noul === "number" ? clarity.noul : undefined;
|
|
388
|
+
const confidence = typeof answer.confidence === "number" ? answer.confidence : undefined;
|
|
389
|
+
|
|
390
|
+
if (calibrated) {
|
|
391
|
+
// A calibrated backend reports a real probability; threshold on it directly.
|
|
392
|
+
if (confidence === undefined || confidence < policy.minDomainConfidence) return null;
|
|
393
|
+
} else {
|
|
394
|
+
// No probability is available. Require an explicit high ordinal instead, and
|
|
395
|
+
// never dress that number up as confidence downstream.
|
|
396
|
+
if (ordinalStrength === undefined || ordinalStrength < policy.minSpecialtyOrdinal) return null;
|
|
397
|
+
}
|
|
398
|
+
|
|
399
|
+
const configured = policy.specialtyModels?.[choice];
|
|
400
|
+
if (configured !== undefined && (typeof configured !== "string" || configured.trim().length > 0)) {
|
|
401
|
+
return { specialty: choice, source: "specialty", value: configured, ordinalStrength, confidence };
|
|
402
|
+
}
|
|
403
|
+
// Legacy compatibility: the old single frontend selector still answers for
|
|
404
|
+
// frontend design when no explicit entry has replaced it.
|
|
405
|
+
const legacy = policy.frontendModel?.trim();
|
|
406
|
+
if (choice === "frontendDesign" && legacy) {
|
|
407
|
+
return { specialty: choice, source: "legacy-frontend", value: legacy, ordinalStrength, confidence };
|
|
408
|
+
}
|
|
409
|
+
return null;
|
|
410
|
+
}
|
|
411
|
+
|
|
130
412
|
/**
|
|
131
413
|
* Decide the model for one subagent spawn, or null to leave the configured one alone.
|
|
132
414
|
*
|
|
@@ -140,9 +422,13 @@ export async function routeTaskModel(
|
|
|
140
422
|
request: TaskRoutingRequest,
|
|
141
423
|
): Promise<TaskRoutingResult | null> {
|
|
142
424
|
const configured = TASK_TIERS.filter(tier => policy.tiers[tier]);
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
425
|
+
const frontendModel = policy.frontendModel?.trim() || undefined;
|
|
426
|
+
const hasSpecialtyModels = TASK_MODEL_SPECIALTY_IDS.some(id => {
|
|
427
|
+
const value = policy.specialtyModels?.[id];
|
|
428
|
+
return Array.isArray(value) ? value.length > 0 : typeof value === "string" && value.trim().length > 0;
|
|
429
|
+
});
|
|
430
|
+
// No axis has anything to move on: no ladder, no specialty model, no legacy one.
|
|
431
|
+
if (configured.length < 2 && !frontendModel && !hasSpecialtyModels) return null;
|
|
146
432
|
|
|
147
433
|
const assignment = request.assignment.trim();
|
|
148
434
|
if (assignment.length < 24) return null;
|
|
@@ -154,43 +440,135 @@ export async function routeTaskModel(
|
|
|
154
440
|
});
|
|
155
441
|
if (!result) return null;
|
|
156
442
|
|
|
157
|
-
const
|
|
158
|
-
|
|
159
|
-
let tier = TASK_TIERS.find(candidate => candidate === answer.choice);
|
|
160
|
-
if (!tier) return null;
|
|
443
|
+
const answers = result.answers;
|
|
444
|
+
const baseline = baselineCandidates(request);
|
|
161
445
|
|
|
446
|
+
// --- Risk floor ---
|
|
447
|
+
//
|
|
162
448
|
// Work that cannot be undone takes the most capable tier available and skips
|
|
163
449
|
// the confidence bars — this one is not a confidence question. It may only
|
|
164
450
|
// ever raise the tier, never lower it, or "this is risky" would end up
|
|
165
|
-
// *downgrading* an assignment already running deep.
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
451
|
+
// *downgrading* an assignment already running deep. It also outranks the
|
|
452
|
+
// specialty axis: a design-strong model is not the safety property being
|
|
453
|
+
// asked for here.
|
|
454
|
+
const riskAnswer = answers.risky;
|
|
455
|
+
const forcedByRisk = riskAnswer?.type === "noul" && typeof riskAnswer.noul === "number" && riskAnswer.noul > 0.7;
|
|
456
|
+
|
|
457
|
+
const tierAnswer = answers.tier;
|
|
458
|
+
const declaredTier =
|
|
459
|
+
tierAnswer?.type === "choice" ? TASK_TIERS.find(candidate => candidate === tierAnswer.choice) : undefined;
|
|
460
|
+
const tierConfidence =
|
|
461
|
+
tierAnswer?.type === "choice" && typeof tierAnswer.confidence === "number" ? tierAnswer.confidence : undefined;
|
|
462
|
+
|
|
463
|
+
if (forcedByRisk && configured.length > 0) {
|
|
169
464
|
const deepest = configured[configured.length - 1] as TaskTier;
|
|
170
465
|
const currentRank = rankOf(request.currentModel, policy.tiers);
|
|
171
466
|
const wantedRank = Math.max(TASK_TIERS.indexOf(deepest), currentRank ?? 0);
|
|
172
|
-
tier = TASK_TIERS[wantedRank] as TaskTier;
|
|
467
|
+
const tier = TASK_TIERS[wantedRank] as TaskTier;
|
|
468
|
+
const model = policy.tiers[tier];
|
|
469
|
+
if (!model || (request.currentModel && matchesModel(model, request.currentModel))) return null;
|
|
470
|
+
const candidates = dedupeRoutingCandidates([toRoutingCandidates(model, "tier", { tier }), baseline]);
|
|
471
|
+
const head = candidates[0];
|
|
472
|
+
if (!head) return null;
|
|
473
|
+
const reason = `${tier}, forced by risk`;
|
|
474
|
+
logger.debug("decisions/task-routing: routed", { agent: request.agentName, model: head.selector, reason });
|
|
475
|
+
return {
|
|
476
|
+
model: head.selector,
|
|
477
|
+
tier,
|
|
478
|
+
reason,
|
|
479
|
+
candidates,
|
|
480
|
+
requestedSource: "tier",
|
|
481
|
+
requestedTier: tier,
|
|
482
|
+
declared: false,
|
|
483
|
+
calibrated: result.calibrated,
|
|
484
|
+
confidence: tierConfidence,
|
|
485
|
+
};
|
|
486
|
+
}
|
|
487
|
+
|
|
488
|
+
// --- Specialty axis: lateral swap for the kind of work ---
|
|
489
|
+
const specialty = resolveSpecialty(policy, request, answers, result.calibrated);
|
|
490
|
+
if (specialty) {
|
|
491
|
+
const specialtyCandidates = toRoutingCandidates(specialty.value, specialty.source, {
|
|
492
|
+
specialty: specialty.specialty,
|
|
493
|
+
});
|
|
494
|
+
const head = specialtyCandidates[0];
|
|
495
|
+
// Already there: nothing to move, and the ladder should not fire either —
|
|
496
|
+
// the user's specialty choice is the standing answer for this work.
|
|
497
|
+
if (head && request.currentModel && matchesModel(head.selector, request.currentModel)) return null;
|
|
498
|
+
if (head) {
|
|
499
|
+
// The tier is still worth composing *behind* the specialty: if the specialty
|
|
500
|
+
// model cannot be authenticated, the difficulty answer is the next best guess.
|
|
501
|
+
const tierSegment =
|
|
502
|
+
declaredTier && policy.tiers[declaredTier]
|
|
503
|
+
? toRoutingCandidates(policy.tiers[declaredTier] as string, "tier", { tier: declaredTier })
|
|
504
|
+
: [];
|
|
505
|
+
const candidates = dedupeRoutingCandidates([specialtyCandidates, tierSegment, baseline]);
|
|
506
|
+
const effectiveHead = candidates[0];
|
|
507
|
+
if (effectiveHead && effectiveHead.source !== "baseline") {
|
|
508
|
+
const strength = result.calibrated
|
|
509
|
+
? `confidence ${specialty.confidence?.toFixed(2) ?? "n/d"}`
|
|
510
|
+
: `clarity ${specialty.ordinalStrength?.toFixed(2) ?? "n/d"}, uncalibrated`;
|
|
511
|
+
const label =
|
|
512
|
+
specialty.source === "legacy-frontend" ? "frontendDesign (legacy selector)" : specialty.specialty;
|
|
513
|
+
const reason = `${label} (${strength})`;
|
|
514
|
+
logger.debug("decisions/task-routing: routed", {
|
|
515
|
+
agent: request.agentName,
|
|
516
|
+
model: effectiveHead.selector,
|
|
517
|
+
reason,
|
|
518
|
+
});
|
|
519
|
+
return {
|
|
520
|
+
model: effectiveHead.selector,
|
|
521
|
+
tier: null,
|
|
522
|
+
reason,
|
|
523
|
+
candidates,
|
|
524
|
+
requestedSource: specialty.source,
|
|
525
|
+
requestedSpecialty: specialty.specialty,
|
|
526
|
+
requestedTier: declaredTier,
|
|
527
|
+
declared: false,
|
|
528
|
+
calibrated: result.calibrated,
|
|
529
|
+
confidence: result.calibrated ? specialty.confidence : undefined,
|
|
530
|
+
ordinalStrength: result.calibrated ? undefined : specialty.ordinalStrength,
|
|
531
|
+
};
|
|
532
|
+
}
|
|
533
|
+
}
|
|
173
534
|
}
|
|
174
535
|
|
|
536
|
+
// --- Difficulty axis: the ladder ---
|
|
537
|
+
if (configured.length < 2) return null;
|
|
538
|
+
if (!declaredTier) return null;
|
|
539
|
+
const tier = declaredTier;
|
|
540
|
+
|
|
175
541
|
const model = policy.tiers[tier];
|
|
176
542
|
if (!model || (request.currentModel && matchesModel(model, request.currentModel))) return null;
|
|
177
543
|
|
|
178
544
|
const currentRank = rankOf(request.currentModel, policy.tiers);
|
|
179
545
|
const wantedRank = TASK_TIERS.indexOf(tier);
|
|
180
|
-
if (!
|
|
546
|
+
if (!allowed(wantedRank, currentRank, tierConfidence, result.calibrated, policy)) {
|
|
181
547
|
logger.debug("decisions/task-routing: below the bar, keeping the configured model", {
|
|
182
548
|
agent: request.agentName,
|
|
183
549
|
wanted: tier,
|
|
184
550
|
current: request.currentModel,
|
|
185
|
-
confidence:
|
|
551
|
+
confidence: tierConfidence,
|
|
552
|
+
declared: false,
|
|
186
553
|
calibrated: result.calibrated,
|
|
187
554
|
});
|
|
188
555
|
return null;
|
|
189
556
|
}
|
|
190
557
|
|
|
191
|
-
const
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
558
|
+
const candidates = dedupeRoutingCandidates([toRoutingCandidates(model, "tier", { tier }), baseline]);
|
|
559
|
+
const head = candidates[0];
|
|
560
|
+
if (!head) return null;
|
|
561
|
+
const reason = `${tier} (confidence ${tierConfidence?.toFixed(2) ?? "n/d"})`;
|
|
562
|
+
logger.debug("decisions/task-routing: routed", { agent: request.agentName, model: head.selector, reason });
|
|
563
|
+
return {
|
|
564
|
+
model: head.selector,
|
|
565
|
+
tier,
|
|
566
|
+
reason,
|
|
567
|
+
candidates,
|
|
568
|
+
requestedSource: "tier",
|
|
569
|
+
requestedTier: tier,
|
|
570
|
+
declared: false,
|
|
571
|
+
calibrated: result.calibrated,
|
|
572
|
+
confidence: tierConfidence,
|
|
573
|
+
};
|
|
196
574
|
}
|
package/src/i18n/messages/de.ts
CHANGED
|
@@ -75,6 +75,22 @@ export const de: Partial<Record<MsgKey, string>> = {
|
|
|
75
75
|
"modelSelector.noMatching": "Keine passenden Modelle.",
|
|
76
76
|
"modelSelector.modelName": "Modellname: {value}",
|
|
77
77
|
"modelSelector.actionFor": "Aktion für: {id}",
|
|
78
|
+
"modelSelector.setAsTarget": "Als {tag} ({name}) festlegen",
|
|
79
|
+
"modelSelector.setForAllRoleAgents": "Fur alle Rollen-Agenten setzen",
|
|
80
|
+
"modelSelector.setForAllTargets": "Fur alle Ziele setzen",
|
|
81
|
+
"modelSelector.hasDetailedUses": "(detaillierte Aufgaben)",
|
|
82
|
+
"modelSelector.detailedUseFor": "Detaillierte Aufgabe fur {target}: {id}",
|
|
83
|
+
"modelSelector.generalRole": "Allgemein (ganze Rolle)",
|
|
84
|
+
"modelSelector.resetSpecialties": "Detaillierte Aufgaben zurucksetzen",
|
|
85
|
+
"modelSelector.specialty.backendArchitecture": "Backend-Architektur",
|
|
86
|
+
"modelSelector.specialty.frontendDesign": "Frontend-Design",
|
|
87
|
+
"modelSelector.specialty.implementation": "Implementierung",
|
|
88
|
+
"modelSelector.specialty.testing": "Tests",
|
|
89
|
+
"modelSelector.specialty.review": "Review",
|
|
90
|
+
"modelSelector.specialtySaved": "{specialty} verwendet {value}.",
|
|
91
|
+
"modelSelector.specialtySavedRoutingOff":
|
|
92
|
+
"{specialty} verwendet {value}, sobald eine Aufgabe diese Arbeit deklariert. Automatische Erkennung aus dem Auftragstext bleibt aus, bis task.modelRouting.enabled aktiv ist.",
|
|
93
|
+
"modelSelector.specialtyCleared": "Detaillierte Aufgaben fur {target} zuruckgesetzt.",
|
|
78
94
|
"modelSelector.reasoningFor": "Reasoning für {target}: {id}",
|
|
79
95
|
"modelSelector.temporaryModel": "temporäres Modell",
|
|
80
96
|
};
|
package/src/i18n/messages/en.ts
CHANGED
|
@@ -95,6 +95,22 @@ export const en = {
|
|
|
95
95
|
"modelSelector.noMatching": "No matching models.",
|
|
96
96
|
"modelSelector.modelName": "Model Name: {value}",
|
|
97
97
|
"modelSelector.actionFor": "Action for: {id}",
|
|
98
|
+
"modelSelector.setAsTarget": "Set as {tag} ({name})",
|
|
99
|
+
"modelSelector.setForAllRoleAgents": "Set for all role agents",
|
|
100
|
+
"modelSelector.setForAllTargets": "Set for all targets",
|
|
101
|
+
"modelSelector.hasDetailedUses": "(detailed uses)",
|
|
102
|
+
"modelSelector.detailedUseFor": "Detailed use for {target}: {id}",
|
|
103
|
+
"modelSelector.generalRole": "General (whole role)",
|
|
104
|
+
"modelSelector.resetSpecialties": "Clear detailed-use overrides",
|
|
105
|
+
"modelSelector.specialty.backendArchitecture": "Backend architecture",
|
|
106
|
+
"modelSelector.specialty.frontendDesign": "Frontend design",
|
|
107
|
+
"modelSelector.specialty.implementation": "Implementation",
|
|
108
|
+
"modelSelector.specialty.testing": "Testing",
|
|
109
|
+
"modelSelector.specialty.review": "Review",
|
|
110
|
+
"modelSelector.specialtySaved": "{specialty} will use {value}.",
|
|
111
|
+
"modelSelector.specialtySavedRoutingOff":
|
|
112
|
+
"{specialty} will use {value} whenever a task declares that work. Auto-detection from assignment text stays off until you enable task.modelRouting.enabled.",
|
|
113
|
+
"modelSelector.specialtyCleared": "Cleared detailed-use overrides for {target}.",
|
|
98
114
|
"modelSelector.reasoningFor": "Reasoning for {target}: {id}",
|
|
99
115
|
"modelSelector.temporaryModel": "temporary model",
|
|
100
116
|
} as const;
|
package/src/i18n/messages/es.ts
CHANGED
|
@@ -75,6 +75,22 @@ export const es: Partial<Record<MsgKey, string>> = {
|
|
|
75
75
|
"modelSelector.noMatching": "No hay modelos coincidentes.",
|
|
76
76
|
"modelSelector.modelName": "Nombre del modelo: {value}",
|
|
77
77
|
"modelSelector.actionFor": "Acción para: {id}",
|
|
78
|
+
"modelSelector.setAsTarget": "Asignar como {tag} ({name})",
|
|
79
|
+
"modelSelector.setForAllRoleAgents": "Asignar a todos los agentes de rol",
|
|
80
|
+
"modelSelector.setForAllTargets": "Asignar a todos los destinos",
|
|
81
|
+
"modelSelector.hasDetailedUses": "(usos detallados)",
|
|
82
|
+
"modelSelector.detailedUseFor": "Uso detallado de {target}: {id}",
|
|
83
|
+
"modelSelector.generalRole": "General (todo el rol)",
|
|
84
|
+
"modelSelector.resetSpecialties": "Borrar los usos detallados",
|
|
85
|
+
"modelSelector.specialty.backendArchitecture": "Arquitectura de backend",
|
|
86
|
+
"modelSelector.specialty.frontendDesign": "Diseño de frontend",
|
|
87
|
+
"modelSelector.specialty.implementation": "Implementación",
|
|
88
|
+
"modelSelector.specialty.testing": "Pruebas",
|
|
89
|
+
"modelSelector.specialty.review": "Revisión",
|
|
90
|
+
"modelSelector.specialtySaved": "{specialty} usará {value}.",
|
|
91
|
+
"modelSelector.specialtySavedRoutingOff":
|
|
92
|
+
"{specialty} usará {value} cuando una tarea declare ese trabajo. La detección automática desde el texto de la tarea sigue desactivada hasta que actives task.modelRouting.enabled.",
|
|
93
|
+
"modelSelector.specialtyCleared": "Se borraron los usos detallados de {target}.",
|
|
78
94
|
"modelSelector.reasoningFor": "Razonamiento para {target}: {id}",
|
|
79
95
|
"modelSelector.temporaryModel": "modelo temporal",
|
|
80
96
|
};
|
package/src/i18n/messages/fr.ts
CHANGED
|
@@ -75,6 +75,22 @@ export const fr: Partial<Record<MsgKey, string>> = {
|
|
|
75
75
|
"modelSelector.noMatching": "Aucun modèle correspondant.",
|
|
76
76
|
"modelSelector.modelName": "Nom du modèle : {value}",
|
|
77
77
|
"modelSelector.actionFor": "Action pour : {id}",
|
|
78
|
+
"modelSelector.setAsTarget": "Définir comme {tag} ({name})",
|
|
79
|
+
"modelSelector.setForAllRoleAgents": "Definir pour tous les agents de role",
|
|
80
|
+
"modelSelector.setForAllTargets": "Definir pour toutes les cibles",
|
|
81
|
+
"modelSelector.hasDetailedUses": "(usages detailles)",
|
|
82
|
+
"modelSelector.detailedUseFor": "Usage detaille pour {target} : {id}",
|
|
83
|
+
"modelSelector.generalRole": "General (tout le role)",
|
|
84
|
+
"modelSelector.resetSpecialties": "Effacer les usages detailles",
|
|
85
|
+
"modelSelector.specialty.backendArchitecture": "Architecture backend",
|
|
86
|
+
"modelSelector.specialty.frontendDesign": "Design frontend",
|
|
87
|
+
"modelSelector.specialty.implementation": "Implementation",
|
|
88
|
+
"modelSelector.specialty.testing": "Tests",
|
|
89
|
+
"modelSelector.specialty.review": "Revue",
|
|
90
|
+
"modelSelector.specialtySaved": "{specialty} utilisera {value}.",
|
|
91
|
+
"modelSelector.specialtySavedRoutingOff":
|
|
92
|
+
"{specialty} utilisera {value} des qu'une tache declare ce travail. La detection automatique depuis le texte reste inactive sans task.modelRouting.enabled.",
|
|
93
|
+
"modelSelector.specialtyCleared": "Usages detailles effaces pour {target}.",
|
|
78
94
|
"modelSelector.reasoningFor": "Raisonnement pour {target} : {id}",
|
|
79
95
|
"modelSelector.temporaryModel": "modèle temporaire",
|
|
80
96
|
};
|
package/src/i18n/messages/ja.ts
CHANGED
|
@@ -79,6 +79,22 @@ export const ja: Partial<Record<MsgKey, string>> = {
|
|
|
79
79
|
"modelSelector.noMatching": "一致するモデルがありません。",
|
|
80
80
|
"modelSelector.modelName": "モデル名: {value}",
|
|
81
81
|
"modelSelector.actionFor": "操作対象: {id}",
|
|
82
|
+
"modelSelector.setAsTarget": "{tag} に設定",
|
|
83
|
+
"modelSelector.setForAllRoleAgents": "すべてのロールエージェントに設定",
|
|
84
|
+
"modelSelector.setForAllTargets": "すべての対象に設定",
|
|
85
|
+
"modelSelector.hasDetailedUses": "(詳細用途あり)",
|
|
86
|
+
"modelSelector.detailedUseFor": "{target} の詳細用途: {id}",
|
|
87
|
+
"modelSelector.generalRole": "一般(ロール全体)",
|
|
88
|
+
"modelSelector.resetSpecialties": "詳細用途の設定を解除",
|
|
89
|
+
"modelSelector.specialty.backendArchitecture": "バックエンド設計",
|
|
90
|
+
"modelSelector.specialty.frontendDesign": "フロントエンドデザイン",
|
|
91
|
+
"modelSelector.specialty.implementation": "実装",
|
|
92
|
+
"modelSelector.specialty.testing": "テスト",
|
|
93
|
+
"modelSelector.specialty.review": "レビュー",
|
|
94
|
+
"modelSelector.specialtySaved": "{specialty} には {value} を使用します。",
|
|
95
|
+
"modelSelector.specialtySavedRoutingOff":
|
|
96
|
+
"{specialty} として宣言されたタスクには {value} を使用します。指示文からの自動判定は task.modelRouting.enabled を有効にするまで無効です。",
|
|
97
|
+
"modelSelector.specialtyCleared": "{target} の詳細用途設定を解除しました。",
|
|
82
98
|
"modelSelector.reasoningFor": "{target} の推論レベル: {id}",
|
|
83
99
|
"modelSelector.temporaryModel": "一時モデル",
|
|
84
100
|
};
|
package/src/i18n/messages/ko.ts
CHANGED
|
@@ -79,6 +79,22 @@ export const ko: Partial<Record<MsgKey, string>> = {
|
|
|
79
79
|
"modelSelector.noMatching": "일치하는 모델이 없습니다.",
|
|
80
80
|
"modelSelector.modelName": "모델 이름: {value}",
|
|
81
81
|
"modelSelector.actionFor": "작업 대상: {id}",
|
|
82
|
+
"modelSelector.setAsTarget": "{tag}에 설정",
|
|
83
|
+
"modelSelector.setForAllRoleAgents": "모든 역할 에이전트에 설정",
|
|
84
|
+
"modelSelector.setForAllTargets": "모든 대상에 설정",
|
|
85
|
+
"modelSelector.hasDetailedUses": "(세부 업무 있음)",
|
|
86
|
+
"modelSelector.detailedUseFor": "{target} 세부 업무: {id}",
|
|
87
|
+
"modelSelector.generalRole": "일반 (역할 전체)",
|
|
88
|
+
"modelSelector.resetSpecialties": "세부 업무 설정 해제",
|
|
89
|
+
"modelSelector.specialty.backendArchitecture": "백엔드 아키텍처",
|
|
90
|
+
"modelSelector.specialty.frontendDesign": "프런트엔드 디자인",
|
|
91
|
+
"modelSelector.specialty.implementation": "구현",
|
|
92
|
+
"modelSelector.specialty.testing": "테스트",
|
|
93
|
+
"modelSelector.specialty.review": "리뷰",
|
|
94
|
+
"modelSelector.specialtySaved": "{specialty} 작업에 {value}을(를) 사용합니다.",
|
|
95
|
+
"modelSelector.specialtySavedRoutingOff":
|
|
96
|
+
"{specialty} 작업으로 선언된 태스크는 {value}을(를) 사용합니다. 지시문에서 자동 감지하려면 task.modelRouting.enabled를 켜야 합니다.",
|
|
97
|
+
"modelSelector.specialtyCleared": "{target}의 세부 업무 설정을 해제했습니다.",
|
|
82
98
|
"modelSelector.reasoningFor": "{target} 추론 수준: {id}",
|
|
83
99
|
"modelSelector.temporaryModel": "임시 모델",
|
|
84
100
|
};
|