@sayknow-cli/coding-agent 0.5.24 → 0.5.26

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -6,14 +6,36 @@
6
6
  * a one-line rename and a migration across twelve files. This asks about the
7
7
  * actual assignment and moves the model when the answer is clear enough.
8
8
  *
9
+ * Two axes, asked in one call:
10
+ *
11
+ * - **Difficulty** — the fast/balanced/deep ladder. Directional, so moving down
12
+ * costs more confidence than moving up.
13
+ * - **Specialty** — the kind of work (backend architecture, frontend design,
14
+ * implementation, test work, review). Lateral, so a single bar applies.
15
+ *
9
16
  * Only subagents. The main loop's model is deliberately out of scope — changing
10
17
  * it mid-session invalidates the prompt cache, and on a long context re-caching
11
18
  * routinely costs more than the cheaper tier saves. A subagent starts with its
12
19
  * own context, so there is nothing to invalidate.
13
20
  */
14
21
  import { logger } from "@sayknow-cli/utils";
22
+ import type { ModelSelectorValue } from "../config/model-selector-value";
23
+ import type { Settings } from "../config/settings";
24
+ import {
25
+ dedupeRoutingCandidates,
26
+ isTaskModelSpecialty,
27
+ specialtySelectorHead,
28
+ specialtySupportsRole,
29
+ TASK_MODEL_SPECIALTY_IDS,
30
+ TASK_MODEL_SPECIALTY_NONE,
31
+ TASK_MODEL_SPECIALTY_ROLES,
32
+ type TaskModelSpecialty,
33
+ type TaskRoutingCandidate,
34
+ type TaskRoutingSource,
35
+ toRoutingCandidates,
36
+ } from "../config/task-model-specialties";
15
37
  import type { DecisionService } from "./index";
16
- import type { Question } from "./types";
38
+ import type { Answer, Question } from "./types";
17
39
 
18
40
  /** Ordered cheapest to most capable. The order *is* the policy's direction. */
19
41
  export const TASK_TIERS = ["fast", "balanced", "deep"] as const;
@@ -37,6 +59,37 @@ export interface TaskRoutingPolicy {
37
59
  * this bar sits higher than the upgrade bar on purpose.
38
60
  */
39
61
  minDowngradeConfidence: number;
62
+ /**
63
+ * Model for frontend planning, when the assignment reads as frontend work.
64
+ *
65
+ * Superseded by `specialtyModels.frontendDesign`; kept as the fallback source
66
+ * so an existing configuration keeps working untouched until it is migrated.
67
+ */
68
+ frontendModel?: string;
69
+ /**
70
+ * Per-specialty models — the work-kind axis.
71
+ *
72
+ * Absent entries inherit the role's own chain, which is why an unset specialty
73
+ * is not an error and does not suppress the difficulty ladder.
74
+ */
75
+ specialtyModels?: Partial<Record<TaskModelSpecialty, ModelSelectorValue>>;
76
+ /**
77
+ * Bar for a lateral swap on a **calibrated** backend. Directional bars do not
78
+ * apply here because neither direction is "spending more": being wrong either
79
+ * way costs quality, symmetrically, so one bar is the whole story.
80
+ */
81
+ minDomainConfidence: number;
82
+ /**
83
+ * Bar for a lateral swap on an **uncalibrated** backend.
84
+ *
85
+ * The ordinary logged-in model cannot report a probability, and asking it for
86
+ * one measurably degrades the answer, so its choice carries no `confidence`.
87
+ * What it *can* report is an ordinal strength. Requiring a high ordinal is not
88
+ * the same guarantee as a calibrated threshold, and the result is recorded as
89
+ * uncalibrated — but refusing to route at all would make a user's explicit
90
+ * specialty selection silently inert on the default backend.
91
+ */
92
+ minSpecialtyOrdinal: number;
40
93
  }
41
94
 
42
95
  export const DEFAULT_TASK_ROUTING_POLICY: Omit<TaskRoutingPolicy, "tiers"> = {
@@ -46,8 +99,57 @@ export const DEFAULT_TASK_ROUTING_POLICY: Omit<TaskRoutingPolicy, "tiers"> = {
46
99
  // overriding one needs a stronger signal in either direction.
47
100
  minUpgradeConfidence: 0.5,
48
101
  minDowngradeConfidence: 0.75,
102
+ minDomainConfidence: 0.6,
103
+ // One step above "probably yes" on the ordinal ladder the uncalibrated backend
104
+ // emits, so "unclear" and "probably yes" both decline.
105
+ minSpecialtyOrdinal: 0.75,
49
106
  };
50
107
 
108
+ /**
109
+ * The settings surface this module reads.
110
+ *
111
+ * Narrowed to `get` so the router cannot quietly start writing settings, and so
112
+ * a caller only has to supply a reader rather than a whole `Settings` instance.
113
+ */
114
+ export type TaskRoutingSettingsReader = Pick<Settings, "get">;
115
+
116
+ /** True when the value names at least one model rather than being blank. */
117
+ function hasConfiguredModel(value: ModelSelectorValue | undefined): boolean {
118
+ if (Array.isArray(value)) return value.some(entry => entry.trim().length > 0);
119
+ return typeof value === "string" && value.trim().length > 0;
120
+ }
121
+
122
+ /**
123
+ * Build the routing policy from settings, or null when routing must not run.
124
+ *
125
+ * Null is returned for two distinct reasons that both mean "leave the configured
126
+ * model alone": the feature is off, or it is on but nothing is configured to
127
+ * route *to*. A lone tier is not an axis — there is nowhere to move from it —
128
+ * so two tiers is the floor unless a specialty or the legacy frontend model
129
+ * supplies a lateral target instead.
130
+ */
131
+ export function buildTaskRoutingPolicyFromSettings(settings: TaskRoutingSettingsReader): TaskRoutingPolicy | null {
132
+ if (!settings.get("task.modelRouting.enabled")) return null;
133
+ const tiers: TaskTierModels = {
134
+ fast: settings.get("task.modelRouting.fastModel") || undefined,
135
+ balanced: settings.get("task.modelRouting.balancedModel") || undefined,
136
+ deep: settings.get("task.modelRouting.deepModel") || undefined,
137
+ };
138
+ const frontendModel = settings.get("task.modelRouting.frontendModel") || undefined;
139
+ const specialtyModels = settings.get("task.modelRouting.specialtyModels") ?? {};
140
+ const hasSpecialty = TASK_MODEL_SPECIALTY_IDS.some(id => hasConfiguredModel(specialtyModels[id]));
141
+ const tierCount = Object.values(tiers).filter(Boolean).length;
142
+ if (tierCount < 2 && !frontendModel && !hasSpecialty) return null;
143
+ return { ...DEFAULT_TASK_ROUTING_POLICY, tiers, frontendModel, specialtyModels };
144
+ }
145
+
146
+ /**
147
+ * Roles whose output is a plan or a design review.
148
+ *
149
+ * Derived from the specialty compatibility map so the two never drift apart.
150
+ */
151
+ export const PLANNING_ROLES: ReadonlySet<string> = new Set(TASK_MODEL_SPECIALTY_ROLES.frontendDesign);
152
+
51
153
  /**
52
154
  * The questions describe the *work*, never a model name.
53
155
  *
@@ -71,6 +173,27 @@ function buildQuestions(): Record<string, Question> {
71
173
  instructions:
72
174
  "Does this assignment touch production, money, credentials, published releases, or state that cannot be undone?",
73
175
  },
176
+ specialty: {
177
+ type: "choice",
178
+ instructions: "Which kind of work is this assignment?",
179
+ criteria: {
180
+ backendArchitecture:
181
+ "Designing or reviewing backend structure: APIs, data models, services, storage, infrastructure.",
182
+ frontendDesign:
183
+ "Designing or reviewing an interface: layout, visual design, interaction, components, styling.",
184
+ implementation:
185
+ "Writing or changing code against an established pattern, where the approach is already settled.",
186
+ testing: "Designing, writing, debugging or running tests and verification.",
187
+ review: "Judging existing work for correctness, regressions, or maintainability.",
188
+ [TASK_MODEL_SPECIALTY_NONE]:
189
+ "General, mixed, or unclear work that does not sit in exactly one of the categories above.",
190
+ },
191
+ },
192
+ specialtyClear: {
193
+ type: "noul",
194
+ instructions:
195
+ "Does this assignment clearly belong to exactly one of those kinds of work, rather than spanning several or being unclear?",
196
+ },
74
197
  };
75
198
  }
76
199
 
@@ -80,13 +203,111 @@ export interface TaskRoutingRequest {
80
203
  assignment: string;
81
204
  /** Whatever the role is configured to use today, used as the direction baseline. */
82
205
  currentModel: string | undefined;
206
+ /**
207
+ * The role's fully resolved chain, in order. The composed candidate list ends
208
+ * with this, so a specialty or tier that cannot be authenticated falls through
209
+ * to the model the role would have used anyway.
210
+ */
211
+ baselineChain?: readonly string[];
83
212
  signal?: AbortSignal;
84
213
  }
85
214
 
86
215
  export interface TaskRoutingResult {
216
+ /** Head of the composed chain — what the spawn runs on if it authenticates. */
87
217
  model: string;
88
- tier: TaskTier;
218
+ /** Null when the move was a specialty swap — that axis has no ladder. */
219
+ tier: TaskTier | null;
89
220
  reason: string;
221
+ /** Ordered, provenance-tagged chain for the existing auth-aware resolver. */
222
+ candidates: TaskRoutingCandidate[];
223
+ /** What the classifier asked for. The *effective* source is only known after resolution. */
224
+ requestedSource: TaskRoutingSource;
225
+ requestedSpecialty?: TaskModelSpecialty;
226
+ requestedTier?: TaskTier;
227
+ /**
228
+ * True when the caller named the specialty on the spawn itself. No classifier
229
+ * ran, so `calibrated`/`confidence`/`ordinalStrength` describe nothing here.
230
+ */
231
+ declared: boolean;
232
+ /** False means `ordinalStrength` ranks, and no probability was available. */
233
+ calibrated: boolean;
234
+ confidence?: number;
235
+ ordinalStrength?: number;
236
+ }
237
+
238
+ export interface DeclaredSpecialtyRequest {
239
+ agentName: string;
240
+ specialty: TaskModelSpecialty;
241
+ /** Whatever the role is configured to use today. */
242
+ currentModel: string | undefined;
243
+ /** The role's fully resolved chain; always the tail so a dead specialty model falls through. */
244
+ baselineChain?: readonly string[];
245
+ }
246
+
247
+ /**
248
+ * Route a spawn whose caller *declared* the kind of work.
249
+ *
250
+ * This is the deterministic half of the specialty axis. Nothing here asks a
251
+ * classifier, reads `task.modelRouting.enabled`, or applies a confidence bar:
252
+ * the user put a model on this specialty in `/model`, the caller says this is
253
+ * that work, and the only remaining reason not to run on it is that it fails —
254
+ * which the child session's fallback chain handles at the transport boundary
255
+ * (429, 5xx, auth, quota) by advancing to the role's baseline behind it.
256
+ *
257
+ * Role eligibility is deliberately not checked. The menu groups specialties
258
+ * under the roles that usually do that work, but the setting is one flat map:
259
+ * a frontend model the user chose for design is the same frontend model they
260
+ * expect when the *implementation* of that frontend is delegated. Refusing
261
+ * here would make "frontend uses a different model" false for exactly the
262
+ * spawns where it matters most.
263
+ *
264
+ * Returns null only when nothing is configured for the specialty, or when the
265
+ * configured model is already what the role would run on anyway.
266
+ */
267
+ export function resolveDeclaredSpecialtyRouting(
268
+ settings: TaskRoutingSettingsReader,
269
+ request: DeclaredSpecialtyRequest,
270
+ ): TaskRoutingResult | null {
271
+ const specialtyModels = settings.get("task.modelRouting.specialtyModels") ?? {};
272
+ const configured = specialtyModels[request.specialty];
273
+ let value: ModelSelectorValue | undefined;
274
+ let source: Extract<TaskRoutingSource, "specialty" | "legacy-frontend"> = "specialty";
275
+ if (hasConfiguredModel(configured)) {
276
+ value = configured;
277
+ } else if (request.specialty === "frontendDesign") {
278
+ // Legacy compatibility: the old single frontend selector still answers for
279
+ // frontend design when no explicit entry has replaced it.
280
+ const legacy = settings.get("task.modelRouting.frontendModel")?.trim();
281
+ if (legacy) {
282
+ value = legacy;
283
+ source = "legacy-frontend";
284
+ }
285
+ }
286
+ if (value === undefined) return null;
287
+
288
+ const specialtyCandidates = toRoutingCandidates(value, source, { specialty: request.specialty });
289
+ const head = specialtyCandidates[0];
290
+ if (!head) return null;
291
+ // Already there: the declared model is the role's own. Reporting a route would
292
+ // claim a swap that never happened.
293
+ if (request.currentModel && matchesModel(head.selector, request.currentModel)) return null;
294
+
295
+ const candidates = dedupeRoutingCandidates([specialtyCandidates, baselineCandidates(request)]);
296
+ const effectiveHead = candidates[0];
297
+ if (!effectiveHead || effectiveHead.source === "baseline") return null;
298
+ const label = source === "legacy-frontend" ? "frontendDesign (legacy selector)" : request.specialty;
299
+ const reason = `${label}, declared by caller`;
300
+ logger.debug("decisions/task-routing: routed", { agent: request.agentName, model: effectiveHead.selector, reason });
301
+ return {
302
+ model: effectiveHead.selector,
303
+ tier: null,
304
+ reason,
305
+ candidates,
306
+ requestedSource: source,
307
+ requestedSpecialty: request.specialty,
308
+ declared: true,
309
+ calibrated: false,
310
+ };
90
311
  }
91
312
 
92
313
  /** Where a concrete model id sits in the ladder, or null when it is not one of ours. */
@@ -103,8 +324,7 @@ function rankOf(model: string | undefined, tiers: TaskTierModels): number | null
103
324
  * may or may not repeat, so compare the part before it.
104
325
  */
105
326
  function matchesModel(a: string, b: string): boolean {
106
- const base = (value: string) => value.split(":")[0]?.trim().toLowerCase() ?? "";
107
- return base(a) === base(b);
327
+ return specialtySelectorHead(a) === specialtySelectorHead(b);
108
328
  }
109
329
 
110
330
  /**
@@ -127,6 +347,68 @@ function allowed(
127
347
  return confidence >= (isDowngrade ? policy.minDowngradeConfidence : policy.minUpgradeConfidence);
128
348
  }
129
349
 
350
+ /** The role's own chain, used as the tail of every composed candidate list. */
351
+ function baselineCandidates(
352
+ request: Pick<TaskRoutingRequest, "baselineChain" | "currentModel">,
353
+ ): TaskRoutingCandidate[] {
354
+ const chain = request.baselineChain?.length ? request.baselineChain : [request.currentModel ?? ""];
355
+ return toRoutingCandidates([...chain], "baseline");
356
+ }
357
+
358
+ interface SpecialtySelection {
359
+ specialty: TaskModelSpecialty;
360
+ source: Extract<TaskRoutingSource, "specialty" | "legacy-frontend">;
361
+ value: ModelSelectorValue;
362
+ ordinalStrength?: number;
363
+ confidence?: number;
364
+ }
365
+
366
+ /**
367
+ * Resolve the specialty axis, or null to leave it to the difficulty ladder.
368
+ *
369
+ * Declines are deliberately quiet and numerous: an unrecognized option, `none`,
370
+ * an incompatible role, no configured model, a below-bar answer, or a missing
371
+ * clarity signal all mean "the user did not ask for this, carry on".
372
+ */
373
+ function resolveSpecialty(
374
+ policy: TaskRoutingPolicy,
375
+ request: TaskRoutingRequest,
376
+ answers: Record<string, Answer>,
377
+ calibrated: boolean,
378
+ ): SpecialtySelection | null {
379
+ const answer = answers.specialty;
380
+ if (answer?.type !== "choice") return null;
381
+ const choice = answer.choice;
382
+ if (typeof choice !== "string" || choice === TASK_MODEL_SPECIALTY_NONE) return null;
383
+ if (!isTaskModelSpecialty(choice)) return null;
384
+ if (!specialtySupportsRole(choice, request.agentName)) return null;
385
+
386
+ const clarity = answers.specialtyClear;
387
+ const ordinalStrength = clarity?.type === "noul" && typeof clarity.noul === "number" ? clarity.noul : undefined;
388
+ const confidence = typeof answer.confidence === "number" ? answer.confidence : undefined;
389
+
390
+ if (calibrated) {
391
+ // A calibrated backend reports a real probability; threshold on it directly.
392
+ if (confidence === undefined || confidence < policy.minDomainConfidence) return null;
393
+ } else {
394
+ // No probability is available. Require an explicit high ordinal instead, and
395
+ // never dress that number up as confidence downstream.
396
+ if (ordinalStrength === undefined || ordinalStrength < policy.minSpecialtyOrdinal) return null;
397
+ }
398
+
399
+ const configured = policy.specialtyModels?.[choice];
400
+ if (configured !== undefined && (typeof configured !== "string" || configured.trim().length > 0)) {
401
+ return { specialty: choice, source: "specialty", value: configured, ordinalStrength, confidence };
402
+ }
403
+ // Legacy compatibility: the old single frontend selector still answers for
404
+ // frontend design when no explicit entry has replaced it.
405
+ const legacy = policy.frontendModel?.trim();
406
+ if (choice === "frontendDesign" && legacy) {
407
+ return { specialty: choice, source: "legacy-frontend", value: legacy, ordinalStrength, confidence };
408
+ }
409
+ return null;
410
+ }
411
+
130
412
  /**
131
413
  * Decide the model for one subagent spawn, or null to leave the configured one alone.
132
414
  *
@@ -140,9 +422,13 @@ export async function routeTaskModel(
140
422
  request: TaskRoutingRequest,
141
423
  ): Promise<TaskRoutingResult | null> {
142
424
  const configured = TASK_TIERS.filter(tier => policy.tiers[tier]);
143
- // One tier is not a ladder; with nothing to move between there is no decision
144
- // worth paying a model call for.
145
- if (configured.length < 2) return null;
425
+ const frontendModel = policy.frontendModel?.trim() || undefined;
426
+ const hasSpecialtyModels = TASK_MODEL_SPECIALTY_IDS.some(id => {
427
+ const value = policy.specialtyModels?.[id];
428
+ return Array.isArray(value) ? value.length > 0 : typeof value === "string" && value.trim().length > 0;
429
+ });
430
+ // No axis has anything to move on: no ladder, no specialty model, no legacy one.
431
+ if (configured.length < 2 && !frontendModel && !hasSpecialtyModels) return null;
146
432
 
147
433
  const assignment = request.assignment.trim();
148
434
  if (assignment.length < 24) return null;
@@ -154,43 +440,135 @@ export async function routeTaskModel(
154
440
  });
155
441
  if (!result) return null;
156
442
 
157
- const answer = result.answers.tier;
158
- if (answer?.type !== "choice") return null;
159
- let tier = TASK_TIERS.find(candidate => candidate === answer.choice);
160
- if (!tier) return null;
443
+ const answers = result.answers;
444
+ const baseline = baselineCandidates(request);
161
445
 
446
+ // --- Risk floor ---
447
+ //
162
448
  // Work that cannot be undone takes the most capable tier available and skips
163
449
  // the confidence bars — this one is not a confidence question. It may only
164
450
  // ever raise the tier, never lower it, or "this is risky" would end up
165
- // *downgrading* an assignment already running deep.
166
- const riskAnswer = result.answers.risky;
167
- const forcedByRisk = riskAnswer?.type === "noul" && riskAnswer.noul > 0.7;
168
- if (forcedByRisk) {
451
+ // *downgrading* an assignment already running deep. It also outranks the
452
+ // specialty axis: a design-strong model is not the safety property being
453
+ // asked for here.
454
+ const riskAnswer = answers.risky;
455
+ const forcedByRisk = riskAnswer?.type === "noul" && typeof riskAnswer.noul === "number" && riskAnswer.noul > 0.7;
456
+
457
+ const tierAnswer = answers.tier;
458
+ const declaredTier =
459
+ tierAnswer?.type === "choice" ? TASK_TIERS.find(candidate => candidate === tierAnswer.choice) : undefined;
460
+ const tierConfidence =
461
+ tierAnswer?.type === "choice" && typeof tierAnswer.confidence === "number" ? tierAnswer.confidence : undefined;
462
+
463
+ if (forcedByRisk && configured.length > 0) {
169
464
  const deepest = configured[configured.length - 1] as TaskTier;
170
465
  const currentRank = rankOf(request.currentModel, policy.tiers);
171
466
  const wantedRank = Math.max(TASK_TIERS.indexOf(deepest), currentRank ?? 0);
172
- tier = TASK_TIERS[wantedRank] as TaskTier;
467
+ const tier = TASK_TIERS[wantedRank] as TaskTier;
468
+ const model = policy.tiers[tier];
469
+ if (!model || (request.currentModel && matchesModel(model, request.currentModel))) return null;
470
+ const candidates = dedupeRoutingCandidates([toRoutingCandidates(model, "tier", { tier }), baseline]);
471
+ const head = candidates[0];
472
+ if (!head) return null;
473
+ const reason = `${tier}, forced by risk`;
474
+ logger.debug("decisions/task-routing: routed", { agent: request.agentName, model: head.selector, reason });
475
+ return {
476
+ model: head.selector,
477
+ tier,
478
+ reason,
479
+ candidates,
480
+ requestedSource: "tier",
481
+ requestedTier: tier,
482
+ declared: false,
483
+ calibrated: result.calibrated,
484
+ confidence: tierConfidence,
485
+ };
486
+ }
487
+
488
+ // --- Specialty axis: lateral swap for the kind of work ---
489
+ const specialty = resolveSpecialty(policy, request, answers, result.calibrated);
490
+ if (specialty) {
491
+ const specialtyCandidates = toRoutingCandidates(specialty.value, specialty.source, {
492
+ specialty: specialty.specialty,
493
+ });
494
+ const head = specialtyCandidates[0];
495
+ // Already there: nothing to move, and the ladder should not fire either —
496
+ // the user's specialty choice is the standing answer for this work.
497
+ if (head && request.currentModel && matchesModel(head.selector, request.currentModel)) return null;
498
+ if (head) {
499
+ // The tier is still worth composing *behind* the specialty: if the specialty
500
+ // model cannot be authenticated, the difficulty answer is the next best guess.
501
+ const tierSegment =
502
+ declaredTier && policy.tiers[declaredTier]
503
+ ? toRoutingCandidates(policy.tiers[declaredTier] as string, "tier", { tier: declaredTier })
504
+ : [];
505
+ const candidates = dedupeRoutingCandidates([specialtyCandidates, tierSegment, baseline]);
506
+ const effectiveHead = candidates[0];
507
+ if (effectiveHead && effectiveHead.source !== "baseline") {
508
+ const strength = result.calibrated
509
+ ? `confidence ${specialty.confidence?.toFixed(2) ?? "n/d"}`
510
+ : `clarity ${specialty.ordinalStrength?.toFixed(2) ?? "n/d"}, uncalibrated`;
511
+ const label =
512
+ specialty.source === "legacy-frontend" ? "frontendDesign (legacy selector)" : specialty.specialty;
513
+ const reason = `${label} (${strength})`;
514
+ logger.debug("decisions/task-routing: routed", {
515
+ agent: request.agentName,
516
+ model: effectiveHead.selector,
517
+ reason,
518
+ });
519
+ return {
520
+ model: effectiveHead.selector,
521
+ tier: null,
522
+ reason,
523
+ candidates,
524
+ requestedSource: specialty.source,
525
+ requestedSpecialty: specialty.specialty,
526
+ requestedTier: declaredTier,
527
+ declared: false,
528
+ calibrated: result.calibrated,
529
+ confidence: result.calibrated ? specialty.confidence : undefined,
530
+ ordinalStrength: result.calibrated ? undefined : specialty.ordinalStrength,
531
+ };
532
+ }
533
+ }
173
534
  }
174
535
 
536
+ // --- Difficulty axis: the ladder ---
537
+ if (configured.length < 2) return null;
538
+ if (!declaredTier) return null;
539
+ const tier = declaredTier;
540
+
175
541
  const model = policy.tiers[tier];
176
542
  if (!model || (request.currentModel && matchesModel(model, request.currentModel))) return null;
177
543
 
178
544
  const currentRank = rankOf(request.currentModel, policy.tiers);
179
545
  const wantedRank = TASK_TIERS.indexOf(tier);
180
- if (!forcedByRisk && !allowed(wantedRank, currentRank, answer.confidence, result.calibrated, policy)) {
546
+ if (!allowed(wantedRank, currentRank, tierConfidence, result.calibrated, policy)) {
181
547
  logger.debug("decisions/task-routing: below the bar, keeping the configured model", {
182
548
  agent: request.agentName,
183
549
  wanted: tier,
184
550
  current: request.currentModel,
185
- confidence: answer.confidence,
551
+ confidence: tierConfidence,
552
+ declared: false,
186
553
  calibrated: result.calibrated,
187
554
  });
188
555
  return null;
189
556
  }
190
557
 
191
- const reason = forcedByRisk
192
- ? `${tier}, forced by risk`
193
- : `${tier} (confidence ${answer.confidence?.toFixed(2) ?? "n/d"})`;
194
- logger.debug("decisions/task-routing: routed", { agent: request.agentName, model, reason });
195
- return { model, tier, reason };
558
+ const candidates = dedupeRoutingCandidates([toRoutingCandidates(model, "tier", { tier }), baseline]);
559
+ const head = candidates[0];
560
+ if (!head) return null;
561
+ const reason = `${tier} (confidence ${tierConfidence?.toFixed(2) ?? "n/d"})`;
562
+ logger.debug("decisions/task-routing: routed", { agent: request.agentName, model: head.selector, reason });
563
+ return {
564
+ model: head.selector,
565
+ tier,
566
+ reason,
567
+ candidates,
568
+ requestedSource: "tier",
569
+ requestedTier: tier,
570
+ declared: false,
571
+ calibrated: result.calibrated,
572
+ confidence: tierConfidence,
573
+ };
196
574
  }
@@ -75,6 +75,22 @@ export const de: Partial<Record<MsgKey, string>> = {
75
75
  "modelSelector.noMatching": "Keine passenden Modelle.",
76
76
  "modelSelector.modelName": "Modellname: {value}",
77
77
  "modelSelector.actionFor": "Aktion für: {id}",
78
+ "modelSelector.setAsTarget": "Als {tag} ({name}) festlegen",
79
+ "modelSelector.setForAllRoleAgents": "Fur alle Rollen-Agenten setzen",
80
+ "modelSelector.setForAllTargets": "Fur alle Ziele setzen",
81
+ "modelSelector.hasDetailedUses": "(detaillierte Aufgaben)",
82
+ "modelSelector.detailedUseFor": "Detaillierte Aufgabe fur {target}: {id}",
83
+ "modelSelector.generalRole": "Allgemein (ganze Rolle)",
84
+ "modelSelector.resetSpecialties": "Detaillierte Aufgaben zurucksetzen",
85
+ "modelSelector.specialty.backendArchitecture": "Backend-Architektur",
86
+ "modelSelector.specialty.frontendDesign": "Frontend-Design",
87
+ "modelSelector.specialty.implementation": "Implementierung",
88
+ "modelSelector.specialty.testing": "Tests",
89
+ "modelSelector.specialty.review": "Review",
90
+ "modelSelector.specialtySaved": "{specialty} verwendet {value}.",
91
+ "modelSelector.specialtySavedRoutingOff":
92
+ "{specialty} verwendet {value}, sobald eine Aufgabe diese Arbeit deklariert. Automatische Erkennung aus dem Auftragstext bleibt aus, bis task.modelRouting.enabled aktiv ist.",
93
+ "modelSelector.specialtyCleared": "Detaillierte Aufgaben fur {target} zuruckgesetzt.",
78
94
  "modelSelector.reasoningFor": "Reasoning für {target}: {id}",
79
95
  "modelSelector.temporaryModel": "temporäres Modell",
80
96
  };
@@ -95,6 +95,22 @@ export const en = {
95
95
  "modelSelector.noMatching": "No matching models.",
96
96
  "modelSelector.modelName": "Model Name: {value}",
97
97
  "modelSelector.actionFor": "Action for: {id}",
98
+ "modelSelector.setAsTarget": "Set as {tag} ({name})",
99
+ "modelSelector.setForAllRoleAgents": "Set for all role agents",
100
+ "modelSelector.setForAllTargets": "Set for all targets",
101
+ "modelSelector.hasDetailedUses": "(detailed uses)",
102
+ "modelSelector.detailedUseFor": "Detailed use for {target}: {id}",
103
+ "modelSelector.generalRole": "General (whole role)",
104
+ "modelSelector.resetSpecialties": "Clear detailed-use overrides",
105
+ "modelSelector.specialty.backendArchitecture": "Backend architecture",
106
+ "modelSelector.specialty.frontendDesign": "Frontend design",
107
+ "modelSelector.specialty.implementation": "Implementation",
108
+ "modelSelector.specialty.testing": "Testing",
109
+ "modelSelector.specialty.review": "Review",
110
+ "modelSelector.specialtySaved": "{specialty} will use {value}.",
111
+ "modelSelector.specialtySavedRoutingOff":
112
+ "{specialty} will use {value} whenever a task declares that work. Auto-detection from assignment text stays off until you enable task.modelRouting.enabled.",
113
+ "modelSelector.specialtyCleared": "Cleared detailed-use overrides for {target}.",
98
114
  "modelSelector.reasoningFor": "Reasoning for {target}: {id}",
99
115
  "modelSelector.temporaryModel": "temporary model",
100
116
  } as const;
@@ -75,6 +75,22 @@ export const es: Partial<Record<MsgKey, string>> = {
75
75
  "modelSelector.noMatching": "No hay modelos coincidentes.",
76
76
  "modelSelector.modelName": "Nombre del modelo: {value}",
77
77
  "modelSelector.actionFor": "Acción para: {id}",
78
+ "modelSelector.setAsTarget": "Asignar como {tag} ({name})",
79
+ "modelSelector.setForAllRoleAgents": "Asignar a todos los agentes de rol",
80
+ "modelSelector.setForAllTargets": "Asignar a todos los destinos",
81
+ "modelSelector.hasDetailedUses": "(usos detallados)",
82
+ "modelSelector.detailedUseFor": "Uso detallado de {target}: {id}",
83
+ "modelSelector.generalRole": "General (todo el rol)",
84
+ "modelSelector.resetSpecialties": "Borrar los usos detallados",
85
+ "modelSelector.specialty.backendArchitecture": "Arquitectura de backend",
86
+ "modelSelector.specialty.frontendDesign": "Diseño de frontend",
87
+ "modelSelector.specialty.implementation": "Implementación",
88
+ "modelSelector.specialty.testing": "Pruebas",
89
+ "modelSelector.specialty.review": "Revisión",
90
+ "modelSelector.specialtySaved": "{specialty} usará {value}.",
91
+ "modelSelector.specialtySavedRoutingOff":
92
+ "{specialty} usará {value} cuando una tarea declare ese trabajo. La detección automática desde el texto de la tarea sigue desactivada hasta que actives task.modelRouting.enabled.",
93
+ "modelSelector.specialtyCleared": "Se borraron los usos detallados de {target}.",
78
94
  "modelSelector.reasoningFor": "Razonamiento para {target}: {id}",
79
95
  "modelSelector.temporaryModel": "modelo temporal",
80
96
  };
@@ -75,6 +75,22 @@ export const fr: Partial<Record<MsgKey, string>> = {
75
75
  "modelSelector.noMatching": "Aucun modèle correspondant.",
76
76
  "modelSelector.modelName": "Nom du modèle : {value}",
77
77
  "modelSelector.actionFor": "Action pour : {id}",
78
+ "modelSelector.setAsTarget": "Définir comme {tag} ({name})",
79
+ "modelSelector.setForAllRoleAgents": "Definir pour tous les agents de role",
80
+ "modelSelector.setForAllTargets": "Definir pour toutes les cibles",
81
+ "modelSelector.hasDetailedUses": "(usages detailles)",
82
+ "modelSelector.detailedUseFor": "Usage detaille pour {target} : {id}",
83
+ "modelSelector.generalRole": "General (tout le role)",
84
+ "modelSelector.resetSpecialties": "Effacer les usages detailles",
85
+ "modelSelector.specialty.backendArchitecture": "Architecture backend",
86
+ "modelSelector.specialty.frontendDesign": "Design frontend",
87
+ "modelSelector.specialty.implementation": "Implementation",
88
+ "modelSelector.specialty.testing": "Tests",
89
+ "modelSelector.specialty.review": "Revue",
90
+ "modelSelector.specialtySaved": "{specialty} utilisera {value}.",
91
+ "modelSelector.specialtySavedRoutingOff":
92
+ "{specialty} utilisera {value} des qu'une tache declare ce travail. La detection automatique depuis le texte reste inactive sans task.modelRouting.enabled.",
93
+ "modelSelector.specialtyCleared": "Usages detailles effaces pour {target}.",
78
94
  "modelSelector.reasoningFor": "Raisonnement pour {target} : {id}",
79
95
  "modelSelector.temporaryModel": "modèle temporaire",
80
96
  };
@@ -79,6 +79,22 @@ export const ja: Partial<Record<MsgKey, string>> = {
79
79
  "modelSelector.noMatching": "一致するモデルがありません。",
80
80
  "modelSelector.modelName": "モデル名: {value}",
81
81
  "modelSelector.actionFor": "操作対象: {id}",
82
+ "modelSelector.setAsTarget": "{tag} に設定",
83
+ "modelSelector.setForAllRoleAgents": "すべてのロールエージェントに設定",
84
+ "modelSelector.setForAllTargets": "すべての対象に設定",
85
+ "modelSelector.hasDetailedUses": "(詳細用途あり)",
86
+ "modelSelector.detailedUseFor": "{target} の詳細用途: {id}",
87
+ "modelSelector.generalRole": "一般(ロール全体)",
88
+ "modelSelector.resetSpecialties": "詳細用途の設定を解除",
89
+ "modelSelector.specialty.backendArchitecture": "バックエンド設計",
90
+ "modelSelector.specialty.frontendDesign": "フロントエンドデザイン",
91
+ "modelSelector.specialty.implementation": "実装",
92
+ "modelSelector.specialty.testing": "テスト",
93
+ "modelSelector.specialty.review": "レビュー",
94
+ "modelSelector.specialtySaved": "{specialty} には {value} を使用します。",
95
+ "modelSelector.specialtySavedRoutingOff":
96
+ "{specialty} として宣言されたタスクには {value} を使用します。指示文からの自動判定は task.modelRouting.enabled を有効にするまで無効です。",
97
+ "modelSelector.specialtyCleared": "{target} の詳細用途設定を解除しました。",
82
98
  "modelSelector.reasoningFor": "{target} の推論レベル: {id}",
83
99
  "modelSelector.temporaryModel": "一時モデル",
84
100
  };
@@ -79,6 +79,22 @@ export const ko: Partial<Record<MsgKey, string>> = {
79
79
  "modelSelector.noMatching": "일치하는 모델이 없습니다.",
80
80
  "modelSelector.modelName": "모델 이름: {value}",
81
81
  "modelSelector.actionFor": "작업 대상: {id}",
82
+ "modelSelector.setAsTarget": "{tag}에 설정",
83
+ "modelSelector.setForAllRoleAgents": "모든 역할 에이전트에 설정",
84
+ "modelSelector.setForAllTargets": "모든 대상에 설정",
85
+ "modelSelector.hasDetailedUses": "(세부 업무 있음)",
86
+ "modelSelector.detailedUseFor": "{target} 세부 업무: {id}",
87
+ "modelSelector.generalRole": "일반 (역할 전체)",
88
+ "modelSelector.resetSpecialties": "세부 업무 설정 해제",
89
+ "modelSelector.specialty.backendArchitecture": "백엔드 아키텍처",
90
+ "modelSelector.specialty.frontendDesign": "프런트엔드 디자인",
91
+ "modelSelector.specialty.implementation": "구현",
92
+ "modelSelector.specialty.testing": "테스트",
93
+ "modelSelector.specialty.review": "리뷰",
94
+ "modelSelector.specialtySaved": "{specialty} 작업에 {value}을(를) 사용합니다.",
95
+ "modelSelector.specialtySavedRoutingOff":
96
+ "{specialty} 작업으로 선언된 태스크는 {value}을(를) 사용합니다. 지시문에서 자동 감지하려면 task.modelRouting.enabled를 켜야 합니다.",
97
+ "modelSelector.specialtyCleared": "{target}의 세부 업무 설정을 해제했습니다.",
82
98
  "modelSelector.reasoningFor": "{target} 추론 수준: {id}",
83
99
  "modelSelector.temporaryModel": "임시 모델",
84
100
  };