@sayknow-cli/coding-agent 0.5.25 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +36 -1
- package/dist/types/config/settings-schema.d.ts +51 -5
- package/dist/types/config/task-model-specialties.d.ts +55 -0
- package/dist/types/decisions/keyword-learning.d.ts +61 -0
- package/dist/types/decisions/llm-backend.d.ts +13 -1
- package/dist/types/decisions/prompt-triage.d.ts +42 -0
- package/dist/types/decisions/skill-routing.d.ts +41 -6
- package/dist/types/decisions/task-routing.d.ts +96 -11
- package/dist/types/hooks/native-prompt-routing.d.ts +21 -0
- package/dist/types/hooks/native-skill-hook.d.ts +3 -0
- package/dist/types/hooks/skill-keywords.d.ts +9 -0
- package/dist/types/hooks/skill-state.d.ts +20 -3
- package/dist/types/hooks/ui-skill-keywords.d.ts +15 -0
- package/dist/types/i18n/messages/en.d.ts +15 -0
- package/dist/types/lsp/index.d.ts +1 -1
- package/dist/types/lsp/types.d.ts +1 -1
- package/dist/types/modes/components/model-selector.d.ts +11 -0
- package/dist/types/sdk/session.d.ts +3 -13
- package/dist/types/session/agent-session.d.ts +8 -0
- package/dist/types/session/auth-storage-discovery.d.ts +13 -0
- package/dist/types/task/index.d.ts +1 -1
- package/dist/types/task/receipt.d.ts +2 -0
- package/dist/types/task/types.d.ts +114 -18
- package/dist/types/tools/browser.d.ts +2 -2
- package/dist/types/tools/subagent.d.ts +2 -2
- package/package.json +7 -7
- package/scripts/eval-skill-routing.ts +37 -12
- package/src/config/settings-schema.ts +64 -12
- package/src/config/task-model-specialties.ts +131 -0
- package/src/decisions/index.ts +8 -2
- package/src/decisions/keyword-learning.ts +678 -0
- package/src/decisions/llm-backend.ts +213 -67
- package/src/decisions/prompt-triage.ts +163 -0
- package/src/decisions/skill-routing.ts +39 -56
- package/src/decisions/task-routing.ts +382 -66
- package/src/decisions/typesafe-backend.ts +3 -0
- package/src/hooks/native-prompt-routing.ts +190 -0
- package/src/hooks/native-skill-hook.ts +21 -12
- package/src/hooks/skill-keywords.ts +9 -0
- package/src/hooks/skill-state.ts +41 -10
- package/src/hooks/ui-skill-keywords.ts +67 -10
- package/src/i18n/messages/de.ts +16 -0
- package/src/i18n/messages/en.ts +16 -0
- package/src/i18n/messages/es.ts +16 -0
- package/src/i18n/messages/fr.ts +16 -0
- package/src/i18n/messages/ja.ts +16 -0
- package/src/i18n/messages/ko.ts +16 -0
- package/src/i18n/messages/zh.ts +16 -0
- package/src/internal-urls/docs-index.generated.ts +1 -1
- package/src/main.ts +1 -1
- package/src/modes/components/model-selector.ts +275 -34
- package/src/modes/controllers/selector-controller.ts +50 -2
- package/src/modes/shared/agent-wire/command-dispatch.ts +1 -1
- package/src/prompts/tools/task.md +1 -0
- package/src/sdk/session.ts +5 -82
- package/src/session/agent-session.ts +137 -37
- package/src/session/auth-storage-discovery.ts +83 -0
- package/src/slash-commands/builtin-registry.ts +11 -9
- package/src/task/index.ts +98 -40
- package/src/task/receipt.ts +3 -0
- package/src/task/types.ts +44 -0
|
@@ -5,6 +5,7 @@ import { getThinkingLevelMetadata } from "../thinking-metadata";
|
|
|
5
5
|
import { EDIT_MODES } from "../utils/edit-mode";
|
|
6
6
|
import { CONFIGURABLE_SEARCH_PROVIDER_IDS } from "../web/search/types";
|
|
7
7
|
import type { ModelSelectorValue } from "./model-selector-value";
|
|
8
|
+
import { TASK_MODEL_SPECIALTY_IDS } from "./task-model-specialties";
|
|
8
9
|
|
|
9
10
|
const THINKING_EFFORTS = ["minimal", "low", "medium", "high", "xhigh", "max"] as readonly Effort[];
|
|
10
11
|
const DEFAULT_THINKING_LEVELS = ["off", ...THINKING_EFFORTS] as const;
|
|
@@ -192,6 +193,11 @@ interface RecordDef<T> {
|
|
|
192
193
|
type: "record";
|
|
193
194
|
default: Record<string, T>;
|
|
194
195
|
valueSchema?: RecordValueDef;
|
|
196
|
+
/**
|
|
197
|
+
* Closed key set. When present, reconciliation rejects any other key instead of
|
|
198
|
+
* silently carrying a typo that no consumer will ever read.
|
|
199
|
+
*/
|
|
200
|
+
keys?: readonly string[];
|
|
195
201
|
ui?: UiBase;
|
|
196
202
|
}
|
|
197
203
|
|
|
@@ -2002,12 +2008,32 @@ export const SETTINGS_SCHEMA = {
|
|
|
2002
2008
|
// Typed decisions
|
|
2003
2009
|
"decisions.enabled": {
|
|
2004
2010
|
type: "boolean",
|
|
2005
|
-
default:
|
|
2011
|
+
default: true,
|
|
2006
2012
|
ui: {
|
|
2007
2013
|
tab: "context",
|
|
2008
2014
|
label: "Typed decisions",
|
|
2009
2015
|
description:
|
|
2010
|
-
"
|
|
2016
|
+
"Model-backed second stage for workflow routing. The keyword table runs first on every turn and costs nothing; this handles the phrasings it cannot express, which is most wording that is not a literal match. Costs one small model call, only on turns the keyword table did not already answer, on the cheapest backend available: TypeSafe when a key is stored, else a local runtime that is running, else the small model of the provider you are chatting with. Any failure falls back to keyword-only behaviour, but a successful answer can also select a different workflow than the deep-interview ambiguity detector would have. Turn off to route on the keyword table alone.",
|
|
2017
|
+
},
|
|
2018
|
+
},
|
|
2019
|
+
|
|
2020
|
+
/**
|
|
2021
|
+
* Feed routing answers back into the deterministic keyword table.
|
|
2022
|
+
*
|
|
2023
|
+
* The hand-written table cannot be grown by hand for Korean — measured recall
|
|
2024
|
+
* was 0/9 — so it grows itself instead: a two-stem pattern that produced the
|
|
2025
|
+
* same routing answer on two distinct prompts, and was never contradicted, is
|
|
2026
|
+
* promoted and thereafter fires for free. One contradiction retracts it
|
|
2027
|
+
* permanently. Stems and hashes are stored; prompt text never is.
|
|
2028
|
+
*/
|
|
2029
|
+
"decisions.keywordLearning": {
|
|
2030
|
+
type: "boolean",
|
|
2031
|
+
default: true,
|
|
2032
|
+
ui: {
|
|
2033
|
+
tab: "context",
|
|
2034
|
+
label: "Learn routing keywords",
|
|
2035
|
+
description:
|
|
2036
|
+
"Remember the phrasings the routing model resolves, so repeating one stops costing a model call. A pattern must give the same answer on two different prompts before it fires on its own, and a single disagreement removes it for good. Only word stems and hashes are written to disk, never your prompts. Turn off to keep the keyword table frozen at its built-in entries.",
|
|
2011
2037
|
},
|
|
2012
2038
|
},
|
|
2013
2039
|
|
|
@@ -2020,12 +2046,14 @@ export const SETTINGS_SCHEMA = {
|
|
|
2020
2046
|
* mid-session invalidates the prompt cache, which on a long context costs
|
|
2021
2047
|
* more than the cheaper tier saves.
|
|
2022
2048
|
*
|
|
2023
|
-
* Needs `decisions.enabled` and at least two tiers configured.
|
|
2024
|
-
*
|
|
2049
|
+
* Needs `decisions.enabled` and at least two tiers configured. The tiers are
|
|
2050
|
+
* empty by default, so this is a no-op until the user sets them — which is why
|
|
2051
|
+
* it defaults on: shipping it off meant the routing code existed and never ran
|
|
2052
|
+
* even for users who had configured the tiers it needs.
|
|
2025
2053
|
*/
|
|
2026
2054
|
"task.modelRouting.enabled": {
|
|
2027
2055
|
type: "boolean",
|
|
2028
|
-
default:
|
|
2056
|
+
default: true,
|
|
2029
2057
|
ui: {
|
|
2030
2058
|
tab: "tasks",
|
|
2031
2059
|
label: "Route subagent models per task",
|
|
@@ -2050,6 +2078,25 @@ export const SETTINGS_SCHEMA = {
|
|
|
2050
2078
|
* frontend, not to write it.
|
|
2051
2079
|
*/
|
|
2052
2080
|
"task.modelRouting.frontendModel": { type: "string", default: "" },
|
|
2081
|
+
/**
|
|
2082
|
+
* Per-specialty models — the **work-kind** axis.
|
|
2083
|
+
*
|
|
2084
|
+
* Keys are the bounded ids in `config/task-model-specialties.ts`; values use the
|
|
2085
|
+
* same selector grammar as any other model setting, so a chain is allowed. A
|
|
2086
|
+
* specialty with no entry inherits the role's resolved model unchanged, which is
|
|
2087
|
+
* why the default is empty rather than pre-populated.
|
|
2088
|
+
*
|
|
2089
|
+
* These are routing hints, never agents: nothing here widens the canonical role
|
|
2090
|
+
* roster, model-profile role keys, tool grants, or spawn permissions. Saving one
|
|
2091
|
+
* does not enable `task.modelRouting.enabled`; the assignment surface reports the
|
|
2092
|
+
* disabled state instead of silently turning routing on.
|
|
2093
|
+
*/
|
|
2094
|
+
"task.modelRouting.specialtyModels": {
|
|
2095
|
+
type: "record",
|
|
2096
|
+
default: {} as Record<string, ModelSelectorValue>,
|
|
2097
|
+
valueSchema: MODEL_SELECTOR_VALUE_SCHEMA,
|
|
2098
|
+
keys: TASK_MODEL_SPECIALTY_IDS,
|
|
2099
|
+
},
|
|
2053
2100
|
|
|
2054
2101
|
// TTSR
|
|
2055
2102
|
"ttsr.enabled": {
|
|
@@ -4079,15 +4126,20 @@ export function reconcileSettingsSchema(raw: Record<string, unknown>): {
|
|
|
4079
4126
|
}
|
|
4080
4127
|
if (!validSettingValue(definition, next))
|
|
4081
4128
|
issues.push({ path, kind: "invalid", detail: `Expected ${definition.type}.` });
|
|
4082
|
-
if (
|
|
4083
|
-
|
|
4084
|
-
"valueSchema" in definition
|
|
4085
|
-
definition.valueSchema &&
|
|
4086
|
-
validSettingValue(definition, next)
|
|
4087
|
-
) {
|
|
4129
|
+
if (definition.type === "record" && validSettingValue(definition, next)) {
|
|
4130
|
+
const allowedKeys = "keys" in definition && definition.keys ? new Set<string>(definition.keys) : undefined;
|
|
4131
|
+
const valueSchema = "valueSchema" in definition ? definition.valueSchema : undefined;
|
|
4088
4132
|
for (const [key, entry] of Object.entries(next as Record<string, unknown>)) {
|
|
4133
|
+
if (allowedKeys && !allowedKeys.has(key)) {
|
|
4134
|
+
issues.push({
|
|
4135
|
+
path: `${path}.${key}`,
|
|
4136
|
+
kind: "invalid",
|
|
4137
|
+
detail: `Unknown key. Expected one of: ${[...allowedKeys].join(", ")}.`,
|
|
4138
|
+
});
|
|
4139
|
+
continue;
|
|
4140
|
+
}
|
|
4089
4141
|
if (
|
|
4090
|
-
|
|
4142
|
+
valueSchema?.type === "model-selector-value" &&
|
|
4091
4143
|
!(typeof entry === "string" || (Array.isArray(entry) && entry.every(item => typeof item === "string")))
|
|
4092
4144
|
) {
|
|
4093
4145
|
issues.push({ path: `${path}.${key}`, kind: "invalid", detail: "Expected model-selector-value." });
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Bounded work-specialty axis for subagent model selection.
|
|
3
|
+
*
|
|
4
|
+
* A canonical role model is a standing guess about the *average* assignment that
|
|
5
|
+
* role gets. The difficulty ladder in `decisions/task-routing.ts` moves that guess
|
|
6
|
+
* when one assignment is unusually hard or unusually mechanical. This axis moves it
|
|
7
|
+
* for a different reason: the assignment belongs to a *kind of work* the user has
|
|
8
|
+
* deliberately picked a model for — backend architecture, frontend design, plain
|
|
9
|
+
* implementation, test work, or review.
|
|
10
|
+
*
|
|
11
|
+
* Specialties are **not agents**. They never appear in agent discovery, task `agent`
|
|
12
|
+
* values, prompts, tool grants, spawn allowlists, or model-profile role keys. The
|
|
13
|
+
* canonical roster stays exactly `executor`, `architect`, `planner`, `critic`, and
|
|
14
|
+
* `default` remains the main-model assignment target. A specialty only ever selects
|
|
15
|
+
* a model for work that is already being routed to one of those roles.
|
|
16
|
+
*/
|
|
17
|
+
import { normalizeModelSelectorValue } from "./model-selector-value";
|
|
18
|
+
|
|
19
|
+
/** Stable configuration/routing identifiers. Display strings are localized separately. */
|
|
20
|
+
export const TASK_MODEL_SPECIALTY_IDS = [
|
|
21
|
+
"backendArchitecture",
|
|
22
|
+
"frontendDesign",
|
|
23
|
+
"implementation",
|
|
24
|
+
"testing",
|
|
25
|
+
"review",
|
|
26
|
+
] as const;
|
|
27
|
+
|
|
28
|
+
export type TaskModelSpecialty = (typeof TASK_MODEL_SPECIALTY_IDS)[number];
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* Which canonical role agents may receive each specialty.
|
|
32
|
+
*
|
|
33
|
+
* Planning roles share the two design specialties on purpose: the user's intent is
|
|
34
|
+
* "a backend-strong model plans backend work", not "planner and architect each get
|
|
35
|
+
* their own private backend model". Implementation roles never take a design
|
|
36
|
+
* specialty, mirroring the existing frontend-domain rule.
|
|
37
|
+
*/
|
|
38
|
+
export const TASK_MODEL_SPECIALTY_ROLES: Readonly<Record<TaskModelSpecialty, readonly string[]>> = {
|
|
39
|
+
backendArchitecture: ["planner", "architect"],
|
|
40
|
+
frontendDesign: ["planner", "architect"],
|
|
41
|
+
implementation: ["executor"],
|
|
42
|
+
testing: ["executor"],
|
|
43
|
+
review: ["critic"],
|
|
44
|
+
};
|
|
45
|
+
|
|
46
|
+
/** Neutral classifier outcome: the work does not clearly belong to one specialty. */
|
|
47
|
+
export const TASK_MODEL_SPECIALTY_NONE = "none" as const;
|
|
48
|
+
|
|
49
|
+
/** Bounded-key predicate for the `task.modelRouting.specialtyModels` record. */
|
|
50
|
+
export function isTaskModelSpecialty(value: unknown): value is TaskModelSpecialty {
|
|
51
|
+
return typeof value === "string" && (TASK_MODEL_SPECIALTY_IDS as readonly string[]).includes(value);
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
/** Specialties a given canonical role agent is allowed to receive, in declaration order. */
|
|
55
|
+
export function specialtiesForRole(agentName: string): TaskModelSpecialty[] {
|
|
56
|
+
return TASK_MODEL_SPECIALTY_IDS.filter(specialty => TASK_MODEL_SPECIALTY_ROLES[specialty].includes(agentName));
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/** Whether a specialty may be applied to work routed to this canonical role agent. */
|
|
60
|
+
export function specialtySupportsRole(specialty: TaskModelSpecialty, agentName: string): boolean {
|
|
61
|
+
return TASK_MODEL_SPECIALTY_ROLES[specialty].includes(agentName);
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/** Where a composed candidate came from, so a receipt never overstates what was selected. */
|
|
65
|
+
export type TaskRoutingSource = "specialty" | "legacy-frontend" | "tier" | "baseline";
|
|
66
|
+
|
|
67
|
+
export interface TaskRoutingCandidate {
|
|
68
|
+
/** The configured selector, normalized but otherwise untouched. */
|
|
69
|
+
selector: string;
|
|
70
|
+
source: TaskRoutingSource;
|
|
71
|
+
/** Set when `source` is `specialty` or `legacy-frontend`. */
|
|
72
|
+
specialty?: TaskModelSpecialty;
|
|
73
|
+
/** Set when `source` is `tier`. */
|
|
74
|
+
tier?: string;
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Identity used for deduplication.
|
|
79
|
+
*
|
|
80
|
+
* An explicit thinking suffix is part of the identity: `provider/model:high` and
|
|
81
|
+
* `provider/model:low` are different configured intents, and collapsing them would
|
|
82
|
+
* silently drop the user's effort choice from a chain.
|
|
83
|
+
*/
|
|
84
|
+
export function specialtySelectorIdentity(selector: string): string {
|
|
85
|
+
return selector.trim().toLowerCase();
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/** The `provider/model` part, ignoring any thinking suffix. Used for "same model" checks. */
|
|
89
|
+
export function specialtySelectorHead(selector: string): string {
|
|
90
|
+
return specialtySelectorIdentity(selector).split(":")[0]?.trim() ?? "";
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/** Expand a configured selector value into normalized candidates tagged with their origin. */
|
|
94
|
+
export function toRoutingCandidates(
|
|
95
|
+
value: string | readonly string[] | undefined,
|
|
96
|
+
source: TaskRoutingSource,
|
|
97
|
+
extra?: { specialty?: TaskModelSpecialty; tier?: string },
|
|
98
|
+
): TaskRoutingCandidate[] {
|
|
99
|
+
return normalizeModelSelectorValue(value)
|
|
100
|
+
.map(selector => selector.trim())
|
|
101
|
+
.filter(selector => selector.length > 0)
|
|
102
|
+
.map(selector => ({ selector, source, ...extra }));
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/**
|
|
106
|
+
* Concatenate candidate segments, keeping the first occurrence of each identity.
|
|
107
|
+
*
|
|
108
|
+
* A selector that also exists in the baseline segment is attributed to `baseline`
|
|
109
|
+
* even when an earlier specialty segment introduced it. Resolving that entry proves
|
|
110
|
+
* only that the role's own configured model was usable — reporting it as a specialty
|
|
111
|
+
* hit would claim a routing decision that never happened.
|
|
112
|
+
*/
|
|
113
|
+
export function dedupeRoutingCandidates(segments: readonly TaskRoutingCandidate[][]): TaskRoutingCandidate[] {
|
|
114
|
+
const flattened = segments.flat();
|
|
115
|
+
const baselineIdentities = new Set(
|
|
116
|
+
flattened.filter(candidate => candidate.source === "baseline").map(c => specialtySelectorIdentity(c.selector)),
|
|
117
|
+
);
|
|
118
|
+
const seen = new Set<string>();
|
|
119
|
+
const deduped: TaskRoutingCandidate[] = [];
|
|
120
|
+
for (const candidate of flattened) {
|
|
121
|
+
const identity = specialtySelectorIdentity(candidate.selector);
|
|
122
|
+
if (seen.has(identity)) continue;
|
|
123
|
+
seen.add(identity);
|
|
124
|
+
deduped.push(
|
|
125
|
+
candidate.source !== "baseline" && baselineIdentities.has(identity)
|
|
126
|
+
? { selector: candidate.selector, source: "baseline" }
|
|
127
|
+
: candidate,
|
|
128
|
+
);
|
|
129
|
+
}
|
|
130
|
+
return deduped;
|
|
131
|
+
}
|
package/src/decisions/index.ts
CHANGED
|
@@ -43,8 +43,9 @@ export function createDecisionService(options: DecisionServiceOptions): Decision
|
|
|
43
43
|
* probabilities, and it resolves `null` immediately when no key is stored, so users
|
|
44
44
|
* who never added one pay nothing for it being in the list.
|
|
45
45
|
*
|
|
46
|
-
* The user's logged-in
|
|
47
|
-
* vendor, no extra key, works offline of TypeSafe entirely.
|
|
46
|
+
* The user's logged-in provider is the fallback and the default experience: no extra
|
|
47
|
+
* vendor, no extra key, works offline of TypeSafe entirely. Inside it, a live local
|
|
48
|
+
* runtime is tried before a hosted small model; see `llm-backend.ts`.
|
|
48
49
|
*/
|
|
49
50
|
const backends = options.backends ?? [createTypeSafeDecisionBackend(options), createLlmDecisionBackend(options)];
|
|
50
51
|
|
|
@@ -53,6 +54,11 @@ export function createDecisionService(options: DecisionServiceOptions): Decision
|
|
|
53
54
|
async decide(request: DecisionRequest): Promise<DecisionResult | null> {
|
|
54
55
|
if (!enabled || backends.length === 0) return null;
|
|
55
56
|
for (const backend of backends) {
|
|
57
|
+
// A listener added to a signal that already fired never runs, and the
|
|
58
|
+
// backends only read their own derived signal. Measured: a caller that
|
|
59
|
+
// aborted during the previous backend's await got a full attempt budget of
|
|
60
|
+
// silence from the next one instead of an immediate null.
|
|
61
|
+
if (request.signal?.aborted) return null;
|
|
56
62
|
const controller = new AbortController();
|
|
57
63
|
const abortOnCallerSignal = () => controller.abort();
|
|
58
64
|
request.signal?.addEventListener("abort", abortOnCallerSignal, { once: true });
|