@sayknow-cli/coding-agent 0.5.25 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +36 -1
- package/dist/types/config/settings-schema.d.ts +51 -5
- package/dist/types/config/task-model-specialties.d.ts +55 -0
- package/dist/types/decisions/keyword-learning.d.ts +61 -0
- package/dist/types/decisions/llm-backend.d.ts +13 -1
- package/dist/types/decisions/prompt-triage.d.ts +42 -0
- package/dist/types/decisions/skill-routing.d.ts +41 -6
- package/dist/types/decisions/task-routing.d.ts +96 -11
- package/dist/types/hooks/native-prompt-routing.d.ts +21 -0
- package/dist/types/hooks/native-skill-hook.d.ts +3 -0
- package/dist/types/hooks/skill-keywords.d.ts +9 -0
- package/dist/types/hooks/skill-state.d.ts +20 -3
- package/dist/types/hooks/ui-skill-keywords.d.ts +15 -0
- package/dist/types/i18n/messages/en.d.ts +15 -0
- package/dist/types/lsp/index.d.ts +1 -1
- package/dist/types/lsp/types.d.ts +1 -1
- package/dist/types/modes/components/model-selector.d.ts +11 -0
- package/dist/types/sdk/session.d.ts +3 -13
- package/dist/types/session/agent-session.d.ts +8 -0
- package/dist/types/session/auth-storage-discovery.d.ts +13 -0
- package/dist/types/task/index.d.ts +1 -1
- package/dist/types/task/receipt.d.ts +2 -0
- package/dist/types/task/types.d.ts +114 -18
- package/dist/types/tools/browser.d.ts +2 -2
- package/dist/types/tools/subagent.d.ts +2 -2
- package/package.json +7 -7
- package/scripts/eval-skill-routing.ts +37 -12
- package/src/config/settings-schema.ts +64 -12
- package/src/config/task-model-specialties.ts +131 -0
- package/src/decisions/index.ts +8 -2
- package/src/decisions/keyword-learning.ts +678 -0
- package/src/decisions/llm-backend.ts +213 -67
- package/src/decisions/prompt-triage.ts +163 -0
- package/src/decisions/skill-routing.ts +39 -56
- package/src/decisions/task-routing.ts +382 -66
- package/src/decisions/typesafe-backend.ts +3 -0
- package/src/hooks/native-prompt-routing.ts +190 -0
- package/src/hooks/native-skill-hook.ts +21 -12
- package/src/hooks/skill-keywords.ts +9 -0
- package/src/hooks/skill-state.ts +41 -10
- package/src/hooks/ui-skill-keywords.ts +67 -10
- package/src/i18n/messages/de.ts +16 -0
- package/src/i18n/messages/en.ts +16 -0
- package/src/i18n/messages/es.ts +16 -0
- package/src/i18n/messages/fr.ts +16 -0
- package/src/i18n/messages/ja.ts +16 -0
- package/src/i18n/messages/ko.ts +16 -0
- package/src/i18n/messages/zh.ts +16 -0
- package/src/internal-urls/docs-index.generated.ts +1 -1
- package/src/main.ts +1 -1
- package/src/modes/components/model-selector.ts +275 -34
- package/src/modes/controllers/selector-controller.ts +50 -2
- package/src/modes/shared/agent-wire/command-dispatch.ts +1 -1
- package/src/prompts/tools/task.md +1 -0
- package/src/sdk/session.ts +5 -82
- package/src/session/agent-session.ts +137 -37
- package/src/session/auth-storage-discovery.ts +83 -0
- package/src/slash-commands/builtin-registry.ts +11 -9
- package/src/task/index.ts +98 -40
- package/src/task/receipt.ts +3 -0
- package/src/task/types.ts +44 -0
|
@@ -43,6 +43,21 @@ export declare function detectUiSkillKeywords(text: string): UiSkillKeywordMatch
|
|
|
43
43
|
* UI/UX work.
|
|
44
44
|
*/
|
|
45
45
|
export declare function buildUiSkillActivationContext(text: string): string | null;
|
|
46
|
+
/**
|
|
47
|
+
* Same directive, for a skill chosen by the semantic stage rather than by a
|
|
48
|
+
* pattern. The regex table caught 5 of 8 real frontend prompts; this is the path
|
|
49
|
+
* for the other three, and it names its source so a wrong pick is traceable to
|
|
50
|
+
* the model rather than to a pattern nobody can find.
|
|
51
|
+
*/
|
|
52
|
+
export declare function buildUiSkillDirectiveForSkill(skill: BundledSkcUiSkillName): string;
|
|
53
|
+
/**
|
|
54
|
+
* What each bundled skill is *for*, in the words a user would recognise.
|
|
55
|
+
*
|
|
56
|
+
* This is the whole contract with the routing model — the skill ids alone carry
|
|
57
|
+
* almost no signal, and `appllama-app-design-skill` carries actively misleading
|
|
58
|
+
* signal. Kept next to the patterns so the two cannot drift apart.
|
|
59
|
+
*/
|
|
60
|
+
export declare const BUNDLED_UI_SKILL_MEANINGS: Record<BundledSkcUiSkillName, string>;
|
|
46
61
|
/**
|
|
47
62
|
* Frontend skills SKC routes to but deliberately does NOT vendor.
|
|
48
63
|
*
|
|
@@ -76,6 +76,21 @@ export declare const en: {
|
|
|
76
76
|
readonly "modelSelector.noMatching": "No matching models.";
|
|
77
77
|
readonly "modelSelector.modelName": "Model Name: {value}";
|
|
78
78
|
readonly "modelSelector.actionFor": "Action for: {id}";
|
|
79
|
+
readonly "modelSelector.setAsTarget": "Set as {tag} ({name})";
|
|
80
|
+
readonly "modelSelector.setForAllRoleAgents": "Set for all role agents";
|
|
81
|
+
readonly "modelSelector.setForAllTargets": "Set for all targets";
|
|
82
|
+
readonly "modelSelector.hasDetailedUses": "(detailed uses)";
|
|
83
|
+
readonly "modelSelector.detailedUseFor": "Detailed use for {target}: {id}";
|
|
84
|
+
readonly "modelSelector.generalRole": "General (whole role)";
|
|
85
|
+
readonly "modelSelector.resetSpecialties": "Clear detailed-use overrides";
|
|
86
|
+
readonly "modelSelector.specialty.backendArchitecture": "Backend architecture";
|
|
87
|
+
readonly "modelSelector.specialty.frontendDesign": "Frontend design";
|
|
88
|
+
readonly "modelSelector.specialty.implementation": "Implementation";
|
|
89
|
+
readonly "modelSelector.specialty.testing": "Testing";
|
|
90
|
+
readonly "modelSelector.specialty.review": "Review";
|
|
91
|
+
readonly "modelSelector.specialtySaved": "{specialty} will use {value}.";
|
|
92
|
+
readonly "modelSelector.specialtySavedRoutingOff": "{specialty} will use {value} whenever a task declares that work. Auto-detection from assignment text stays off until you enable task.modelRouting.enabled.";
|
|
93
|
+
readonly "modelSelector.specialtyCleared": "Cleared detailed-use overrides for {target}.";
|
|
79
94
|
readonly "modelSelector.reasoningFor": "Reasoning for {target}: {id}";
|
|
80
95
|
readonly "modelSelector.temporaryModel": "temporary model";
|
|
81
96
|
};
|
|
@@ -95,6 +95,7 @@ export declare class LspTool implements AgentTool<typeof lspSchema, LspToolDetai
|
|
|
95
95
|
readonly description: string;
|
|
96
96
|
readonly parameters: import("zod").ZodObject<{
|
|
97
97
|
action: import("zod").ZodEnum<{
|
|
98
|
+
implementation: "implementation";
|
|
98
99
|
status: "status";
|
|
99
100
|
diagnostics: "diagnostics";
|
|
100
101
|
definition: "definition";
|
|
@@ -105,7 +106,6 @@ export declare class LspTool implements AgentTool<typeof lspSchema, LspToolDetai
|
|
|
105
106
|
rename_file: "rename_file";
|
|
106
107
|
code_actions: "code_actions";
|
|
107
108
|
type_definition: "type_definition";
|
|
108
|
-
implementation: "implementation";
|
|
109
109
|
reload: "reload";
|
|
110
110
|
capabilities: "capabilities";
|
|
111
111
|
request: "request";
|
|
@@ -3,6 +3,7 @@ import * as z from "zod/v4";
|
|
|
3
3
|
import type { OwnedProcess } from "../runtime/process-lifecycle";
|
|
4
4
|
export declare const lspSchema: z.ZodObject<{
|
|
5
5
|
action: z.ZodEnum<{
|
|
6
|
+
implementation: "implementation";
|
|
6
7
|
status: "status";
|
|
7
8
|
diagnostics: "diagnostics";
|
|
8
9
|
definition: "definition";
|
|
@@ -13,7 +14,6 @@ export declare const lspSchema: z.ZodObject<{
|
|
|
13
14
|
rename_file: "rename_file";
|
|
14
15
|
code_actions: "code_actions";
|
|
15
16
|
type_definition: "type_definition";
|
|
16
|
-
implementation: "implementation";
|
|
17
17
|
reload: "reload";
|
|
18
18
|
capabilities: "capabilities";
|
|
19
19
|
request: "request";
|
|
@@ -5,6 +5,7 @@ import type { ModelRegistry, SkcModelAssignmentTargetId } from "../../config/mod
|
|
|
5
5
|
import { type ScopedModelSelection } from "../../config/model-resolver";
|
|
6
6
|
import type { ModelProfileConfig } from "../../config/models-config-schema";
|
|
7
7
|
import type { Settings } from "../../config/settings";
|
|
8
|
+
import { type TaskModelSpecialty } from "../../config/task-model-specialties";
|
|
8
9
|
type ScopedModelItem = ScopedModelSelection;
|
|
9
10
|
export type ModelSelectorSelection = {
|
|
10
11
|
kind: "assignment";
|
|
@@ -17,6 +18,16 @@ export type ModelSelectorSelection = {
|
|
|
17
18
|
kind: "profile";
|
|
18
19
|
profileName: string;
|
|
19
20
|
setDefault: boolean;
|
|
21
|
+
} | {
|
|
22
|
+
kind: "specialtyAssignment";
|
|
23
|
+
model: Model;
|
|
24
|
+
role: SkcModelAssignmentTargetId;
|
|
25
|
+
specialty: TaskModelSpecialty;
|
|
26
|
+
thinkingLevel?: ThinkingLevel;
|
|
27
|
+
selector?: string;
|
|
28
|
+
} | {
|
|
29
|
+
kind: "specialtyReset";
|
|
30
|
+
role: SkcModelAssignmentTargetId;
|
|
20
31
|
} | {
|
|
21
32
|
kind: "createProfile";
|
|
22
33
|
profile: ModelProfileConfig;
|
|
@@ -19,7 +19,8 @@ import { type LocalProtocolOptions } from "../internal-urls";
|
|
|
19
19
|
import { AgentRegistry } from "../registry/agent-registry";
|
|
20
20
|
import { MCPManager } from "../runtime-mcp";
|
|
21
21
|
import { AgentSession, type ForkContextSeed } from "../session/agent-session";
|
|
22
|
-
import { AuthStorage } from "../session/auth-storage";
|
|
22
|
+
import type { AuthStorage } from "../session/auth-storage";
|
|
23
|
+
import { discoverAuthStorage } from "../session/auth-storage-discovery";
|
|
23
24
|
import { SessionManager } from "../session/session-manager";
|
|
24
25
|
import { type BuildSystemPromptResult } from "../system-prompt";
|
|
25
26
|
import { BashTool, BUILTIN_TOOLS, createTools, EditTool, EvalTool, FindTool, HIDDEN_TOOLS, type LspStartupServerInfo, loadSshTool, ReadTool, ResolveTool, SearchTool, type Tool, type ToolSession, WebSearchTool, WriteTool } from "../tools";
|
|
@@ -218,18 +219,7 @@ export type { FileSlashCommand } from "../extensibility/slash-commands";
|
|
|
218
219
|
export type { Tool } from "../tools";
|
|
219
220
|
export { buildDirectoryTree, buildWorkspaceTree, type DirectoryTree, type WorkspaceTree } from "../workspace-tree";
|
|
220
221
|
export { BashTool, BUILTIN_TOOLS, createTools, EditTool, EvalTool, FindTool, HIDDEN_TOOLS, loadSshTool, ReadTool, ResolveTool, SearchTool, type ToolSession, WebSearchTool, WriteTool, };
|
|
221
|
-
|
|
222
|
-
* Create an AuthStorage instance.
|
|
223
|
-
*
|
|
224
|
-
* Default: local SQLite store at `<agentDir>/agent.db`.
|
|
225
|
-
*
|
|
226
|
-
* Broker mode: when `SKC_AUTH_BROKER_URL` is set, credentials are pulled from
|
|
227
|
-
* a remote auth-broker over the wire. Refresh tokens never leave the broker;
|
|
228
|
-
* the client receives access tokens with `refresh = "__remote__"` and calls
|
|
229
|
-
* back into the broker through the {@link AuthStorageOptions.refreshOAuthCredential}
|
|
230
|
-
* override to re-mint access tokens when needed.
|
|
231
|
-
*/
|
|
232
|
-
export declare function discoverAuthStorage(agentDir?: string): Promise<AuthStorage>;
|
|
222
|
+
export { discoverAuthStorage };
|
|
233
223
|
/**
|
|
234
224
|
* Discover extensions from cwd.
|
|
235
225
|
*/
|
|
@@ -223,6 +223,14 @@ export interface AgentSessionConfig {
|
|
|
223
223
|
toolRegistry?: Map<string, AgentTool>;
|
|
224
224
|
/** Tool-session factory context used to lazily attach workflow-gate-only tools. */
|
|
225
225
|
workflowGateToolSession?: ToolSession;
|
|
226
|
+
/**
|
|
227
|
+
* Stage two of workflow routing: before a genuine user turn, ask a small model
|
|
228
|
+
* through the session's own transport which SKC workflow the prompt calls for.
|
|
229
|
+
* Hosts that own an interactive user (`createAgentSession`) turn this on; a bare
|
|
230
|
+
* session leaves it off so a scripted or injected transport is never consumed by a
|
|
231
|
+
* call the host did not script. `decisions.enabled` still governs the user side.
|
|
232
|
+
*/
|
|
233
|
+
semanticWorkflowRouting?: boolean;
|
|
226
234
|
/** Current session pre-LLM message transform pipeline */
|
|
227
235
|
transformContext?: (messages: AgentMessage[], signal?: AbortSignal) => AgentMessage[] | Promise<AgentMessage[]>;
|
|
228
236
|
/** Provider payload hook used by the active session request path */
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
import { AuthStorage } from "./auth-storage";
|
|
2
|
+
/**
|
|
3
|
+
* Create an AuthStorage instance.
|
|
4
|
+
*
|
|
5
|
+
* Default: local SQLite store at `<agentDir>/agent.db`.
|
|
6
|
+
*
|
|
7
|
+
* Broker mode: when `SKC_AUTH_BROKER_URL` is set, credentials are pulled from
|
|
8
|
+
* a remote auth-broker over the wire. Refresh tokens never leave the broker;
|
|
9
|
+
* the client receives access tokens with `refresh = "__remote__"` and calls
|
|
10
|
+
* back into the broker through the {@link AuthStorageOptions.refreshOAuthCredential}
|
|
11
|
+
* override to re-mint access tokens when needed.
|
|
12
|
+
*/
|
|
13
|
+
export declare function discoverAuthStorage(agentDir?: string): Promise<AuthStorage>;
|
|
@@ -12,7 +12,7 @@ export { isValidAllocatedTaskId, isValidTaskId, TASK_ID_DESCRIPTION, TASK_ID_PAT
|
|
|
12
12
|
export { AgentOutputManager } from "./output-manager";
|
|
13
13
|
export type { TaskResultReceipt } from "./receipt";
|
|
14
14
|
export { assertNoRawTaskFields, buildTaskReceipt, buildTaskRoi, buildTaskRoiSummary, findRawTaskLeakKeys, sanitizeTaskToolDetails, } from "./receipt";
|
|
15
|
-
export type { AgentDefinition, AgentProgress, SingleResult, SubagentLifecyclePayload, SubagentProgressPayload, TaskParams, TaskToolDetails, } from "./types";
|
|
15
|
+
export type { AgentDefinition, AgentProgress, SingleResult, SubagentLifecyclePayload, SubagentProgressPayload, TaskParams, TaskRoutingAttribution, TaskToolDetails, } from "./types";
|
|
16
16
|
export { TASK_SUBAGENT_EVENT_CHANNEL, TASK_SUBAGENT_LIFECYCLE_CHANNEL, TASK_SUBAGENT_PROGRESS_CHANNEL, taskSchema, } from "./types";
|
|
17
17
|
export declare function resolveForkContextMaxTokens(configured: number, model: Model | undefined): number;
|
|
18
18
|
/**
|
|
@@ -28,6 +28,8 @@ export interface TaskResultReceipt {
|
|
|
28
28
|
contextTokens?: number;
|
|
29
29
|
contextWindow?: number;
|
|
30
30
|
modelOverride?: string | string[];
|
|
31
|
+
/** What the router asked for, kept separate from what the spawn ran on. */
|
|
32
|
+
routing?: SingleResult["routing"];
|
|
31
33
|
modelSubstitutionWarning?: SingleResult["modelSubstitutionWarning"];
|
|
32
34
|
usage?: SingleResult["usage"];
|
|
33
35
|
cost?: number;
|
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
import type { ThinkingLevel } from "@sayknow-cli/agent-core";
|
|
2
2
|
import type { Usage } from "@sayknow-cli/ai";
|
|
3
3
|
import * as z from "zod/v4";
|
|
4
|
+
import { type TaskModelSpecialty, type TaskRoutingSource } from "../config/task-model-specialties";
|
|
5
|
+
import type { TaskTier } from "../decisions/task-routing";
|
|
4
6
|
import type { TaskResultReceipt } from "./receipt";
|
|
5
7
|
import type { SpawnRoiReconciliation } from "./roi-reconciliation";
|
|
6
8
|
import { type TaskSimpleMode } from "./simple-mode";
|
|
@@ -49,12 +51,19 @@ export declare const taskItemSchema: z.ZodObject<{
|
|
|
49
51
|
default: "default";
|
|
50
52
|
"ultragoal-red-team": "ultragoal-red-team";
|
|
51
53
|
}>>;
|
|
54
|
+
specialty: z.ZodOptional<z.ZodEnum<{
|
|
55
|
+
backendArchitecture: "backendArchitecture";
|
|
56
|
+
frontendDesign: "frontendDesign";
|
|
57
|
+
implementation: "implementation";
|
|
58
|
+
testing: "testing";
|
|
59
|
+
review: "review";
|
|
60
|
+
}>>;
|
|
52
61
|
inheritContext: z.ZodOptional<z.ZodEnum<{
|
|
53
62
|
none: "none";
|
|
54
|
-
|
|
63
|
+
full: "full";
|
|
55
64
|
"last-turn": "last-turn";
|
|
65
|
+
receipt: "receipt";
|
|
56
66
|
bounded: "bounded";
|
|
57
|
-
full: "full";
|
|
58
67
|
}>>;
|
|
59
68
|
repositoryBinding: z.ZodOptional<z.ZodObject<{
|
|
60
69
|
schema: z.ZodLiteral<"skc.repository_binding.v1">;
|
|
@@ -77,12 +86,19 @@ export declare const taskSchema: z.ZodObject<{
|
|
|
77
86
|
default: "default";
|
|
78
87
|
"ultragoal-red-team": "ultragoal-red-team";
|
|
79
88
|
}>>;
|
|
89
|
+
specialty: z.ZodOptional<z.ZodEnum<{
|
|
90
|
+
backendArchitecture: "backendArchitecture";
|
|
91
|
+
frontendDesign: "frontendDesign";
|
|
92
|
+
implementation: "implementation";
|
|
93
|
+
testing: "testing";
|
|
94
|
+
review: "review";
|
|
95
|
+
}>>;
|
|
80
96
|
inheritContext: z.ZodOptional<z.ZodEnum<{
|
|
81
97
|
none: "none";
|
|
82
|
-
|
|
98
|
+
full: "full";
|
|
83
99
|
"last-turn": "last-turn";
|
|
100
|
+
receipt: "receipt";
|
|
84
101
|
bounded: "bounded";
|
|
85
|
-
full: "full";
|
|
86
102
|
}>>;
|
|
87
103
|
repositoryBinding: z.ZodOptional<z.ZodObject<{
|
|
88
104
|
schema: z.ZodLiteral<"skc.repository_binding.v1">;
|
|
@@ -112,12 +128,19 @@ export declare const taskSchemaNoIsolation: z.ZodObject<{
|
|
|
112
128
|
default: "default";
|
|
113
129
|
"ultragoal-red-team": "ultragoal-red-team";
|
|
114
130
|
}>>;
|
|
131
|
+
specialty: z.ZodOptional<z.ZodEnum<{
|
|
132
|
+
backendArchitecture: "backendArchitecture";
|
|
133
|
+
frontendDesign: "frontendDesign";
|
|
134
|
+
implementation: "implementation";
|
|
135
|
+
testing: "testing";
|
|
136
|
+
review: "review";
|
|
137
|
+
}>>;
|
|
115
138
|
inheritContext: z.ZodOptional<z.ZodEnum<{
|
|
116
139
|
none: "none";
|
|
117
|
-
|
|
140
|
+
full: "full";
|
|
118
141
|
"last-turn": "last-turn";
|
|
142
|
+
receipt: "receipt";
|
|
119
143
|
bounded: "bounded";
|
|
120
|
-
full: "full";
|
|
121
144
|
}>>;
|
|
122
145
|
repositoryBinding: z.ZodOptional<z.ZodObject<{
|
|
123
146
|
schema: z.ZodLiteral<"skc.repository_binding.v1">;
|
|
@@ -147,12 +170,19 @@ declare const ALL_TASK_SCHEMAS: readonly [z.ZodObject<{
|
|
|
147
170
|
default: "default";
|
|
148
171
|
"ultragoal-red-team": "ultragoal-red-team";
|
|
149
172
|
}>>;
|
|
173
|
+
specialty: z.ZodOptional<z.ZodEnum<{
|
|
174
|
+
backendArchitecture: "backendArchitecture";
|
|
175
|
+
frontendDesign: "frontendDesign";
|
|
176
|
+
implementation: "implementation";
|
|
177
|
+
testing: "testing";
|
|
178
|
+
review: "review";
|
|
179
|
+
}>>;
|
|
150
180
|
inheritContext: z.ZodOptional<z.ZodEnum<{
|
|
151
181
|
none: "none";
|
|
152
|
-
|
|
182
|
+
full: "full";
|
|
153
183
|
"last-turn": "last-turn";
|
|
184
|
+
receipt: "receipt";
|
|
154
185
|
bounded: "bounded";
|
|
155
|
-
full: "full";
|
|
156
186
|
}>>;
|
|
157
187
|
repositoryBinding: z.ZodOptional<z.ZodObject<{
|
|
158
188
|
schema: z.ZodLiteral<"skc.repository_binding.v1">;
|
|
@@ -181,12 +211,19 @@ declare const ALL_TASK_SCHEMAS: readonly [z.ZodObject<{
|
|
|
181
211
|
default: "default";
|
|
182
212
|
"ultragoal-red-team": "ultragoal-red-team";
|
|
183
213
|
}>>;
|
|
214
|
+
specialty: z.ZodOptional<z.ZodEnum<{
|
|
215
|
+
backendArchitecture: "backendArchitecture";
|
|
216
|
+
frontendDesign: "frontendDesign";
|
|
217
|
+
implementation: "implementation";
|
|
218
|
+
testing: "testing";
|
|
219
|
+
review: "review";
|
|
220
|
+
}>>;
|
|
184
221
|
inheritContext: z.ZodOptional<z.ZodEnum<{
|
|
185
222
|
none: "none";
|
|
186
|
-
|
|
223
|
+
full: "full";
|
|
187
224
|
"last-turn": "last-turn";
|
|
225
|
+
receipt: "receipt";
|
|
188
226
|
bounded: "bounded";
|
|
189
|
-
full: "full";
|
|
190
227
|
}>>;
|
|
191
228
|
repositoryBinding: z.ZodOptional<z.ZodObject<{
|
|
192
229
|
schema: z.ZodLiteral<"skc.repository_binding.v1">;
|
|
@@ -215,12 +252,19 @@ declare const ALL_TASK_SCHEMAS: readonly [z.ZodObject<{
|
|
|
215
252
|
default: "default";
|
|
216
253
|
"ultragoal-red-team": "ultragoal-red-team";
|
|
217
254
|
}>>;
|
|
255
|
+
specialty: z.ZodOptional<z.ZodEnum<{
|
|
256
|
+
backendArchitecture: "backendArchitecture";
|
|
257
|
+
frontendDesign: "frontendDesign";
|
|
258
|
+
implementation: "implementation";
|
|
259
|
+
testing: "testing";
|
|
260
|
+
review: "review";
|
|
261
|
+
}>>;
|
|
218
262
|
inheritContext: z.ZodOptional<z.ZodEnum<{
|
|
219
263
|
none: "none";
|
|
220
|
-
|
|
264
|
+
full: "full";
|
|
221
265
|
"last-turn": "last-turn";
|
|
266
|
+
receipt: "receipt";
|
|
222
267
|
bounded: "bounded";
|
|
223
|
-
full: "full";
|
|
224
268
|
}>>;
|
|
225
269
|
repositoryBinding: z.ZodOptional<z.ZodObject<{
|
|
226
270
|
schema: z.ZodLiteral<"skc.repository_binding.v1">;
|
|
@@ -249,12 +293,19 @@ declare const ALL_TASK_SCHEMAS: readonly [z.ZodObject<{
|
|
|
249
293
|
default: "default";
|
|
250
294
|
"ultragoal-red-team": "ultragoal-red-team";
|
|
251
295
|
}>>;
|
|
296
|
+
specialty: z.ZodOptional<z.ZodEnum<{
|
|
297
|
+
backendArchitecture: "backendArchitecture";
|
|
298
|
+
frontendDesign: "frontendDesign";
|
|
299
|
+
implementation: "implementation";
|
|
300
|
+
testing: "testing";
|
|
301
|
+
review: "review";
|
|
302
|
+
}>>;
|
|
252
303
|
inheritContext: z.ZodOptional<z.ZodEnum<{
|
|
253
304
|
none: "none";
|
|
254
|
-
|
|
305
|
+
full: "full";
|
|
255
306
|
"last-turn": "last-turn";
|
|
307
|
+
receipt: "receipt";
|
|
256
308
|
bounded: "bounded";
|
|
257
|
-
full: "full";
|
|
258
309
|
}>>;
|
|
259
310
|
repositoryBinding: z.ZodOptional<z.ZodObject<{
|
|
260
311
|
schema: z.ZodLiteral<"skc.repository_binding.v1">;
|
|
@@ -283,12 +334,19 @@ declare const ALL_TASK_SCHEMAS: readonly [z.ZodObject<{
|
|
|
283
334
|
default: "default";
|
|
284
335
|
"ultragoal-red-team": "ultragoal-red-team";
|
|
285
336
|
}>>;
|
|
337
|
+
specialty: z.ZodOptional<z.ZodEnum<{
|
|
338
|
+
backendArchitecture: "backendArchitecture";
|
|
339
|
+
frontendDesign: "frontendDesign";
|
|
340
|
+
implementation: "implementation";
|
|
341
|
+
testing: "testing";
|
|
342
|
+
review: "review";
|
|
343
|
+
}>>;
|
|
286
344
|
inheritContext: z.ZodOptional<z.ZodEnum<{
|
|
287
345
|
none: "none";
|
|
288
|
-
|
|
346
|
+
full: "full";
|
|
289
347
|
"last-turn": "last-turn";
|
|
348
|
+
receipt: "receipt";
|
|
290
349
|
bounded: "bounded";
|
|
291
|
-
full: "full";
|
|
292
350
|
}>>;
|
|
293
351
|
repositoryBinding: z.ZodOptional<z.ZodObject<{
|
|
294
352
|
schema: z.ZodLiteral<"skc.repository_binding.v1">;
|
|
@@ -317,12 +375,19 @@ declare const ALL_TASK_SCHEMAS: readonly [z.ZodObject<{
|
|
|
317
375
|
default: "default";
|
|
318
376
|
"ultragoal-red-team": "ultragoal-red-team";
|
|
319
377
|
}>>;
|
|
378
|
+
specialty: z.ZodOptional<z.ZodEnum<{
|
|
379
|
+
backendArchitecture: "backendArchitecture";
|
|
380
|
+
frontendDesign: "frontendDesign";
|
|
381
|
+
implementation: "implementation";
|
|
382
|
+
testing: "testing";
|
|
383
|
+
review: "review";
|
|
384
|
+
}>>;
|
|
320
385
|
inheritContext: z.ZodOptional<z.ZodEnum<{
|
|
321
386
|
none: "none";
|
|
322
|
-
|
|
387
|
+
full: "full";
|
|
323
388
|
"last-turn": "last-turn";
|
|
389
|
+
receipt: "receipt";
|
|
324
390
|
bounded: "bounded";
|
|
325
|
-
full: "full";
|
|
326
391
|
}>>;
|
|
327
392
|
repositoryBinding: z.ZodOptional<z.ZodObject<{
|
|
328
393
|
schema: z.ZodLiteral<"skc.repository_binding.v1">;
|
|
@@ -402,6 +467,33 @@ export interface ModelSubstitutionWarning {
|
|
|
402
467
|
effective: string;
|
|
403
468
|
reason: "auth_unavailable" | "assistant_model_mismatch";
|
|
404
469
|
}
|
|
470
|
+
/**
|
|
471
|
+
* What the model router *asked for* on one child — deliberately not what it ran on.
|
|
472
|
+
*
|
|
473
|
+
* The dispatched value is a fallback chain, so the head can lose to a later
|
|
474
|
+
* candidate when it fails to authenticate. Recording the request separately is
|
|
475
|
+
* what keeps a receipt from claiming a specialty model was used when the spawn
|
|
476
|
+
* actually fell through to the role's baseline. `ModelSubstitutionWarning`
|
|
477
|
+
* covers the disagreement; this covers the intent.
|
|
478
|
+
*/
|
|
479
|
+
export interface TaskRoutingAttribution {
|
|
480
|
+
/** Axis the head came from. `baseline` means the router declined to move. */
|
|
481
|
+
source: TaskRoutingSource;
|
|
482
|
+
/** Bounded specialty id, present only when the specialty axis won. */
|
|
483
|
+
specialty?: TaskModelSpecialty;
|
|
484
|
+
/** Tier the classifier settled on. Null for a specialty swap, which has no ladder. */
|
|
485
|
+
tier?: TaskTier;
|
|
486
|
+
/** True when the caller declared the specialty on the spawn; no classifier ran. */
|
|
487
|
+
declared: boolean;
|
|
488
|
+
/** False for ordinary LLM backends, which return no probabilities at all. */
|
|
489
|
+
calibrated: boolean;
|
|
490
|
+
/** Probability. Present only when `calibrated` is true — never synthesised. */
|
|
491
|
+
confidence?: number;
|
|
492
|
+
/** Ordinal clarity in [0,1], recorded in place of a probability when uncalibrated. */
|
|
493
|
+
ordinalStrength?: number;
|
|
494
|
+
/** Router's own explanation, surfaced verbatim on the receipt. */
|
|
495
|
+
reason: string;
|
|
496
|
+
}
|
|
405
497
|
/** Progress tracking for a single agent */
|
|
406
498
|
export interface AgentProgress {
|
|
407
499
|
index: number;
|
|
@@ -439,6 +531,8 @@ export interface AgentProgress {
|
|
|
439
531
|
durationMs: number;
|
|
440
532
|
modelOverride?: string | string[];
|
|
441
533
|
modelSubstitutionWarning?: ModelSubstitutionWarning;
|
|
534
|
+
/** What the router asked for on this child. See {@link TaskRoutingAttribution}. */
|
|
535
|
+
routing?: TaskRoutingAttribution;
|
|
442
536
|
/** Data extracted by registered subprocess tool handlers (keyed by tool name) */
|
|
443
537
|
extractedToolData?: Record<string, unknown[]>;
|
|
444
538
|
/**
|
|
@@ -500,6 +594,8 @@ export interface SingleResult {
|
|
|
500
594
|
/** Model's context window in tokens, when known. */
|
|
501
595
|
contextWindow?: number;
|
|
502
596
|
modelOverride?: string | string[];
|
|
597
|
+
/** What the router asked for on this child. See {@link TaskRoutingAttribution}. */
|
|
598
|
+
routing?: TaskRoutingAttribution;
|
|
503
599
|
modelSubstitutionWarning?: ModelSubstitutionWarning;
|
|
504
600
|
error?: string;
|
|
505
601
|
aborted?: boolean;
|
|
@@ -8,9 +8,9 @@ export { extractReadableFromHtml, type ReadableFormat, type ReadableResult } fro
|
|
|
8
8
|
export type { Observation, ObservationEntry } from "./browser/tab-protocol";
|
|
9
9
|
declare const browserSchema: z.ZodObject<{
|
|
10
10
|
action: z.ZodEnum<{
|
|
11
|
+
run: "run";
|
|
11
12
|
open: "open";
|
|
12
13
|
close: "close";
|
|
13
|
-
run: "run";
|
|
14
14
|
act: "act";
|
|
15
15
|
}>;
|
|
16
16
|
name: z.ZodOptional<z.ZodString>;
|
|
@@ -122,9 +122,9 @@ export declare class BrowserTool implements AgentTool<typeof browserSchema, Brow
|
|
|
122
122
|
readonly summary = "Control a headless browser to navigate and interact with web pages";
|
|
123
123
|
readonly parameters: z.ZodObject<{
|
|
124
124
|
action: z.ZodEnum<{
|
|
125
|
+
run: "run";
|
|
125
126
|
open: "open";
|
|
126
127
|
close: "close";
|
|
127
|
-
run: "run";
|
|
128
128
|
act: "act";
|
|
129
129
|
}>;
|
|
130
130
|
name: z.ZodOptional<z.ZodString>;
|
|
@@ -19,8 +19,8 @@ declare const subagentSchema: z.ZodObject<{
|
|
|
19
19
|
timeout_ms: z.ZodOptional<z.ZodNumber>;
|
|
20
20
|
limit: z.ZodOptional<z.ZodNumber>;
|
|
21
21
|
verbosity: z.ZodOptional<z.ZodEnum<{
|
|
22
|
-
receipt: "receipt";
|
|
23
22
|
full: "full";
|
|
23
|
+
receipt: "receipt";
|
|
24
24
|
preview: "preview";
|
|
25
25
|
}>>;
|
|
26
26
|
}, z.core.$strip>;
|
|
@@ -88,8 +88,8 @@ export declare class SubagentTool implements AgentTool<typeof subagentSchema, Su
|
|
|
88
88
|
timeout_ms: z.ZodOptional<z.ZodNumber>;
|
|
89
89
|
limit: z.ZodOptional<z.ZodNumber>;
|
|
90
90
|
verbosity: z.ZodOptional<z.ZodEnum<{
|
|
91
|
-
receipt: "receipt";
|
|
92
91
|
full: "full";
|
|
92
|
+
receipt: "receipt";
|
|
93
93
|
preview: "preview";
|
|
94
94
|
}>>;
|
|
95
95
|
}, z.core.$strip>;
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@sayknow-cli/coding-agent",
|
|
4
|
-
"version": "0.
|
|
4
|
+
"version": "0.6.0",
|
|
5
5
|
"description": "Sayknow-CLI CLI with read, bash, edit, write tools and session management",
|
|
6
6
|
"homepage": "https://sayknow-cli.com",
|
|
7
7
|
"author": "jaybeyond",
|
|
@@ -54,12 +54,12 @@
|
|
|
54
54
|
"@agentclientprotocol/sdk": "1.3.0",
|
|
55
55
|
"@babel/parser": "^7.29.3",
|
|
56
56
|
"@mozilla/readability": "^0.6.0",
|
|
57
|
-
"@sayknow-cli/stats": "0.
|
|
58
|
-
"@sayknow-cli/agent-core": "0.
|
|
59
|
-
"@sayknow-cli/ai": "0.
|
|
60
|
-
"@sayknow-cli/natives": "0.
|
|
61
|
-
"@sayknow-cli/tui": "0.
|
|
62
|
-
"@sayknow-cli/utils": "0.
|
|
57
|
+
"@sayknow-cli/stats": "0.6.0",
|
|
58
|
+
"@sayknow-cli/agent-core": "0.6.0",
|
|
59
|
+
"@sayknow-cli/ai": "0.6.0",
|
|
60
|
+
"@sayknow-cli/natives": "0.6.0",
|
|
61
|
+
"@sayknow-cli/tui": "0.6.0",
|
|
62
|
+
"@sayknow-cli/utils": "0.6.0",
|
|
63
63
|
"@puppeteer/browsers": "^2.13.0",
|
|
64
64
|
"@types/turndown": "5.0.6",
|
|
65
65
|
"@xterm/headless": "^6.0.0",
|
|
@@ -19,7 +19,7 @@ import { ModelRegistry } from "../src/config/model-registry";
|
|
|
19
19
|
import { resolveRoleSelection } from "../src/config/model-resolver";
|
|
20
20
|
import { Settings } from "../src/config/settings";
|
|
21
21
|
import { createDecisionService, createLlmDecisionBackend, createTypeSafeDecisionBackend } from "../src/decisions";
|
|
22
|
-
import { createSemanticSkillRouter } from "../src/decisions/
|
|
22
|
+
import { createSemanticSkillRouter } from "../src/decisions/prompt-triage";
|
|
23
23
|
import { detectPrimarySkillKeyword } from "../src/hooks/skill-state";
|
|
24
24
|
import { discoverAuthStorage } from "../src/sdk";
|
|
25
25
|
|
|
@@ -27,7 +27,8 @@ type Expected = "deep-interview" | "ralplan" | "ultragoal" | "team" | null;
|
|
|
27
27
|
interface Case {
|
|
28
28
|
prompt: string;
|
|
29
29
|
expect: Expected;
|
|
30
|
-
|
|
30
|
+
/** BCP 47 primary subtag. The keyword table only knows ko and en; every other row measures the semantic stage alone. */
|
|
31
|
+
lang: string;
|
|
31
32
|
}
|
|
32
33
|
|
|
33
34
|
const CASES: Case[] = [
|
|
@@ -54,6 +55,25 @@ const CASES: Case[] = [
|
|
|
54
55
|
{ prompt: "우리 서비스에 이 모델 붙이면 뭐가 좋아?", expect: null, lang: "ko" },
|
|
55
56
|
{ prompt: "fix the failing lint rule in src/utils.ts", expect: null, lang: "en" },
|
|
56
57
|
{ prompt: "what does this regex do?", expect: null, lang: "en" },
|
|
58
|
+
// Languages the hand-written table has no entries for. Routing here is the
|
|
59
|
+
// semantic stage or nothing, which is what the per-language column shows.
|
|
60
|
+
{ prompt: "需求还不清楚,先通过提问把规格问出来", expect: "deep-interview", lang: "zh" },
|
|
61
|
+
{ prompt: "架构风险很大,先给我一个需要审批的详细计划", expect: "ralplan", lang: "zh" },
|
|
62
|
+
{ prompt: "把这个目标登记下来,持续跟踪直到全部交付验证完", expect: "ultragoal", lang: "zh" },
|
|
63
|
+
{ prompt: "任务太大了,拆成几个并行的工作者一起做", expect: "team", lang: "zh" },
|
|
64
|
+
{ prompt: "这个测试为什么会挂?", expect: null, lang: "zh" },
|
|
65
|
+
{ prompt: "修一下 README 里的错别字", expect: null, lang: "zh" },
|
|
66
|
+
{ prompt: "要件がまだ曖昧なので、質問して仕様を引き出して", expect: "deep-interview", lang: "ja" },
|
|
67
|
+
{ prompt: "実装前に設計案を比較した計画書を作って承認を待って", expect: "ralplan", lang: "ja" },
|
|
68
|
+
{ prompt: "この目標を最後まで追跡して、途中で忘れないで", expect: "ultragoal", lang: "ja" },
|
|
69
|
+
{ prompt: "作業が大きいのでワーカーを複数立てて並列で進めて", expect: "team", lang: "ja" },
|
|
70
|
+
{ prompt: "この関数は何をしているか説明して", expect: null, lang: "ja" },
|
|
71
|
+
{ prompt: "Hazme preguntas hasta que los requisitos estén claros", expect: "deep-interview", lang: "es" },
|
|
72
|
+
{ prompt: "Prepara un plan detallado y espera mi aprobación antes de tocar código", expect: "ralplan", lang: "es" },
|
|
73
|
+
{ prompt: "Arregla el error de lint en src/utils.ts", expect: null, lang: "es" },
|
|
74
|
+
{ prompt: "Составь согласованный план миграции и жди моего одобрения", expect: "ralplan", lang: "ru" },
|
|
75
|
+
{ prompt: "Разбей работу на несколько параллельных воркеров", expect: "team", lang: "ru" },
|
|
76
|
+
{ prompt: "Что делает эта регулярка?", expect: null, lang: "ru" },
|
|
57
77
|
];
|
|
58
78
|
|
|
59
79
|
function pct(hit: number, total: number): string {
|
|
@@ -120,18 +140,23 @@ async function main(): Promise<void> {
|
|
|
120
140
|
};
|
|
121
141
|
const positives = (row: (typeof rows)[number]) => row.case.expect !== null;
|
|
122
142
|
const negatives = (row: (typeof rows)[number]) => row.case.expect === null;
|
|
123
|
-
const
|
|
124
|
-
const en = (row: (typeof rows)[number]) => positives(row) && row.case.lang === "en";
|
|
143
|
+
const langs = [...new Set(CASES.map(testCase => testCase.lang))];
|
|
125
144
|
const hybrid = (row: (typeof rows)[number]) => row.keyword ?? row.semantic;
|
|
126
145
|
|
|
127
146
|
console.log("\n=== stage comparison ===");
|
|
147
|
+
// Both hosts ship the hybrid: the session in `AgentSession#routeWorkflowSemantically`,
|
|
148
|
+
// the Codex hook in `hooks/native-prompt-routing.ts`. The single-stage rows show
|
|
149
|
+
// what each stage contributes on its own.
|
|
128
150
|
for (const [label, pick] of [
|
|
129
|
-
["keyword only
|
|
130
|
-
["semantic only
|
|
131
|
-
["keyword+semantic (
|
|
151
|
+
["keyword only", (row: (typeof rows)[number]) => row.keyword],
|
|
152
|
+
["semantic only", (row: (typeof rows)[number]) => row.semantic],
|
|
153
|
+
["keyword+semantic (SHIPPED)", hybrid],
|
|
132
154
|
] as const) {
|
|
155
|
+
const perLang = langs
|
|
156
|
+
.map(lang => `${lang} ${pct(...score(pick, row => positives(row) && row.case.lang === lang))}`)
|
|
157
|
+
.join(" ");
|
|
133
158
|
console.log(
|
|
134
|
-
`${label.padEnd(32)} all ${pct(...score(pick, () => true))}
|
|
159
|
+
`${label.padEnd(32)} all ${pct(...score(pick, () => true))} ${perLang} clean-negatives ${pct(...score(pick, negatives))}`,
|
|
135
160
|
);
|
|
136
161
|
}
|
|
137
162
|
|
|
@@ -140,13 +165,13 @@ async function main(): Promise<void> {
|
|
|
140
165
|
`\nlatency p50 ${latencies[Math.floor(latencies.length / 2)]}ms p95 ${latencies[Math.max(0, Math.ceil(latencies.length * 0.95) - 1)]}ms max ${latencies.at(-1)}ms`,
|
|
141
166
|
);
|
|
142
167
|
|
|
143
|
-
// Report against what
|
|
144
|
-
const misses = rows.filter(row => row
|
|
168
|
+
// Report against what ships: keyword first, the model for what it missed.
|
|
169
|
+
const misses = rows.filter(row => hybrid(row) !== row.case.expect);
|
|
145
170
|
if (misses.length > 0) {
|
|
146
|
-
console.log("\n=== misses in shipped configuration (semantic
|
|
171
|
+
console.log("\n=== misses in shipped configuration (keyword+semantic) ===");
|
|
147
172
|
for (const row of misses)
|
|
148
173
|
console.log(
|
|
149
|
-
` [${row.case.lang}] want=${row.case.expect ?? "none"} got=${row
|
|
174
|
+
` [${row.case.lang}] want=${row.case.expect ?? "none"} got=${hybrid(row) ?? "none"} :: ${row.case.prompt}`,
|
|
150
175
|
);
|
|
151
176
|
}
|
|
152
177
|
|