@sayknow-cli/coding-agent 0.5.25 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +36 -1
- package/dist/types/config/settings-schema.d.ts +51 -5
- package/dist/types/config/task-model-specialties.d.ts +55 -0
- package/dist/types/decisions/keyword-learning.d.ts +61 -0
- package/dist/types/decisions/llm-backend.d.ts +13 -1
- package/dist/types/decisions/prompt-triage.d.ts +42 -0
- package/dist/types/decisions/skill-routing.d.ts +41 -6
- package/dist/types/decisions/task-routing.d.ts +96 -11
- package/dist/types/hooks/native-prompt-routing.d.ts +21 -0
- package/dist/types/hooks/native-skill-hook.d.ts +3 -0
- package/dist/types/hooks/skill-keywords.d.ts +9 -0
- package/dist/types/hooks/skill-state.d.ts +20 -3
- package/dist/types/hooks/ui-skill-keywords.d.ts +15 -0
- package/dist/types/i18n/messages/en.d.ts +15 -0
- package/dist/types/lsp/index.d.ts +1 -1
- package/dist/types/lsp/types.d.ts +1 -1
- package/dist/types/modes/components/model-selector.d.ts +11 -0
- package/dist/types/sdk/session.d.ts +3 -13
- package/dist/types/session/agent-session.d.ts +8 -0
- package/dist/types/session/auth-storage-discovery.d.ts +13 -0
- package/dist/types/task/index.d.ts +1 -1
- package/dist/types/task/receipt.d.ts +2 -0
- package/dist/types/task/types.d.ts +114 -18
- package/dist/types/tools/browser.d.ts +2 -2
- package/dist/types/tools/subagent.d.ts +2 -2
- package/package.json +7 -7
- package/scripts/eval-skill-routing.ts +37 -12
- package/src/config/settings-schema.ts +64 -12
- package/src/config/task-model-specialties.ts +131 -0
- package/src/decisions/index.ts +8 -2
- package/src/decisions/keyword-learning.ts +678 -0
- package/src/decisions/llm-backend.ts +213 -67
- package/src/decisions/prompt-triage.ts +163 -0
- package/src/decisions/skill-routing.ts +39 -56
- package/src/decisions/task-routing.ts +382 -66
- package/src/decisions/typesafe-backend.ts +3 -0
- package/src/hooks/native-prompt-routing.ts +190 -0
- package/src/hooks/native-skill-hook.ts +21 -12
- package/src/hooks/skill-keywords.ts +9 -0
- package/src/hooks/skill-state.ts +41 -10
- package/src/hooks/ui-skill-keywords.ts +67 -10
- package/src/i18n/messages/de.ts +16 -0
- package/src/i18n/messages/en.ts +16 -0
- package/src/i18n/messages/es.ts +16 -0
- package/src/i18n/messages/fr.ts +16 -0
- package/src/i18n/messages/ja.ts +16 -0
- package/src/i18n/messages/ko.ts +16 -0
- package/src/i18n/messages/zh.ts +16 -0
- package/src/internal-urls/docs-index.generated.ts +1 -1
- package/src/main.ts +1 -1
- package/src/modes/components/model-selector.ts +275 -34
- package/src/modes/controllers/selector-controller.ts +50 -2
- package/src/modes/shared/agent-wire/command-dispatch.ts +1 -1
- package/src/prompts/tools/task.md +1 -0
- package/src/sdk/session.ts +5 -82
- package/src/session/agent-session.ts +137 -37
- package/src/session/auth-storage-discovery.ts +83 -0
- package/src/slash-commands/builtin-registry.ts +11 -9
- package/src/task/index.ts +98 -40
- package/src/task/receipt.ts +3 -0
- package/src/task/types.ts +44 -0
|
@@ -154,11 +154,21 @@ export interface LlmBackendDeps {
|
|
|
154
154
|
maxInputCostPerMTok?: number;
|
|
155
155
|
/** Injected in tests to make the local-runtime probe deterministic. */
|
|
156
156
|
fetchImpl?: typeof fetch;
|
|
157
|
+
/** Injected in tests to script provider answers without a network. */
|
|
158
|
+
completeImpl?: typeof completeSimple;
|
|
159
|
+
/** Per-candidate deadline; injected in tests so a hung provider does not cost real seconds. */
|
|
160
|
+
attemptTimeoutMs?: number;
|
|
157
161
|
registry: ModelRegistry;
|
|
158
162
|
settings: Settings;
|
|
159
163
|
sessionId?: string;
|
|
160
164
|
/** Overrides role resolution; used by callers that already picked a model. */
|
|
161
165
|
model?: Model<Api>;
|
|
166
|
+
/**
|
|
167
|
+
* Provider of the model the caller is already talking to. Its small model is tried
|
|
168
|
+
* first when nothing was configured; see `rankSmallModels` for why. A thunk is
|
|
169
|
+
* accepted so a long-lived backend follows the session when the user switches model.
|
|
170
|
+
*/
|
|
171
|
+
preferredProvider?: string | (() => string | undefined);
|
|
162
172
|
}
|
|
163
173
|
|
|
164
174
|
/**
|
|
@@ -188,6 +198,23 @@ function isTextOnly(model: Model<Api>): boolean {
|
|
|
188
198
|
return (model.input ?? ["text"]).includes("text") && !(model.output ?? ["text"]).includes("image");
|
|
189
199
|
}
|
|
190
200
|
|
|
201
|
+
/**
|
|
202
|
+
* Local runtimes list embedding models next to chat models with identical metadata.
|
|
203
|
+
* Measured: `ollama/nomic-embed-text:latest` was ranked as a candidate and answered
|
|
204
|
+
* HTTP 400 "does not support chat". Nothing in `Model` says so; the id does.
|
|
205
|
+
*/
|
|
206
|
+
const EMBEDDING_MODEL_ID = /embed/i;
|
|
207
|
+
|
|
208
|
+
/**
|
|
209
|
+
* Parameter count from a local model id (`qwen3:1.7b`, `gemma4:e4b`, `lfm2-24b-a2b`),
|
|
210
|
+
* used only to order local candidates: smaller loads faster and answers faster, and a
|
|
211
|
+
* five-way choice does not need a 30b model. Unparseable ids sort last.
|
|
212
|
+
*/
|
|
213
|
+
function localModelSize(id: string): number {
|
|
214
|
+
const match = /(\d+(?:\.\d+)?)b(?![a-z])/i.exec(id);
|
|
215
|
+
return match ? Number(match[1]) : Number.POSITIVE_INFINITY;
|
|
216
|
+
}
|
|
217
|
+
|
|
191
218
|
/**
|
|
192
219
|
* Locally hosted runtimes. A decision answered here costs no tokens at all and the
|
|
193
220
|
* state never leaves the machine, which is the strongest possible fit for this feature.
|
|
@@ -236,7 +263,50 @@ async function isLocalRuntimeAlive(baseUrl: string, fetchImpl: typeof fetch = fe
|
|
|
236
263
|
}
|
|
237
264
|
|
|
238
265
|
/**
|
|
239
|
-
*
|
|
266
|
+
* A model that answered a decision with an error or a timeout, skipped for a while.
|
|
267
|
+
*
|
|
268
|
+
* The catalog lists models the provider no longer serves — `claude-3-haiku-20240307`
|
|
269
|
+
* was the cheapest Anthropic entry and answered 404 on every call — and it lists
|
|
270
|
+
* "free" reasoning tiers that ignore `disableReasoning` and take ~9s to answer. Either
|
|
271
|
+
* one, chosen blindly, turned the fallback into a fixed ~1-9s stall that answered
|
|
272
|
+
* nothing, on every turn, forever. Remembering the failure means one bad turn per
|
|
273
|
+
* model per TTL, not one per prompt.
|
|
274
|
+
*/
|
|
275
|
+
const DEAD_MODEL_TTL_MS = 10 * 60_000;
|
|
276
|
+
const deadModels = new Map<string, { at: number; reason: string }>();
|
|
277
|
+
|
|
278
|
+
/** Reset between tests. */
|
|
279
|
+
export function clearDeadModelCache(): void {
|
|
280
|
+
deadModels.clear();
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
function modelKey(model: Model<Api>): string {
|
|
284
|
+
return `${model.provider}/${model.id}`;
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
function isDead(model: Model<Api>): boolean {
|
|
288
|
+
const entry = deadModels.get(modelKey(model));
|
|
289
|
+
if (!entry) return false;
|
|
290
|
+
if (Date.now() - entry.at < DEAD_MODEL_TTL_MS) return true;
|
|
291
|
+
deadModels.delete(modelKey(model));
|
|
292
|
+
return false;
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
/**
|
|
296
|
+
* Candidates tried per decision when nothing was configured. Two attempts of
|
|
297
|
+
* `ATTEMPT_TIMEOUT_MS` fit inside the service's 8s deadline with room for the
|
|
298
|
+
* liveness probe; a third only runs when the earlier ones failed fast (404, auth).
|
|
299
|
+
*/
|
|
300
|
+
const MAX_AUTO_ATTEMPTS = 3;
|
|
301
|
+
/**
|
|
302
|
+
* Per-attempt budget. The hosted small models measured here answer in 0.3-1.3s; the
|
|
303
|
+
* ones that blow this are reasoning tiers that think despite being told not to, and
|
|
304
|
+
* the right response to those is the next candidate, not a longer wait.
|
|
305
|
+
*/
|
|
306
|
+
const ATTEMPT_TIMEOUT_MS = 3_500;
|
|
307
|
+
|
|
308
|
+
/**
|
|
309
|
+
* Rank small, fast text models, best first.
|
|
240
310
|
*
|
|
241
311
|
* Sorting by price alone is a trap, and it was measured: the cheapest qualifying model
|
|
242
312
|
* on this registry is free but took **4.8s** per routing decision — three times slower
|
|
@@ -244,29 +314,44 @@ async function isLocalRuntimeAlive(baseUrl: string, fetchImpl: typeof fetch = fe
|
|
|
244
314
|
* are dominated by reasoning models. A decision service that is cheap and slow has
|
|
245
315
|
* missed the point twice over.
|
|
246
316
|
*
|
|
247
|
-
*
|
|
248
|
-
*
|
|
249
|
-
*
|
|
317
|
+
* The provider the user is already chatting with goes first. Measured on a registry
|
|
318
|
+
* with Anthropic, Codex and Z.ai keys: price order put four `zai/glm-*` tiers ahead of
|
|
319
|
+
* `claude-haiku-4-5`, and the first of them took 8.9s — past the service deadline, so
|
|
320
|
+
* the answer was nothing. The user's own provider is the credential known to work, the
|
|
321
|
+
* bill they expect to see, and (Haiku next to Opus, mini next to GPT) the small model
|
|
322
|
+
* they would have picked by hand. Within a provider, non-reasoning wins, then price.
|
|
323
|
+
* Ties break by id so the choice is stable across runs; a backend that silently changed
|
|
324
|
+
* model between turns would make routing non-reproducible, which is most of what this
|
|
325
|
+
* feature is for.
|
|
250
326
|
*/
|
|
251
|
-
async function
|
|
327
|
+
async function rankSmallModels(
|
|
252
328
|
available: Model<Api>[],
|
|
253
329
|
costCeiling: number,
|
|
330
|
+
preferredProvider: string | undefined,
|
|
254
331
|
fetchImpl?: typeof fetch,
|
|
255
|
-
): Promise<Model<Api>
|
|
256
|
-
// A local runtime
|
|
257
|
-
//
|
|
258
|
-
//
|
|
332
|
+
): Promise<Model<Api>[]> {
|
|
333
|
+
// A live local runtime goes first: zero tokens, zero egress. One candidate only —
|
|
334
|
+
// measured with Ollama on CPU, every loaded model blew the attempt budget on a cold
|
|
335
|
+
// start, and with five of them ranked ahead of every hosted model the hosted
|
|
336
|
+
// fallback was never reached before the service deadline. One local miss now costs
|
|
337
|
+
// one attempt, after which the dead-model cache sends the next ten minutes of
|
|
338
|
+
// decisions straight to the hosted candidate.
|
|
339
|
+
const ranked: Model<Api>[] = [];
|
|
259
340
|
const local = available
|
|
260
|
-
.filter(model => LOCAL_PROVIDERS.has(model.provider) && isTextOnly(model))
|
|
261
|
-
.sort((a, b) => a.id.localeCompare(b.id));
|
|
341
|
+
.filter(model => LOCAL_PROVIDERS.has(model.provider) && isTextOnly(model) && !EMBEDDING_MODEL_ID.test(model.id))
|
|
342
|
+
.sort((a, b) => localModelSize(a.id) - localModelSize(b.id) || a.id.localeCompare(b.id));
|
|
262
343
|
for (const model of local) {
|
|
344
|
+
// The smallest one missing parks the whole runtime for the TTL; trying the next
|
|
345
|
+
// size up would only repeat the cold-start stall on the next turn.
|
|
346
|
+
if (isDead(model)) break;
|
|
263
347
|
if (await isLocalRuntimeAlive(model.baseUrl, fetchImpl)) {
|
|
264
|
-
logger.debug("decisions/llm: using local runtime", { id:
|
|
265
|
-
|
|
348
|
+
logger.debug("decisions/llm: using local runtime", { id: modelKey(model) });
|
|
349
|
+
ranked.push(model);
|
|
350
|
+
break;
|
|
266
351
|
}
|
|
267
352
|
}
|
|
268
353
|
|
|
269
|
-
|
|
354
|
+
const hosted = available
|
|
270
355
|
.filter(
|
|
271
356
|
model =>
|
|
272
357
|
isTextOnly(model) &&
|
|
@@ -276,12 +361,18 @@ async function pickSmallModel(
|
|
|
276
361
|
)
|
|
277
362
|
.sort(
|
|
278
363
|
(a, b) =>
|
|
279
|
-
Number(
|
|
280
|
-
|
|
364
|
+
Number(b.provider === preferredProvider) - Number(a.provider === preferredProvider) ||
|
|
365
|
+
Number(!!a.reasoning) - Number(!!b.reasoning) ||
|
|
366
|
+
a.cost.input - b.cost.input ||
|
|
367
|
+
a.id.localeCompare(b.id),
|
|
368
|
+
);
|
|
369
|
+
return ranked.concat(hosted);
|
|
281
370
|
}
|
|
282
371
|
|
|
283
372
|
export function createLlmDecisionBackend(deps: LlmBackendDeps): DecisionBackend {
|
|
284
373
|
const costCeiling = deps.maxInputCostPerMTok ?? DEFAULT_MAX_INPUT_COST_PER_MTOK;
|
|
374
|
+
const complete = deps.completeImpl ?? completeSimple;
|
|
375
|
+
const attemptTimeoutMs = deps.attemptTimeoutMs ?? ATTEMPT_TIMEOUT_MS;
|
|
285
376
|
return {
|
|
286
377
|
name: "llm",
|
|
287
378
|
async decide(request: DecisionRequest): Promise<DecisionResult | null> {
|
|
@@ -290,67 +381,122 @@ export function createLlmDecisionBackend(deps: LlmBackendDeps): DecisionBackend
|
|
|
290
381
|
// Resolution order, cheapest intent first:
|
|
291
382
|
// 1. an explicit override — the caller already decided
|
|
292
383
|
// 2. the `smol` role — the user already decided
|
|
293
|
-
// 3. the
|
|
384
|
+
// 3. the ranked small models on hand — nobody decided, so decide safely,
|
|
385
|
+
// and move on when one is dead or slow
|
|
294
386
|
// `default` is deliberately absent: it is whatever the user chats with, which is
|
|
295
387
|
// exactly the frontier model this feature exists to avoid spending on.
|
|
296
|
-
const
|
|
297
|
-
deps.model ??
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
388
|
+
const configured =
|
|
389
|
+
deps.model ?? resolveRoleSelection(["smol"], deps.settings, available, deps.registry)?.model;
|
|
390
|
+
const preferredProvider =
|
|
391
|
+
typeof deps.preferredProvider === "function" ? deps.preferredProvider() : deps.preferredProvider;
|
|
392
|
+
const candidates = configured
|
|
393
|
+
? [configured]
|
|
394
|
+
: (await rankSmallModels(available, costCeiling, preferredProvider, deps.fetchImpl))
|
|
395
|
+
.filter(model => !isDead(model))
|
|
396
|
+
.slice(0, MAX_AUTO_ATTEMPTS);
|
|
397
|
+
if (candidates.length === 0) {
|
|
301
398
|
logger.debug("decisions/llm: no small model available; leaving the decision to existing behaviour");
|
|
302
399
|
return null;
|
|
303
400
|
}
|
|
304
|
-
const model = chosen;
|
|
305
|
-
// The ceiling still applies to an explicitly configured `smol` role — a role can
|
|
306
|
-
// point anywhere, including at a frontier model.
|
|
307
|
-
if (!deps.model && model.cost.input > costCeiling) {
|
|
308
|
-
logger.debug("decisions/llm: declining, model too expensive for a decision", {
|
|
309
|
-
id: `${model.provider}/${model.id}`,
|
|
310
|
-
inputCostPerMTok: model.cost.input,
|
|
311
|
-
ceiling: costCeiling,
|
|
312
|
-
});
|
|
313
|
-
return null;
|
|
314
|
-
}
|
|
315
|
-
const apiKey = await deps.registry.getApiKey(model, deps.sessionId);
|
|
316
|
-
if (!apiKey) {
|
|
317
|
-
logger.debug("decisions/llm: no credential", { provider: model.provider, id: model.id });
|
|
318
|
-
return null;
|
|
319
|
-
}
|
|
320
401
|
|
|
321
402
|
const text = stateToText(request.state);
|
|
322
403
|
const state = text.length > MAX_STATE_CHARS ? `${text.slice(0, MAX_STATE_CHARS)}…` : text;
|
|
323
|
-
const started = Date.now();
|
|
324
|
-
const response = await completeSimple(
|
|
325
|
-
model,
|
|
326
|
-
{
|
|
327
|
-
systemPrompt: [SYSTEM_PROMPT],
|
|
328
|
-
messages: [{ role: "user", content: `<state>\n${state}\n</state>`, timestamp: Date.now() }],
|
|
329
|
-
tools: [buildTool(request.questions)],
|
|
330
|
-
},
|
|
331
|
-
{
|
|
332
|
-
apiKey,
|
|
333
|
-
maxTokens: model.reasoning ? Math.max(MAX_TOKENS, REASONING_SAFE_MAX_TOKENS) : MAX_TOKENS,
|
|
334
|
-
disableReasoning: true,
|
|
335
|
-
toolChoice: { type: "tool", name: TOOL_NAME },
|
|
336
|
-
signal: request.signal,
|
|
337
|
-
},
|
|
338
|
-
);
|
|
339
404
|
|
|
340
|
-
const
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
405
|
+
for (const model of candidates) {
|
|
406
|
+
if (request.signal?.aborted) return null;
|
|
407
|
+
// The ceiling still applies to an explicitly configured `smol` role — a role can
|
|
408
|
+
// point anywhere, including at a frontier model.
|
|
409
|
+
if (!deps.model && model.cost.input > costCeiling) {
|
|
410
|
+
logger.debug("decisions/llm: declining, model too expensive for a decision", {
|
|
411
|
+
id: modelKey(model),
|
|
412
|
+
inputCostPerMTok: model.cost.input,
|
|
413
|
+
ceiling: costCeiling,
|
|
414
|
+
});
|
|
415
|
+
return null;
|
|
416
|
+
}
|
|
417
|
+
const apiKey = await deps.registry.getApiKey(model, deps.sessionId);
|
|
418
|
+
if (!apiKey) {
|
|
419
|
+
logger.debug("decisions/llm: no credential", { provider: model.provider, id: model.id });
|
|
420
|
+
return null;
|
|
421
|
+
}
|
|
422
|
+
// The credential lookup awaited; an abort in that window would be missed by
|
|
423
|
+
// the listener registered below.
|
|
424
|
+
if (request.signal?.aborted) return null;
|
|
425
|
+
|
|
426
|
+
const controller = new AbortController();
|
|
427
|
+
const abortOnCaller = () => controller.abort();
|
|
428
|
+
request.signal?.addEventListener("abort", abortOnCaller, { once: true });
|
|
429
|
+
let timedOut = false;
|
|
430
|
+
const timer = setTimeout(() => {
|
|
431
|
+
timedOut = true;
|
|
432
|
+
controller.abort();
|
|
433
|
+
}, attemptTimeoutMs);
|
|
434
|
+
const started = Date.now();
|
|
435
|
+
let response: AssistantMessage;
|
|
436
|
+
try {
|
|
437
|
+
response = await complete(
|
|
438
|
+
model,
|
|
439
|
+
{
|
|
440
|
+
systemPrompt: [SYSTEM_PROMPT],
|
|
441
|
+
messages: [{ role: "user", content: `<state>\n${state}\n</state>`, timestamp: Date.now() }],
|
|
442
|
+
tools: [buildTool(request.questions)],
|
|
443
|
+
},
|
|
444
|
+
{
|
|
445
|
+
apiKey,
|
|
446
|
+
maxTokens: model.reasoning ? Math.max(MAX_TOKENS, REASONING_SAFE_MAX_TOKENS) : MAX_TOKENS,
|
|
447
|
+
disableReasoning: true,
|
|
448
|
+
toolChoice: { type: "tool", name: TOOL_NAME },
|
|
449
|
+
signal: controller.signal,
|
|
450
|
+
},
|
|
451
|
+
);
|
|
452
|
+
} catch (error) {
|
|
453
|
+
// A thrown transport error is as dead as a 404 for our purposes.
|
|
454
|
+
response = {
|
|
455
|
+
role: "assistant",
|
|
456
|
+
content: [],
|
|
457
|
+
stopReason: "error",
|
|
458
|
+
errorMessage: String(error),
|
|
459
|
+
} as unknown as AssistantMessage;
|
|
460
|
+
} finally {
|
|
461
|
+
clearTimeout(timer);
|
|
462
|
+
request.signal?.removeEventListener("abort", abortOnCaller);
|
|
463
|
+
}
|
|
464
|
+
|
|
465
|
+
if (request.signal?.aborted) return null;
|
|
466
|
+
if (timedOut || response.stopReason === "error" || response.stopReason === "aborted") {
|
|
467
|
+
const reason = timedOut
|
|
468
|
+
? `timeout after ${attemptTimeoutMs}ms`
|
|
469
|
+
: `${response.errorStatus ?? response.stopReason}: ${(response.errorMessage ?? "").slice(0, 200)}`;
|
|
470
|
+
deadModels.set(modelKey(model), { at: Date.now(), reason });
|
|
471
|
+
logger.debug("decisions/llm: model failed, trying the next candidate", {
|
|
472
|
+
id: modelKey(model),
|
|
473
|
+
reason,
|
|
474
|
+
durationMs: Date.now() - started,
|
|
475
|
+
});
|
|
476
|
+
continue;
|
|
477
|
+
}
|
|
478
|
+
|
|
479
|
+
const args = readToolArguments(response.content);
|
|
480
|
+
if (!args) {
|
|
481
|
+
// The provider answered but ignored the forced tool: a model answer, not an
|
|
482
|
+
// availability problem, so it is neither retried nor remembered as dead.
|
|
483
|
+
logger.debug("decisions/llm: model did not emit the forced tool call", {
|
|
484
|
+
id: modelKey(model),
|
|
485
|
+
stopReason: response.stopReason,
|
|
486
|
+
});
|
|
487
|
+
return null;
|
|
488
|
+
}
|
|
489
|
+
const answers = toAnswers(request.questions, args);
|
|
490
|
+
if (Object.keys(answers).length === 0) return null;
|
|
491
|
+
return {
|
|
492
|
+
answers,
|
|
493
|
+
backend: "llm",
|
|
494
|
+
model: modelKey(model),
|
|
495
|
+
calibrated: false,
|
|
496
|
+
durationMs: Date.now() - started,
|
|
497
|
+
};
|
|
344
498
|
}
|
|
345
|
-
|
|
346
|
-
if (Object.keys(answers).length === 0) return null;
|
|
347
|
-
return {
|
|
348
|
-
answers,
|
|
349
|
-
backend: "llm",
|
|
350
|
-
model: `${model.provider}/${model.id}`,
|
|
351
|
-
calibrated: false,
|
|
352
|
-
durationMs: Date.now() - started,
|
|
353
|
-
};
|
|
499
|
+
return null;
|
|
354
500
|
},
|
|
355
501
|
};
|
|
356
502
|
}
|
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* One typed decision per user turn, answering every routing question SKC has.
|
|
3
|
+
*
|
|
4
|
+
* Typed decisions started out wired to exactly one question — a five-way
|
|
5
|
+
* workflow enum — while the thirteen bundled UI skills were still selected by
|
|
6
|
+
* twenty hand-written regexes whose own comment measured them at 5 of 8 real
|
|
7
|
+
* frontend prompts and described the result as "not activation, it is hope".
|
|
8
|
+
*
|
|
9
|
+
* Both questions are about the same sentence, so they belong in the same call.
|
|
10
|
+
* The decision API takes a map of questions and returns a map of answers, which
|
|
11
|
+
* means adding the UI question costs one extra criteria block in the prompt and
|
|
12
|
+
* **zero** extra round trips: same latency budget, same deadline, same backend.
|
|
13
|
+
*
|
|
14
|
+
* Each question is skipped when a free deterministic stage already answered it,
|
|
15
|
+
* so the call shrinks to whatever is genuinely unknown — and vanishes entirely
|
|
16
|
+
* when nothing is.
|
|
17
|
+
*/
|
|
18
|
+
import { logger } from "@sayknow-cli/utils";
|
|
19
|
+
import { BUNDLED_SKC_UI_SKILL_NAMES, type BundledSkcUiSkillName } from "../defaults/skc-ui-skills";
|
|
20
|
+
import { BUNDLED_UI_SKILL_MEANINGS } from "../hooks/ui-skill-keywords";
|
|
21
|
+
import { CANONICAL_SKC_WORKFLOW_SKILLS, type CanonicalSkcWorkflowSkill } from "../skill-state/active-state";
|
|
22
|
+
import type { DecisionService } from "./index";
|
|
23
|
+
import {
|
|
24
|
+
buildRoutingCriteria,
|
|
25
|
+
MAX_PROMPT_CHARS,
|
|
26
|
+
MIN_CALIBRATED_CONFIDENCE,
|
|
27
|
+
MIN_PROMPT_CHARS,
|
|
28
|
+
NONE_CHOICE,
|
|
29
|
+
promptSignalLength,
|
|
30
|
+
} from "./skill-routing";
|
|
31
|
+
import type { Question } from "./types";
|
|
32
|
+
|
|
33
|
+
const UI_INSTRUCTIONS =
|
|
34
|
+
"Which bundled UI craft skill should be loaded for this request? Choose none unless the request is about building, reviewing, or polishing a user-visible interface.";
|
|
35
|
+
|
|
36
|
+
/** Exported so tests can assert the contract the model is actually given. */
|
|
37
|
+
export function buildUiSkillCriteria(): Record<string, string> {
|
|
38
|
+
const criteria: Record<string, string> = {};
|
|
39
|
+
for (const skill of BUNDLED_SKC_UI_SKILL_NAMES) criteria[skill] = BUNDLED_UI_SKILL_MEANINGS[skill];
|
|
40
|
+
criteria[NONE_CHOICE] =
|
|
41
|
+
"Not interface work: backend, data, infrastructure, tooling, SKC's own terminal UI, or a question with no surface to build.";
|
|
42
|
+
return criteria;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
export interface PromptTriage {
|
|
46
|
+
/** Null means the router deliberately chose no workflow, not that it failed. */
|
|
47
|
+
workflow: CanonicalSkcWorkflowSkill | null;
|
|
48
|
+
uiSkill: BundledSkcUiSkillName | null;
|
|
49
|
+
/** Only meaningful when {@link calibrated} is true. */
|
|
50
|
+
workflowConfidence: number | undefined;
|
|
51
|
+
calibrated: boolean;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
export interface PromptTriageRequest {
|
|
55
|
+
text: string;
|
|
56
|
+
/** Skip the workflow question — the keyword table already answered it. */
|
|
57
|
+
skipWorkflow?: boolean;
|
|
58
|
+
/** Skip the UI question — the regex table already matched. */
|
|
59
|
+
skipUiSkill?: boolean;
|
|
60
|
+
signal?: AbortSignal | undefined;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
export type PromptTriager = (request: PromptTriageRequest) => Promise<PromptTriage | null>;
|
|
64
|
+
|
|
65
|
+
function resolveChoice<T extends string>(
|
|
66
|
+
answer: unknown,
|
|
67
|
+
allowed: readonly T[],
|
|
68
|
+
calibrated: boolean,
|
|
69
|
+
label: string,
|
|
70
|
+
): { value: T | null; confidence: number | undefined } {
|
|
71
|
+
const choice = answer as { type?: string; choice?: string; confidence?: number } | undefined;
|
|
72
|
+
if (choice?.type !== "choice") return { value: null, confidence: undefined };
|
|
73
|
+
if (choice.choice === NONE_CHOICE) return { value: null, confidence: choice.confidence };
|
|
74
|
+
const match = allowed.find(candidate => candidate === choice.choice);
|
|
75
|
+
if (!match) return { value: null, confidence: choice.confidence };
|
|
76
|
+
if (calibrated && (choice.confidence ?? 0) < MIN_CALIBRATED_CONFIDENCE) {
|
|
77
|
+
// A floor is only meaningful against a calibrated probability; against an
|
|
78
|
+
// ordinal score it would reject answers that are simply scaled differently.
|
|
79
|
+
logger.debug("decisions/prompt-triage: below confidence floor", {
|
|
80
|
+
question: label,
|
|
81
|
+
choice: choice.choice,
|
|
82
|
+
confidence: choice.confidence,
|
|
83
|
+
floor: MIN_CALIBRATED_CONFIDENCE,
|
|
84
|
+
});
|
|
85
|
+
return { value: null, confidence: choice.confidence };
|
|
86
|
+
}
|
|
87
|
+
return { value: match, confidence: choice.confidence };
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* Build the per-turn triager.
|
|
92
|
+
*
|
|
93
|
+
* Returns null when the service is disabled, the prompt is too short to carry
|
|
94
|
+
* intent, every question was already answered for free, or the backend did not
|
|
95
|
+
* produce a usable result. Null means "no information", which is different from
|
|
96
|
+
* a result whose fields are all null — that one is the router saying "none", and
|
|
97
|
+
* the keyword learner treats it as a negative example.
|
|
98
|
+
*/
|
|
99
|
+
export function createPromptTriage(service: DecisionService): PromptTriager {
|
|
100
|
+
const workflowCriteria = buildRoutingCriteria();
|
|
101
|
+
const uiCriteria = buildUiSkillCriteria();
|
|
102
|
+
return async (request: PromptTriageRequest): Promise<PromptTriage | null> => {
|
|
103
|
+
if (!service.enabled) return null;
|
|
104
|
+
const trimmed = request.text.trim();
|
|
105
|
+
if (promptSignalLength(trimmed) < MIN_PROMPT_CHARS) return null;
|
|
106
|
+
const state = trimmed.length > MAX_PROMPT_CHARS ? trimmed.slice(0, MAX_PROMPT_CHARS) : trimmed;
|
|
107
|
+
|
|
108
|
+
const questions: Record<string, Question> = {};
|
|
109
|
+
if (!request.skipWorkflow) {
|
|
110
|
+
questions.workflow = {
|
|
111
|
+
type: "choice",
|
|
112
|
+
instructions:
|
|
113
|
+
"Which workflow should handle this user request? Choose none unless the request clearly calls for one of the workflows.",
|
|
114
|
+
criteria: workflowCriteria,
|
|
115
|
+
};
|
|
116
|
+
}
|
|
117
|
+
if (!request.skipUiSkill) {
|
|
118
|
+
questions.uiSkill = { type: "choice", instructions: UI_INSTRUCTIONS, criteria: uiCriteria };
|
|
119
|
+
}
|
|
120
|
+
if (Object.keys(questions).length === 0) return null;
|
|
121
|
+
|
|
122
|
+
const result = await service.decide({ state, questions, signal: request.signal });
|
|
123
|
+
if (!result) return null;
|
|
124
|
+
|
|
125
|
+
const workflow = request.skipWorkflow
|
|
126
|
+
? { value: null, confidence: undefined }
|
|
127
|
+
: resolveChoice(result.answers.workflow, CANONICAL_SKC_WORKFLOW_SKILLS, result.calibrated, "workflow");
|
|
128
|
+
const uiSkill = request.skipUiSkill
|
|
129
|
+
? { value: null, confidence: undefined }
|
|
130
|
+
: resolveChoice(result.answers.uiSkill, BUNDLED_SKC_UI_SKILL_NAMES, result.calibrated, "uiSkill");
|
|
131
|
+
|
|
132
|
+
logger.debug("decisions/prompt-triage: answered", {
|
|
133
|
+
workflow: workflow.value,
|
|
134
|
+
uiSkill: uiSkill.value,
|
|
135
|
+
backend: result.backend,
|
|
136
|
+
model: result.model,
|
|
137
|
+
calibrated: result.calibrated,
|
|
138
|
+
durationMs: result.durationMs,
|
|
139
|
+
});
|
|
140
|
+
return {
|
|
141
|
+
workflow: workflow.value,
|
|
142
|
+
uiSkill: uiSkill.value,
|
|
143
|
+
workflowConfidence: workflow.confidence,
|
|
144
|
+
calibrated: result.calibrated,
|
|
145
|
+
};
|
|
146
|
+
};
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
export type SkillRouter = (text: string, signal?: AbortSignal) => Promise<CanonicalSkcWorkflowSkill | null>;
|
|
150
|
+
|
|
151
|
+
/**
|
|
152
|
+
* Workflow-only view of the triager.
|
|
153
|
+
*
|
|
154
|
+
* A few lines over the same implementation rather than a second one: the eval
|
|
155
|
+
* harness in `scripts/eval-skill-routing.ts` and the routing tests want
|
|
156
|
+
* text-in/skill-out and have no UI question to ask, and a parallel router would
|
|
157
|
+
* be free to drift away from the thresholds this one enforces.
|
|
158
|
+
*/
|
|
159
|
+
export function createSemanticSkillRouter(service: DecisionService): SkillRouter {
|
|
160
|
+
const triage = createPromptTriage(service);
|
|
161
|
+
return async (text: string, signal?: AbortSignal): Promise<CanonicalSkcWorkflowSkill | null> =>
|
|
162
|
+
(await triage({ text, skipUiSkill: true, signal }))?.workflow ?? null;
|
|
163
|
+
}
|
|
@@ -1,25 +1,29 @@
|
|
|
1
1
|
/**
|
|
2
|
-
*
|
|
2
|
+
* The workflow-routing question: what the model is asked, and the thresholds the
|
|
3
|
+
* answer is held to.
|
|
3
4
|
*
|
|
4
|
-
* The keyword table in `hooks/skill-keywords.ts` is
|
|
5
|
+
* The keyword table in `hooks/skill-keywords.ts` is a list of literal strings. It is
|
|
5
6
|
* exact and free, and it is the right first stage — but measured against realistic
|
|
6
7
|
* paraphrases it recalls 4/17, and **0/9 in Korean**, which is most of our users. A
|
|
7
8
|
* miss is not fatal (the model still sees the routing rules in the system prompt), but
|
|
8
9
|
* it means the deterministic gate simply does not exist for those prompts.
|
|
9
10
|
*
|
|
10
|
-
*
|
|
11
|
+
* The staging that closes the gap lives in `prompt-triage.ts`:
|
|
11
12
|
*
|
|
12
|
-
* keyword (exact, free) ->
|
|
13
|
+
* keyword + learned (exact, free) -> one typed decision -> system prompt (as today)
|
|
13
14
|
*
|
|
14
|
-
* The
|
|
15
|
-
* the same 22 prompts, the literal stage is the one that catches `ultragoal this`
|
|
16
|
-
* `consensus plan`; the semantic stage is the one that catches everything Korean.
|
|
15
|
+
* The stages fail in opposite directions, which is why all of them are kept. Measured
|
|
16
|
+
* on the same 22 prompts, the literal stage is the one that catches `ultragoal this`
|
|
17
|
+
* and `consensus plan`; the semantic stage is the one that catches everything Korean.
|
|
18
|
+
*
|
|
19
|
+
* This file holds only the contract and the numbers, so the criteria cannot drift away
|
|
20
|
+
* from the thresholds that judge answers against them.
|
|
17
21
|
*/
|
|
18
|
-
import { logger } from "@sayknow-cli/utils";
|
|
19
22
|
import { CANONICAL_SKC_WORKFLOW_SKILLS, type CanonicalSkcWorkflowSkill } from "../skill-state/active-state";
|
|
20
|
-
import type { DecisionService } from "./index";
|
|
21
23
|
|
|
22
|
-
|
|
24
|
+
/** Shared across every routing question so one answer shape covers them all. */
|
|
25
|
+
export const NONE_CHOICE = "none";
|
|
26
|
+
const NONE = NONE_CHOICE;
|
|
23
27
|
|
|
24
28
|
/**
|
|
25
29
|
* What each workflow is *for*, in the words a user would recognise. These descriptions
|
|
@@ -40,7 +44,7 @@ const WORKFLOW_MEANINGS: Record<CanonicalSkcWorkflowSkill, string> = {
|
|
|
40
44
|
team: "The work is large enough to split across several coordinated workers running in parallel.",
|
|
41
45
|
};
|
|
42
46
|
|
|
43
|
-
const ROUTING_INSTRUCTIONS =
|
|
47
|
+
export const ROUTING_INSTRUCTIONS =
|
|
44
48
|
"Which workflow should handle this user request? Choose none unless the request clearly calls for one of the workflows.";
|
|
45
49
|
|
|
46
50
|
/** Exported so tests can assert the contract the model is actually given. */
|
|
@@ -52,10 +56,30 @@ export function buildRoutingCriteria(): Record<string, string> {
|
|
|
52
56
|
return criteria;
|
|
53
57
|
}
|
|
54
58
|
|
|
55
|
-
/**
|
|
56
|
-
|
|
59
|
+
/**
|
|
60
|
+
* Prompts below this length never carry enough signal to justify a model round-trip.
|
|
61
|
+
* Measured with {@link promptSignalLength}, not `String.length`: the floor was fitted
|
|
62
|
+
* to English and Korean, and in code units a complete Chinese request is shorter than
|
|
63
|
+
* "ok thanks".
|
|
64
|
+
*/
|
|
65
|
+
export const MIN_PROMPT_CHARS = 12;
|
|
57
66
|
/** Only the opening of a prompt decides its workflow; the rest is payload. */
|
|
58
|
-
const MAX_PROMPT_CHARS = 4_000;
|
|
67
|
+
export const MAX_PROMPT_CHARS = 4_000;
|
|
68
|
+
|
|
69
|
+
const DENSE_SCRIPT_PATTERN = /[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Hangul}]/gu;
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Prompt length in Latin-letter equivalents.
|
|
73
|
+
*
|
|
74
|
+
* One Han character, kana or Hangul syllable carries what two or three Latin letters
|
|
75
|
+
* do, so each counts double. "先做架构设计" (6 code units) is a whole request; "谢谢" and
|
|
76
|
+
* "고마워요" still fall under the floor, which is the point of having one.
|
|
77
|
+
*/
|
|
78
|
+
export function promptSignalLength(text: string): number {
|
|
79
|
+
let dense = 0;
|
|
80
|
+
for (const _ of text.matchAll(DENSE_SCRIPT_PATTERN)) dense++;
|
|
81
|
+
return text.length + dense;
|
|
82
|
+
}
|
|
59
83
|
|
|
60
84
|
/**
|
|
61
85
|
* Minimum calibrated confidence required to activate a workflow.
|
|
@@ -79,45 +103,4 @@ const MAX_PROMPT_CHARS = 4_000;
|
|
|
79
103
|
* through a forced enum has no meaningful confidence to compare against, so gating on a
|
|
80
104
|
* number it did not really produce would just be superstition.
|
|
81
105
|
*/
|
|
82
|
-
const MIN_CALIBRATED_CONFIDENCE = 0.75;
|
|
83
|
-
|
|
84
|
-
export type SkillRouter = (text: string) => Promise<CanonicalSkcWorkflowSkill | null>;
|
|
85
|
-
|
|
86
|
-
/**
|
|
87
|
-
* Build the semantic router. Returns null-resolving function when the service is
|
|
88
|
-
* disabled so the caller keeps its existing behaviour with no branching.
|
|
89
|
-
*/
|
|
90
|
-
export function createSemanticSkillRouter(service: DecisionService): SkillRouter {
|
|
91
|
-
const criteria = buildRoutingCriteria();
|
|
92
|
-
return async (text: string): Promise<CanonicalSkcWorkflowSkill | null> => {
|
|
93
|
-
if (!service.enabled) return null;
|
|
94
|
-
const trimmed = text.trim();
|
|
95
|
-
if (trimmed.length < MIN_PROMPT_CHARS) return null;
|
|
96
|
-
const state = trimmed.length > MAX_PROMPT_CHARS ? trimmed.slice(0, MAX_PROMPT_CHARS) : trimmed;
|
|
97
|
-
|
|
98
|
-
const result = await service.decide({
|
|
99
|
-
state,
|
|
100
|
-
questions: { workflow: { type: "choice", instructions: ROUTING_INSTRUCTIONS, criteria } },
|
|
101
|
-
});
|
|
102
|
-
const answer = result?.answers.workflow;
|
|
103
|
-
if (!result || answer?.type !== "choice" || answer.choice === NONE) return null;
|
|
104
|
-
const skill = CANONICAL_SKC_WORKFLOW_SKILLS.find(candidate => candidate === answer.choice);
|
|
105
|
-
if (!skill) return null;
|
|
106
|
-
if (result.calibrated && (answer.confidence ?? 0) < MIN_CALIBRATED_CONFIDENCE) {
|
|
107
|
-
logger.debug("decisions/skill-routing: below confidence floor, leaving routing alone", {
|
|
108
|
-
skill,
|
|
109
|
-
confidence: answer.confidence,
|
|
110
|
-
floor: MIN_CALIBRATED_CONFIDENCE,
|
|
111
|
-
});
|
|
112
|
-
return null;
|
|
113
|
-
}
|
|
114
|
-
logger.debug("decisions/skill-routing: semantic match", {
|
|
115
|
-
skill,
|
|
116
|
-
backend: result.backend,
|
|
117
|
-
confidence: answer.confidence,
|
|
118
|
-
calibrated: result.calibrated,
|
|
119
|
-
durationMs: result.durationMs,
|
|
120
|
-
});
|
|
121
|
-
return skill;
|
|
122
|
-
};
|
|
123
|
-
}
|
|
106
|
+
export const MIN_CALIBRATED_CONFIDENCE = 0.75;
|