@sayknow-cli/coding-agent 0.5.20 → 0.5.22

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. package/CHANGELOG.md +22 -0
  2. package/dist/types/cli/auth-gateway-cli.d.ts +24 -0
  3. package/dist/types/cli/setup-cli.d.ts +15 -1
  4. package/dist/types/commands/auth-gateway.d.ts +2 -1
  5. package/dist/types/commands/setup.d.ts +6 -0
  6. package/dist/types/config/settings-schema.d.ts +9 -0
  7. package/dist/types/decisions/index.d.ts +18 -0
  8. package/dist/types/decisions/llm-backend.d.ts +51 -0
  9. package/dist/types/decisions/skill-routing.d.ts +8 -0
  10. package/dist/types/decisions/types.d.ts +91 -0
  11. package/dist/types/decisions/typesafe-backend.d.ts +15 -0
  12. package/dist/types/hooks/skill-state.d.ts +6 -0
  13. package/dist/types/modes/components/provider-onboarding-selector.d.ts +1 -1
  14. package/dist/types/modes/components/typesafe-key-prompt.d.ts +23 -0
  15. package/dist/types/sdk/bus/native-runtime-compatibility.d.ts +3 -1
  16. package/dist/types/session/agent-session.d.ts +0 -9
  17. package/dist/types/setup/decision-provider.d.ts +24 -0
  18. package/package.json +7 -7
  19. package/scripts/eval-skill-routing.ts +172 -0
  20. package/src/cli/auth-gateway-cli.ts +128 -85
  21. package/src/cli/setup-cli.ts +53 -1
  22. package/src/commands/auth-gateway.ts +7 -5
  23. package/src/commands/setup.ts +5 -0
  24. package/src/config/settings-schema.ts +12 -0
  25. package/src/decisions/index.ts +84 -0
  26. package/src/decisions/llm-backend.ts +356 -0
  27. package/src/decisions/skill-routing.ts +83 -0
  28. package/src/decisions/types.ts +119 -0
  29. package/src/decisions/typesafe-backend.ts +168 -0
  30. package/src/hooks/skill-keywords.ts +56 -0
  31. package/src/hooks/skill-state.ts +18 -2
  32. package/src/internal-urls/docs-index.generated.ts +1 -1
  33. package/src/modes/components/provider-onboarding-selector.ts +13 -1
  34. package/src/modes/components/typesafe-key-prompt.ts +108 -0
  35. package/src/modes/controllers/selector-controller.ts +44 -0
  36. package/src/sdk/bus/native-runtime-compatibility.ts +30 -3
  37. package/src/session/agent-session.ts +51 -1
  38. package/src/setup/decision-provider.ts +94 -0
package/CHANGELOG.md CHANGED
@@ -2,6 +2,28 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [0.5.22] - 2026-09-19
6
+
7
+ ### Added
8
+
9
+ - TypeSafe (hosted System One `jev`) as a typed-decision backend. Add the key with `TYPESAFE_API_KEY=<key> skc setup typesafe`; it is verified against the live API before being stored, because the decision service fails open and an unverified bad key would be swallowed silently forever. Removing it (`--remove`) falls back to your logged-in model. The key is read from the environment, never a flag — this repo already refuses raw `--api-key` values because they leak into shell history and the process list.
10
+ - Decision backends now resolve in order: TypeSafe when a key exists (the only backend that returns calibrated probabilities), otherwise the model you are already logged into. TypeSafe is deliberately **not** registered in the chat-provider registry: it has no stream, no messages and no text output, so giving it a `Model` shape would put a non-chat endpoint in the model picker.
11
+ - Typed decisions (`decisions.enabled`, default off): a small, reusable service that asks the model you are already logged into a typed question and gets back a value your code can branch on. Type safety comes from a forced tool call with enum-constrained fields, so an option outside the declared set cannot reach the caller. No extra vendor, no extra key.
12
+ - Workflow-skill routing now has a semantic second stage. The literal keyword table stays first and free; the model is consulted only when it matches nothing. Measured over 23 prompts × 3 runs against `claude-opus-5`: keyword-only **0/27 in Korean** (43% overall), hybrid **69/69 (100%)** with zero false activations on the 18 negative cases, p50 1.46s. Reproduce with `bun scripts/eval-skill-routing.ts --repeat 3`.
13
+
14
+ ### Fixed
15
+
16
+ - `NativeRuntimeCompatibilityError` now lists only the causes that actually fired and names the file the process loaded `@sayknow-cli/natives` from. A pure version mismatch used to read "required workflow arbitration methods are **available**", which described a healthy runtime while refusing to start every session and extension.
17
+ - `dev:link` / `dev:doctor` fail on (and `dev:link` removes) nested installs under `packages/<pkg>/node_modules` that shadow a workspace package, and verify that `@sayknow-cli/natives` resolves to the same version as the `coding-agent` runtime that loads it. A published `@sayknow-cli/natives` copy left inside `packages/coding-agent/node_modules` wins resolution over the workspace link and breaks every SDK session.
18
+ - `dev:doctor` accepts this checkout's bun-linked `bin/skc.js` wrapper — the one `install:dev` itself creates — instead of reporting it as drift, but only when the wrapper is byte-identical to the expected workspace wrapper and `@sayknow-cli/coding-agent/cli` resolves back to this checkout's `src/cli.ts`.
19
+
20
+ ## [0.5.21] - 2026-09-17
21
+
22
+ ### Changed
23
+
24
+ - `skc auth-gateway` (`serve`, `status`, `check`) now falls back to the local SQLite credential store when no broker is configured, matching `discoverAuthStorage()` precedence. A configured broker still wins and never degrades to local credentials. Single-machine users no longer have to run `skc auth-broker serve` alongside the gateway.
25
+ - `skc auth-gateway status --json` reports `source` (`broker`/`local`), `dbPath`, and `credentialCount` in both modes; `ready` now requires a bearer token **and** at least one credential (`reason`: `token_missing`, `no_credentials`, `broker_unavailable`, `local_store_unavailable`).
26
+
5
27
  ## [0.5.20] - 2026-09-17
6
28
 
7
29
  ### Fixed
@@ -1,3 +1,4 @@
1
+ import { AuthStorage } from "@sayknow-cli/ai";
1
2
  export type AuthGatewayAction = "serve" | "token" | "status" | "check";
2
3
  export interface AuthGatewayCommandArgs {
3
4
  action: AuthGatewayAction;
@@ -14,5 +15,28 @@ export interface AuthGatewayCommandArgs {
14
15
  };
15
16
  }
16
17
  declare const ACTIONS: readonly AuthGatewayAction[];
18
+ /**
19
+ * Credential source the gateway is serving from. `broker` mirrors the previous
20
+ * behaviour; `local` is the single-machine path that makes `skc auth-broker
21
+ * serve` optional.
22
+ */
23
+ export interface GatewayCredentialSource {
24
+ storage: AuthStorage;
25
+ kind: "broker" | "local";
26
+ /** Broker URL in broker mode, `null` in local mode. */
27
+ brokerUrl: string | null;
28
+ /** `<agentDir>/agent.db` in local mode, `null` in broker mode. */
29
+ dbPath: string | null;
30
+ /** `broker <url>` / `local <dbPath>` — also used as the AuthStorage sourceLabel. */
31
+ label: string;
32
+ }
33
+ /**
34
+ * Open the credential store the gateway should serve from.
35
+ *
36
+ * Same precedence as `discoverAuthStorage()` in sdk/session.ts: a configured
37
+ * broker wins, otherwise fall back to the local SQLite store. Callers own the
38
+ * returned `storage` and must `close()` it.
39
+ */
40
+ export declare function openGatewayCredentialSource(): Promise<GatewayCredentialSource>;
17
41
  export declare function runAuthGatewayCommand(cmd: AuthGatewayCommandArgs): Promise<void>;
18
42
  export { ACTIONS as AUTH_GATEWAY_ACTIONS };
@@ -5,7 +5,7 @@
5
5
  */
6
6
  import { AuthStorage, SqliteAuthCredentialStore } from "@sayknow-cli/ai";
7
7
  import { runExternalCredentialAutoImport } from "../setup/credential-auto-import";
8
- export type SetupComponent = "claude" | "codex" | "credentials" | "defaults" | "hermes" | "hooks" | "provider" | "python" | "stt" | "ui-skills";
8
+ export type SetupComponent = "claude" | "codex" | "credentials" | "defaults" | "hermes" | "hooks" | "provider" | "python" | "stt" | "typesafe" | "ui-skills";
9
9
  export interface SetupCommandArgs {
10
10
  component: SetupComponent;
11
11
  flags: {
@@ -37,6 +37,8 @@ export interface SetupCommandArgs {
37
37
  yes?: boolean;
38
38
  dryRun?: boolean;
39
39
  keychain?: boolean;
40
+ skipVerify?: boolean;
41
+ remove?: boolean;
40
42
  };
41
43
  }
42
44
  /**
@@ -71,3 +73,15 @@ export declare function handleCredentialsSetup(flags: {
71
73
  * Print setup command help.
72
74
  */
73
75
  export declare function printSetupHelp(): void;
76
+ /**
77
+ * `skc setup typesafe` — enable the hosted System One model for typed decisions.
78
+ *
79
+ * The key is read from `TYPESAFE_API_KEY`, never from a flag: this repo already refuses
80
+ * raw `--api-key` values because they land in shell history and in the process list of
81
+ * every user on the machine. Same rule applies here.
82
+ *
83
+ * Verified against the live API before storing. The decision service fails open, so an
84
+ * unverified bad key would be swallowed forever — the user would believe TypeSafe was
85
+ * active while every decision quietly came from their own model.
86
+ */
87
+ export declare function handleTypeSafeSetup(flags: SetupCommandArgs["flags"]): Promise<void>;
@@ -1,5 +1,6 @@
1
1
  /**
2
- * `skc auth-gateway` — run a forward proxy that injects auth from the broker.
2
+ * `skc auth-gateway` — run a forward proxy that injects auth from the broker,
3
+ * or from the local credential store when no broker is configured.
3
4
  */
4
5
  import { Command } from "@sayknow-cli/utils/cli";
5
6
  import { type AuthGatewayAction } from "../cli/auth-gateway-cli";
@@ -100,6 +100,12 @@ export default class Setup extends Command {
100
100
  "dry-run": import("@sayknow-cli/utils/cli").FlagDescriptor<"boolean"> & {
101
101
  description: string;
102
102
  };
103
+ "skip-verify": import("@sayknow-cli/utils/cli").FlagDescriptor<"boolean"> & {
104
+ description: string;
105
+ };
106
+ remove: import("@sayknow-cli/utils/cli").FlagDescriptor<"boolean"> & {
107
+ description: string;
108
+ };
103
109
  };
104
110
  run(): Promise<void>;
105
111
  }
@@ -2412,6 +2412,15 @@ export declare const SETTINGS_SCHEMA: {
2412
2412
  readonly type: "number";
2413
2413
  readonly default: 16000;
2414
2414
  };
2415
+ readonly "decisions.enabled": {
2416
+ readonly type: "boolean";
2417
+ readonly default: false;
2418
+ readonly ui: {
2419
+ readonly tab: "context";
2420
+ readonly label: "Typed decisions";
2421
+ readonly description: "Let a cheap model answer typed questions the deterministic rules cannot. Currently routes workflow skills when the keyword table finds nothing — which is every non-English phrasing. Costs one small model call on those prompts; every failure falls back to today's behaviour.";
2422
+ };
2423
+ };
2415
2424
  readonly "ttsr.enabled": {
2416
2425
  readonly type: "boolean";
2417
2426
  readonly default: true;
@@ -0,0 +1,18 @@
1
+ import { type LlmBackendDeps } from "./llm-backend";
2
+ import type { DecisionBackend, DecisionRequest, DecisionResult } from "./types";
3
+ export { createLlmDecisionBackend } from "./llm-backend";
4
+ export * from "./types";
5
+ export { createTypeSafeDecisionBackend, TYPESAFE_PROVIDER } from "./typesafe-backend";
6
+ export interface DecisionServiceOptions extends LlmBackendDeps {
7
+ /** Off by default; callers opt in per feature. */
8
+ enabled?: boolean;
9
+ timeoutMs?: number;
10
+ /** Injection point for tests and for the self-hosted/hosted backends. */
11
+ backends?: DecisionBackend[];
12
+ }
13
+ export interface DecisionService {
14
+ readonly enabled: boolean;
15
+ /** Resolves null when disabled, unavailable, timed out, or the model misbehaved. */
16
+ decide(request: DecisionRequest): Promise<DecisionResult | null>;
17
+ }
18
+ export declare function createDecisionService(options: DecisionServiceOptions): DecisionService;
@@ -0,0 +1,51 @@
1
+ /**
2
+ * Decision backend that runs on the model the user is already logged into.
3
+ *
4
+ * No extra API key, no extra vendor, no data leaving the providers the user already
5
+ * trusts. The type safety comes from a **forced tool call with enum-constrained
6
+ * properties**: the provider itself rejects any value outside the declared set, so a
7
+ * malformed or hallucinated option cannot reach our code — the same guarantee the
8
+ * hosted System One model gives, enforced one layer up.
9
+ *
10
+ * Two deliberate omissions:
11
+ *
12
+ * 1. We never ask the model to emit probabilities. Measured elsewhere on this exact
13
+ * task shape, writing probabilities collapses accuracy (~0.35 vs ~0.90 for picking
14
+ * a constrained option), and the numbers are not calibrated anyway. Answers from
15
+ * this backend carry `calibrated: false` and no `probabilities` map.
16
+ * 2. Every question goes in **one** call. Splitting them multiplies cost and latency
17
+ * while the enum constraint already keeps each field independent.
18
+ */
19
+ import { type Api, type Model } from "@sayknow-cli/ai";
20
+ import type { ModelRegistry } from "../config/model-registry";
21
+ import type { Settings } from "../config/settings";
22
+ import { type DecisionBackend } from "./types";
23
+ export interface LlmBackendDeps {
24
+ /**
25
+ * Refuse to spend more than this per million input tokens on a decision.
26
+ *
27
+ * The whole premise of a typed-decision service is judgment cheap enough to put in
28
+ * places you could not previously afford it. Routing a prompt through a frontier
29
+ * model inverts that: the deterministic path it replaces costs effectively nothing
30
+ * (the routing rules already sit in the cached system prompt), so a decision call on
31
+ * an expensive model is a pure cost *increase* for a few points of accuracy.
32
+ *
33
+ * Measured: a routing decision on claude-opus-5 costs ~$0.0063 and 1.46s; the same
34
+ * decision on the hosted System One model costs ~$0.000018 and 0.31s.
35
+ *
36
+ * Above the cap this backend declines, which leaves routing to the system prompt —
37
+ * exactly the behaviour before typed decisions existed. Configure a `smol` role with
38
+ * a cheap model to turn it back on.
39
+ */
40
+ maxInputCostPerMTok?: number;
41
+ /** Injected in tests to make the local-runtime probe deterministic. */
42
+ fetchImpl?: typeof fetch;
43
+ registry: ModelRegistry;
44
+ settings: Settings;
45
+ sessionId?: string;
46
+ /** Overrides role resolution; used by callers that already picked a model. */
47
+ model?: Model<Api>;
48
+ }
49
+ /** Reset between tests; also lets a caller force a fresh probe after starting a runtime. */
50
+ export declare function clearLocalRuntimeLivenessCache(): void;
51
+ export declare function createLlmDecisionBackend(deps: LlmBackendDeps): DecisionBackend;
@@ -0,0 +1,8 @@
1
+ import { type CanonicalSkcWorkflowSkill } from "../skill-state/active-state";
2
+ import type { DecisionService } from "./index";
3
+ export type SkillRouter = (text: string) => Promise<CanonicalSkcWorkflowSkill | null>;
4
+ /**
5
+ * Build the semantic router. Returns null-resolving function when the service is
6
+ * disabled so the caller keeps its existing behaviour with no branching.
7
+ */
8
+ export declare function createSemanticSkillRouter(service: DecisionService): SkillRouter;
@@ -0,0 +1,91 @@
1
+ /**
2
+ * Typed decisions — "Jev-shaped" structured judgments for code to branch on.
3
+ *
4
+ * The request/response shape follows TypeSafe's System One API so a backend can be
5
+ * swapped without touching call sites: the hosted `jev` model, a self-hosted OpenJev
6
+ * daemon, or — the default — the model the user is already logged into.
7
+ *
8
+ * What this is NOT: a probability oracle. Only the hosted model returns calibrated
9
+ * probabilities. Backends that constrain an ordinary LLM return an ordinal value and
10
+ * report `calibrated: false`; treat those numbers as a ranking, never as P(correct).
11
+ */
12
+ /** Pick exactly one option from a closed set. */
13
+ export interface ChoiceQuestion {
14
+ type: "choice";
15
+ instructions: string;
16
+ /** option id -> what that option means. At least two. */
17
+ criteria: Record<string, string>;
18
+ }
19
+ /** Rate the state against ordered levels. Level 0 is the lowest. */
20
+ export interface ScoreQuestion {
21
+ type: "score";
22
+ instructions: string;
23
+ /** Ordered level descriptions, lowest first. At least two. */
24
+ criteria: string[];
25
+ }
26
+ /** Is this statement true of the state? */
27
+ export interface NoulQuestion {
28
+ type: "noul";
29
+ instructions: string;
30
+ }
31
+ export type Question = ChoiceQuestion | ScoreQuestion | NoulQuestion;
32
+ export interface ChoiceAnswer {
33
+ type: "choice";
34
+ /** The selected option id. Always one of the declared `criteria` keys. */
35
+ choice: string;
36
+ /** Present only when the backend exposes a distribution. */
37
+ probabilities?: Record<string, number>;
38
+ /**
39
+ * How certain the backend is, 0..1. Only meaningful when the result reports
40
+ * `calibrated: true` — that is the difference between a number you can threshold on
41
+ * and a number that merely ranks. Absent when the backend cannot supply one.
42
+ */
43
+ confidence?: number;
44
+ }
45
+ export interface ScoreAnswer {
46
+ type: "score";
47
+ /** Level index. Fractional only when the backend returns a distribution. */
48
+ score: number;
49
+ /** Selected level index. */
50
+ level: number;
51
+ legend: Record<string, string>;
52
+ probabilities?: Record<string, number>;
53
+ /** See {@link ChoiceAnswer.confidence}. */
54
+ confidence?: number;
55
+ }
56
+ export interface NoulAnswer {
57
+ type: "noul";
58
+ /** 0 (no) .. 1 (yes). Ordinal unless `calibrated` is true. */
59
+ noul: number;
60
+ }
61
+ export type Answer = ChoiceAnswer | ScoreAnswer | NoulAnswer;
62
+ export interface DecisionResult {
63
+ answers: Record<string, Answer>;
64
+ /** Which backend answered, for logging and A/B comparison. */
65
+ backend: string;
66
+ /** Model identifier the backend used. */
67
+ model: string;
68
+ /**
69
+ * False means the numbers are ordinal rankings, not probabilities.
70
+ * Only the hosted System One model reports true.
71
+ */
72
+ calibrated: boolean;
73
+ durationMs: number;
74
+ }
75
+ export interface DecisionRequest {
76
+ /** The content to judge. Plain text, or JSON-serialisable structured state. */
77
+ state: string | Record<string, unknown> | unknown[];
78
+ /** Question id -> question. Answers come back under the same ids. */
79
+ questions: Record<string, Question>;
80
+ signal?: AbortSignal;
81
+ }
82
+ export interface DecisionBackend {
83
+ readonly name: string;
84
+ /** Resolves null when the backend is unavailable (no credentials, offline, disabled). */
85
+ decide(request: DecisionRequest): Promise<DecisionResult | null>;
86
+ }
87
+ export declare const MIN_OPTIONS = 2;
88
+ /** Matches OpenJev's letter-slot ceiling so a graph stays portable across backends. */
89
+ export declare const MAX_OPTIONS = 16;
90
+ export declare function validateQuestions(questions: Record<string, Question>): void;
91
+ export declare function stateToText(state: DecisionRequest["state"]): string;
@@ -0,0 +1,15 @@
1
+ import type { ModelRegistry } from "../config/model-registry";
2
+ import { type DecisionBackend } from "./types";
3
+ /** Provider id under which the key is stored and surfaced in the model list. */
4
+ export declare const TYPESAFE_PROVIDER = "typesafe";
5
+ export interface TypeSafeBackendDeps {
6
+ registry: ModelRegistry;
7
+ sessionId?: string;
8
+ /** Override for self-hosted or proxied deployments. */
9
+ baseUrl?: string;
10
+ /** Model id sent in the request body. Named to avoid colliding with the LLM backend's `model`. */
11
+ modelId?: string;
12
+ /** Injected in tests; defaults to global fetch. */
13
+ fetchImpl?: typeof fetch;
14
+ }
15
+ export declare function createTypeSafeDecisionBackend(deps: TypeSafeBackendDeps): DecisionBackend;
@@ -38,6 +38,12 @@ export interface RecordSkillActivationInput {
38
38
  turnId?: string;
39
39
  nowIso?: string;
40
40
  stateDir?: string;
41
+ /**
42
+ * Semantic fallback, consulted only when no keyword matched. Supplying it turns the
43
+ * literal keyword table into a two-stage router; omitting it keeps the historical
44
+ * keyword-only behaviour byte for byte.
45
+ */
46
+ resolveSkillSemantically?: (text: string) => Promise<SkcWorkflowSkill | null>;
41
47
  }
42
48
  export interface StopHookInput {
43
49
  cwd: string;
@@ -1,5 +1,5 @@
1
1
  import { Container } from "@sayknow-cli/tui";
2
- export type ProviderOnboardingAction = "custom-provider-wizard" | "oauth-login" | "import-credentials" | "api-guide";
2
+ export type ProviderOnboardingAction = "custom-provider-wizard" | "oauth-login" | "import-credentials" | "api-guide" | "typesafe-key";
3
3
  export declare class ProviderOnboardingSelectorComponent extends Container {
4
4
  #private;
5
5
  constructor(onSelect: (action: ProviderOnboardingAction) => void, onCancel: () => void);
@@ -0,0 +1,23 @@
1
+ /**
2
+ * Key entry for TypeSafe, reachable from the same place models are added.
3
+ *
4
+ * TypeSafe is not a chat model, so it never appears in the model picker — but the place
5
+ * users look when they want to "add a model with a key" is this onboarding list, and
6
+ * making them find a CLI subcommand instead would mean most users never enable it.
7
+ *
8
+ * The key is taken through {@link SecretInput} and consumed once: it is never rendered,
9
+ * never placed in a flag, and never written anywhere but the credential store.
10
+ */
11
+ import { Container } from "@sayknow-cli/tui";
12
+ export interface TypeSafeKeyPromptResult {
13
+ apiKey: string;
14
+ }
15
+ export declare class TypeSafeKeyPromptComponent extends Container {
16
+ #private;
17
+ constructor(onSubmit: (result: TypeSafeKeyPromptResult) => void, onCancel: () => void, onRender?: () => void);
18
+ /** Shown while the key is being checked against the live API. */
19
+ setBusy(busy: boolean): void;
20
+ /** Keeps the prompt open so a rejected key can be corrected without restarting. */
21
+ setError(message: string): void;
22
+ handleInput(keyData: string): void;
23
+ }
@@ -2,12 +2,14 @@ export declare class NativeRuntimeCompatibilityError extends Error {
2
2
  readonly runtimeVersion: string;
3
3
  readonly nativeVersion: string;
4
4
  readonly workflowArbitrationAvailable: boolean;
5
+ readonly nativeModulePath: string | null;
5
6
  readonly code = "native_runtime_incompatible";
6
7
  readonly retryable = false;
7
- constructor(runtimeVersion: string, nativeVersion: string, workflowArbitrationAvailable: boolean);
8
+ constructor(runtimeVersion: string, nativeVersion: string, workflowArbitrationAvailable: boolean, nativeModulePath?: string | null);
8
9
  }
9
10
  export declare function assertNativeRuntimeCompatibility(input: {
10
11
  runtimeVersion: string;
11
12
  nativeVersion: string;
12
13
  notificationServer: unknown;
14
+ nativeModulePath?: string | null;
13
15
  }): void;
@@ -872,15 +872,6 @@ export declare class AgentSession {
872
872
  get customCommands(): ReadonlyArray<LoadedCustomCommand>;
873
873
  /** Update the MCP prompt commands list. Called when server prompts are (re)loaded. */
874
874
  setMCPPromptCommands(commands: LoadedCustomCommand[]): void;
875
- /**
876
- * Send a prompt to the agent.
877
- * - Handles extension commands (registered via pi.registerCommand) immediately, even during streaming
878
- * - Expands file-based prompt templates by default
879
- * - During streaming, queues via steer() or followUp() based on streamingBehavior option
880
- * - Validates model and API key before sending (when not streaming)
881
- * @throws Error if streaming and no streamingBehavior specified
882
- * @throws Error if no model selected or no API key available (when not streaming)
883
- */
884
875
  prompt(text: string, options?: PromptOptions): Promise<void>;
885
876
  promptCustomMessage<T = unknown>(message: Pick<CustomMessage<T>, "customType" | "content" | "display" | "details" | "attribution">, options?: Pick<PromptOptions, "streamingBehavior" | "toolChoice" | "followUpQueuePolicy" | "onPreflightAccepted" | "onPreflightAcceptCommit">): Promise<void>;
886
877
  /**
@@ -0,0 +1,24 @@
1
+ export interface TypeSafeKeySetupResult {
2
+ provider: string;
3
+ verified: boolean;
4
+ /** Present when verification ran and failed; the key is not stored in that case. */
5
+ error?: string;
6
+ }
7
+ /**
8
+ * Verify a key against the live API before storing it.
9
+ *
10
+ * Storing an unverified key is worse than refusing it: the decision service fails open,
11
+ * so a bad key produces no error anywhere — it silently falls back to the user's own
12
+ * model forever, and the user believes TypeSafe is active.
13
+ */
14
+ export declare function verifyTypeSafeKey(apiKey: string, fetchImpl?: typeof fetch): Promise<string | null>;
15
+ export interface SetTypeSafeKeyOptions {
16
+ apiKey: string;
17
+ /** Skip the live check. Only for offline setup; the key may be wrong. */
18
+ skipVerify?: boolean;
19
+ fetchImpl?: typeof fetch;
20
+ dbPath?: string;
21
+ }
22
+ export declare function setTypeSafeKey(options: SetTypeSafeKeyOptions): Promise<TypeSafeKeySetupResult>;
23
+ export declare function removeTypeSafeKey(dbPath?: string): Promise<void>;
24
+ export declare function formatTypeSafeKeyResult(result: TypeSafeKeySetupResult): string;
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@sayknow-cli/coding-agent",
4
- "version": "0.5.20",
4
+ "version": "0.5.22",
5
5
  "description": "Sayknow-CLI CLI with read, bash, edit, write tools and session management",
6
6
  "homepage": "https://sayknow-cli.com",
7
7
  "author": "jaybeyond",
@@ -54,12 +54,12 @@
54
54
  "@agentclientprotocol/sdk": "1.3.0",
55
55
  "@babel/parser": "^7.29.3",
56
56
  "@mozilla/readability": "^0.6.0",
57
- "@sayknow-cli/stats": "0.5.20",
58
- "@sayknow-cli/agent-core": "0.5.20",
59
- "@sayknow-cli/ai": "0.5.20",
60
- "@sayknow-cli/natives": "0.5.20",
61
- "@sayknow-cli/tui": "0.5.20",
62
- "@sayknow-cli/utils": "0.5.20",
57
+ "@sayknow-cli/stats": "0.5.22",
58
+ "@sayknow-cli/agent-core": "0.5.22",
59
+ "@sayknow-cli/ai": "0.5.22",
60
+ "@sayknow-cli/natives": "0.5.22",
61
+ "@sayknow-cli/tui": "0.5.22",
62
+ "@sayknow-cli/utils": "0.5.22",
63
63
  "@puppeteer/browsers": "^2.13.0",
64
64
  "@types/turndown": "5.0.6",
65
65
  "@xterm/headless": "^6.0.0",
@@ -0,0 +1,172 @@
1
+ #!/usr/bin/env bun
2
+ /**
3
+ * Measure workflow routing against the model the user is logged into.
4
+ *
5
+ * Runs the real `decisions/llm-backend` — same forced tool call, same enum
6
+ * constraint, same model role resolution the session uses — over a fixed set of
7
+ * prompts, and compares it with the deterministic keyword stage it is meant to
8
+ * back up.
9
+ *
10
+ * bun scripts/eval-skill-routing.ts [--repeat N] [--json out.json] [--backend typesafe|llm]
11
+ *
12
+ * `--backend` pins one backend so the two can be compared head to head. Without it the
13
+ * normal resolution order applies (TypeSafe when a key exists, else the logged-in model).
14
+ *
15
+ * The keyword baseline is recomputed here rather than quoted, so the comparison
16
+ * can never drift away from what `skill-keywords.ts` currently contains.
17
+ */
18
+ import { ModelRegistry } from "../src/config/model-registry";
19
+ import { resolveRoleSelection } from "../src/config/model-resolver";
20
+ import { Settings } from "../src/config/settings";
21
+ import { createDecisionService, createLlmDecisionBackend, createTypeSafeDecisionBackend } from "../src/decisions";
22
+ import { createSemanticSkillRouter } from "../src/decisions/skill-routing";
23
+ import { detectPrimarySkillKeyword } from "../src/hooks/skill-state";
24
+ import { discoverAuthStorage } from "../src/sdk";
25
+
26
+ type Expected = "deep-interview" | "ralplan" | "ultragoal" | "team" | null;
27
+ interface Case {
28
+ prompt: string;
29
+ expect: Expected;
30
+ lang: "ko" | "en";
31
+ }
32
+
33
+ const CASES: Case[] = [
34
+ { prompt: "요구사항이 아직 흐릿한데 나한테 질문해서 스펙을 뽑아줘", expect: "deep-interview", lang: "ko" },
35
+ { prompt: "뭘 만들지 정리가 안 됐어. 인터뷰하듯 파고들어줘", expect: "deep-interview", lang: "ko" },
36
+ { prompt: "추측하지 말고 모르는 건 다 물어봐", expect: "deep-interview", lang: "ko" },
37
+ { prompt: "Ask me questions until the requirements are actually clear", expect: "deep-interview", lang: "en" },
38
+ { prompt: "don't assume anything, dig into what I actually need", expect: "deep-interview", lang: "en" },
39
+ { prompt: "이거 아키텍처 리스크 커. 실행 전에 합의된 계획부터 세워줘", expect: "ralplan", lang: "ko" },
40
+ { prompt: "여러 안 비교해서 검토받을 계획서 만들어줘", expect: "ralplan", lang: "ko" },
41
+ { prompt: "Draft a deliberate plan and stop for my approval before touching code", expect: "ralplan", lang: "en" },
42
+ { prompt: "consensus plan for the migration", expect: "ralplan", lang: "en" },
43
+ { prompt: "이 목표 끝까지 추적해줘. 중간에 잊지 말고", expect: "ultragoal", lang: "ko" },
44
+ { prompt: "장기 목표로 등록해두고 진행상황 계속 관리해", expect: "ultragoal", lang: "ko" },
45
+ { prompt: "Track this objective until every deliverable is verified", expect: "ultragoal", lang: "en" },
46
+ { prompt: "ultragoal this and keep the ledger updated", expect: "ultragoal", lang: "en" },
47
+ { prompt: "작업 크니까 워커 여러 개로 나눠서 병렬로 돌려줘", expect: "team", lang: "ko" },
48
+ { prompt: "팀 구성해서 각자 파트 맡아 진행하게 해", expect: "team", lang: "ko" },
49
+ { prompt: "Spin up coordinated workers for these three slices", expect: "team", lang: "en" },
50
+ { prompt: "coordinated team run on the backlog", expect: "team", lang: "en" },
51
+ { prompt: "이 테스트 왜 깨지는지 봐줘", expect: null, lang: "ko" },
52
+ { prompt: "README 오타 하나 고쳐", expect: null, lang: "ko" },
53
+ { prompt: "이 함수 뭐하는 건지 설명해줘", expect: null, lang: "ko" },
54
+ { prompt: "우리 서비스에 이 모델 붙이면 뭐가 좋아?", expect: null, lang: "ko" },
55
+ { prompt: "fix the failing lint rule in src/utils.ts", expect: null, lang: "en" },
56
+ { prompt: "what does this regex do?", expect: null, lang: "en" },
57
+ ];
58
+
59
+ function pct(hit: number, total: number): string {
60
+ return total === 0 ? "n/a" : `${hit}/${total} (${((hit / total) * 100).toFixed(0)}%)`;
61
+ }
62
+
63
+ async function main(): Promise<void> {
64
+ const repeat = Math.max(1, Number(Bun.argv[Bun.argv.indexOf("--repeat") + 1]) || 1);
65
+ const jsonIndex = Bun.argv.indexOf("--json");
66
+ const jsonPath = jsonIndex > 0 ? Bun.argv[jsonIndex + 1] : undefined;
67
+
68
+ const settings = await Settings.init();
69
+ const registry = new ModelRegistry(await discoverAuthStorage());
70
+ await registry.refresh();
71
+ registry.applyConfiguredModelBindings(settings);
72
+ const backendIndex = Bun.argv.indexOf("--backend");
73
+ const pinned = backendIndex > 0 ? Bun.argv[backendIndex + 1] : undefined;
74
+ const model = resolveRoleSelection(["smol", "default"], settings, registry.getAvailable(), registry)?.model;
75
+ if (!model && pinned !== "typesafe") throw new Error("no model available — log in first");
76
+
77
+ // `--model provider/id` pins one concrete model so candidates can be compared on
78
+ // measured accuracy and latency instead of on guesses about what "small" means.
79
+ const modelIndex = Bun.argv.indexOf("--model");
80
+ const wanted = modelIndex > 0 ? Bun.argv[modelIndex + 1] : undefined;
81
+ const forced = wanted
82
+ ? registry.getAvailable().find(m => `${m.provider}/${m.id}` === wanted || m.id === wanted)
83
+ : undefined;
84
+ if (wanted && !forced) throw new Error(`model not available: ${wanted}`);
85
+
86
+ const backends =
87
+ pinned === "typesafe"
88
+ ? [createTypeSafeDecisionBackend({ registry })]
89
+ : pinned === "llm" || forced
90
+ ? [createLlmDecisionBackend({ registry, settings, model: forced })]
91
+ : undefined;
92
+ // The llm backend now selects its own small model, so the script cannot label the run
93
+ // from role resolution — doing so reported opus while a 4B model actually answered.
94
+ const label =
95
+ pinned === "typesafe" ? "typesafe/jev" : pinned === "llm" ? "llm/auto-small" : `${model?.provider}/${model?.id}`;
96
+ console.log(`backend: ${pinned ?? "auto"} model: ${label} repeat: ${repeat}\n`);
97
+
98
+ const route = createSemanticSkillRouter(
99
+ createDecisionService({ registry, settings, enabled: true, timeoutMs: 30_000, backends }),
100
+ );
101
+
102
+ const rows: Array<{ case: Case; keyword: Expected; semantic: Expected; ms: number }> = [];
103
+ for (const testCase of CASES) {
104
+ for (let run = 0; run < repeat; run++) {
105
+ const keyword = (detectPrimarySkillKeyword(testCase.prompt)?.skill ?? null) as Expected;
106
+ const started = Date.now();
107
+ const semantic = (await route(testCase.prompt)) as Expected;
108
+ rows.push({ case: testCase, keyword, semantic, ms: Date.now() - started });
109
+ const hybrid = keyword ?? semantic;
110
+ const mark = hybrid === testCase.expect ? "OK " : "MISS";
111
+ console.log(
112
+ `${mark} [${testCase.lang}] want=${testCase.expect ?? "none"} kw=${keyword ?? "-"} sem=${semantic ?? "none"} ${Date.now() - started}ms :: ${testCase.prompt.slice(0, 44)}`,
113
+ );
114
+ }
115
+ }
116
+
117
+ const score = (pick: (row: (typeof rows)[number]) => Expected, filter: (row: (typeof rows)[number]) => boolean) => {
118
+ const subset = rows.filter(filter);
119
+ return [subset.filter(row => pick(row) === row.case.expect).length, subset.length] as const;
120
+ };
121
+ const positives = (row: (typeof rows)[number]) => row.case.expect !== null;
122
+ const negatives = (row: (typeof rows)[number]) => row.case.expect === null;
123
+ const ko = (row: (typeof rows)[number]) => positives(row) && row.case.lang === "ko";
124
+ const en = (row: (typeof rows)[number]) => positives(row) && row.case.lang === "en";
125
+ const hybrid = (row: (typeof rows)[number]) => row.keyword ?? row.semantic;
126
+
127
+ console.log("\n=== stage comparison ===");
128
+ for (const [label, pick] of [
129
+ ["keyword only (today)", (row: (typeof rows)[number]) => row.keyword],
130
+ ["semantic only", (row: (typeof rows)[number]) => row.semantic],
131
+ ["hybrid (shipped)", hybrid],
132
+ ] as const) {
133
+ console.log(
134
+ `${label.padEnd(22)} all ${pct(...score(pick, () => true))} ko ${pct(...score(pick, ko))} en ${pct(...score(pick, en))} clean-negatives ${pct(...score(pick, negatives))}`,
135
+ );
136
+ }
137
+
138
+ const latencies = rows.map(row => row.ms).sort((a, b) => a - b);
139
+ console.log(
140
+ `\nlatency p50 ${latencies[Math.floor(latencies.length / 2)]}ms p95 ${latencies[Math.max(0, Math.ceil(latencies.length * 0.95) - 1)]}ms max ${latencies.at(-1)}ms`,
141
+ );
142
+
143
+ const misses = rows.filter(row => hybrid(row) !== row.case.expect);
144
+ if (misses.length > 0) {
145
+ console.log("\n=== hybrid misses ===");
146
+ for (const row of misses)
147
+ console.log(
148
+ ` [${row.case.lang}] want=${row.case.expect ?? "none"} got=${hybrid(row) ?? "none"} :: ${row.case.prompt}`,
149
+ );
150
+ }
151
+
152
+ if (jsonPath) {
153
+ await Bun.write(
154
+ jsonPath,
155
+ JSON.stringify(
156
+ {
157
+ // Must be the backend that actually answered, not the chat model that was
158
+ // resolved for the fallback path — a mislabelled run poisons later comparisons.
159
+ model: label,
160
+ backend: pinned ?? "auto",
161
+ repeat,
162
+ rows: rows.map(r => ({ ...r.case, ...r, case: undefined })),
163
+ },
164
+ null,
165
+ 2,
166
+ ),
167
+ );
168
+ console.log(`\nwrote ${jsonPath}`);
169
+ }
170
+ }
171
+
172
+ await main();