@sayknow-cli/coding-agent 0.5.21 → 0.5.23
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +17 -0
- package/dist/types/cli/setup-cli.d.ts +15 -1
- package/dist/types/commands/setup.d.ts +6 -0
- package/dist/types/config/settings-schema.d.ts +9 -0
- package/dist/types/decisions/index.d.ts +18 -0
- package/dist/types/decisions/llm-backend.d.ts +51 -0
- package/dist/types/decisions/skill-routing.d.ts +10 -0
- package/dist/types/decisions/types.d.ts +91 -0
- package/dist/types/decisions/typesafe-backend.d.ts +15 -0
- package/dist/types/hooks/skill-state.d.ts +6 -0
- package/dist/types/modes/components/provider-onboarding-selector.d.ts +1 -1
- package/dist/types/modes/components/typesafe-key-prompt.d.ts +23 -0
- package/dist/types/modes/controllers/selector-controller.d.ts +9 -0
- package/dist/types/modes/interactive-mode.d.ts +1 -0
- package/dist/types/modes/types.d.ts +2 -0
- package/dist/types/sdk/bus/native-runtime-compatibility.d.ts +3 -1
- package/dist/types/session/agent-session.d.ts +0 -9
- package/dist/types/setup/decision-provider.d.ts +24 -0
- package/dist/types/setup/model-onboarding-guidance.d.ts +5 -0
- package/package.json +7 -7
- package/scripts/eval-skill-routing.ts +173 -0
- package/src/cli/setup-cli.ts +53 -1
- package/src/commands/setup.ts +5 -0
- package/src/config/settings-schema.ts +12 -0
- package/src/decisions/index.ts +84 -0
- package/src/decisions/llm-backend.ts +356 -0
- package/src/decisions/skill-routing.ts +123 -0
- package/src/decisions/types.ts +119 -0
- package/src/decisions/typesafe-backend.ts +168 -0
- package/src/hooks/skill-keywords.ts +56 -0
- package/src/hooks/skill-state.ts +18 -2
- package/src/modes/components/provider-onboarding-selector.ts +13 -1
- package/src/modes/components/typesafe-key-prompt.ts +108 -0
- package/src/modes/controllers/selector-controller.ts +44 -0
- package/src/modes/interactive-mode.ts +4 -0
- package/src/modes/types.ts +2 -0
- package/src/prompts/agents/architect.md +1 -1
- package/src/prompts/agents/critic.md +1 -1
- package/src/prompts/agents/planner.md +1 -1
- package/src/sdk/bus/native-runtime-compatibility.ts +30 -3
- package/src/session/agent-session.ts +135 -2
- package/src/setup/decision-provider.ts +94 -0
- package/src/setup/model-onboarding-guidance.ts +7 -1
- package/src/slash-commands/builtin-registry.ts +18 -1
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,23 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [0.5.23] - 2026-09-20
|
|
6
|
+
|
|
7
|
+
## [0.5.22] - 2026-09-19
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
|
|
11
|
+
- TypeSafe (hosted System One `jev`) as a typed-decision backend. Add the key with `TYPESAFE_API_KEY=<key> skc setup typesafe`; it is verified against the live API before being stored, because the decision service fails open and an unverified bad key would be swallowed silently forever. Removing it (`--remove`) falls back to your logged-in model. The key is read from the environment, never a flag — this repo already refuses raw `--api-key` values because they leak into shell history and the process list.
|
|
12
|
+
- Decision backends now resolve in order: TypeSafe when a key exists (the only backend that returns calibrated probabilities), otherwise the model you are already logged into. TypeSafe is deliberately **not** registered in the chat-provider registry: it has no stream, no messages and no text output, so giving it a `Model` shape would put a non-chat endpoint in the model picker.
|
|
13
|
+
- Typed decisions (`decisions.enabled`, default off): a small, reusable service that asks the model you are already logged into a typed question and gets back a value your code can branch on. Type safety comes from a forced tool call with enum-constrained fields, so an option outside the declared set cannot reach the caller. No extra vendor, no extra key.
|
|
14
|
+
- Workflow-skill routing now has a semantic second stage. The literal keyword table stays first and free; the model is consulted only when it matches nothing. Measured over 23 prompts × 3 runs against `claude-opus-5`: keyword-only **0/27 in Korean** (43% overall), hybrid **69/69 (100%)** with zero false activations on the 18 negative cases, p50 1.46s. Reproduce with `bun scripts/eval-skill-routing.ts --repeat 3`.
|
|
15
|
+
|
|
16
|
+
### Fixed
|
|
17
|
+
|
|
18
|
+
- `NativeRuntimeCompatibilityError` now lists only the causes that actually fired and names the file the process loaded `@sayknow-cli/natives` from. A pure version mismatch used to read "required workflow arbitration methods are **available**", which described a healthy runtime while refusing to start every session and extension.
|
|
19
|
+
- `dev:link` / `dev:doctor` fail on (and `dev:link` removes) nested installs under `packages/<pkg>/node_modules` that shadow a workspace package, and verify that `@sayknow-cli/natives` resolves to the same version as the `coding-agent` runtime that loads it. A published `@sayknow-cli/natives` copy left inside `packages/coding-agent/node_modules` wins resolution over the workspace link and breaks every SDK session.
|
|
20
|
+
- `dev:doctor` accepts this checkout's bun-linked `bin/skc.js` wrapper — the one `install:dev` itself creates — instead of reporting it as drift, but only when the wrapper is byte-identical to the expected workspace wrapper and `@sayknow-cli/coding-agent/cli` resolves back to this checkout's `src/cli.ts`.
|
|
21
|
+
|
|
5
22
|
## [0.5.21] - 2026-09-17
|
|
6
23
|
|
|
7
24
|
### Changed
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
*/
|
|
6
6
|
import { AuthStorage, SqliteAuthCredentialStore } from "@sayknow-cli/ai";
|
|
7
7
|
import { runExternalCredentialAutoImport } from "../setup/credential-auto-import";
|
|
8
|
-
export type SetupComponent = "claude" | "codex" | "credentials" | "defaults" | "hermes" | "hooks" | "provider" | "python" | "stt" | "ui-skills";
|
|
8
|
+
export type SetupComponent = "claude" | "codex" | "credentials" | "defaults" | "hermes" | "hooks" | "provider" | "python" | "stt" | "typesafe" | "ui-skills";
|
|
9
9
|
export interface SetupCommandArgs {
|
|
10
10
|
component: SetupComponent;
|
|
11
11
|
flags: {
|
|
@@ -37,6 +37,8 @@ export interface SetupCommandArgs {
|
|
|
37
37
|
yes?: boolean;
|
|
38
38
|
dryRun?: boolean;
|
|
39
39
|
keychain?: boolean;
|
|
40
|
+
skipVerify?: boolean;
|
|
41
|
+
remove?: boolean;
|
|
40
42
|
};
|
|
41
43
|
}
|
|
42
44
|
/**
|
|
@@ -71,3 +73,15 @@ export declare function handleCredentialsSetup(flags: {
|
|
|
71
73
|
* Print setup command help.
|
|
72
74
|
*/
|
|
73
75
|
export declare function printSetupHelp(): void;
|
|
76
|
+
/**
|
|
77
|
+
* `skc setup typesafe` — enable the hosted System One model for typed decisions.
|
|
78
|
+
*
|
|
79
|
+
* The key is read from `TYPESAFE_API_KEY`, never from a flag: this repo already refuses
|
|
80
|
+
* raw `--api-key` values because they land in shell history and in the process list of
|
|
81
|
+
* every user on the machine. Same rule applies here.
|
|
82
|
+
*
|
|
83
|
+
* Verified against the live API before storing. The decision service fails open, so an
|
|
84
|
+
* unverified bad key would be swallowed forever — the user would believe TypeSafe was
|
|
85
|
+
* active while every decision quietly came from their own model.
|
|
86
|
+
*/
|
|
87
|
+
export declare function handleTypeSafeSetup(flags: SetupCommandArgs["flags"]): Promise<void>;
|
|
@@ -100,6 +100,12 @@ export default class Setup extends Command {
|
|
|
100
100
|
"dry-run": import("@sayknow-cli/utils/cli").FlagDescriptor<"boolean"> & {
|
|
101
101
|
description: string;
|
|
102
102
|
};
|
|
103
|
+
"skip-verify": import("@sayknow-cli/utils/cli").FlagDescriptor<"boolean"> & {
|
|
104
|
+
description: string;
|
|
105
|
+
};
|
|
106
|
+
remove: import("@sayknow-cli/utils/cli").FlagDescriptor<"boolean"> & {
|
|
107
|
+
description: string;
|
|
108
|
+
};
|
|
103
109
|
};
|
|
104
110
|
run(): Promise<void>;
|
|
105
111
|
}
|
|
@@ -2412,6 +2412,15 @@ export declare const SETTINGS_SCHEMA: {
|
|
|
2412
2412
|
readonly type: "number";
|
|
2413
2413
|
readonly default: 16000;
|
|
2414
2414
|
};
|
|
2415
|
+
readonly "decisions.enabled": {
|
|
2416
|
+
readonly type: "boolean";
|
|
2417
|
+
readonly default: false;
|
|
2418
|
+
readonly ui: {
|
|
2419
|
+
readonly tab: "context";
|
|
2420
|
+
readonly label: "Typed decisions";
|
|
2421
|
+
readonly description: "Add a model-backed second stage to workflow routing. The keyword table already runs on every turn and costs nothing; this handles the phrasings it cannot express, which is most wording that is not a literal match. Costs one small model call, and only on turns the keyword table did not already answer. Any failure falls back to keyword-only behaviour, but a successful answer can also select a different workflow than the deep-interview ambiguity detector would have.";
|
|
2422
|
+
};
|
|
2423
|
+
};
|
|
2415
2424
|
readonly "ttsr.enabled": {
|
|
2416
2425
|
readonly type: "boolean";
|
|
2417
2426
|
readonly default: true;
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
import { type LlmBackendDeps } from "./llm-backend";
|
|
2
|
+
import type { DecisionBackend, DecisionRequest, DecisionResult } from "./types";
|
|
3
|
+
export { createLlmDecisionBackend } from "./llm-backend";
|
|
4
|
+
export * from "./types";
|
|
5
|
+
export { createTypeSafeDecisionBackend, TYPESAFE_PROVIDER } from "./typesafe-backend";
|
|
6
|
+
export interface DecisionServiceOptions extends LlmBackendDeps {
|
|
7
|
+
/** Off by default; callers opt in per feature. */
|
|
8
|
+
enabled?: boolean;
|
|
9
|
+
timeoutMs?: number;
|
|
10
|
+
/** Injection point for tests and for the self-hosted/hosted backends. */
|
|
11
|
+
backends?: DecisionBackend[];
|
|
12
|
+
}
|
|
13
|
+
export interface DecisionService {
|
|
14
|
+
readonly enabled: boolean;
|
|
15
|
+
/** Resolves null when disabled, unavailable, timed out, or the model misbehaved. */
|
|
16
|
+
decide(request: DecisionRequest): Promise<DecisionResult | null>;
|
|
17
|
+
}
|
|
18
|
+
export declare function createDecisionService(options: DecisionServiceOptions): DecisionService;
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Decision backend that runs on the model the user is already logged into.
|
|
3
|
+
*
|
|
4
|
+
* No extra API key, no extra vendor, no data leaving the providers the user already
|
|
5
|
+
* trusts. The type safety comes from a **forced tool call with enum-constrained
|
|
6
|
+
* properties**: the provider itself rejects any value outside the declared set, so a
|
|
7
|
+
* malformed or hallucinated option cannot reach our code — the same guarantee the
|
|
8
|
+
* hosted System One model gives, enforced one layer up.
|
|
9
|
+
*
|
|
10
|
+
* Two deliberate omissions:
|
|
11
|
+
*
|
|
12
|
+
* 1. We never ask the model to emit probabilities. Measured elsewhere on this exact
|
|
13
|
+
* task shape, writing probabilities collapses accuracy (~0.35 vs ~0.90 for picking
|
|
14
|
+
* a constrained option), and the numbers are not calibrated anyway. Answers from
|
|
15
|
+
* this backend carry `calibrated: false` and no `probabilities` map.
|
|
16
|
+
* 2. Every question goes in **one** call. Splitting them multiplies cost and latency
|
|
17
|
+
* while the enum constraint already keeps each field independent.
|
|
18
|
+
*/
|
|
19
|
+
import { type Api, type Model } from "@sayknow-cli/ai";
|
|
20
|
+
import type { ModelRegistry } from "../config/model-registry";
|
|
21
|
+
import type { Settings } from "../config/settings";
|
|
22
|
+
import { type DecisionBackend } from "./types";
|
|
23
|
+
export interface LlmBackendDeps {
|
|
24
|
+
/**
|
|
25
|
+
* Refuse to spend more than this per million input tokens on a decision.
|
|
26
|
+
*
|
|
27
|
+
* The whole premise of a typed-decision service is judgment cheap enough to put in
|
|
28
|
+
* places you could not previously afford it. Routing a prompt through a frontier
|
|
29
|
+
* model inverts that: the deterministic path it replaces costs effectively nothing
|
|
30
|
+
* (the routing rules already sit in the cached system prompt), so a decision call on
|
|
31
|
+
* an expensive model is a pure cost *increase* for a few points of accuracy.
|
|
32
|
+
*
|
|
33
|
+
* Measured: a routing decision on claude-opus-5 costs ~$0.0063 and 1.46s; the same
|
|
34
|
+
* decision on the hosted System One model costs ~$0.000018 and 0.31s.
|
|
35
|
+
*
|
|
36
|
+
* Above the cap this backend declines, which leaves routing to the system prompt —
|
|
37
|
+
* exactly the behaviour before typed decisions existed. Configure a `smol` role with
|
|
38
|
+
* a cheap model to turn it back on.
|
|
39
|
+
*/
|
|
40
|
+
maxInputCostPerMTok?: number;
|
|
41
|
+
/** Injected in tests to make the local-runtime probe deterministic. */
|
|
42
|
+
fetchImpl?: typeof fetch;
|
|
43
|
+
registry: ModelRegistry;
|
|
44
|
+
settings: Settings;
|
|
45
|
+
sessionId?: string;
|
|
46
|
+
/** Overrides role resolution; used by callers that already picked a model. */
|
|
47
|
+
model?: Model<Api>;
|
|
48
|
+
}
|
|
49
|
+
/** Reset between tests; also lets a caller force a fresh probe after starting a runtime. */
|
|
50
|
+
export declare function clearLocalRuntimeLivenessCache(): void;
|
|
51
|
+
export declare function createLlmDecisionBackend(deps: LlmBackendDeps): DecisionBackend;
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import { type CanonicalSkcWorkflowSkill } from "../skill-state/active-state";
|
|
2
|
+
import type { DecisionService } from "./index";
|
|
3
|
+
/** Exported so tests can assert the contract the model is actually given. */
|
|
4
|
+
export declare function buildRoutingCriteria(): Record<string, string>;
|
|
5
|
+
export type SkillRouter = (text: string) => Promise<CanonicalSkcWorkflowSkill | null>;
|
|
6
|
+
/**
|
|
7
|
+
* Build the semantic router. Returns null-resolving function when the service is
|
|
8
|
+
* disabled so the caller keeps its existing behaviour with no branching.
|
|
9
|
+
*/
|
|
10
|
+
export declare function createSemanticSkillRouter(service: DecisionService): SkillRouter;
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Typed decisions — "Jev-shaped" structured judgments for code to branch on.
|
|
3
|
+
*
|
|
4
|
+
* The request/response shape follows TypeSafe's System One API so a backend can be
|
|
5
|
+
* swapped without touching call sites: the hosted `jev` model, a self-hosted OpenJev
|
|
6
|
+
* daemon, or — the default — the model the user is already logged into.
|
|
7
|
+
*
|
|
8
|
+
* What this is NOT: a probability oracle. Only the hosted model returns calibrated
|
|
9
|
+
* probabilities. Backends that constrain an ordinary LLM return an ordinal value and
|
|
10
|
+
* report `calibrated: false`; treat those numbers as a ranking, never as P(correct).
|
|
11
|
+
*/
|
|
12
|
+
/** Pick exactly one option from a closed set. */
|
|
13
|
+
export interface ChoiceQuestion {
|
|
14
|
+
type: "choice";
|
|
15
|
+
instructions: string;
|
|
16
|
+
/** option id -> what that option means. At least two. */
|
|
17
|
+
criteria: Record<string, string>;
|
|
18
|
+
}
|
|
19
|
+
/** Rate the state against ordered levels. Level 0 is the lowest. */
|
|
20
|
+
export interface ScoreQuestion {
|
|
21
|
+
type: "score";
|
|
22
|
+
instructions: string;
|
|
23
|
+
/** Ordered level descriptions, lowest first. At least two. */
|
|
24
|
+
criteria: string[];
|
|
25
|
+
}
|
|
26
|
+
/** Is this statement true of the state? */
|
|
27
|
+
export interface NoulQuestion {
|
|
28
|
+
type: "noul";
|
|
29
|
+
instructions: string;
|
|
30
|
+
}
|
|
31
|
+
export type Question = ChoiceQuestion | ScoreQuestion | NoulQuestion;
|
|
32
|
+
export interface ChoiceAnswer {
|
|
33
|
+
type: "choice";
|
|
34
|
+
/** The selected option id. Always one of the declared `criteria` keys. */
|
|
35
|
+
choice: string;
|
|
36
|
+
/** Present only when the backend exposes a distribution. */
|
|
37
|
+
probabilities?: Record<string, number>;
|
|
38
|
+
/**
|
|
39
|
+
* How certain the backend is, 0..1. Only meaningful when the result reports
|
|
40
|
+
* `calibrated: true` — that is the difference between a number you can threshold on
|
|
41
|
+
* and a number that merely ranks. Absent when the backend cannot supply one.
|
|
42
|
+
*/
|
|
43
|
+
confidence?: number;
|
|
44
|
+
}
|
|
45
|
+
export interface ScoreAnswer {
|
|
46
|
+
type: "score";
|
|
47
|
+
/** Level index. Fractional only when the backend returns a distribution. */
|
|
48
|
+
score: number;
|
|
49
|
+
/** Selected level index. */
|
|
50
|
+
level: number;
|
|
51
|
+
legend: Record<string, string>;
|
|
52
|
+
probabilities?: Record<string, number>;
|
|
53
|
+
/** See {@link ChoiceAnswer.confidence}. */
|
|
54
|
+
confidence?: number;
|
|
55
|
+
}
|
|
56
|
+
export interface NoulAnswer {
|
|
57
|
+
type: "noul";
|
|
58
|
+
/** 0 (no) .. 1 (yes). Ordinal unless `calibrated` is true. */
|
|
59
|
+
noul: number;
|
|
60
|
+
}
|
|
61
|
+
export type Answer = ChoiceAnswer | ScoreAnswer | NoulAnswer;
|
|
62
|
+
export interface DecisionResult {
|
|
63
|
+
answers: Record<string, Answer>;
|
|
64
|
+
/** Which backend answered, for logging and A/B comparison. */
|
|
65
|
+
backend: string;
|
|
66
|
+
/** Model identifier the backend used. */
|
|
67
|
+
model: string;
|
|
68
|
+
/**
|
|
69
|
+
* False means the numbers are ordinal rankings, not probabilities.
|
|
70
|
+
* Only the hosted System One model reports true.
|
|
71
|
+
*/
|
|
72
|
+
calibrated: boolean;
|
|
73
|
+
durationMs: number;
|
|
74
|
+
}
|
|
75
|
+
export interface DecisionRequest {
|
|
76
|
+
/** The content to judge. Plain text, or JSON-serialisable structured state. */
|
|
77
|
+
state: string | Record<string, unknown> | unknown[];
|
|
78
|
+
/** Question id -> question. Answers come back under the same ids. */
|
|
79
|
+
questions: Record<string, Question>;
|
|
80
|
+
signal?: AbortSignal;
|
|
81
|
+
}
|
|
82
|
+
export interface DecisionBackend {
|
|
83
|
+
readonly name: string;
|
|
84
|
+
/** Resolves null when the backend is unavailable (no credentials, offline, disabled). */
|
|
85
|
+
decide(request: DecisionRequest): Promise<DecisionResult | null>;
|
|
86
|
+
}
|
|
87
|
+
export declare const MIN_OPTIONS = 2;
|
|
88
|
+
/** Matches OpenJev's letter-slot ceiling so a graph stays portable across backends. */
|
|
89
|
+
export declare const MAX_OPTIONS = 16;
|
|
90
|
+
export declare function validateQuestions(questions: Record<string, Question>): void;
|
|
91
|
+
export declare function stateToText(state: DecisionRequest["state"]): string;
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
import type { ModelRegistry } from "../config/model-registry";
|
|
2
|
+
import { type DecisionBackend } from "./types";
|
|
3
|
+
/** Provider id under which the key is stored and surfaced in the model list. */
|
|
4
|
+
export declare const TYPESAFE_PROVIDER = "typesafe";
|
|
5
|
+
export interface TypeSafeBackendDeps {
|
|
6
|
+
registry: ModelRegistry;
|
|
7
|
+
sessionId?: string;
|
|
8
|
+
/** Override for self-hosted or proxied deployments. */
|
|
9
|
+
baseUrl?: string;
|
|
10
|
+
/** Model id sent in the request body. Named to avoid colliding with the LLM backend's `model`. */
|
|
11
|
+
modelId?: string;
|
|
12
|
+
/** Injected in tests; defaults to global fetch. */
|
|
13
|
+
fetchImpl?: typeof fetch;
|
|
14
|
+
}
|
|
15
|
+
export declare function createTypeSafeDecisionBackend(deps: TypeSafeBackendDeps): DecisionBackend;
|
|
@@ -38,6 +38,12 @@ export interface RecordSkillActivationInput {
|
|
|
38
38
|
turnId?: string;
|
|
39
39
|
nowIso?: string;
|
|
40
40
|
stateDir?: string;
|
|
41
|
+
/**
|
|
42
|
+
* Semantic fallback, consulted only when no keyword matched. Supplying it turns the
|
|
43
|
+
* literal keyword table into a two-stage router; omitting it keeps the historical
|
|
44
|
+
* keyword-only behaviour byte for byte.
|
|
45
|
+
*/
|
|
46
|
+
resolveSkillSemantically?: (text: string) => Promise<SkcWorkflowSkill | null>;
|
|
41
47
|
}
|
|
42
48
|
export interface StopHookInput {
|
|
43
49
|
cwd: string;
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { Container } from "@sayknow-cli/tui";
|
|
2
|
-
export type ProviderOnboardingAction = "custom-provider-wizard" | "oauth-login" | "import-credentials" | "api-guide";
|
|
2
|
+
export type ProviderOnboardingAction = "custom-provider-wizard" | "oauth-login" | "import-credentials" | "api-guide" | "typesafe-key";
|
|
3
3
|
export declare class ProviderOnboardingSelectorComponent extends Container {
|
|
4
4
|
#private;
|
|
5
5
|
constructor(onSelect: (action: ProviderOnboardingAction) => void, onCancel: () => void);
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Key entry for TypeSafe, reachable from the same place models are added.
|
|
3
|
+
*
|
|
4
|
+
* TypeSafe is not a chat model, so it never appears in the model picker — but the place
|
|
5
|
+
* users look when they want to "add a model with a key" is this onboarding list, and
|
|
6
|
+
* making them find a CLI subcommand instead would mean most users never enable it.
|
|
7
|
+
*
|
|
8
|
+
* The key is taken through {@link SecretInput} and consumed once: it is never rendered,
|
|
9
|
+
* never placed in a flag, and never written anywhere but the credential store.
|
|
10
|
+
*/
|
|
11
|
+
import { Container } from "@sayknow-cli/tui";
|
|
12
|
+
export interface TypeSafeKeyPromptResult {
|
|
13
|
+
apiKey: string;
|
|
14
|
+
}
|
|
15
|
+
export declare class TypeSafeKeyPromptComponent extends Container {
|
|
16
|
+
#private;
|
|
17
|
+
constructor(onSubmit: (result: TypeSafeKeyPromptResult) => void, onCancel: () => void, onRender?: () => void);
|
|
18
|
+
/** Shown while the key is being checked against the live API. */
|
|
19
|
+
setBusy(busy: boolean): void;
|
|
20
|
+
/** Keeps the prompt open so a rejected key can be corrected without restarting. */
|
|
21
|
+
setError(message: string): void;
|
|
22
|
+
handleInput(keyData: string): void;
|
|
23
|
+
}
|
|
@@ -81,6 +81,15 @@ export declare class SelectorController {
|
|
|
81
81
|
}): void;
|
|
82
82
|
showCommandPalette(commands: SlashCommand[], actions: CommandPaletteAction[], executeSlashCommand: (name: string) => Promise<void>): void;
|
|
83
83
|
showProviderOnboarding(): void;
|
|
84
|
+
/**
|
|
85
|
+
* Take a TypeSafe key and verify it before storing.
|
|
86
|
+
*
|
|
87
|
+
* Verification is not optional here. The decision service fails open by design, so an
|
|
88
|
+
* unverified bad key produces no error anywhere: decisions silently keep coming from
|
|
89
|
+
* the user's own model while the UI claims TypeSafe is on. Better to keep the prompt
|
|
90
|
+
* open and say the key was rejected.
|
|
91
|
+
*/
|
|
92
|
+
showTypeSafeKeyPrompt(): void;
|
|
84
93
|
showCustomModelPresetWizard(snapshot: ModelProfileConfig): void;
|
|
85
94
|
showCustomProviderWizard(): void;
|
|
86
95
|
showEffortSelector(): void;
|
|
@@ -264,6 +264,7 @@ export declare class InteractiveMode implements InteractiveModeContext {
|
|
|
264
264
|
temporaryOnly?: boolean;
|
|
265
265
|
}): void;
|
|
266
266
|
showEffortSelector(): void;
|
|
267
|
+
showTypeSafeKeyPrompt(): void;
|
|
267
268
|
showProviderOnboarding(): void;
|
|
268
269
|
showPluginSelector(mode?: "install" | "uninstall"): void;
|
|
269
270
|
showUserMessageSelector(): void;
|
|
@@ -280,6 +280,8 @@ export interface InteractiveModeContext {
|
|
|
280
280
|
}): void;
|
|
281
281
|
showEffortSelector(): void;
|
|
282
282
|
showProviderOnboarding(): void;
|
|
283
|
+
/** Open the TypeSafe key prompt (typed decisions; not a chat model). */
|
|
284
|
+
showTypeSafeKeyPrompt(): void;
|
|
283
285
|
showPluginSelector(mode?: "install" | "uninstall"): void;
|
|
284
286
|
showUserMessageSelector(): void;
|
|
285
287
|
showTreeSelector(): void;
|
|
@@ -2,12 +2,14 @@ export declare class NativeRuntimeCompatibilityError extends Error {
|
|
|
2
2
|
readonly runtimeVersion: string;
|
|
3
3
|
readonly nativeVersion: string;
|
|
4
4
|
readonly workflowArbitrationAvailable: boolean;
|
|
5
|
+
readonly nativeModulePath: string | null;
|
|
5
6
|
readonly code = "native_runtime_incompatible";
|
|
6
7
|
readonly retryable = false;
|
|
7
|
-
constructor(runtimeVersion: string, nativeVersion: string, workflowArbitrationAvailable: boolean);
|
|
8
|
+
constructor(runtimeVersion: string, nativeVersion: string, workflowArbitrationAvailable: boolean, nativeModulePath?: string | null);
|
|
8
9
|
}
|
|
9
10
|
export declare function assertNativeRuntimeCompatibility(input: {
|
|
10
11
|
runtimeVersion: string;
|
|
11
12
|
nativeVersion: string;
|
|
12
13
|
notificationServer: unknown;
|
|
14
|
+
nativeModulePath?: string | null;
|
|
13
15
|
}): void;
|
|
@@ -872,15 +872,6 @@ export declare class AgentSession {
|
|
|
872
872
|
get customCommands(): ReadonlyArray<LoadedCustomCommand>;
|
|
873
873
|
/** Update the MCP prompt commands list. Called when server prompts are (re)loaded. */
|
|
874
874
|
setMCPPromptCommands(commands: LoadedCustomCommand[]): void;
|
|
875
|
-
/**
|
|
876
|
-
* Send a prompt to the agent.
|
|
877
|
-
* - Handles extension commands (registered via pi.registerCommand) immediately, even during streaming
|
|
878
|
-
* - Expands file-based prompt templates by default
|
|
879
|
-
* - During streaming, queues via steer() or followUp() based on streamingBehavior option
|
|
880
|
-
* - Validates model and API key before sending (when not streaming)
|
|
881
|
-
* @throws Error if streaming and no streamingBehavior specified
|
|
882
|
-
* @throws Error if no model selected or no API key available (when not streaming)
|
|
883
|
-
*/
|
|
884
875
|
prompt(text: string, options?: PromptOptions): Promise<void>;
|
|
885
876
|
promptCustomMessage<T = unknown>(message: Pick<CustomMessage<T>, "customType" | "content" | "display" | "details" | "attribution">, options?: Pick<PromptOptions, "streamingBehavior" | "toolChoice" | "followUpQueuePolicy" | "onPreflightAccepted" | "onPreflightAcceptCommit">): Promise<void>;
|
|
886
877
|
/**
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
export interface TypeSafeKeySetupResult {
|
|
2
|
+
provider: string;
|
|
3
|
+
verified: boolean;
|
|
4
|
+
/** Present when verification ran and failed; the key is not stored in that case. */
|
|
5
|
+
error?: string;
|
|
6
|
+
}
|
|
7
|
+
/**
|
|
8
|
+
* Verify a key against the live API before storing it.
|
|
9
|
+
*
|
|
10
|
+
* Storing an unverified key is worse than refusing it: the decision service fails open,
|
|
11
|
+
* so a bad key produces no error anywhere — it silently falls back to the user's own
|
|
12
|
+
* model forever, and the user believes TypeSafe is active.
|
|
13
|
+
*/
|
|
14
|
+
export declare function verifyTypeSafeKey(apiKey: string, fetchImpl?: typeof fetch): Promise<string | null>;
|
|
15
|
+
export interface SetTypeSafeKeyOptions {
|
|
16
|
+
apiKey: string;
|
|
17
|
+
/** Skip the live check. Only for offline setup; the key may be wrong. */
|
|
18
|
+
skipVerify?: boolean;
|
|
19
|
+
fetchImpl?: typeof fetch;
|
|
20
|
+
dbPath?: string;
|
|
21
|
+
}
|
|
22
|
+
export declare function setTypeSafeKey(options: SetTypeSafeKeyOptions): Promise<TypeSafeKeySetupResult>;
|
|
23
|
+
export declare function removeTypeSafeKey(dbPath?: string): Promise<void>;
|
|
24
|
+
export declare function formatTypeSafeKeyResult(result: TypeSafeKeySetupResult): string;
|
|
@@ -2,6 +2,11 @@ export declare const MODEL_ONBOARDING_API_PROVIDER_COMMAND = "/provider add --co
|
|
|
2
2
|
export declare const MODEL_ONBOARDING_PROVIDER_PRESET_COMMAND = "/provider add --preset <minimax|minimax-cn|glm>";
|
|
3
3
|
export declare const MODEL_ONBOARDING_SETUP_COMMAND = "skc setup provider";
|
|
4
4
|
export declare const MODEL_ONBOARDING_OAUTH_COMMAND = "/provider login [provider-id] or /login [provider-id]";
|
|
5
|
+
/**
|
|
6
|
+
* TypeSafe is not a chat model and never appears in the model list, so the only way a
|
|
7
|
+
* user learns it exists is from the surfaces where they go to add credentials.
|
|
8
|
+
*/
|
|
9
|
+
export declare const MODEL_ONBOARDING_TYPESAFE_COMMAND = "/provider typesafe";
|
|
5
10
|
export declare function formatModelOnboardingGuidance(): string;
|
|
6
11
|
export declare function formatModelOnboardingInlineHint(): string;
|
|
7
12
|
export declare function formatNoModelOnboardingError(): string;
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@sayknow-cli/coding-agent",
|
|
4
|
-
"version": "0.5.
|
|
4
|
+
"version": "0.5.23",
|
|
5
5
|
"description": "Sayknow-CLI CLI with read, bash, edit, write tools and session management",
|
|
6
6
|
"homepage": "https://sayknow-cli.com",
|
|
7
7
|
"author": "jaybeyond",
|
|
@@ -54,12 +54,12 @@
|
|
|
54
54
|
"@agentclientprotocol/sdk": "1.3.0",
|
|
55
55
|
"@babel/parser": "^7.29.3",
|
|
56
56
|
"@mozilla/readability": "^0.6.0",
|
|
57
|
-
"@sayknow-cli/stats": "0.5.
|
|
58
|
-
"@sayknow-cli/agent-core": "0.5.
|
|
59
|
-
"@sayknow-cli/ai": "0.5.
|
|
60
|
-
"@sayknow-cli/natives": "0.5.
|
|
61
|
-
"@sayknow-cli/tui": "0.5.
|
|
62
|
-
"@sayknow-cli/utils": "0.5.
|
|
57
|
+
"@sayknow-cli/stats": "0.5.23",
|
|
58
|
+
"@sayknow-cli/agent-core": "0.5.23",
|
|
59
|
+
"@sayknow-cli/ai": "0.5.23",
|
|
60
|
+
"@sayknow-cli/natives": "0.5.23",
|
|
61
|
+
"@sayknow-cli/tui": "0.5.23",
|
|
62
|
+
"@sayknow-cli/utils": "0.5.23",
|
|
63
63
|
"@puppeteer/browsers": "^2.13.0",
|
|
64
64
|
"@types/turndown": "5.0.6",
|
|
65
65
|
"@xterm/headless": "^6.0.0",
|
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
#!/usr/bin/env bun
|
|
2
|
+
/**
|
|
3
|
+
* Measure workflow routing against the model the user is logged into.
|
|
4
|
+
*
|
|
5
|
+
* Runs the real `decisions/llm-backend` — same forced tool call, same enum
|
|
6
|
+
* constraint, same model role resolution the session uses — over a fixed set of
|
|
7
|
+
* prompts, and compares it with the deterministic keyword stage it is meant to
|
|
8
|
+
* back up.
|
|
9
|
+
*
|
|
10
|
+
* bun scripts/eval-skill-routing.ts [--repeat N] [--json out.json] [--backend typesafe|llm]
|
|
11
|
+
*
|
|
12
|
+
* `--backend` pins one backend so the two can be compared head to head. Without it the
|
|
13
|
+
* normal resolution order applies (TypeSafe when a key exists, else the logged-in model).
|
|
14
|
+
*
|
|
15
|
+
* The keyword baseline is recomputed here rather than quoted, so the comparison
|
|
16
|
+
* can never drift away from what `skill-keywords.ts` currently contains.
|
|
17
|
+
*/
|
|
18
|
+
import { ModelRegistry } from "../src/config/model-registry";
|
|
19
|
+
import { resolveRoleSelection } from "../src/config/model-resolver";
|
|
20
|
+
import { Settings } from "../src/config/settings";
|
|
21
|
+
import { createDecisionService, createLlmDecisionBackend, createTypeSafeDecisionBackend } from "../src/decisions";
|
|
22
|
+
import { createSemanticSkillRouter } from "../src/decisions/skill-routing";
|
|
23
|
+
import { detectPrimarySkillKeyword } from "../src/hooks/skill-state";
|
|
24
|
+
import { discoverAuthStorage } from "../src/sdk";
|
|
25
|
+
|
|
26
|
+
type Expected = "deep-interview" | "ralplan" | "ultragoal" | "team" | null;
|
|
27
|
+
interface Case {
|
|
28
|
+
prompt: string;
|
|
29
|
+
expect: Expected;
|
|
30
|
+
lang: "ko" | "en";
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
const CASES: Case[] = [
|
|
34
|
+
{ prompt: "요구사항이 아직 흐릿한데 나한테 질문해서 스펙을 뽑아줘", expect: "deep-interview", lang: "ko" },
|
|
35
|
+
{ prompt: "뭘 만들지 정리가 안 됐어. 인터뷰하듯 파고들어줘", expect: "deep-interview", lang: "ko" },
|
|
36
|
+
{ prompt: "추측하지 말고 모르는 건 다 물어봐", expect: "deep-interview", lang: "ko" },
|
|
37
|
+
{ prompt: "Ask me questions until the requirements are actually clear", expect: "deep-interview", lang: "en" },
|
|
38
|
+
{ prompt: "don't assume anything, dig into what I actually need", expect: "deep-interview", lang: "en" },
|
|
39
|
+
{ prompt: "이거 아키텍처 리스크 커. 실행 전에 합의된 계획부터 세워줘", expect: "ralplan", lang: "ko" },
|
|
40
|
+
{ prompt: "여러 안 비교해서 검토받을 계획서 만들어줘", expect: "ralplan", lang: "ko" },
|
|
41
|
+
{ prompt: "Draft a deliberate plan and stop for my approval before touching code", expect: "ralplan", lang: "en" },
|
|
42
|
+
{ prompt: "consensus plan for the migration", expect: "ralplan", lang: "en" },
|
|
43
|
+
{ prompt: "이 목표 끝까지 추적해줘. 중간에 잊지 말고", expect: "ultragoal", lang: "ko" },
|
|
44
|
+
{ prompt: "장기 목표로 등록해두고 진행상황 계속 관리해", expect: "ultragoal", lang: "ko" },
|
|
45
|
+
{ prompt: "Track this objective until every deliverable is verified", expect: "ultragoal", lang: "en" },
|
|
46
|
+
{ prompt: "ultragoal this and keep the ledger updated", expect: "ultragoal", lang: "en" },
|
|
47
|
+
{ prompt: "작업 크니까 워커 여러 개로 나눠서 병렬로 돌려줘", expect: "team", lang: "ko" },
|
|
48
|
+
{ prompt: "팀 구성해서 각자 파트 맡아 진행하게 해", expect: "team", lang: "ko" },
|
|
49
|
+
{ prompt: "Spin up coordinated workers for these three slices", expect: "team", lang: "en" },
|
|
50
|
+
{ prompt: "coordinated team run on the backlog", expect: "team", lang: "en" },
|
|
51
|
+
{ prompt: "이 테스트 왜 깨지는지 봐줘", expect: null, lang: "ko" },
|
|
52
|
+
{ prompt: "README 오타 하나 고쳐", expect: null, lang: "ko" },
|
|
53
|
+
{ prompt: "이 함수 뭐하는 건지 설명해줘", expect: null, lang: "ko" },
|
|
54
|
+
{ prompt: "우리 서비스에 이 모델 붙이면 뭐가 좋아?", expect: null, lang: "ko" },
|
|
55
|
+
{ prompt: "fix the failing lint rule in src/utils.ts", expect: null, lang: "en" },
|
|
56
|
+
{ prompt: "what does this regex do?", expect: null, lang: "en" },
|
|
57
|
+
];
|
|
58
|
+
|
|
59
|
+
function pct(hit: number, total: number): string {
|
|
60
|
+
return total === 0 ? "n/a" : `${hit}/${total} (${((hit / total) * 100).toFixed(0)}%)`;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
async function main(): Promise<void> {
|
|
64
|
+
const repeat = Math.max(1, Number(Bun.argv[Bun.argv.indexOf("--repeat") + 1]) || 1);
|
|
65
|
+
const jsonIndex = Bun.argv.indexOf("--json");
|
|
66
|
+
const jsonPath = jsonIndex > 0 ? Bun.argv[jsonIndex + 1] : undefined;
|
|
67
|
+
|
|
68
|
+
const settings = await Settings.init();
|
|
69
|
+
const registry = new ModelRegistry(await discoverAuthStorage());
|
|
70
|
+
await registry.refresh();
|
|
71
|
+
registry.applyConfiguredModelBindings(settings);
|
|
72
|
+
const backendIndex = Bun.argv.indexOf("--backend");
|
|
73
|
+
const pinned = backendIndex > 0 ? Bun.argv[backendIndex + 1] : undefined;
|
|
74
|
+
const model = resolveRoleSelection(["smol", "default"], settings, registry.getAvailable(), registry)?.model;
|
|
75
|
+
if (!model && pinned !== "typesafe") throw new Error("no model available — log in first");
|
|
76
|
+
|
|
77
|
+
// `--model provider/id` pins one concrete model so candidates can be compared on
|
|
78
|
+
// measured accuracy and latency instead of on guesses about what "small" means.
|
|
79
|
+
const modelIndex = Bun.argv.indexOf("--model");
|
|
80
|
+
const wanted = modelIndex > 0 ? Bun.argv[modelIndex + 1] : undefined;
|
|
81
|
+
const forced = wanted
|
|
82
|
+
? registry.getAvailable().find(m => `${m.provider}/${m.id}` === wanted || m.id === wanted)
|
|
83
|
+
: undefined;
|
|
84
|
+
if (wanted && !forced) throw new Error(`model not available: ${wanted}`);
|
|
85
|
+
|
|
86
|
+
const backends =
|
|
87
|
+
pinned === "typesafe"
|
|
88
|
+
? [createTypeSafeDecisionBackend({ registry })]
|
|
89
|
+
: pinned === "llm" || forced
|
|
90
|
+
? [createLlmDecisionBackend({ registry, settings, model: forced })]
|
|
91
|
+
: undefined;
|
|
92
|
+
// The llm backend now selects its own small model, so the script cannot label the run
|
|
93
|
+
// from role resolution — doing so reported opus while a 4B model actually answered.
|
|
94
|
+
const label =
|
|
95
|
+
pinned === "typesafe" ? "typesafe/jev" : pinned === "llm" ? "llm/auto-small" : `${model?.provider}/${model?.id}`;
|
|
96
|
+
console.log(`backend: ${pinned ?? "auto"} model: ${label} repeat: ${repeat}\n`);
|
|
97
|
+
|
|
98
|
+
const route = createSemanticSkillRouter(
|
|
99
|
+
createDecisionService({ registry, settings, enabled: true, timeoutMs: 30_000, backends }),
|
|
100
|
+
);
|
|
101
|
+
|
|
102
|
+
const rows: Array<{ case: Case; keyword: Expected; semantic: Expected; ms: number }> = [];
|
|
103
|
+
for (const testCase of CASES) {
|
|
104
|
+
for (let run = 0; run < repeat; run++) {
|
|
105
|
+
const keyword = (detectPrimarySkillKeyword(testCase.prompt)?.skill ?? null) as Expected;
|
|
106
|
+
const started = Date.now();
|
|
107
|
+
const semantic = (await route(testCase.prompt)) as Expected;
|
|
108
|
+
rows.push({ case: testCase, keyword, semantic, ms: Date.now() - started });
|
|
109
|
+
const hybrid = keyword ?? semantic;
|
|
110
|
+
const mark = hybrid === testCase.expect ? "OK " : "MISS";
|
|
111
|
+
console.log(
|
|
112
|
+
`${mark} [${testCase.lang}] want=${testCase.expect ?? "none"} kw=${keyword ?? "-"} sem=${semantic ?? "none"} ${Date.now() - started}ms :: ${testCase.prompt.slice(0, 44)}`,
|
|
113
|
+
);
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
const score = (pick: (row: (typeof rows)[number]) => Expected, filter: (row: (typeof rows)[number]) => boolean) => {
|
|
118
|
+
const subset = rows.filter(filter);
|
|
119
|
+
return [subset.filter(row => pick(row) === row.case.expect).length, subset.length] as const;
|
|
120
|
+
};
|
|
121
|
+
const positives = (row: (typeof rows)[number]) => row.case.expect !== null;
|
|
122
|
+
const negatives = (row: (typeof rows)[number]) => row.case.expect === null;
|
|
123
|
+
const ko = (row: (typeof rows)[number]) => positives(row) && row.case.lang === "ko";
|
|
124
|
+
const en = (row: (typeof rows)[number]) => positives(row) && row.case.lang === "en";
|
|
125
|
+
const hybrid = (row: (typeof rows)[number]) => row.keyword ?? row.semantic;
|
|
126
|
+
|
|
127
|
+
console.log("\n=== stage comparison ===");
|
|
128
|
+
for (const [label, pick] of [
|
|
129
|
+
["keyword only (Codex hook)", (row: (typeof rows)[number]) => row.keyword],
|
|
130
|
+
["semantic only (SHIPPED)", (row: (typeof rows)[number]) => row.semantic],
|
|
131
|
+
["keyword+semantic (upper bound)", hybrid],
|
|
132
|
+
] as const) {
|
|
133
|
+
console.log(
|
|
134
|
+
`${label.padEnd(32)} all ${pct(...score(pick, () => true))} ko ${pct(...score(pick, ko))} en ${pct(...score(pick, en))} clean-negatives ${pct(...score(pick, negatives))}`,
|
|
135
|
+
);
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
const latencies = rows.map(row => row.ms).sort((a, b) => a - b);
|
|
139
|
+
console.log(
|
|
140
|
+
`\nlatency p50 ${latencies[Math.floor(latencies.length / 2)]}ms p95 ${latencies[Math.max(0, Math.ceil(latencies.length * 0.95) - 1)]}ms max ${latencies.at(-1)}ms`,
|
|
141
|
+
);
|
|
142
|
+
|
|
143
|
+
// Report against what actually ships in this host, not against the upper bound.
|
|
144
|
+
const misses = rows.filter(row => row.semantic !== row.case.expect);
|
|
145
|
+
if (misses.length > 0) {
|
|
146
|
+
console.log("\n=== misses in shipped configuration (semantic only) ===");
|
|
147
|
+
for (const row of misses)
|
|
148
|
+
console.log(
|
|
149
|
+
` [${row.case.lang}] want=${row.case.expect ?? "none"} got=${row.semantic ?? "none"} :: ${row.case.prompt}`,
|
|
150
|
+
);
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
if (jsonPath) {
|
|
154
|
+
await Bun.write(
|
|
155
|
+
jsonPath,
|
|
156
|
+
JSON.stringify(
|
|
157
|
+
{
|
|
158
|
+
// Must be the backend that actually answered, not the chat model that was
|
|
159
|
+
// resolved for the fallback path — a mislabelled run poisons later comparisons.
|
|
160
|
+
model: label,
|
|
161
|
+
backend: pinned ?? "auto",
|
|
162
|
+
repeat,
|
|
163
|
+
rows: rows.map(r => ({ ...r.case, ...r, case: undefined })),
|
|
164
|
+
},
|
|
165
|
+
null,
|
|
166
|
+
2,
|
|
167
|
+
),
|
|
168
|
+
);
|
|
169
|
+
console.log(`\nwrote ${jsonPath}`);
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
await main();
|