@sayknow-cli/coding-agent 0.5.21 → 0.5.23
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +17 -0
- package/dist/types/cli/setup-cli.d.ts +15 -1
- package/dist/types/commands/setup.d.ts +6 -0
- package/dist/types/config/settings-schema.d.ts +9 -0
- package/dist/types/decisions/index.d.ts +18 -0
- package/dist/types/decisions/llm-backend.d.ts +51 -0
- package/dist/types/decisions/skill-routing.d.ts +10 -0
- package/dist/types/decisions/types.d.ts +91 -0
- package/dist/types/decisions/typesafe-backend.d.ts +15 -0
- package/dist/types/hooks/skill-state.d.ts +6 -0
- package/dist/types/modes/components/provider-onboarding-selector.d.ts +1 -1
- package/dist/types/modes/components/typesafe-key-prompt.d.ts +23 -0
- package/dist/types/modes/controllers/selector-controller.d.ts +9 -0
- package/dist/types/modes/interactive-mode.d.ts +1 -0
- package/dist/types/modes/types.d.ts +2 -0
- package/dist/types/sdk/bus/native-runtime-compatibility.d.ts +3 -1
- package/dist/types/session/agent-session.d.ts +0 -9
- package/dist/types/setup/decision-provider.d.ts +24 -0
- package/dist/types/setup/model-onboarding-guidance.d.ts +5 -0
- package/package.json +7 -7
- package/scripts/eval-skill-routing.ts +173 -0
- package/src/cli/setup-cli.ts +53 -1
- package/src/commands/setup.ts +5 -0
- package/src/config/settings-schema.ts +12 -0
- package/src/decisions/index.ts +84 -0
- package/src/decisions/llm-backend.ts +356 -0
- package/src/decisions/skill-routing.ts +123 -0
- package/src/decisions/types.ts +119 -0
- package/src/decisions/typesafe-backend.ts +168 -0
- package/src/hooks/skill-keywords.ts +56 -0
- package/src/hooks/skill-state.ts +18 -2
- package/src/modes/components/provider-onboarding-selector.ts +13 -1
- package/src/modes/components/typesafe-key-prompt.ts +108 -0
- package/src/modes/controllers/selector-controller.ts +44 -0
- package/src/modes/interactive-mode.ts +4 -0
- package/src/modes/types.ts +2 -0
- package/src/prompts/agents/architect.md +1 -1
- package/src/prompts/agents/critic.md +1 -1
- package/src/prompts/agents/planner.md +1 -1
- package/src/sdk/bus/native-runtime-compatibility.ts +30 -3
- package/src/session/agent-session.ts +135 -2
- package/src/setup/decision-provider.ts +94 -0
- package/src/setup/model-onboarding-guidance.ts +7 -1
- package/src/slash-commands/builtin-registry.ts +18 -1
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Semantic fallback for workflow-skill routing.
|
|
3
|
+
*
|
|
4
|
+
* The keyword table in `hooks/skill-keywords.ts` is thirteen literal strings. It is
|
|
5
|
+
* exact and free, and it is the right first stage — but measured against realistic
|
|
6
|
+
* paraphrases it recalls 4/17, and **0/9 in Korean**, which is most of our users. A
|
|
7
|
+
* miss is not fatal (the model still sees the routing rules in the system prompt), but
|
|
8
|
+
* it means the deterministic gate simply does not exist for those prompts.
|
|
9
|
+
*
|
|
10
|
+
* This module fills that gap only where the keyword stage produced nothing:
|
|
11
|
+
*
|
|
12
|
+
* keyword (exact, free) -> semantic (this, one cheap call) -> system prompt (as today)
|
|
13
|
+
*
|
|
14
|
+
* The two stages fail in opposite directions, which is why both are kept. Measured on
|
|
15
|
+
* the same 22 prompts, the literal stage is the one that catches `ultragoal this` and
|
|
16
|
+
* `consensus plan`; the semantic stage is the one that catches everything Korean.
|
|
17
|
+
*/
|
|
18
|
+
import { logger } from "@sayknow-cli/utils";
|
|
19
|
+
import { CANONICAL_SKC_WORKFLOW_SKILLS, type CanonicalSkcWorkflowSkill } from "../skill-state/active-state";
|
|
20
|
+
import type { DecisionService } from "./index";
|
|
21
|
+
|
|
22
|
+
const NONE = "none";
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* What each workflow is *for*, in the words a user would recognise. These descriptions
|
|
26
|
+
* are the whole contract with the model — the enum ids alone carry almost no signal.
|
|
27
|
+
*/
|
|
28
|
+
const WORKFLOW_MEANINGS: Record<CanonicalSkcWorkflowSkill, string> = {
|
|
29
|
+
// Scoped to the *behaviour* the user is asking for, not only to the state of the
|
|
30
|
+
// request. The earlier wording ("vague about what to build") described a property of
|
|
31
|
+
// the spec, so a direct instruction to ask rather than assume — the request is not
|
|
32
|
+
// vague, it is an order about how to proceed — landed on `none`. Measured: that one
|
|
33
|
+
// prompt was the sole miss in the 23-case set.
|
|
34
|
+
"deep-interview":
|
|
35
|
+
"The user wants requirements drawn out of them by questioning before anything is built. Includes explicit instructions to ask rather than assume.",
|
|
36
|
+
ralplan:
|
|
37
|
+
"The user wants a deliberate plan, design comparison, or approval before any code is touched. Architecture or sequencing risk is involved.",
|
|
38
|
+
ultragoal:
|
|
39
|
+
"The user wants an objective tracked in a durable ledger across many turns until every deliverable is verified.",
|
|
40
|
+
team: "The work is large enough to split across several coordinated workers running in parallel.",
|
|
41
|
+
};
|
|
42
|
+
|
|
43
|
+
const ROUTING_INSTRUCTIONS =
|
|
44
|
+
"Which workflow should handle this user request? Choose none unless the request clearly calls for one of the workflows.";
|
|
45
|
+
|
|
46
|
+
/** Exported so tests can assert the contract the model is actually given. */
|
|
47
|
+
export function buildRoutingCriteria(): Record<string, string> {
|
|
48
|
+
const criteria: Record<string, string> = {};
|
|
49
|
+
for (const skill of CANONICAL_SKC_WORKFLOW_SKILLS) criteria[skill] = WORKFLOW_MEANINGS[skill];
|
|
50
|
+
criteria[NONE] =
|
|
51
|
+
"An ordinary request: a question, a bug fix, a small edit, or anything that should just be handled directly.";
|
|
52
|
+
return criteria;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/** Prompts below this length never carry enough signal to justify a model round-trip. */
|
|
56
|
+
const MIN_PROMPT_CHARS = 12;
|
|
57
|
+
/** Only the opening of a prompt decides its workflow; the rest is payload. */
|
|
58
|
+
const MAX_PROMPT_CHARS = 4_000;
|
|
59
|
+
|
|
60
|
+
/**
|
|
61
|
+
* Minimum calibrated confidence required to activate a workflow.
|
|
62
|
+
*
|
|
63
|
+
* Activation is a strong move: it switches on the mutation guard, the Stop hook and the
|
|
64
|
+
* ask tool. Getting it wrong is worse than missing, because the user did not ask for any
|
|
65
|
+
* of that and has no obvious way to see why it appeared.
|
|
66
|
+
*
|
|
67
|
+
* Measured over ten routing prompts against the hosted model: every answer it reported
|
|
68
|
+
* at 1.00 was correct, and its single wrong answer reported 0.71. The lowest *correct*
|
|
69
|
+
* confidence was 0.67 — and that case was "none", so gating it out costs nothing. A
|
|
70
|
+
* floor here therefore removes the observed error without removing a real activation.
|
|
71
|
+
*
|
|
72
|
+
* One prompt sits close to this line. "추측하지 말고 모르는 건 다 물어봐" resolves to
|
|
73
|
+
* deep-interview in 8/8 samples but at 0.76-0.83, so the floor has roughly 0.01 of
|
|
74
|
+
* headroom on it. Raising the floor would drop a correct activation; lowering it would
|
|
75
|
+
* re-admit the 0.71 error. Treat 0.75 as fitted to a small sample and re-derive it from
|
|
76
|
+
* real usage rather than nudging it on a hunch.
|
|
77
|
+
*
|
|
78
|
+
* Only applied when the backend reports `calibrated: true`. An ordinary LLM answering
|
|
79
|
+
* through a forced enum has no meaningful confidence to compare against, so gating on a
|
|
80
|
+
* number it did not really produce would just be superstition.
|
|
81
|
+
*/
|
|
82
|
+
const MIN_CALIBRATED_CONFIDENCE = 0.75;
|
|
83
|
+
|
|
84
|
+
export type SkillRouter = (text: string) => Promise<CanonicalSkcWorkflowSkill | null>;
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* Build the semantic router. Returns null-resolving function when the service is
|
|
88
|
+
* disabled so the caller keeps its existing behaviour with no branching.
|
|
89
|
+
*/
|
|
90
|
+
export function createSemanticSkillRouter(service: DecisionService): SkillRouter {
|
|
91
|
+
const criteria = buildRoutingCriteria();
|
|
92
|
+
return async (text: string): Promise<CanonicalSkcWorkflowSkill | null> => {
|
|
93
|
+
if (!service.enabled) return null;
|
|
94
|
+
const trimmed = text.trim();
|
|
95
|
+
if (trimmed.length < MIN_PROMPT_CHARS) return null;
|
|
96
|
+
const state = trimmed.length > MAX_PROMPT_CHARS ? trimmed.slice(0, MAX_PROMPT_CHARS) : trimmed;
|
|
97
|
+
|
|
98
|
+
const result = await service.decide({
|
|
99
|
+
state,
|
|
100
|
+
questions: { workflow: { type: "choice", instructions: ROUTING_INSTRUCTIONS, criteria } },
|
|
101
|
+
});
|
|
102
|
+
const answer = result?.answers.workflow;
|
|
103
|
+
if (!result || answer?.type !== "choice" || answer.choice === NONE) return null;
|
|
104
|
+
const skill = CANONICAL_SKC_WORKFLOW_SKILLS.find(candidate => candidate === answer.choice);
|
|
105
|
+
if (!skill) return null;
|
|
106
|
+
if (result.calibrated && (answer.confidence ?? 0) < MIN_CALIBRATED_CONFIDENCE) {
|
|
107
|
+
logger.debug("decisions/skill-routing: below confidence floor, leaving routing alone", {
|
|
108
|
+
skill,
|
|
109
|
+
confidence: answer.confidence,
|
|
110
|
+
floor: MIN_CALIBRATED_CONFIDENCE,
|
|
111
|
+
});
|
|
112
|
+
return null;
|
|
113
|
+
}
|
|
114
|
+
logger.debug("decisions/skill-routing: semantic match", {
|
|
115
|
+
skill,
|
|
116
|
+
backend: result.backend,
|
|
117
|
+
confidence: answer.confidence,
|
|
118
|
+
calibrated: result.calibrated,
|
|
119
|
+
durationMs: result.durationMs,
|
|
120
|
+
});
|
|
121
|
+
return skill;
|
|
122
|
+
};
|
|
123
|
+
}
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Typed decisions — "Jev-shaped" structured judgments for code to branch on.
|
|
3
|
+
*
|
|
4
|
+
* The request/response shape follows TypeSafe's System One API so a backend can be
|
|
5
|
+
* swapped without touching call sites: the hosted `jev` model, a self-hosted OpenJev
|
|
6
|
+
* daemon, or — the default — the model the user is already logged into.
|
|
7
|
+
*
|
|
8
|
+
* What this is NOT: a probability oracle. Only the hosted model returns calibrated
|
|
9
|
+
* probabilities. Backends that constrain an ordinary LLM return an ordinal value and
|
|
10
|
+
* report `calibrated: false`; treat those numbers as a ranking, never as P(correct).
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
/** Pick exactly one option from a closed set. */
|
|
14
|
+
export interface ChoiceQuestion {
|
|
15
|
+
type: "choice";
|
|
16
|
+
instructions: string;
|
|
17
|
+
/** option id -> what that option means. At least two. */
|
|
18
|
+
criteria: Record<string, string>;
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
/** Rate the state against ordered levels. Level 0 is the lowest. */
|
|
22
|
+
export interface ScoreQuestion {
|
|
23
|
+
type: "score";
|
|
24
|
+
instructions: string;
|
|
25
|
+
/** Ordered level descriptions, lowest first. At least two. */
|
|
26
|
+
criteria: string[];
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
/** Is this statement true of the state? */
|
|
30
|
+
export interface NoulQuestion {
|
|
31
|
+
type: "noul";
|
|
32
|
+
instructions: string;
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
export type Question = ChoiceQuestion | ScoreQuestion | NoulQuestion;
|
|
36
|
+
|
|
37
|
+
export interface ChoiceAnswer {
|
|
38
|
+
type: "choice";
|
|
39
|
+
/** The selected option id. Always one of the declared `criteria` keys. */
|
|
40
|
+
choice: string;
|
|
41
|
+
/** Present only when the backend exposes a distribution. */
|
|
42
|
+
probabilities?: Record<string, number>;
|
|
43
|
+
/**
|
|
44
|
+
* How certain the backend is, 0..1. Only meaningful when the result reports
|
|
45
|
+
* `calibrated: true` — that is the difference between a number you can threshold on
|
|
46
|
+
* and a number that merely ranks. Absent when the backend cannot supply one.
|
|
47
|
+
*/
|
|
48
|
+
confidence?: number;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
export interface ScoreAnswer {
|
|
52
|
+
type: "score";
|
|
53
|
+
/** Level index. Fractional only when the backend returns a distribution. */
|
|
54
|
+
score: number;
|
|
55
|
+
/** Selected level index. */
|
|
56
|
+
level: number;
|
|
57
|
+
legend: Record<string, string>;
|
|
58
|
+
probabilities?: Record<string, number>;
|
|
59
|
+
/** See {@link ChoiceAnswer.confidence}. */
|
|
60
|
+
confidence?: number;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
export interface NoulAnswer {
|
|
64
|
+
type: "noul";
|
|
65
|
+
/** 0 (no) .. 1 (yes). Ordinal unless `calibrated` is true. */
|
|
66
|
+
noul: number;
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
export type Answer = ChoiceAnswer | ScoreAnswer | NoulAnswer;
|
|
70
|
+
|
|
71
|
+
export interface DecisionResult {
|
|
72
|
+
answers: Record<string, Answer>;
|
|
73
|
+
/** Which backend answered, for logging and A/B comparison. */
|
|
74
|
+
backend: string;
|
|
75
|
+
/** Model identifier the backend used. */
|
|
76
|
+
model: string;
|
|
77
|
+
/**
|
|
78
|
+
* False means the numbers are ordinal rankings, not probabilities.
|
|
79
|
+
* Only the hosted System One model reports true.
|
|
80
|
+
*/
|
|
81
|
+
calibrated: boolean;
|
|
82
|
+
durationMs: number;
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
export interface DecisionRequest {
|
|
86
|
+
/** The content to judge. Plain text, or JSON-serialisable structured state. */
|
|
87
|
+
state: string | Record<string, unknown> | unknown[];
|
|
88
|
+
/** Question id -> question. Answers come back under the same ids. */
|
|
89
|
+
questions: Record<string, Question>;
|
|
90
|
+
signal?: AbortSignal;
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
export interface DecisionBackend {
|
|
94
|
+
readonly name: string;
|
|
95
|
+
/** Resolves null when the backend is unavailable (no credentials, offline, disabled). */
|
|
96
|
+
decide(request: DecisionRequest): Promise<DecisionResult | null>;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
export const MIN_OPTIONS = 2;
|
|
100
|
+
/** Matches OpenJev's letter-slot ceiling so a graph stays portable across backends. */
|
|
101
|
+
export const MAX_OPTIONS = 16;
|
|
102
|
+
|
|
103
|
+
export function validateQuestions(questions: Record<string, Question>): void {
|
|
104
|
+
const entries = Object.entries(questions);
|
|
105
|
+
if (entries.length === 0) throw new Error("decisions: questions must not be empty");
|
|
106
|
+
for (const [key, question] of entries) {
|
|
107
|
+
if (question.type === "noul") {
|
|
108
|
+
if (!question.instructions?.trim()) throw new Error(`decisions: ${key} needs instructions`);
|
|
109
|
+
continue;
|
|
110
|
+
}
|
|
111
|
+
const size = question.type === "choice" ? Object.keys(question.criteria).length : question.criteria.length;
|
|
112
|
+
if (size < MIN_OPTIONS || size > MAX_OPTIONS)
|
|
113
|
+
throw new Error(`decisions: ${key} needs ${MIN_OPTIONS}-${MAX_OPTIONS} options, got ${size}`);
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
export function stateToText(state: DecisionRequest["state"]): string {
|
|
118
|
+
return typeof state === "string" ? state : JSON.stringify(state);
|
|
119
|
+
}
|
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Decision backend backed by TypeSafe's hosted System One model (`jev`).
|
|
3
|
+
*
|
|
4
|
+
* Unlike the LLM backend, this one returns **calibrated probabilities**: the model is
|
|
5
|
+
* trained to make the number mean what it says, so `confidence` is a value code can
|
|
6
|
+
* threshold on rather than a ranking. That is the whole reason to prefer it when a key
|
|
7
|
+
* is present — the type safety alone we already get from enum-constrained tool calls.
|
|
8
|
+
*
|
|
9
|
+
* Credentials ride the existing store (`ModelRegistry.getApiKeyForProvider`), so a key
|
|
10
|
+
* added through the normal "add model" flow turns this backend on and removing it turns
|
|
11
|
+
* it off. There is nothing extra to configure.
|
|
12
|
+
*
|
|
13
|
+
* Deliberately *not* registered in `packages/ai`'s provider registry: that registry is
|
|
14
|
+
* for streaming chat APIs, and this endpoint has no stream, no messages, and no text
|
|
15
|
+
* output. Wiring it there would force a `Model` shape onto something that is not a chat
|
|
16
|
+
* model. It stays a plain HTTP client behind the `DecisionBackend` interface.
|
|
17
|
+
*/
|
|
18
|
+
import { logger } from "@sayknow-cli/utils";
|
|
19
|
+
import type { ModelRegistry } from "../config/model-registry";
|
|
20
|
+
import {
|
|
21
|
+
type Answer,
|
|
22
|
+
type DecisionBackend,
|
|
23
|
+
type DecisionRequest,
|
|
24
|
+
type DecisionResult,
|
|
25
|
+
type Question,
|
|
26
|
+
validateQuestions,
|
|
27
|
+
} from "./types";
|
|
28
|
+
|
|
29
|
+
/** Provider id under which the key is stored and surfaced in the model list. */
|
|
30
|
+
export const TYPESAFE_PROVIDER = "typesafe";
|
|
31
|
+
const DEFAULT_BASE_URL = "https://api.typesafe.ai";
|
|
32
|
+
const DEFAULT_MODEL = "jev-latest";
|
|
33
|
+
/** The hosted model answers in well under a second; anything slower is a network fault. */
|
|
34
|
+
const REQUEST_TIMEOUT_MS = 10_000;
|
|
35
|
+
|
|
36
|
+
interface TypeSafeAnswer {
|
|
37
|
+
type?: string;
|
|
38
|
+
noul?: number;
|
|
39
|
+
choice?: string;
|
|
40
|
+
score?: number;
|
|
41
|
+
probabilities?: Record<string, number>;
|
|
42
|
+
legend?: Record<string, string>;
|
|
43
|
+
confidence?: number;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
interface TypeSafeResponse {
|
|
47
|
+
model?: string;
|
|
48
|
+
answers?: Record<string, TypeSafeAnswer>;
|
|
49
|
+
usage?: { input_tokens?: number; output_tokens?: number };
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* Our question shape is already the System One shape, so this is a rename rather than a
|
|
54
|
+
* translation — `noul` carries no criteria, `choice` an id→meaning map, `score` an
|
|
55
|
+
* ordered level array. Keeping them aligned is what lets a caller switch backends
|
|
56
|
+
* without touching the call site.
|
|
57
|
+
*/
|
|
58
|
+
function toWireQuestions(questions: Record<string, Question>): Record<string, unknown> {
|
|
59
|
+
const wire: Record<string, unknown> = {};
|
|
60
|
+
for (const [key, question] of Object.entries(questions)) {
|
|
61
|
+
wire[key] =
|
|
62
|
+
question.type === "noul"
|
|
63
|
+
? { type: "noul", instructions: question.instructions }
|
|
64
|
+
: { type: question.type, instructions: question.instructions, criteria: question.criteria };
|
|
65
|
+
}
|
|
66
|
+
return wire;
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* Map a wire answer onto our typed answer.
|
|
71
|
+
*
|
|
72
|
+
* Anything that does not match the question we asked is dropped rather than coerced: a
|
|
73
|
+
* decision that looks typed but is not the one we requested is worse than a missing one,
|
|
74
|
+
* because the caller cannot tell the difference.
|
|
75
|
+
*/
|
|
76
|
+
function toAnswer(question: Question, raw: TypeSafeAnswer | undefined): Answer | null {
|
|
77
|
+
if (!raw) return null;
|
|
78
|
+
if (question.type === "noul") {
|
|
79
|
+
return typeof raw.noul === "number" ? { type: "noul", noul: raw.noul } : null;
|
|
80
|
+
}
|
|
81
|
+
if (question.type === "choice") {
|
|
82
|
+
if (typeof raw.choice !== "string" || !(raw.choice in question.criteria)) return null;
|
|
83
|
+
return {
|
|
84
|
+
type: "choice",
|
|
85
|
+
choice: raw.choice,
|
|
86
|
+
...(raw.probabilities ? { probabilities: raw.probabilities } : {}),
|
|
87
|
+
...(typeof raw.confidence === "number" ? { confidence: raw.confidence } : {}),
|
|
88
|
+
};
|
|
89
|
+
}
|
|
90
|
+
if (typeof raw.score !== "number") return null;
|
|
91
|
+
const level = Math.min(question.criteria.length - 1, Math.max(0, Math.round(raw.score)));
|
|
92
|
+
return {
|
|
93
|
+
type: "score",
|
|
94
|
+
score: raw.score,
|
|
95
|
+
level,
|
|
96
|
+
legend: raw.legend ?? Object.fromEntries(question.criteria.map((meaning, index) => [String(index), meaning])),
|
|
97
|
+
...(raw.probabilities ? { probabilities: raw.probabilities } : {}),
|
|
98
|
+
...(typeof raw.confidence === "number" ? { confidence: raw.confidence } : {}),
|
|
99
|
+
};
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
export interface TypeSafeBackendDeps {
|
|
103
|
+
registry: ModelRegistry;
|
|
104
|
+
sessionId?: string;
|
|
105
|
+
/** Override for self-hosted or proxied deployments. */
|
|
106
|
+
baseUrl?: string;
|
|
107
|
+
/** Model id sent in the request body. Named to avoid colliding with the LLM backend's `model`. */
|
|
108
|
+
modelId?: string;
|
|
109
|
+
/** Injected in tests; defaults to global fetch. */
|
|
110
|
+
fetchImpl?: typeof fetch;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
export function createTypeSafeDecisionBackend(deps: TypeSafeBackendDeps): DecisionBackend {
|
|
114
|
+
const baseUrl = (deps.baseUrl ?? DEFAULT_BASE_URL).replace(/\/+$/, "");
|
|
115
|
+
const model = deps.modelId ?? DEFAULT_MODEL;
|
|
116
|
+
const doFetch = deps.fetchImpl ?? fetch;
|
|
117
|
+
|
|
118
|
+
return {
|
|
119
|
+
name: "typesafe",
|
|
120
|
+
async decide(request: DecisionRequest): Promise<DecisionResult | null> {
|
|
121
|
+
validateQuestions(request.questions);
|
|
122
|
+
const apiKey = await deps.registry.getApiKeyForProvider(TYPESAFE_PROVIDER, deps.sessionId);
|
|
123
|
+
// No key means the user never added TypeSafe. That is not an error — the next
|
|
124
|
+
// backend (their logged-in model) handles it.
|
|
125
|
+
if (!apiKey) return null;
|
|
126
|
+
|
|
127
|
+
const controller = new AbortController();
|
|
128
|
+
const abortOnCaller = () => controller.abort();
|
|
129
|
+
request.signal?.addEventListener("abort", abortOnCaller, { once: true });
|
|
130
|
+
const timer = setTimeout(() => controller.abort(), REQUEST_TIMEOUT_MS);
|
|
131
|
+
const started = Date.now();
|
|
132
|
+
try {
|
|
133
|
+
const response = await doFetch(`${baseUrl}/v1/systemone`, {
|
|
134
|
+
method: "POST",
|
|
135
|
+
headers: { Authorization: `Bearer ${apiKey}`, "Content-Type": "application/json" },
|
|
136
|
+
body: JSON.stringify({ state: request.state, model, questions: toWireQuestions(request.questions) }),
|
|
137
|
+
signal: controller.signal,
|
|
138
|
+
});
|
|
139
|
+
if (!response.ok) {
|
|
140
|
+
logger.debug("decisions/typesafe: request failed", {
|
|
141
|
+
status: response.status,
|
|
142
|
+
body: (await response.text().catch(() => "")).slice(0, 300),
|
|
143
|
+
});
|
|
144
|
+
return null;
|
|
145
|
+
}
|
|
146
|
+
const payload = (await response.json()) as TypeSafeResponse;
|
|
147
|
+
const answers: Record<string, Answer> = {};
|
|
148
|
+
for (const [key, question] of Object.entries(request.questions)) {
|
|
149
|
+
const answer = toAnswer(question, payload.answers?.[key]);
|
|
150
|
+
if (answer) answers[key] = answer;
|
|
151
|
+
}
|
|
152
|
+
if (Object.keys(answers).length === 0) return null;
|
|
153
|
+
return {
|
|
154
|
+
answers,
|
|
155
|
+
backend: "typesafe",
|
|
156
|
+
model: payload.model ?? model,
|
|
157
|
+
// The hosted System One model is trained for calibration; this is the one
|
|
158
|
+
// backend allowed to claim it.
|
|
159
|
+
calibrated: true,
|
|
160
|
+
durationMs: Date.now() - started,
|
|
161
|
+
};
|
|
162
|
+
} finally {
|
|
163
|
+
clearTimeout(timer);
|
|
164
|
+
request.signal?.removeEventListener("abort", abortOnCaller);
|
|
165
|
+
}
|
|
166
|
+
},
|
|
167
|
+
};
|
|
168
|
+
}
|
|
@@ -36,6 +36,26 @@ export const SKC_SKILL_KEYWORD_DEFINITIONS: readonly SkillKeywordDefinition[] =
|
|
|
36
36
|
priority: 8,
|
|
37
37
|
guidance: "Activate SKC deep-interview requirements workflow",
|
|
38
38
|
},
|
|
39
|
+
// Korean counterparts. The table was English-only, which is why the deterministic
|
|
40
|
+
// stage recalled 0/9 on Korean prompts while scoring 4/8 on English ones — the gap
|
|
41
|
+
// was never about phrasing being harder to detect, it was about nobody enumerating it.
|
|
42
|
+
//
|
|
43
|
+
// These stay deliberately narrow. A keyword fires with full authority and no
|
|
44
|
+
// confidence to fall back on, so a loose phrase here activates a workflow the user
|
|
45
|
+
// never asked for — worse than missing one, because the semantic stage still catches
|
|
46
|
+
// paraphrases behind it.
|
|
47
|
+
{
|
|
48
|
+
keyword: "추측하지 말",
|
|
49
|
+
skill: "deep-interview",
|
|
50
|
+
priority: 8,
|
|
51
|
+
guidance: "Activate SKC deep-interview requirements workflow",
|
|
52
|
+
},
|
|
53
|
+
{
|
|
54
|
+
keyword: "인터뷰하듯",
|
|
55
|
+
skill: "deep-interview",
|
|
56
|
+
priority: 8,
|
|
57
|
+
guidance: "Activate SKC deep-interview requirements workflow",
|
|
58
|
+
},
|
|
39
59
|
{
|
|
40
60
|
keyword: "$ralplan",
|
|
41
61
|
skill: "ralplan",
|
|
@@ -48,6 +68,18 @@ export const SKC_SKILL_KEYWORD_DEFINITIONS: readonly SkillKeywordDefinition[] =
|
|
|
48
68
|
priority: 9,
|
|
49
69
|
guidance: "Activate SKC ralplan planning workflow",
|
|
50
70
|
},
|
|
71
|
+
{
|
|
72
|
+
keyword: "합의된 계획",
|
|
73
|
+
skill: "ralplan",
|
|
74
|
+
priority: 9,
|
|
75
|
+
guidance: "Activate SKC ralplan planning workflow",
|
|
76
|
+
},
|
|
77
|
+
{
|
|
78
|
+
keyword: "계획서 만들",
|
|
79
|
+
skill: "ralplan",
|
|
80
|
+
priority: 9,
|
|
81
|
+
guidance: "Activate SKC ralplan planning workflow",
|
|
82
|
+
},
|
|
51
83
|
{
|
|
52
84
|
keyword: "$ultragoal",
|
|
53
85
|
skill: "ultragoal",
|
|
@@ -60,6 +92,18 @@ export const SKC_SKILL_KEYWORD_DEFINITIONS: readonly SkillKeywordDefinition[] =
|
|
|
60
92
|
priority: 8,
|
|
61
93
|
guidance: "Activate SKC ultragoal durable goal workflow",
|
|
62
94
|
},
|
|
95
|
+
{
|
|
96
|
+
keyword: "끝까지 추적",
|
|
97
|
+
skill: "ultragoal",
|
|
98
|
+
priority: 8,
|
|
99
|
+
guidance: "Activate SKC ultragoal durable goal workflow",
|
|
100
|
+
},
|
|
101
|
+
{
|
|
102
|
+
keyword: "장기 목표로",
|
|
103
|
+
skill: "ultragoal",
|
|
104
|
+
priority: 8,
|
|
105
|
+
guidance: "Activate SKC ultragoal durable goal workflow",
|
|
106
|
+
},
|
|
63
107
|
{
|
|
64
108
|
keyword: "$team",
|
|
65
109
|
skill: "team",
|
|
@@ -72,6 +116,18 @@ export const SKC_SKILL_KEYWORD_DEFINITIONS: readonly SkillKeywordDefinition[] =
|
|
|
72
116
|
priority: 8,
|
|
73
117
|
guidance: "Activate SKC team workflow",
|
|
74
118
|
},
|
|
119
|
+
{
|
|
120
|
+
keyword: "병렬로 돌려",
|
|
121
|
+
skill: "team",
|
|
122
|
+
priority: 8,
|
|
123
|
+
guidance: "Activate SKC team workflow",
|
|
124
|
+
},
|
|
125
|
+
{
|
|
126
|
+
keyword: "팀 구성해서",
|
|
127
|
+
skill: "team",
|
|
128
|
+
priority: 8,
|
|
129
|
+
guidance: "Activate SKC team workflow",
|
|
130
|
+
},
|
|
75
131
|
] as const;
|
|
76
132
|
|
|
77
133
|
export function isSkcWorkflowSkill(value: string): value is SkcWorkflowSkill {
|
package/src/hooks/skill-state.ts
CHANGED
|
@@ -121,6 +121,12 @@ export interface RecordSkillActivationInput {
|
|
|
121
121
|
turnId?: string;
|
|
122
122
|
nowIso?: string;
|
|
123
123
|
stateDir?: string;
|
|
124
|
+
/**
|
|
125
|
+
* Semantic fallback, consulted only when no keyword matched. Supplying it turns the
|
|
126
|
+
* literal keyword table into a two-stage router; omitting it keeps the historical
|
|
127
|
+
* keyword-only behaviour byte for byte.
|
|
128
|
+
*/
|
|
129
|
+
resolveSkillSemantically?: (text: string) => Promise<SkcWorkflowSkill | null>;
|
|
124
130
|
}
|
|
125
131
|
|
|
126
132
|
export interface StopHookInput {
|
|
@@ -471,8 +477,18 @@ async function seedSkillActivationState(
|
|
|
471
477
|
// real /skill dispatch paths resolve sub-skill activation before prompt construction.
|
|
472
478
|
export async function recordSkillActivation(input: RecordSkillActivationInput): Promise<SkillActiveState | null> {
|
|
473
479
|
const match = detectPrimarySkillKeyword(input.text);
|
|
474
|
-
if (
|
|
475
|
-
|
|
480
|
+
if (match) return await seedSkillActivationState(match.skill, match.keyword, "skc-skill-state-hook", input);
|
|
481
|
+
if (!input.resolveSkillSemantically) return null;
|
|
482
|
+
// The semantic stage is advisory: any failure leaves routing to the system prompt,
|
|
483
|
+
// exactly as it behaved before this stage existed.
|
|
484
|
+
let semantic: SkcWorkflowSkill | null = null;
|
|
485
|
+
try {
|
|
486
|
+
semantic = await input.resolveSkillSemantically(input.text);
|
|
487
|
+
} catch {
|
|
488
|
+
return null;
|
|
489
|
+
}
|
|
490
|
+
if (!semantic) return null;
|
|
491
|
+
return await seedSkillActivationState(semantic, `semantic:${semantic}`, "skc-skill-decision", input);
|
|
476
492
|
}
|
|
477
493
|
|
|
478
494
|
export interface EnsureWorkflowSkillActivationInput {
|
|
@@ -4,7 +4,12 @@ import { matchesSelectCancel } from "../../modes/utils/keybinding-matchers";
|
|
|
4
4
|
import { formatModelOnboardingGuidance } from "../../setup/model-onboarding-guidance";
|
|
5
5
|
import { DynamicBorder } from "./dynamic-border";
|
|
6
6
|
|
|
7
|
-
export type ProviderOnboardingAction =
|
|
7
|
+
export type ProviderOnboardingAction =
|
|
8
|
+
| "custom-provider-wizard"
|
|
9
|
+
| "oauth-login"
|
|
10
|
+
| "import-credentials"
|
|
11
|
+
| "api-guide"
|
|
12
|
+
| "typesafe-key";
|
|
8
13
|
|
|
9
14
|
interface ProviderOnboardingOption {
|
|
10
15
|
label: string;
|
|
@@ -28,6 +33,13 @@ const PROVIDER_ONBOARDING_OPTIONS: ProviderOnboardingOption[] = [
|
|
|
28
33
|
description: "Show the /provider add and skc setup provider commands.",
|
|
29
34
|
action: "api-guide",
|
|
30
35
|
},
|
|
36
|
+
{
|
|
37
|
+
// Not a chat model, so it never shows in the model picker — but this list is where
|
|
38
|
+
// users come to add a key, and a CLI-only path means nobody turns it on.
|
|
39
|
+
label: "Add TypeSafe key (typed decisions)",
|
|
40
|
+
description: "Route workflow decisions through the hosted System One model. Off entirely without a key.",
|
|
41
|
+
action: "typesafe-key",
|
|
42
|
+
},
|
|
31
43
|
{
|
|
32
44
|
label: "Import existing credentials",
|
|
33
45
|
description: "Detect and import Claude Code / Codex CLI logins already on this machine.",
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Key entry for TypeSafe, reachable from the same place models are added.
|
|
3
|
+
*
|
|
4
|
+
* TypeSafe is not a chat model, so it never appears in the model picker — but the place
|
|
5
|
+
* users look when they want to "add a model with a key" is this onboarding list, and
|
|
6
|
+
* making them find a CLI subcommand instead would mean most users never enable it.
|
|
7
|
+
*
|
|
8
|
+
* The key is taken through {@link SecretInput} and consumed once: it is never rendered,
|
|
9
|
+
* never placed in a flag, and never written anywhere but the credential store.
|
|
10
|
+
*/
|
|
11
|
+
import { Container, type Input, matchesKey, SecretInput, Spacer, Text, TruncatedText } from "@sayknow-cli/tui";
|
|
12
|
+
import { theme } from "../theme/theme";
|
|
13
|
+
import { matchesSelectCancel } from "../utils/keybinding-matchers";
|
|
14
|
+
import { DynamicBorder } from "./dynamic-border";
|
|
15
|
+
|
|
16
|
+
export interface TypeSafeKeyPromptResult {
|
|
17
|
+
apiKey: string;
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
export class TypeSafeKeyPromptComponent extends Container {
|
|
21
|
+
#content: Container;
|
|
22
|
+
#input: SecretInput | null = null;
|
|
23
|
+
#onSubmit: (result: TypeSafeKeyPromptResult) => void;
|
|
24
|
+
#onCancel: () => void;
|
|
25
|
+
#onRender: () => void;
|
|
26
|
+
#busy = false;
|
|
27
|
+
#error: string | null = null;
|
|
28
|
+
|
|
29
|
+
constructor(
|
|
30
|
+
onSubmit: (result: TypeSafeKeyPromptResult) => void,
|
|
31
|
+
onCancel: () => void,
|
|
32
|
+
onRender: () => void = () => {},
|
|
33
|
+
) {
|
|
34
|
+
super();
|
|
35
|
+
this.#onSubmit = onSubmit;
|
|
36
|
+
this.#onCancel = onCancel;
|
|
37
|
+
this.#onRender = onRender;
|
|
38
|
+
this.#content = new Container();
|
|
39
|
+
this.addChild(new DynamicBorder());
|
|
40
|
+
this.addChild(new Spacer(1));
|
|
41
|
+
this.addChild(new TruncatedText(theme.bold("TypeSafe (typed decisions)")));
|
|
42
|
+
this.addChild(
|
|
43
|
+
new TruncatedText(
|
|
44
|
+
theme.fg(
|
|
45
|
+
"muted",
|
|
46
|
+
" Routes workflow decisions through the hosted System One model instead of your chat model.",
|
|
47
|
+
),
|
|
48
|
+
0,
|
|
49
|
+
0,
|
|
50
|
+
),
|
|
51
|
+
);
|
|
52
|
+
this.addChild(
|
|
53
|
+
new TruncatedText(theme.fg("muted", " Without a key this stays off and nothing else changes."), 0, 0),
|
|
54
|
+
);
|
|
55
|
+
this.addChild(new Spacer(1));
|
|
56
|
+
this.addChild(this.#content);
|
|
57
|
+
this.addChild(new Spacer(1));
|
|
58
|
+
this.addChild(new DynamicBorder());
|
|
59
|
+
this.#render();
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/** Shown while the key is being checked against the live API. */
|
|
63
|
+
setBusy(busy: boolean): void {
|
|
64
|
+
this.#busy = busy;
|
|
65
|
+
this.#render();
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/** Keeps the prompt open so a rejected key can be corrected without restarting. */
|
|
69
|
+
setError(message: string): void {
|
|
70
|
+
this.#busy = false;
|
|
71
|
+
this.#error = message;
|
|
72
|
+
this.#render();
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
#render(): void {
|
|
76
|
+
this.#content.clear();
|
|
77
|
+
if (this.#busy) {
|
|
78
|
+
this.#input = null;
|
|
79
|
+
this.#content.addChild(new Text(theme.fg("muted", "Verifying key against TypeSafe…"), 0, 0));
|
|
80
|
+
this.#onRender();
|
|
81
|
+
return;
|
|
82
|
+
}
|
|
83
|
+
if (this.#error) {
|
|
84
|
+
this.#content.addChild(new Text(theme.fg("error", this.#error), 0, 0));
|
|
85
|
+
this.#content.addChild(new Spacer(1));
|
|
86
|
+
}
|
|
87
|
+
this.#content.addChild(new Text("Paste your TypeSafe API key:", 0, 0));
|
|
88
|
+
this.#content.addChild(new Spacer(1));
|
|
89
|
+
const input = new SecretInput();
|
|
90
|
+
input.onSubmit = secret => {
|
|
91
|
+
const apiKey = secret.consume().trim();
|
|
92
|
+
if (!apiKey) return;
|
|
93
|
+
this.#error = null;
|
|
94
|
+
this.#onSubmit({ apiKey });
|
|
95
|
+
};
|
|
96
|
+
this.#input = input;
|
|
97
|
+
this.#content.addChild(input);
|
|
98
|
+
this.#onRender();
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
handleInput(keyData: string): void {
|
|
102
|
+
if (matchesSelectCancel(keyData) || matchesKey(keyData, "escape")) {
|
|
103
|
+
this.#onCancel();
|
|
104
|
+
return;
|
|
105
|
+
}
|
|
106
|
+
(this.#input as Input | SecretInput | null)?.handleInput(keyData);
|
|
107
|
+
}
|
|
108
|
+
}
|