@selesai/code 0.13.28 → 0.13.30-beta.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/CHANGELOG.md +14 -0
  2. package/dist/core/model-registry.d.ts +13 -1
  3. package/dist/core/model-registry.js +16 -0
  4. package/dist/defaults/models.json +38 -0
  5. package/dist/defaults/settings.json +8 -13
  6. package/dist/extensions/capability-gateway/catalog.ts +2 -2
  7. package/dist/extensions/capability-gateway/index.ts +103 -18
  8. package/dist/extensions/capability-gateway/integration.test.ts +394 -8
  9. package/dist/extensions/capability-gateway/routing.test.ts +415 -0
  10. package/dist/extensions/capability-gateway/routing.ts +221 -0
  11. package/dist/extensions/grep-app/index.ts +10 -0
  12. package/dist/extensions/jev/decisions.test.ts +316 -0
  13. package/dist/extensions/jev/decisions.ts +527 -0
  14. package/dist/extensions/jev/test-support.ts +233 -0
  15. package/dist/extensions/jev-advisory-lifecycle.test.ts +206 -0
  16. package/dist/extensions/jev-advisory-memory.test.ts +191 -0
  17. package/dist/extensions/jev-advisory-recommendations.test.ts +240 -0
  18. package/dist/extensions/jev-advisory-routing.ts +539 -0
  19. package/dist/extensions/package.json +2 -2
  20. package/dist/extensions/pi-hermes-memory/src/memory-search-bridge.ts +40 -0
  21. package/dist/extensions/pi-hermes-memory/src/tools/memory-search-tool.ts +57 -46
  22. package/dist/extensions/pi-hermes-memory/src/tools/memory-tool.ts +17 -0
  23. package/dist/extensions/pi-hermes-memory/src/tools/session-search-tool.ts +10 -0
  24. package/dist/extensions/pi-hermes-memory/src/tools/skill-tool.ts +5 -0
  25. package/dist/extensions/pi-hermes-memory/tests/tools/memory-search-tool.test.ts +25 -0
  26. package/dist/extensions/pi-intercom/index.ts +10 -0
  27. package/dist/extensions/pi-subagents/src/extension/fanout-child.ts +5 -0
  28. package/dist/extensions/pi-subagents/src/extension/index.ts +5 -0
  29. package/dist/extensions/pi-subagents/src/intercom/native-supervisor-channel.ts +10 -0
  30. package/dist/extensions/pi-subagents/src/runs/background/wait-tool.ts +10 -0
  31. package/dist/extensions/pi-web-agent/src/extension.ts +5 -0
  32. package/dist/extensions/question/index.ts +5 -0
  33. package/dist/extensions/tokenin-onboarding.ts +185 -0
  34. package/docs/settings.md +68 -31
  35. package/package.json +3 -3
  36. package/dist/extensions/auto-model.test.ts +0 -438
  37. package/dist/extensions/auto-model.ts +0 -357
package/docs/settings.md CHANGED
@@ -46,49 +46,86 @@ Use `/trust` in interactive mode to save a project trust decision for future ses
46
46
  }
47
47
  ```
48
48
 
49
- ### Automatic Model Routing
49
+ ### Jev Advisory Routing
50
50
 
51
- The bundled `auto-model` extension can classify each idle, top-level prompt as `simple`, `medium`,
52
- `complex`, or `reasoning` and switch to the model configured for that tier before the turn starts.
53
- Classification uses the Jev decisions model through the Token-In gateway. Routing stays off until
54
- `autoModel.enabled` is `true`.
51
+ The bundled `jev-advisory-routing` extension uses the Jev decisions model for bounded opt-in
52
+ routing. Every route stays off until it is enabled in `jevAdvisory`; the two routes share one
53
+ Jev provider/model pair:
54
+
55
+ - `memory` — only after an explicit durable-memory cue (for example “the convention we
56
+ decided” or “don't repeat the past failure”), Jev chooses one read-only local
57
+ `memory_search` target. Jev never sees memory contents, and no memory is written.
58
+ - `recommendations` — one discovered skill or prompt workflow that fits, or a proportionate
59
+ verification level. Nothing is loaded, started, or executed: the recommendation is context for the
60
+ agent, and required project/workflow gates are unchanged.
55
61
 
56
62
  | Setting | Type | Default | Description |
57
63
  |---------|------|---------|-------------|
58
- | `autoModel.enabled` | boolean | `false` | Enable automatic per-prompt routing |
59
- | `autoModel.classifier.provider` | string | `"tokenin"` | Provider serving the Jev decisions deployment |
60
- | `autoModel.classifier.model` | string | `"jev-1.13"` | Decisions model the classifier calls |
61
- | `autoModel.classifier.baseUrl` | string | inherited | Base URL override; defaults to any registered model of `classifier.provider` |
62
- | `autoModel.classifier.timeoutMs` | number | `10000` | Classifier request timeout (ms) |
63
- | `autoModel.classifier.minConfidence` | number | `0.5` | Below this Jev confidence the fallback tier is used |
64
- | `autoModel.classifier.contextTurns` | number | `4` | Prior user turns sent as classifier context |
65
- | `autoModel.classifier.contextChars` | number | `4000` | Character budget for that context |
66
- | `autoModel.tiers.simple` | string | `"tokenin/deepseek-v4.1-flash"` | Model for greetings, lookups, and tiny transformations |
67
- | `autoModel.tiers.medium` | string | `"tokenin/celestial-pro"` | Model for routine coding, edits, and explanations (the default fallback) |
68
- | `autoModel.tiers.complex` | string | `"tokenin/celestial-max"` | Model for non-trivial engineering and root-cause debugging |
69
- | `autoModel.tiers.reasoning` | string | `"tokenin/celestial-ultra"` | Model for open-ended reasoning and tradeoffs |
70
- | `autoModel.fallbackTier` | string | `"medium"` | Tier used when the classifier is unavailable |
71
-
72
- Each tier value is `provider/modelId`, with an optional `:thinkingLevel` suffix (for example
73
- `tokenin/celestial-max:max`). Only idle, top-level prompts are routed: queued steering/follow-up
74
- messages, extension-injected messages, and slash commands are left alone, because the session model
75
- is global. Selecting a model with `/model` suspends routing for the rest of the session.
64
+ | `jevAdvisory.provider` | string | `"tokenin"` | Provider serving the Jev decisions deployment |
65
+ | `jevAdvisory.model` | string | `"jev-1.13"` | Decisions model every route calls |
66
+ | `jevAdvisory.baseUrl` | string | inherited | Base URL override; defaults to any registered model of `provider` |
67
+ | `jevAdvisory.routes.memory.enabled` | boolean | `false` | Enable the memory-lookup route |
68
+ | `jevAdvisory.routes.recommendations.enabled` | boolean | `false` | Enable skill/workflow and verification recommendations |
69
+ | `<route>.timeoutMs` | number | `8000` | Route request timeout (ms); memory is capped at 750ms |
70
+ | `<route>.minConfidence` | number | `0.6` | Below this Jev confidence the route abstains |
71
+ | `<route>.contextTurns` | number | `4` | Prior user turns sent as recommendation context; memory sends only the current bounded prompt |
72
+ | `<route>.contextChars` | number | `4000` | Character budget for recommendation context |
73
+ | `<route>.payloadBytes` | number | `8192` | Hard cap on the serialized decision request |
74
+
75
+ Only idle, top-level, interactive prompts are routed: queued steering/follow-up input, slash
76
+ commands, and extension-injected turns are skipped, and each turn is routed at most once. The memory
77
+ route sends no history to Jev; on an accepted target it runs one local, read-only lookup (at most five
78
+ results, bounded before injection). A missing Token-In subscription, a timeout, a malformed or
79
+ low-confidence answer, an unknown candidate, unavailable/empty local memory, or an oversized payload
80
+ is an ordinary abstention that leaves the existing memory policy and verification requirements in force.
76
81
 
77
82
  ```json
78
83
  {
79
- "autoModel": {
80
- "enabled": true,
81
- "classifier": { "provider": "tokenin", "model": "jev-1.13" },
82
- "tiers": {
83
- "simple": "tokenin/deepseek-v4.1-flash",
84
- "medium": "tokenin/celestial-pro",
85
- "complex": "tokenin/celestial-max",
86
- "reasoning": "tokenin/celestial-ultra"
84
+ "jevAdvisory": {
85
+ "provider": "tokenin",
86
+ "model": "jev-1.13",
87
+ "routes": {
88
+ "memory": { "enabled": true },
89
+ "recommendations": { "enabled": true }
87
90
  }
88
91
  }
89
92
  }
90
93
  ```
91
94
 
95
+ ### Capability Gateway (experimental)
96
+
97
+ The bundled `capability-gateway` extension keeps optional extension tools dormant until they are
98
+ needed: a compact `capability_catalog` lists them, `capability_discover` activates one for the
99
+ current run, and `capability_skill_show` loads one skill's full instructions. Set
100
+ `SELESAI_CAPABILITY_GATEWAY=0` to disable the gateway and keep every tool visible.
101
+
102
+ Routing has two rungs:
103
+
104
+ 1. A deterministic router activates a tool when the prompt uniquely matches its name, alias, or
105
+ discovery summary. Skills are never auto-loaded or auto-selected.
106
+ 2. Opt-in Jev tie-breaking: when the deterministic router returns an ambiguous lexical hint among
107
+ optional tools, the Jev decisions model is asked which of two or three hinted tools (or `none`)
108
+ should be exposed. Jev sees only the bounded current prompt and each hinted tool's compact
109
+ discovery line — never conversation history, tool schemas, or the full catalog. A prompt with no
110
+ lexical signal, a unique activation, and a skill-only match never reach Jev.
111
+
112
+ | Setting | Type | Default | Description |
113
+ |---------|------|---------|-------------|
114
+ | `capabilityGateway.routing.jev.enabled` | boolean | `false` | Enable Jev tie-breaking for ambiguous tool hints |
115
+ | `capabilityGateway.routing.jev.provider` | string | `"tokenin"` | Provider serving the Jev decisions deployment |
116
+ | `capabilityGateway.routing.jev.model` | string | `"jev-1.13"` | Decisions model the tie-breaker calls |
117
+ | `capabilityGateway.routing.jev.baseUrl` | string | inherited | Base URL override; defaults to any registered model of `provider` |
118
+ | `capabilityGateway.routing.jev.timeoutMs` | number | `1000` | Pre-turn request timeout (ms); hard-capped at `2000` |
119
+ | `capabilityGateway.routing.jev.minConfidence` | number | `0.6` | Below this Jev confidence the tie-breaker abstains |
120
+ | `capabilityGateway.routing.jev.payloadBytes` | number | `8192` | Hard cap on the serialized decision request |
121
+
122
+ This area is independent of `jevAdvisory`: gateway routing reads only
123
+ `capabilityGateway.routing.jev` and shares just the Jev provider/model deployment identity. A
124
+ failure, timeout, invalid answer, low confidence, or `none` is an ordinary abstention that leaves
125
+ the deterministic behavior in place. Temporary activations reset when the run settles, and
126
+ content-free telemetry on the `capability-gateway` event channel records the route outcome and
127
+ whether an activated tool was actually invoked.
128
+
92
129
  ### UI & Display
93
130
 
94
131
  | Setting | Type | Default | Description |
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@selesai/code",
3
- "version": "0.13.28",
3
+ "version": "0.13.30-beta.0",
4
4
  "description": "Maintained, extension-first Pi coding agent with built-in workflows, subagents, web research, questions, skills, and an enhanced terminal UI.",
5
5
  "type": "module",
6
6
  "engines": {
@@ -55,8 +55,8 @@
55
55
  "clean": "shx rm -rf dist",
56
56
  "dev": "tsx src/cli.ts",
57
57
  "dev:print": "tsx src/cli.ts --print",
58
- "test": "vitest run src/extensions/undo.test.ts src/extensions/auto-model.test.ts src/extensions/tps.test.ts src/extensions/pi-graft src/extensions/agent-browser.test.ts src/extensions/auto-session-name.test.ts src/extensions/context-compaction-reminder.test.ts src/extensions/copy-turn.test.ts src/extensions/grep-app/index.test.ts src/extensions/inline-skills.test.ts src/extensions/rtk.test.ts src/extensions/handoff-new.test.ts src/extensions/question/tests src/extensions/ponytail/test src/__tests__/tokenin-search.test.ts src/__tests__/tokenin-onboarding.test.ts src/__tests__/model-registry-defaults.test.ts src/core/system-prompt.test.ts src/core/session-format.test.ts test/settings-manager-compaction.test.ts test/suite/agent-session-runtime.test.ts test/suite/agent-session-prompt.test.ts test/suite/agent-session-queue.test.ts test/suite/agent-session-retry-events.test.ts test/suite/agent-session-compaction.test.ts test/suite/agent-session-compaction-model-overrides.test.ts test/suite/agent-session-bash-persistence.test.ts test/suite/agent-session-model-extension.test.ts test/clipboard.test.ts test/clipboard-command.test.ts test/clipboard-image.test.ts test/clipboard-image-bmp-conversion.test.ts test/clipboard-image-native-errors.test.ts src/cli/args.test.ts",
59
- "test:coverage": "vitest run --coverage src/extensions/undo.test.ts src/extensions/auto-model.test.ts src/extensions/tps.test.ts src/extensions/agent-browser.test.ts src/extensions/auto-session-name.test.ts src/extensions/context-compaction-reminder.test.ts src/extensions/copy-turn.test.ts src/extensions/grep-app/index.test.ts src/extensions/inline-skills.test.ts src/extensions/rtk.test.ts src/extensions/handoff-new.test.ts src/extensions/question/tests src/extensions/ponytail/test src/__tests__/tokenin-search.test.ts src/__tests__/tokenin-onboarding.test.ts src/__tests__/model-registry-defaults.test.ts src/core/system-prompt.test.ts src/core/session-format.test.ts test/settings-manager-compaction.test.ts test/suite/agent-session-runtime.test.ts test/suite/agent-session-prompt.test.ts test/suite/agent-session-queue.test.ts test/suite/agent-session-retry-events.test.ts test/suite/agent-session-compaction.test.ts test/suite/agent-session-compaction-model-overrides.test.ts test/suite/agent-session-bash-persistence.test.ts test/suite/agent-session-model-extension.test.ts test/clipboard.test.ts test/clipboard-command.test.ts test/clipboard-image.test.ts test/clipboard-image-bmp-conversion.test.ts test/clipboard-image-native-errors.test.ts src/cli/args.test.ts",
58
+ "test": "vitest run src/__tests__/model-registry-completion.test.ts src/extensions/jev/decisions.test.ts src/extensions/capability-gateway/routing.test.ts src/extensions/jev-advisory-memory.test.ts src/extensions/jev-advisory-recommendations.test.ts src/extensions/jev-advisory-lifecycle.test.ts src/extensions/undo.test.ts src/extensions/tps.test.ts src/extensions/pi-graft src/extensions/agent-browser.test.ts src/extensions/auto-session-name.test.ts src/extensions/context-compaction-reminder.test.ts src/extensions/copy-turn.test.ts src/extensions/grep-app/index.test.ts src/extensions/inline-skills.test.ts src/extensions/rtk.test.ts src/extensions/handoff-new.test.ts src/extensions/question/tests src/extensions/ponytail/test src/__tests__/tokenin-search.test.ts src/__tests__/tokenin-onboarding.test.ts src/__tests__/model-registry-defaults.test.ts src/core/system-prompt.test.ts src/core/session-format.test.ts test/settings-manager-compaction.test.ts test/suite/agent-session-runtime.test.ts test/suite/agent-session-prompt.test.ts test/suite/agent-session-queue.test.ts test/suite/agent-session-retry-events.test.ts test/suite/agent-session-compaction.test.ts test/suite/agent-session-compaction-model-overrides.test.ts test/suite/agent-session-bash-persistence.test.ts test/suite/agent-session-model-extension.test.ts test/clipboard.test.ts test/clipboard-command.test.ts test/clipboard-image.test.ts test/clipboard-image-bmp-conversion.test.ts test/clipboard-image-native-errors.test.ts src/cli/args.test.ts",
59
+ "test:coverage": "vitest run --coverage src/__tests__/model-registry-completion.test.ts src/extensions/jev/decisions.test.ts src/extensions/capability-gateway/routing.test.ts src/extensions/jev-advisory-memory.test.ts src/extensions/jev-advisory-recommendations.test.ts src/extensions/jev-advisory-lifecycle.test.ts src/extensions/undo.test.ts src/extensions/tps.test.ts src/extensions/agent-browser.test.ts src/extensions/auto-session-name.test.ts src/extensions/context-compaction-reminder.test.ts src/extensions/copy-turn.test.ts src/extensions/grep-app/index.test.ts src/extensions/inline-skills.test.ts src/extensions/rtk.test.ts src/extensions/handoff-new.test.ts src/extensions/question/tests src/extensions/ponytail/test src/__tests__/tokenin-search.test.ts src/__tests__/tokenin-onboarding.test.ts src/__tests__/model-registry-defaults.test.ts src/core/system-prompt.test.ts src/core/session-format.test.ts test/settings-manager-compaction.test.ts test/suite/agent-session-runtime.test.ts test/suite/agent-session-prompt.test.ts test/suite/agent-session-queue.test.ts test/suite/agent-session-retry-events.test.ts test/suite/agent-session-compaction.test.ts test/suite/agent-session-compaction-model-overrides.test.ts test/suite/agent-session-bash-persistence.test.ts test/suite/agent-session-model-extension.test.ts test/clipboard.test.ts test/clipboard-command.test.ts test/clipboard-image.test.ts test/clipboard-image-bmp-conversion.test.ts test/clipboard-image-native-errors.test.ts src/cli/args.test.ts",
60
60
  "prepare": "npm run build",
61
61
  "build": "npm run clean && tsgo -p tsconfig.build.json && shx chmod +x dist/cli.js dist/rpc-entry.js && npm run copy-assets",
62
62
  "copy-assets": "shx mkdir -p dist/modes/interactive/theme && shx cp src/modes/interactive/theme/*.json dist/modes/interactive/theme/ && shx mkdir -p dist/modes/interactive/assets && shx cp src/modes/interactive/assets/*.png dist/modes/interactive/assets/ && shx mkdir -p dist/core/export-html/vendor && shx cp src/core/export-html/template.html src/core/export-html/template.css src/core/export-html/template.js dist/core/export-html/ && shx cp src/core/export-html/vendor/*.js dist/core/export-html/vendor/ && shx mkdir -p dist/defaults && shx cp src/defaults/* dist/defaults/ && shx mkdir -p dist/extensions && node scripts/copy-extensions.mjs && shx mkdir -p dist/themes && shx cp -r src/themes/. dist/themes/ && shx mkdir -p dist/skills && shx cp -r src/skills/. dist/skills/"
@@ -1,438 +0,0 @@
1
- import { mkdtempSync, writeFileSync } from "node:fs";
2
- import { tmpdir } from "node:os";
3
- import { join } from "node:path";
4
- import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
5
- import type { ExtensionAPI, ExtensionContext, SessionEntry } from "@selesai/code";
6
-
7
- const state = vi.hoisted(() => ({ settingsPath: "" }));
8
-
9
- vi.mock("@selesai/code", () => ({
10
- getSettingsPath: () => state.settingsPath,
11
- }));
12
-
13
- vi.mock("@earendil-works/pi-ai/compat", async (importOriginal) => {
14
- const actual = await importOriginal<typeof import("@earendil-works/pi-ai/compat")>();
15
- return { ...actual, complete: vi.fn() };
16
- });
17
-
18
- import { complete } from "@earendil-works/pi-ai/compat";
19
- import autoModelExtension, {
20
- buildConversation,
21
- buildJevPayload,
22
- classifierModel,
23
- classifyTier,
24
- DEFAULT_AUTO_MODEL_CONFIG,
25
- parseModelRef,
26
- readAutoModelConfig,
27
- TIERS,
28
- tierFromJevResponse,
29
- type AutoModelConfig,
30
- } from "./auto-model.ts";
31
-
32
- const completeMock = vi.mocked(complete);
33
-
34
- type Handler = (event: unknown, ctx: ExtensionContext) => unknown;
35
-
36
- function templateModel(provider = "tokenin", id = "celestial-pro") {
37
- return {
38
- provider,
39
- id,
40
- name: id,
41
- api: "openai-completions",
42
- baseUrl: "https://lite.andlet.me/v1",
43
- reasoning: true,
44
- input: ["text"],
45
- cost: { input: 1, output: 2, cacheRead: 0, cacheWrite: 0 },
46
- contextWindow: 393216,
47
- maxTokens: 64000,
48
- };
49
- }
50
-
51
- function userEntry(text: string, id = "u1", content?: unknown): SessionEntry {
52
- return {
53
- id,
54
- type: "message",
55
- message: { role: "user", content: content ?? [{ type: "text", text }], timestamp: 1 },
56
- parentId: "root",
57
- timestamp: "2025-01-01T00:00:00.000Z",
58
- } as unknown as SessionEntry;
59
- }
60
-
61
- function assistantEntry(text: string, id = "a1"): SessionEntry {
62
- return {
63
- id,
64
- type: "message",
65
- message: { role: "assistant", content: [{ type: "text", text }], timestamp: 1 },
66
- parentId: "root",
67
- timestamp: "2025-01-01T00:00:00.000Z",
68
- } as unknown as SessionEntry;
69
- }
70
-
71
- function createHarness(branch: SessionEntry[] = [], scopedModels: unknown[] = []) {
72
- const handlers = new Map<string, Handler>();
73
- const setModel = vi.fn().mockResolvedValue(true);
74
- const setThinkingLevel = vi.fn();
75
- const pi = {
76
- on: vi.fn((event: string, handler: Handler) => handlers.set(event, handler)),
77
- setModel,
78
- setThinkingLevel,
79
- } as unknown as ExtensionAPI;
80
- const target = templateModel();
81
- const known = new Set(["celestial-pro", "celestial-max", "celestial-ultra", "deepseek-v4.1-flash"]);
82
- const ctx = {
83
- modelRegistry: {
84
- getAll: () => [templateModel()],
85
- find: vi.fn((provider: string, id: string) =>
86
- provider === "tokenin" && known.has(id) ? { ...target, id } : undefined,
87
- ),
88
- getApiKeyAndHeaders: vi.fn().mockResolvedValue({ ok: true, apiKey: "key", headers: {} }),
89
- },
90
- sessionManager: { getBranch: () => branch },
91
- scopedModels,
92
- getSystemPrompt: () => "system prompt",
93
- } as unknown as ExtensionContext;
94
- autoModelExtension(pi);
95
- return { handlers, ctx, setModel, setThinkingLevel };
96
- }
97
-
98
- function writeSettings(value: unknown): void {
99
- writeFileSync(state.settingsPath, typeof value === "string" ? value : JSON.stringify(value), "utf-8");
100
- }
101
-
102
- function jevAnswer(choice: unknown, confidence?: unknown): string {
103
- return JSON.stringify({ answers: { complexity: { choice, confidence } } });
104
- }
105
-
106
- function textResponse(text: string) {
107
- return { content: [{ type: "text", text }], stopReason: "stop" } as never;
108
- }
109
-
110
- beforeEach(() => {
111
- vi.clearAllMocks();
112
- state.settingsPath = join(mkdtempSync(join(tmpdir(), "auto-model-")), "settings.json");
113
- writeSettings({ autoModel: { enabled: true } });
114
- });
115
-
116
- afterEach(() => {
117
- vi.restoreAllMocks();
118
- });
119
-
120
- describe("readAutoModelConfig", () => {
121
- it("falls back to the disabled defaults on a missing or malformed file", () => {
122
- expect(readAutoModelConfig(join(tmpdir(), "auto-model-does-not-exist", "settings.json"))).toEqual(
123
- DEFAULT_AUTO_MODEL_CONFIG,
124
- );
125
-
126
- writeSettings("{ not json");
127
- expect(readAutoModelConfig()).toEqual(DEFAULT_AUTO_MODEL_CONFIG);
128
-
129
- writeSettings([1, 2, 3]);
130
- expect(readAutoModelConfig()).toEqual(DEFAULT_AUTO_MODEL_CONFIG);
131
-
132
- writeSettings({ autoModel: "nope" });
133
- expect(readAutoModelConfig()).toEqual(DEFAULT_AUTO_MODEL_CONFIG);
134
- });
135
-
136
- it("merges partial config over the defaults and honors a valid fallback tier", () => {
137
- writeSettings({
138
- autoModel: {
139
- enabled: true,
140
- classifier: {
141
- model: "jev-9",
142
- baseUrl: "https://custom/v1",
143
- timeoutMs: 1,
144
- minConfidence: 0.2,
145
- contextTurns: 2,
146
- contextChars: "bad",
147
- },
148
- tiers: { simple: "tokenin/x" },
149
- fallbackTier: "reasoning",
150
- },
151
- });
152
- const config = readAutoModelConfig();
153
- expect(config.enabled).toBe(true);
154
- expect(config.classifier).toEqual({
155
- provider: "tokenin",
156
- model: "jev-9",
157
- baseUrl: "https://custom/v1",
158
- timeoutMs: 1,
159
- minConfidence: 0.2,
160
- contextTurns: 2,
161
- contextChars: DEFAULT_AUTO_MODEL_CONFIG.classifier.contextChars,
162
- });
163
- expect(config.tiers).toEqual({ ...DEFAULT_AUTO_MODEL_CONFIG.tiers, simple: "tokenin/x" });
164
- expect(config.fallbackTier).toBe("reasoning");
165
- });
166
-
167
- it("rejects bad field types and an unknown fallback tier", () => {
168
- writeSettings({
169
- autoModel: {
170
- enabled: "yes",
171
- classifier: { provider: 1, model: "", baseUrl: "", timeoutMs: -1, minConfidence: "x", contextTurns: 0, contextChars: 0 },
172
- tiers: { nonTier: "ignored" },
173
- fallbackTier: "nonsense",
174
- },
175
- });
176
- const config = readAutoModelConfig();
177
- expect(config.enabled).toBe(false);
178
- expect(config.classifier.provider).toBe("tokenin");
179
- expect(config.classifier.baseUrl).toBeUndefined();
180
- expect(config.fallbackTier).toBe("medium");
181
- expect(config.tiers).toEqual(DEFAULT_AUTO_MODEL_CONFIG.tiers);
182
- });
183
- });
184
-
185
- describe("buildJevPayload", () => {
186
- it("puts the turns in state and the four tiers in the choice question", () => {
187
- const payload = buildJevPayload([{ role: "user", text: "hi" }], " be terse ", 100) as any;
188
- expect(payload.state).toEqual({ conversation: [{ role: "user", text: "hi" }], system_prompt: "be terse" });
189
- expect(Object.keys(payload.questions.complexity.criteria)).toEqual(["SIMPLE", "MEDIUM", "COMPLEX", "REASONING"]);
190
- expect(payload.questions.complexity.type).toBe("choice");
191
- });
192
-
193
- it("omits an empty or missing system prompt and caps a long one", () => {
194
- expect((buildJevPayload([], undefined, 10) as any).state.system_prompt).toBeUndefined();
195
- expect((buildJevPayload([], " ", 10) as any).state.system_prompt).toBeUndefined();
196
- expect((buildJevPayload([], "x".repeat(50), 10) as any).state.system_prompt).toHaveLength(10);
197
- });
198
- });
199
-
200
- describe("tierFromJevResponse", () => {
201
- it("reads the chosen tier and rejects unusable or unsure answers", () => {
202
- expect(tierFromJevResponse(jevAnswer("COMPLEX", 0.9), 0.5)).toBe("complex");
203
- expect(tierFromJevResponse(jevAnswer("simple"), 0.5)).toBe("simple");
204
- expect(tierFromJevResponse(jevAnswer("REASONING", 0.5), 0.5)).toBe("reasoning");
205
- expect(tierFromJevResponse(jevAnswer("MEDIUM", 0.4), 0.5)).toBeUndefined();
206
- expect(tierFromJevResponse(jevAnswer("MEDIUM", "high"), 0.5)).toBe("medium");
207
- expect(tierFromJevResponse(jevAnswer("UNKNOWN", 0.9), 0.5)).toBeUndefined();
208
- expect(tierFromJevResponse(jevAnswer(7, 0.9), 0.5)).toBeUndefined();
209
- expect(tierFromJevResponse("{ bad", 0.5)).toBeUndefined();
210
- expect(tierFromJevResponse(JSON.stringify({ answers: null }), 0.5)).toBeUndefined();
211
- expect(tierFromJevResponse(JSON.stringify({ answers: { complexity: 1 } }), 0.5)).toBeUndefined();
212
- expect(tierFromJevResponse("42", 0.5)).toBeUndefined();
213
- });
214
- });
215
-
216
- describe("parseModelRef", () => {
217
- it("splits provider, id, and an optional thinking level", () => {
218
- expect(parseModelRef("tokenin/celestial-max")).toEqual({ provider: "tokenin", id: "celestial-max" });
219
- expect(parseModelRef("tokenin/celestial-max:max")).toEqual({
220
- provider: "tokenin",
221
- id: "celestial-max",
222
- thinking: "max",
223
- });
224
- expect(parseModelRef("no-slash")).toBeUndefined();
225
- expect(parseModelRef("/leading")).toBeUndefined();
226
- expect(parseModelRef("trailing/")).toBeUndefined();
227
- expect(parseModelRef("tokenin/:max")).toEqual({ provider: "tokenin", id: ":max" });
228
- });
229
- });
230
-
231
- describe("buildConversation", () => {
232
- it("keeps the current ask last and pulls prior user turns under the budget", () => {
233
- const branch = [
234
- userEntry("oldest", "u1"),
235
- assistantEntry("ignored narration"),
236
- { id: "x", type: "model_change" } as unknown as SessionEntry,
237
- userEntry("previous", "u2"),
238
- ];
239
- const turns = buildConversation("current", branch, { contextTurns: 4, contextChars: 1000 });
240
- expect(turns.map((t) => t.text)).toEqual(["oldest", "previous", "current"]);
241
- });
242
-
243
- it("drops empty turns, array-content text, and enforces turn and char caps", () => {
244
- const branch = [
245
- userEntry("", "e1"),
246
- userEntry("", "e2", [{ type: "image", data: "x" }]),
247
- userEntry("", "e3", 42),
248
- userEntry("", "e4", "plain string"),
249
- userEntry("kept", "u2"),
250
- ];
251
- expect(buildConversation(" ", branch, { contextTurns: 4, contextChars: 100 })).toEqual([
252
- { role: "user", text: "plain string" },
253
- { role: "user", text: "kept" },
254
- ]);
255
- expect(buildConversation("current", branch, { contextTurns: 2, contextChars: 100 }).map((t) => t.text)).toEqual([
256
- "kept",
257
- "current",
258
- ]);
259
- const truncated = buildConversation("abcdef", [], { contextTurns: 4, contextChars: 3 });
260
- expect(truncated).toEqual([{ role: "user", text: "def" }]);
261
- const exhausted = buildConversation("abc", [userEntry("later", "u1")], { contextTurns: 4, contextChars: 3 });
262
- expect(exhausted).toEqual([{ role: "user", text: "abc" }]);
263
- });
264
- });
265
-
266
- describe("classifierModel", () => {
267
- it("inherits the provider template, or uses an explicit baseUrl", () => {
268
- const registry = { getAll: () => [templateModel()] } as unknown as ExtensionContext["modelRegistry"];
269
- const model = classifierModel(registry, DEFAULT_AUTO_MODEL_CONFIG.classifier);
270
- expect(model).toMatchObject({
271
- id: "jev-1.13",
272
- provider: "tokenin",
273
- baseUrl: "https://lite.andlet.me/v1",
274
- contextWindow: 393216,
275
- });
276
-
277
- const bare = { getAll: () => [] } as unknown as ExtensionContext["modelRegistry"];
278
- expect(classifierModel(bare, DEFAULT_AUTO_MODEL_CONFIG.classifier)).toBeUndefined();
279
- const override = classifierModel(bare, { ...DEFAULT_AUTO_MODEL_CONFIG.classifier, baseUrl: "https://x/v1" });
280
- expect(override).toMatchObject({ baseUrl: "https://x/v1", reasoning: false, input: ["text"] });
281
- });
282
- });
283
-
284
- describe("classifyTier", () => {
285
- const config: AutoModelConfig = { ...DEFAULT_AUTO_MODEL_CONFIG, enabled: true };
286
-
287
- function ctxWith(overrides: Partial<ExtensionContext> = {}): ExtensionContext {
288
- return {
289
- modelRegistry: {
290
- getAll: () => [templateModel()],
291
- getApiKeyAndHeaders: vi.fn().mockResolvedValue({ ok: true, apiKey: "key", headers: {} }),
292
- },
293
- sessionManager: { getBranch: () => [] },
294
- getSystemPrompt: () => "sys",
295
- ...overrides,
296
- } as unknown as ExtensionContext;
297
- }
298
-
299
- it("returns the tier Jev answers", async () => {
300
- completeMock.mockResolvedValue(textResponse(jevAnswer("COMPLEX", 0.9)));
301
- await expect(classifyTier(ctxWith(), config, "fix the parser")).resolves.toBe("complex");
302
- });
303
-
304
- it("returns undefined when the classifier is unusable", async () => {
305
- const noTemplate = { modelRegistry: { getAll: () => [] } } as unknown as ExtensionContext;
306
- await expect(classifyTier(noTemplate, config, "hi")).resolves.toBeUndefined();
307
-
308
- const noAuth = ctxWith({
309
- modelRegistry: {
310
- getAll: () => [templateModel()],
311
- getApiKeyAndHeaders: vi.fn().mockResolvedValue({ ok: false, error: "no key" }),
312
- } as unknown as ExtensionContext["modelRegistry"],
313
- });
314
- await expect(classifyTier(noAuth, config, "hi")).resolves.toBeUndefined();
315
- });
316
- });
317
-
318
- describe("auto-model extension", () => {
319
- it("registers the session, model, and input handlers", () => {
320
- const { handlers } = createHarness();
321
- expect([...handlers.keys()]).toEqual(["session_start", "model_select", "input"]);
322
- });
323
-
324
- it("routes an idle prompt to the classified tier model", async () => {
325
- completeMock.mockResolvedValue(textResponse(jevAnswer("COMPLEX", 0.9)));
326
- const { handlers, ctx, setModel, setThinkingLevel } = createHarness();
327
- await handlers.get("input")!({ type: "input", text: "fix the parser", source: "interactive" }, ctx);
328
- expect(setModel).toHaveBeenCalledWith(expect.objectContaining({ provider: "tokenin" }));
329
- expect(setThinkingLevel).not.toHaveBeenCalled();
330
- });
331
-
332
- it("applies a per-tier thinking level when the mapping carries one", async () => {
333
- writeSettings({ autoModel: { enabled: true, tiers: { complex: "tokenin/celestial-pro:max" } } });
334
- completeMock.mockResolvedValue(textResponse(jevAnswer("COMPLEX", 0.9)));
335
- const { handlers, ctx, setThinkingLevel } = createHarness();
336
- await handlers.get("input")!({ type: "input", text: "fix", source: "interactive" }, ctx);
337
- expect(setThinkingLevel).toHaveBeenCalledWith("max");
338
- });
339
-
340
- it("skips non-eligible input and disabled routing", async () => {
341
- const { handlers, ctx, setModel } = createHarness();
342
- const input = handlers.get("input")!;
343
- await input({ type: "input", text: "hi", source: "extension" }, ctx);
344
- await input({ type: "input", text: "hi", source: "interactive", streamingBehavior: "steer" }, ctx);
345
- await input({ type: "input", text: " ", source: "interactive" }, ctx);
346
- await input({ type: "input", text: "/model", source: "interactive" }, ctx);
347
- writeSettings({ autoModel: { enabled: false } });
348
- await input({ type: "input", text: "hi", source: "interactive" }, ctx);
349
- expect(setModel).not.toHaveBeenCalled();
350
- expect(completeMock).not.toHaveBeenCalled();
351
- });
352
-
353
- it("suspends after a manual model change and resumes on the next session", async () => {
354
- completeMock.mockResolvedValue(textResponse(jevAnswer("SIMPLE", 0.9)));
355
- const { handlers, ctx, setModel } = createHarness();
356
- await handlers.get("model_select")!({ type: "model_select", model: templateModel(), source: "cycle" }, ctx);
357
- await handlers.get("input")!({ type: "input", text: "hi", source: "interactive" }, ctx);
358
- expect(setModel).not.toHaveBeenCalled();
359
-
360
- handlers.get("session_start")!({ type: "session_start" }, ctx);
361
- await handlers.get("input")!({ type: "input", text: "hi", source: "interactive" }, ctx);
362
- expect(setModel).toHaveBeenCalledTimes(1);
363
- });
364
-
365
- it("ignores its own routing switch and session restore", async () => {
366
- completeMock.mockResolvedValue(textResponse(jevAnswer("SIMPLE", 0.9)));
367
- const { handlers, ctx, setModel } = createHarness();
368
- const modelSelect = handlers.get("model_select")!;
369
- modelSelect({ type: "model_select", model: templateModel(), source: "restore" }, ctx);
370
- // The model_select emitted by our own setModel must not suspend the next prompt.
371
- setModel.mockImplementation(async () => {
372
- modelSelect({ type: "model_select", model: templateModel(), source: "set" }, ctx);
373
- return true;
374
- });
375
- await handlers.get("input")!({ type: "input", text: "hi", source: "interactive" }, ctx);
376
- await handlers.get("input")!({ type: "input", text: "again", source: "interactive" }, ctx);
377
- expect(setModel).toHaveBeenCalledTimes(2);
378
- });
379
-
380
- it("keeps the current model when no mapping resolves or the target is out of scope", async () => {
381
- completeMock.mockResolvedValue(textResponse(jevAnswer("COMPLEX", 0.9)));
382
- const unparsable = createHarness();
383
- writeSettings({ autoModel: { enabled: true, tiers: { complex: "no-slash-here" } } });
384
- await unparsable.handlers.get("input")!({ type: "input", text: "go", source: "interactive" }, unparsable.ctx);
385
- expect(unparsable.setModel).not.toHaveBeenCalled();
386
-
387
- writeSettings({ autoModel: { enabled: true, tiers: { complex: "tokenin/missing" } } });
388
- const missing = createHarness();
389
- await missing.handlers.get("input")!({ type: "input", text: "go", source: "interactive" }, missing.ctx);
390
- expect(missing.setModel).not.toHaveBeenCalled();
391
-
392
- const scoped = createHarness([], [{ model: templateModel("other", "x") }]);
393
- writeSettings({ autoModel: { enabled: true } });
394
- await scoped.handlers.get("input")!({ type: "input", text: "go", source: "interactive" }, scoped.ctx);
395
- expect(scoped.setModel).not.toHaveBeenCalled();
396
-
397
- const sameProvider = createHarness([], [{ model: templateModel("tokenin", "other-id") }]);
398
- await sameProvider.handlers.get("input")!({ type: "input", text: "go", source: "interactive" }, sameProvider.ctx);
399
- expect(sameProvider.setModel).not.toHaveBeenCalled();
400
- });
401
-
402
- it("falls back to the fallback tier and survives classifier and switch failures", async () => {
403
- writeSettings({ autoModel: { enabled: true, fallbackTier: "medium", tiers: { medium: "tokenin/celestial-pro" } } });
404
- completeMock.mockRejectedValue(new Error("network down"));
405
- const failed = createHarness();
406
- await failed.handlers.get("input")!({ type: "input", text: "go", source: "interactive" }, failed.ctx);
407
- expect(failed.setModel).toHaveBeenCalledWith(expect.objectContaining({ id: "celestial-pro" }));
408
-
409
- writeSettings({ autoModel: { enabled: true, tiers: { medium: "tokenin/celestial-pro" } } });
410
- completeMock.mockResolvedValue(textResponse(jevAnswer("COMPLEX", 0.9)));
411
- const refused = createHarness();
412
- refused.setModel.mockResolvedValue(false);
413
- await refused.handlers.get("input")!({ type: "input", text: "go", source: "interactive" }, refused.ctx);
414
- expect(refused.setThinkingLevel).not.toHaveBeenCalled();
415
- });
416
-
417
- it("serializes concurrent prompts so only one classification runs", async () => {
418
- let release: (() => void) | undefined;
419
- completeMock.mockImplementation(
420
- () =>
421
- new Promise((resolve) => {
422
- release = () => resolve(textResponse(jevAnswer("SIMPLE", 0.9)));
423
- }) as never,
424
- );
425
- const { handlers, ctx, setModel } = createHarness();
426
- const input = handlers.get("input")!;
427
- const first = input({ type: "input", text: "one", source: "interactive" }, ctx);
428
- const second = input({ type: "input", text: "two", source: "interactive" }, ctx);
429
- await vi.waitFor(() => expect(completeMock).toHaveBeenCalledTimes(1));
430
- release?.();
431
- await Promise.all([first, second]);
432
- expect(setModel).toHaveBeenCalledTimes(1);
433
- });
434
-
435
- it("exposes the tier list used by the payload", () => {
436
- expect(TIERS).toEqual(["simple", "medium", "complex", "reasoning"]);
437
- });
438
- });