@selesai/code 0.8.10 → 0.8.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +36 -0
- package/dist/core/agent-session.d.ts +24 -0
- package/dist/core/agent-session.js +151 -1
- package/dist/core/package-manager.d.ts +1 -0
- package/dist/core/package-manager.js +8 -8
- package/dist/core/package-manager.test.js +41 -2
- package/dist/core/resource-loader.d.ts +10 -1
- package/dist/core/resource-loader.js +45 -1
- package/dist/core/settings-manager.d.ts +6 -0
- package/dist/core/settings-manager.js +27 -0
- package/dist/core/tools/read-vision.test.d.ts +1 -0
- package/dist/core/tools/read-vision.test.js +122 -0
- package/dist/core/tools/read.d.ts +13 -1
- package/dist/core/tools/read.js +51 -1
- package/dist/core/vision-caption.d.ts +30 -0
- package/dist/core/vision-caption.js +62 -0
- package/dist/defaults/models.json +28 -5
- package/dist/extensions/model-prompt-injector/config.json +1 -1
- package/dist/modes/interactive/components/settings-selector.d.ts +15 -0
- package/dist/modes/interactive/components/settings-selector.js +85 -2
- package/dist/modes/interactive/components/status-indicator.d.ts +4 -1
- package/dist/modes/interactive/components/status-indicator.js +5 -0
- package/dist/modes/interactive/interactive-mode.d.ts +4 -0
- package/dist/modes/interactive/interactive-mode.js +97 -1
- package/docs/settings.md +2 -0
- package/package.json +3 -1
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
import { beforeEach, describe, expect, it, vi } from "vitest";
|
|
2
|
+
import { captionImageWithModel } from "../vision-caption.js";
|
|
3
|
+
import { captionImage } from "./read.js";
|
|
4
|
+
const completeMock = vi.hoisted(() => vi.fn());
|
|
5
|
+
vi.mock("@earendil-works/pi-ai/compat", async () => {
|
|
6
|
+
const actual = await vi.importActual("@earendil-works/pi-ai/compat");
|
|
7
|
+
return { ...actual, complete: completeMock };
|
|
8
|
+
});
|
|
9
|
+
const image = { type: "image", data: "aGVsbG8=", mimeType: "image/png" };
|
|
10
|
+
function visionModel() {
|
|
11
|
+
return {
|
|
12
|
+
id: "kimi-k3",
|
|
13
|
+
name: "Kimi K3",
|
|
14
|
+
api: "openai-completions",
|
|
15
|
+
provider: "tokenin",
|
|
16
|
+
baseUrl: "https://example.com/v1",
|
|
17
|
+
reasoning: true,
|
|
18
|
+
input: ["text", "image"],
|
|
19
|
+
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
|
20
|
+
contextWindow: 512000,
|
|
21
|
+
maxTokens: 64000,
|
|
22
|
+
};
|
|
23
|
+
}
|
|
24
|
+
function makeCtx() {
|
|
25
|
+
const find = vi.fn().mockReturnValue(visionModel());
|
|
26
|
+
const auth = vi.fn().mockResolvedValue({ ok: true, apiKey: "sk-test", headers: {} });
|
|
27
|
+
return {
|
|
28
|
+
ctx: {
|
|
29
|
+
model: { ...visionModel(), input: ["text"] },
|
|
30
|
+
modelRegistry: { find, getApiKeyAndHeaders: auth },
|
|
31
|
+
},
|
|
32
|
+
find,
|
|
33
|
+
auth,
|
|
34
|
+
};
|
|
35
|
+
}
|
|
36
|
+
describe("captionImage", () => {
|
|
37
|
+
beforeEach(() => {
|
|
38
|
+
completeMock.mockReset();
|
|
39
|
+
});
|
|
40
|
+
it("returns null when no caption model is configured", async () => {
|
|
41
|
+
const { ctx } = makeCtx();
|
|
42
|
+
const result = await captionImage(image, undefined, ctx, undefined);
|
|
43
|
+
expect(result).toBeNull();
|
|
44
|
+
expect(completeMock).not.toHaveBeenCalled();
|
|
45
|
+
});
|
|
46
|
+
it("returns null when the caption model cannot accept images", async () => {
|
|
47
|
+
const { ctx, find } = makeCtx();
|
|
48
|
+
find.mockReturnValue({ ...visionModel(), input: ["text"] });
|
|
49
|
+
const result = await captionImage(image, "tokenin/kimi-k3", ctx, undefined);
|
|
50
|
+
expect(result).toBeNull();
|
|
51
|
+
expect(completeMock).not.toHaveBeenCalled();
|
|
52
|
+
});
|
|
53
|
+
it("returns null when credentials are unavailable", async () => {
|
|
54
|
+
const { ctx, auth } = makeCtx();
|
|
55
|
+
auth.mockResolvedValue({ ok: false, error: "no key" });
|
|
56
|
+
const result = await captionImage(image, "tokenin/kimi-k3", ctx, undefined);
|
|
57
|
+
expect(result).toBeNull();
|
|
58
|
+
});
|
|
59
|
+
it("returns the caption text from the vision model", async () => {
|
|
60
|
+
completeMock.mockResolvedValue({
|
|
61
|
+
stopReason: "end_turn",
|
|
62
|
+
content: [{ type: "text", text: "A login form with a username field." }],
|
|
63
|
+
});
|
|
64
|
+
const { ctx, find } = makeCtx();
|
|
65
|
+
const result = await captionImage(image, "tokenin/kimi-k3", ctx, undefined);
|
|
66
|
+
expect(result).toBe("A login form with a username field.");
|
|
67
|
+
expect(find).toHaveBeenCalledWith("tokenin", "kimi-k3");
|
|
68
|
+
// The image block must be forwarded to the vision model.
|
|
69
|
+
const sentMessages = completeMock.mock.calls[0][1].messages;
|
|
70
|
+
expect(sentMessages[0].content).toContainEqual(image);
|
|
71
|
+
});
|
|
72
|
+
it("returns null and does not throw when the vision request errors", async () => {
|
|
73
|
+
completeMock.mockRejectedValue(new Error("boom"));
|
|
74
|
+
const { ctx } = makeCtx();
|
|
75
|
+
const result = await captionImage(image, "tokenin/kimi-k3", ctx, undefined);
|
|
76
|
+
expect(result).toBeNull();
|
|
77
|
+
});
|
|
78
|
+
});
|
|
79
|
+
describe("captionImageWithModel", () => {
|
|
80
|
+
beforeEach(() => {
|
|
81
|
+
completeMock.mockReset();
|
|
82
|
+
});
|
|
83
|
+
it("returns the caption text with a minimal context", async () => {
|
|
84
|
+
completeMock.mockResolvedValue({
|
|
85
|
+
stopReason: "end_turn",
|
|
86
|
+
content: [{ type: "text", text: "A red button labeled Save." }],
|
|
87
|
+
});
|
|
88
|
+
const result = await captionImageWithModel(visionModel(), image, {
|
|
89
|
+
apiKey: "sk-test",
|
|
90
|
+
});
|
|
91
|
+
expect(result).toBe("A red button labeled Save.");
|
|
92
|
+
const sentContext = completeMock.mock.calls[0][1];
|
|
93
|
+
// The caption request must not include any main-agent conversation history.
|
|
94
|
+
expect(sentContext.messages).toHaveLength(1);
|
|
95
|
+
expect(sentContext.messages[0].content).toContainEqual(image);
|
|
96
|
+
expect(completeMock.mock.calls[0][2]).toMatchObject({ apiKey: "sk-test" });
|
|
97
|
+
});
|
|
98
|
+
it("returns null on aborted stop reason", async () => {
|
|
99
|
+
completeMock.mockResolvedValue({ stopReason: "aborted", content: [] });
|
|
100
|
+
const result = await captionImageWithModel(visionModel(), image);
|
|
101
|
+
expect(result).toBeNull();
|
|
102
|
+
});
|
|
103
|
+
it("returns null on request failure", async () => {
|
|
104
|
+
completeMock.mockRejectedValue(new Error("network down"));
|
|
105
|
+
const result = await captionImageWithModel(visionModel(), image);
|
|
106
|
+
expect(result).toBeNull();
|
|
107
|
+
});
|
|
108
|
+
it("includes the user prompt and bounded context text", async () => {
|
|
109
|
+
completeMock.mockResolvedValue({
|
|
110
|
+
stopReason: "end_turn",
|
|
111
|
+
content: [{ type: "text", text: "A slider" }],
|
|
112
|
+
});
|
|
113
|
+
const result = await captionImageWithModel(visionModel(), image, {
|
|
114
|
+
userPrompt: "make this UI element bigger",
|
|
115
|
+
contextText: "User: we are fixing the settings panel",
|
|
116
|
+
});
|
|
117
|
+
expect(result).toBe("A slider");
|
|
118
|
+
const sentText = completeMock.mock.calls[0][1].messages[0].content[0].text;
|
|
119
|
+
expect(sentText).toContain("make this UI element bigger");
|
|
120
|
+
expect(sentText).toContain("we are fixing the settings panel");
|
|
121
|
+
});
|
|
122
|
+
});
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import type { AgentTool } from "@earendil-works/pi-agent-core";
|
|
2
|
+
import type { ImageContent } from "@earendil-works/pi-ai";
|
|
2
3
|
import { type Static, Type } from "typebox";
|
|
3
|
-
import type { ToolDefinition } from "../extensions/types.ts";
|
|
4
|
+
import type { ExtensionContext, ToolDefinition } from "../extensions/types.ts";
|
|
4
5
|
import { type TruncationResult } from "./truncate.ts";
|
|
5
6
|
declare const readSchema: Type.TObject<{
|
|
6
7
|
path: Type.TString;
|
|
@@ -26,9 +27,20 @@ export interface ReadOperations {
|
|
|
26
27
|
export interface ReadToolOptions {
|
|
27
28
|
/** Whether to auto-resize images to 2000x2000 max. Default: true */
|
|
28
29
|
autoResizeImages?: boolean;
|
|
30
|
+
/**
|
|
31
|
+
* Vision model ("provider/modelId") used to describe images when the active
|
|
32
|
+
* model cannot accept image input. The caption text is returned in place of
|
|
33
|
+
* the image block. Default: unset (images are omitted as before).
|
|
34
|
+
*/
|
|
35
|
+
imageCaptionModel?: string;
|
|
29
36
|
/** Custom operations for file reading. Default: local filesystem */
|
|
30
37
|
operations?: ReadOperations;
|
|
31
38
|
}
|
|
39
|
+
/**
|
|
40
|
+
* Ask a vision model to describe an image. Returns the caption text, or null if
|
|
41
|
+
* the caption model/credentials are unavailable or the request fails.
|
|
42
|
+
*/
|
|
43
|
+
export declare function captionImage(image: ImageContent, captionModelId: string | undefined, ctx: ExtensionContext | undefined, signal: AbortSignal | undefined): Promise<string | null>;
|
|
32
44
|
export declare function createReadToolDefinition(cwd: string, options?: ReadToolOptions): ToolDefinition<typeof readSchema, ReadToolDetails | undefined>;
|
|
33
45
|
export declare function createReadTool(cwd: string, options?: ReadToolOptions): AgentTool<typeof readSchema>;
|
|
34
46
|
export {};
|
package/dist/core/tools/read.js
CHANGED
|
@@ -10,6 +10,7 @@ import { processImage } from "../../utils/image-process.js";
|
|
|
10
10
|
import { detectSupportedImageMimeTypeFromFile } from "../../utils/mime.js";
|
|
11
11
|
import { formatPathRelativeToCwdOrAbsolute } from "../../utils/paths.js";
|
|
12
12
|
import { getExperimentalToolSampling } from "../experimental.js";
|
|
13
|
+
import { captionImageWithModel } from "../vision-caption.js";
|
|
13
14
|
import { resolveReadPathAsync, resolveToCwd } from "./path-utils.js";
|
|
14
15
|
import { getTextOutput, renderToolPath, replaceTabs, str } from "./render-utils.js";
|
|
15
16
|
import { wrapToolDefinition } from "./tool-definition-wrapper.js";
|
|
@@ -49,6 +50,36 @@ function getNonVisionImageNote(model) {
|
|
|
49
50
|
}
|
|
50
51
|
return "[Current model does not support images. The image will be omitted from this request.]";
|
|
51
52
|
}
|
|
53
|
+
/**
|
|
54
|
+
* Ask a vision model to describe an image. Returns the caption text, or null if
|
|
55
|
+
* the caption model/credentials are unavailable or the request fails.
|
|
56
|
+
*/
|
|
57
|
+
export async function captionImage(image, captionModelId, ctx, signal) {
|
|
58
|
+
if (!captionModelId || !ctx?.modelRegistry || !ctx.model)
|
|
59
|
+
return null;
|
|
60
|
+
const slash = captionModelId.indexOf("/");
|
|
61
|
+
if (slash <= 0 || slash === captionModelId.length - 1)
|
|
62
|
+
return null;
|
|
63
|
+
const provider = captionModelId.slice(0, slash);
|
|
64
|
+
const modelId = captionModelId.slice(slash + 1);
|
|
65
|
+
const captionModel = ctx.modelRegistry.find(provider, modelId);
|
|
66
|
+
if (!captionModel || !captionModel.input.includes("image"))
|
|
67
|
+
return null;
|
|
68
|
+
let auth;
|
|
69
|
+
try {
|
|
70
|
+
auth = await ctx.modelRegistry.getApiKeyAndHeaders(captionModel);
|
|
71
|
+
}
|
|
72
|
+
catch {
|
|
73
|
+
return null;
|
|
74
|
+
}
|
|
75
|
+
if (!auth.ok)
|
|
76
|
+
return null;
|
|
77
|
+
return captionImageWithModel(captionModel, image, {
|
|
78
|
+
apiKey: auth.apiKey,
|
|
79
|
+
headers: auth.headers,
|
|
80
|
+
signal,
|
|
81
|
+
});
|
|
82
|
+
}
|
|
52
83
|
function toPosixPath(filePath) {
|
|
53
84
|
return filePath.split(sep).join("/");
|
|
54
85
|
}
|
|
@@ -130,6 +161,7 @@ function formatReadResult(args, result, options, theme, showImages, _cwd, isErro
|
|
|
130
161
|
}
|
|
131
162
|
export function createReadToolDefinition(cwd, options) {
|
|
132
163
|
const autoResizeImages = options?.autoResizeImages ?? true;
|
|
164
|
+
const imageCaptionModel = options?.imageCaptionModel;
|
|
133
165
|
const ops = options?.operations ?? defaultReadOperations;
|
|
134
166
|
return {
|
|
135
167
|
name: "read",
|
|
@@ -178,11 +210,29 @@ export function createReadToolDefinition(cwd, options) {
|
|
|
178
210
|
let textNote = `Read image file [${processed.mimeType}]`;
|
|
179
211
|
if (processed.hints.length > 0)
|
|
180
212
|
textNote += `\n${processed.hints.join("\n")}`;
|
|
213
|
+
const image = { type: "image", data: processed.data, mimeType: processed.mimeType };
|
|
214
|
+
// Vision relay: when the active model cannot accept images, ask a
|
|
215
|
+
// configured vision model to describe the image and use that text instead.
|
|
216
|
+
if (nonVisionImageNote) {
|
|
217
|
+
const caption = await captionImage(image, imageCaptionModel, ctx, signal);
|
|
218
|
+
if (caption) {
|
|
219
|
+
content = [
|
|
220
|
+
{
|
|
221
|
+
type: "text",
|
|
222
|
+
text: `Read image file [${processed.mimeType}] (described by vision model)\n${caption}`,
|
|
223
|
+
},
|
|
224
|
+
];
|
|
225
|
+
if (aborted)
|
|
226
|
+
return;
|
|
227
|
+
resolve({ content, details });
|
|
228
|
+
return;
|
|
229
|
+
}
|
|
230
|
+
}
|
|
181
231
|
if (nonVisionImageNote)
|
|
182
232
|
textNote += `\n${nonVisionImageNote}`;
|
|
183
233
|
content = [
|
|
184
234
|
{ type: "text", text: textNote },
|
|
185
|
-
|
|
235
|
+
image,
|
|
186
236
|
];
|
|
187
237
|
}
|
|
188
238
|
}
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Shared vision-caption relay: ask a vision-capable model to describe an image
|
|
3
|
+
* as plain text, so a text-only main model can still work from it.
|
|
4
|
+
*/
|
|
5
|
+
import type { Api, ImageContent, Model, ProviderHeaders } from "@earendil-works/pi-ai";
|
|
6
|
+
export declare const VISION_CAPTION_SYSTEM_PROMPT: string;
|
|
7
|
+
export interface VisionCaptionRequestOptions {
|
|
8
|
+
apiKey?: string;
|
|
9
|
+
headers?: ProviderHeaders;
|
|
10
|
+
signal?: AbortSignal;
|
|
11
|
+
/** Called with the underlying error message when the caption request fails. */
|
|
12
|
+
onError?: (message: string) => void;
|
|
13
|
+
/**
|
|
14
|
+
* The user's current instruction (e.g. "make this UI element bigger").
|
|
15
|
+
* Included so the caption is targeted at the task, not generic.
|
|
16
|
+
*/
|
|
17
|
+
userPrompt?: string;
|
|
18
|
+
/**
|
|
19
|
+
* A small, already-bounded slice of recent conversation text (or other relevant
|
|
20
|
+
* context) to help the caption model understand intent. Must be small enough to
|
|
21
|
+
* stay safely under the caption model's context window. Omit for none.
|
|
22
|
+
*/
|
|
23
|
+
contextText?: string;
|
|
24
|
+
}
|
|
25
|
+
/**
|
|
26
|
+
* Ask a vision model to describe an image. Returns the caption text, or null if
|
|
27
|
+
* the request fails or is aborted. The request context is minimal (prompt +
|
|
28
|
+
* image only), so it is independent of any main-agent context usage.
|
|
29
|
+
*/
|
|
30
|
+
export declare function captionImageWithModel(captionModel: Model<Api>, image: ImageContent, options?: VisionCaptionRequestOptions): Promise<string | null>;
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Shared vision-caption relay: ask a vision-capable model to describe an image
|
|
3
|
+
* as plain text, so a text-only main model can still work from it.
|
|
4
|
+
*/
|
|
5
|
+
import { complete } from "@earendil-works/pi-ai/compat";
|
|
6
|
+
export const VISION_CAPTION_SYSTEM_PROMPT = "You are an image captioner for a coding agent whose main model cannot see images. " +
|
|
7
|
+
"Describe the image as completely and precisely as possible in plain text so that a programmer " +
|
|
8
|
+
"can work from your description alone, as if looking at the image themselves. " +
|
|
9
|
+
"Read any visible text, code, UI labels, error messages, terminal output, or diagrams verbatim. " +
|
|
10
|
+
"Cover layout, colors, spatial relationships, relative positions and sizes of elements, " +
|
|
11
|
+
"window or panel structure, and anything relevant to debugging or implementing. " +
|
|
12
|
+
"When a simple ASCII sketch clarifies the layout, include it.";
|
|
13
|
+
function buildUserPrompt(userPrompt, contextText) {
|
|
14
|
+
const lines = [];
|
|
15
|
+
if (userPrompt) {
|
|
16
|
+
lines.push(`The user said:\n${userPrompt}`);
|
|
17
|
+
}
|
|
18
|
+
if (contextText) {
|
|
19
|
+
lines.push(`Relevant context (recent conversation):\n${contextText}`);
|
|
20
|
+
}
|
|
21
|
+
lines.push("Describe this image in detail:");
|
|
22
|
+
return lines.join("\n\n");
|
|
23
|
+
}
|
|
24
|
+
/**
|
|
25
|
+
* Ask a vision model to describe an image. Returns the caption text, or null if
|
|
26
|
+
* the request fails or is aborted. The request context is minimal (prompt +
|
|
27
|
+
* image only), so it is independent of any main-agent context usage.
|
|
28
|
+
*/
|
|
29
|
+
export async function captionImageWithModel(captionModel, image, options) {
|
|
30
|
+
const aiContext = {
|
|
31
|
+
systemPrompt: VISION_CAPTION_SYSTEM_PROMPT,
|
|
32
|
+
messages: [
|
|
33
|
+
{
|
|
34
|
+
role: "user",
|
|
35
|
+
timestamp: Date.now(),
|
|
36
|
+
content: [
|
|
37
|
+
{ type: "text", text: buildUserPrompt(options?.userPrompt, options?.contextText) },
|
|
38
|
+
image,
|
|
39
|
+
],
|
|
40
|
+
},
|
|
41
|
+
],
|
|
42
|
+
};
|
|
43
|
+
try {
|
|
44
|
+
const response = await complete(captionModel, aiContext, {
|
|
45
|
+
apiKey: options?.apiKey,
|
|
46
|
+
headers: options?.headers,
|
|
47
|
+
signal: options?.signal,
|
|
48
|
+
});
|
|
49
|
+
if (response.stopReason === "aborted")
|
|
50
|
+
return null;
|
|
51
|
+
const text = response.content
|
|
52
|
+
.filter((c) => c.type === "text")
|
|
53
|
+
.map((c) => c.text)
|
|
54
|
+
.join("\n")
|
|
55
|
+
.trim();
|
|
56
|
+
return text.length > 0 ? text : null;
|
|
57
|
+
}
|
|
58
|
+
catch (error) {
|
|
59
|
+
options?.onError?.(error instanceof Error ? error.message : String(error));
|
|
60
|
+
return null;
|
|
61
|
+
}
|
|
62
|
+
}
|
|
@@ -24,11 +24,22 @@
|
|
|
24
24
|
"high": null,
|
|
25
25
|
"xhigh": null,
|
|
26
26
|
"max": "max"
|
|
27
|
-
}
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
27
|
+
}
|
|
28
|
+
},
|
|
29
|
+
{
|
|
30
|
+
"id": "auto-premium",
|
|
31
|
+
"name": "Auto Premium",
|
|
32
|
+
"reasoning": true,
|
|
33
|
+
"input": ["text"],
|
|
34
|
+
"contextWindow": 512000,
|
|
35
|
+
"maxTokens": 64000,
|
|
36
|
+
"thinkingLevelMap": {
|
|
37
|
+
"minimal": null,
|
|
38
|
+
"low": null,
|
|
39
|
+
"medium": null,
|
|
40
|
+
"high": null,
|
|
41
|
+
"xhigh": null,
|
|
42
|
+
"max": "max"
|
|
32
43
|
}
|
|
33
44
|
},
|
|
34
45
|
{
|
|
@@ -132,6 +143,18 @@
|
|
|
132
143
|
"supportsDeveloperRole": false,
|
|
133
144
|
"supportsReasoningEffort": true
|
|
134
145
|
}
|
|
146
|
+
},
|
|
147
|
+
{
|
|
148
|
+
"id": "gemma-4",
|
|
149
|
+
"name": "Gemma 4 31B (Vision)",
|
|
150
|
+
"reasoning": false,
|
|
151
|
+
"input": ["text", "image"],
|
|
152
|
+
"contextWindow": 256000,
|
|
153
|
+
"maxTokens": 8192,
|
|
154
|
+
"compat": {
|
|
155
|
+
"supportsDeveloperRole": false,
|
|
156
|
+
"supportsReasoningEffort": false
|
|
157
|
+
}
|
|
135
158
|
}
|
|
136
159
|
]
|
|
137
160
|
}
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"rules": [
|
|
3
3
|
{
|
|
4
|
-
"match": ["deepseek-v4
|
|
4
|
+
"match": ["deepseek-v4-*"],
|
|
5
5
|
"mode": "prepend",
|
|
6
6
|
"prompt": "text: You are a helpful software engineer assistant.\n\ndescription: Bootstrap with shell/read, then expose the full Standard tool catalog after the first durable tool call.\n\nwhen you thought, start with `we need...`",
|
|
7
7
|
"enabled": true
|
|
@@ -3,6 +3,13 @@ import type { Transport } from "@earendil-works/pi-ai";
|
|
|
3
3
|
import { Container, type ScrollViewScrollbar, SettingsList } from "@earendil-works/pi-tui";
|
|
4
4
|
import type { DefaultProjectTrust, FullscreenExitOutput, TuiMode, WarningSettings } from "../../../core/settings-manager.ts";
|
|
5
5
|
import { type TerminalTheme } from "../theme/theme.ts";
|
|
6
|
+
export interface SkillToggleItem {
|
|
7
|
+
name: string;
|
|
8
|
+
enabled: boolean;
|
|
9
|
+
pattern: string;
|
|
10
|
+
scope: "user" | "project";
|
|
11
|
+
id: string;
|
|
12
|
+
}
|
|
6
13
|
export interface SettingsConfig {
|
|
7
14
|
autoCompact: boolean;
|
|
8
15
|
autoHandoffEnabled: boolean;
|
|
@@ -11,6 +18,9 @@ export interface SettingsConfig {
|
|
|
11
18
|
imageWidthCells: number;
|
|
12
19
|
autoResizeImages: boolean;
|
|
13
20
|
blockImages: boolean;
|
|
21
|
+
imageCaptionModel: string | undefined;
|
|
22
|
+
imageCaptionContextTokens: number;
|
|
23
|
+
visionModels: string[];
|
|
14
24
|
steeringMode: "all" | "one-at-a-time";
|
|
15
25
|
followUpMode: "all" | "one-at-a-time";
|
|
16
26
|
transport: Transport;
|
|
@@ -38,6 +48,7 @@ export interface SettingsConfig {
|
|
|
38
48
|
fullscreenExitOutput: FullscreenExitOutput;
|
|
39
49
|
fullscreenScrollbar: ScrollViewScrollbar;
|
|
40
50
|
warnings: WarningSettings;
|
|
51
|
+
skills: SkillToggleItem[];
|
|
41
52
|
}
|
|
42
53
|
export interface SettingsCallbacks {
|
|
43
54
|
onAutoCompactChange: (enabled: boolean) => void;
|
|
@@ -47,6 +58,8 @@ export interface SettingsCallbacks {
|
|
|
47
58
|
onImageWidthCellsChange: (width: number) => void;
|
|
48
59
|
onAutoResizeImagesChange: (enabled: boolean) => void;
|
|
49
60
|
onBlockImagesChange: (blocked: boolean) => void;
|
|
61
|
+
onImageCaptionModelChange: (modelId: string | undefined) => void;
|
|
62
|
+
onImageCaptionContextTokensChange: (tokens: number) => void;
|
|
50
63
|
onSteeringModeChange: (mode: "all" | "one-at-a-time") => void;
|
|
51
64
|
onFollowUpModeChange: (mode: "all" | "one-at-a-time") => void;
|
|
52
65
|
onTransportChange: (transport: Transport) => void;
|
|
@@ -72,6 +85,8 @@ export interface SettingsCallbacks {
|
|
|
72
85
|
onFullscreenExitOutputChange: (output: FullscreenExitOutput) => void;
|
|
73
86
|
onFullscreenScrollbarChange: (mode: ScrollViewScrollbar) => void;
|
|
74
87
|
onWarningsChange: (warnings: WarningSettings) => void;
|
|
88
|
+
onSkillsToggle: (item: SkillToggleItem, enabled: boolean) => void;
|
|
89
|
+
onSkillsReload: () => void;
|
|
75
90
|
onCancel: () => void;
|
|
76
91
|
}
|
|
77
92
|
/**
|
|
@@ -22,6 +22,46 @@ const DEFAULT_PROJECT_TRUST_LABELS = {
|
|
|
22
22
|
never: "Never trust",
|
|
23
23
|
};
|
|
24
24
|
const DEFAULT_PROJECT_TRUST_BY_LABEL = new Map(Object.entries(DEFAULT_PROJECT_TRUST_LABELS).map(([value, label]) => [label, value]));
|
|
25
|
+
class SkillsSubmenu extends Container {
|
|
26
|
+
settingsList;
|
|
27
|
+
constructor(skills, onToggle, onCancel) {
|
|
28
|
+
super();
|
|
29
|
+
const initial = skills.map((s) => ({ ...s }));
|
|
30
|
+
const nameCounts = new Map();
|
|
31
|
+
for (const skill of initial) {
|
|
32
|
+
nameCounts.set(skill.name, (nameCounts.get(skill.name) ?? 0) + 1);
|
|
33
|
+
}
|
|
34
|
+
const items = initial.map((skill) => ({
|
|
35
|
+
id: skill.id,
|
|
36
|
+
label: (nameCounts.get(skill.name) ?? 0) > 1 ? `${skill.name} (${skill.pattern})` : skill.name,
|
|
37
|
+
description: `Scope: ${skill.scope}${(nameCounts.get(skill.name) ?? 0) > 1 ? ` — ${skill.pattern}` : ""}`,
|
|
38
|
+
currentValue: skill.enabled ? "true" : "false",
|
|
39
|
+
values: ["true", "false"],
|
|
40
|
+
}));
|
|
41
|
+
if (items.length === 0) {
|
|
42
|
+
items.push({
|
|
43
|
+
id: "no-skills",
|
|
44
|
+
label: "No skills discovered",
|
|
45
|
+
currentValue: "",
|
|
46
|
+
});
|
|
47
|
+
}
|
|
48
|
+
this.settingsList = new SettingsList(items, Math.min(items.length, 10), getSettingsListTheme(), (id, newValue) => {
|
|
49
|
+
const skill = initial.find((s) => s.id === id);
|
|
50
|
+
if (skill) {
|
|
51
|
+
const enabled = newValue === "true";
|
|
52
|
+
skill.enabled = enabled;
|
|
53
|
+
const source = skills.find((s) => s.id === id);
|
|
54
|
+
if (source)
|
|
55
|
+
source.enabled = enabled;
|
|
56
|
+
onToggle(skill, enabled);
|
|
57
|
+
}
|
|
58
|
+
}, onCancel);
|
|
59
|
+
this.addChild(this.settingsList);
|
|
60
|
+
}
|
|
61
|
+
handleInput(data) {
|
|
62
|
+
this.settingsList.handleInput(data);
|
|
63
|
+
}
|
|
64
|
+
}
|
|
25
65
|
/**
|
|
26
66
|
* A submenu component for selecting from a list of options.
|
|
27
67
|
*/
|
|
@@ -272,6 +312,7 @@ export class SettingsSelectorComponent extends Container {
|
|
|
272
312
|
const supportsImages = getCapabilities().images;
|
|
273
313
|
const followUpKey = keyDisplayText("app.message.followUp");
|
|
274
314
|
let currentWarnings = { ...config.warnings };
|
|
315
|
+
let skillsChanged = false;
|
|
275
316
|
const items = [
|
|
276
317
|
{
|
|
277
318
|
id: "autocompact",
|
|
@@ -378,6 +419,16 @@ export class SettingsSelectorComponent extends Container {
|
|
|
378
419
|
currentValue: config.treeFilterMode,
|
|
379
420
|
values: ["default", "no-tools", "user-only", "labeled-only", "all"],
|
|
380
421
|
},
|
|
422
|
+
{
|
|
423
|
+
id: "skills",
|
|
424
|
+
label: "Skills",
|
|
425
|
+
description: "Choose which skills are loaded (opt-in). Skills apply and reload when you close this menu.",
|
|
426
|
+
currentValue: `${config.skills.filter((s) => s.enabled).length}/${config.skills.length} enabled`,
|
|
427
|
+
submenu: (_currentValue, done) => new SkillsSubmenu(config.skills, (item, enabled) => {
|
|
428
|
+
callbacks.onSkillsToggle(item, enabled);
|
|
429
|
+
skillsChanged = true;
|
|
430
|
+
}, () => done()),
|
|
431
|
+
},
|
|
381
432
|
{
|
|
382
433
|
id: "warnings",
|
|
383
434
|
label: "Warnings",
|
|
@@ -466,9 +517,30 @@ export class SettingsSelectorComponent extends Container {
|
|
|
466
517
|
currentValue: config.blockImages ? "true" : "false",
|
|
467
518
|
values: ["true", "false"],
|
|
468
519
|
});
|
|
469
|
-
//
|
|
520
|
+
// Vision caption model picker (image -> text relay for text-only main models).
|
|
521
|
+
// Values are the available vision-capable models plus "off" for none.
|
|
522
|
+
const visionModelValues = ["off", ...config.visionModels];
|
|
523
|
+
const currentCaptionModel = config.imageCaptionModel ?? "off";
|
|
470
524
|
const blockImagesIndex = items.findIndex((item) => item.id === "block-images");
|
|
471
525
|
items.splice(blockImagesIndex + 1, 0, {
|
|
526
|
+
id: "image-caption-model",
|
|
527
|
+
label: "Vision caption model",
|
|
528
|
+
description: "Model used to describe images when the main model can't see them (off disables)",
|
|
529
|
+
currentValue: currentCaptionModel,
|
|
530
|
+
values: visionModelValues,
|
|
531
|
+
});
|
|
532
|
+
// Vision caption context-token budget
|
|
533
|
+
const captionModelIndex = items.findIndex((item) => item.id === "image-caption-model");
|
|
534
|
+
items.splice(captionModelIndex + 1, 0, {
|
|
535
|
+
id: "image-caption-context-tokens",
|
|
536
|
+
label: "Vision context tokens",
|
|
537
|
+
description: "Budget (approx) for recent-conversation context sent to the vision model",
|
|
538
|
+
currentValue: String(config.imageCaptionContextTokens),
|
|
539
|
+
values: ["0", "4096", "16384", "32768", "65536"],
|
|
540
|
+
});
|
|
541
|
+
// Hardware cursor toggle (insert after the image-caption settings)
|
|
542
|
+
const visionSettingsIndex = items.findIndex((item) => item.id === "image-caption-context-tokens");
|
|
543
|
+
items.splice(visionSettingsIndex + 1, 0, {
|
|
472
544
|
id: "show-hardware-cursor",
|
|
473
545
|
label: "Show hardware cursor",
|
|
474
546
|
description: "Show the terminal cursor while still positioning it for IME support",
|
|
@@ -545,6 +617,12 @@ export class SettingsSelectorComponent extends Container {
|
|
|
545
617
|
case "block-images":
|
|
546
618
|
callbacks.onBlockImagesChange(newValue === "true");
|
|
547
619
|
break;
|
|
620
|
+
case "image-caption-model":
|
|
621
|
+
callbacks.onImageCaptionModelChange(newValue === "off" ? undefined : newValue);
|
|
622
|
+
break;
|
|
623
|
+
case "image-caption-context-tokens":
|
|
624
|
+
callbacks.onImageCaptionContextTokensChange(parseInt(newValue, 10) || 0);
|
|
625
|
+
break;
|
|
548
626
|
case "steering-mode":
|
|
549
627
|
callbacks.onSteeringModeChange(newValue);
|
|
550
628
|
break;
|
|
@@ -620,7 +698,12 @@ export class SettingsSelectorComponent extends Container {
|
|
|
620
698
|
callbacks.onThemeChange(newValue);
|
|
621
699
|
break;
|
|
622
700
|
}
|
|
623
|
-
},
|
|
701
|
+
}, () => {
|
|
702
|
+
if (skillsChanged) {
|
|
703
|
+
callbacks.onSkillsReload();
|
|
704
|
+
}
|
|
705
|
+
callbacks.onCancel();
|
|
706
|
+
}, { enableSearch: true });
|
|
624
707
|
this.addChild(this.settingsList);
|
|
625
708
|
this.addChild(new DynamicBorder());
|
|
626
709
|
}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { type Component, Loader, type TUI } from "@earendil-works/pi-tui";
|
|
2
2
|
import type { WorkingIndicatorOptions } from "../../../core/extensions/index.ts";
|
|
3
|
-
export type StatusIndicatorKind = "working" | "retry" | "compaction" | "branchSummary";
|
|
3
|
+
export type StatusIndicatorKind = "working" | "retry" | "compaction" | "branchSummary" | "imageCaptioning";
|
|
4
4
|
export declare class StatusIndicator extends Loader {
|
|
5
5
|
readonly kind: StatusIndicatorKind;
|
|
6
6
|
constructor(kind: StatusIndicatorKind, ui: TUI, spinnerColorFn: (str: string) => string, messageColorFn: (str: string) => string, message: string, indicator?: WorkingIndicatorOptions);
|
|
@@ -9,6 +9,9 @@ export declare class StatusIndicator extends Loader {
|
|
|
9
9
|
export declare class WorkingStatusIndicator extends StatusIndicator {
|
|
10
10
|
constructor(ui: TUI, message: string, indicator?: WorkingIndicatorOptions);
|
|
11
11
|
}
|
|
12
|
+
export declare class ImageCaptioningStatusIndicator extends StatusIndicator {
|
|
13
|
+
constructor(ui: TUI);
|
|
14
|
+
}
|
|
12
15
|
export declare class RetryStatusIndicator extends StatusIndicator {
|
|
13
16
|
private countdown;
|
|
14
17
|
constructor(ui: TUI, attempt: number, maxAttempts: number, delayMs: number);
|
|
@@ -17,6 +17,11 @@ export class WorkingStatusIndicator extends StatusIndicator {
|
|
|
17
17
|
super("working", ui, (spinner) => theme.fg("accent", spinner), (text) => theme.fg("muted", text), message, indicator);
|
|
18
18
|
}
|
|
19
19
|
}
|
|
20
|
+
export class ImageCaptioningStatusIndicator extends StatusIndicator {
|
|
21
|
+
constructor(ui) {
|
|
22
|
+
super("imageCaptioning", ui, (spinner) => theme.fg("accent", spinner), (text) => theme.fg("muted", text), "Reading image with vision model...");
|
|
23
|
+
}
|
|
24
|
+
}
|
|
20
25
|
export class RetryStatusIndicator extends StatusIndicator {
|
|
21
26
|
countdown;
|
|
22
27
|
constructor(ui, attempt, maxAttempts, delayMs) {
|
|
@@ -373,6 +373,10 @@ export declare class InteractiveMode {
|
|
|
373
373
|
*/
|
|
374
374
|
private showSelector;
|
|
375
375
|
private showSettingsSelector;
|
|
376
|
+
private getVisionModels;
|
|
377
|
+
private buildSkillToggleItems;
|
|
378
|
+
private toggleSkillSetting;
|
|
379
|
+
private reloadAfterSkillChange;
|
|
376
380
|
private handleModelCommand;
|
|
377
381
|
private findExactModelMatch;
|
|
378
382
|
private getModelCandidates;
|