@selesai/code 0.8.10 → 0.8.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,122 @@
1
+ import { beforeEach, describe, expect, it, vi } from "vitest";
2
+ import { captionImageWithModel } from "../vision-caption.js";
3
+ import { captionImage } from "./read.js";
4
+ const completeMock = vi.hoisted(() => vi.fn());
5
+ vi.mock("@earendil-works/pi-ai/compat", async () => {
6
+ const actual = await vi.importActual("@earendil-works/pi-ai/compat");
7
+ return { ...actual, complete: completeMock };
8
+ });
9
+ const image = { type: "image", data: "aGVsbG8=", mimeType: "image/png" };
10
+ function visionModel() {
11
+ return {
12
+ id: "kimi-k3",
13
+ name: "Kimi K3",
14
+ api: "openai-completions",
15
+ provider: "tokenin",
16
+ baseUrl: "https://example.com/v1",
17
+ reasoning: true,
18
+ input: ["text", "image"],
19
+ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
20
+ contextWindow: 512000,
21
+ maxTokens: 64000,
22
+ };
23
+ }
24
+ function makeCtx() {
25
+ const find = vi.fn().mockReturnValue(visionModel());
26
+ const auth = vi.fn().mockResolvedValue({ ok: true, apiKey: "sk-test", headers: {} });
27
+ return {
28
+ ctx: {
29
+ model: { ...visionModel(), input: ["text"] },
30
+ modelRegistry: { find, getApiKeyAndHeaders: auth },
31
+ },
32
+ find,
33
+ auth,
34
+ };
35
+ }
36
+ describe("captionImage", () => {
37
+ beforeEach(() => {
38
+ completeMock.mockReset();
39
+ });
40
+ it("returns null when no caption model is configured", async () => {
41
+ const { ctx } = makeCtx();
42
+ const result = await captionImage(image, undefined, ctx, undefined);
43
+ expect(result).toBeNull();
44
+ expect(completeMock).not.toHaveBeenCalled();
45
+ });
46
+ it("returns null when the caption model cannot accept images", async () => {
47
+ const { ctx, find } = makeCtx();
48
+ find.mockReturnValue({ ...visionModel(), input: ["text"] });
49
+ const result = await captionImage(image, "tokenin/kimi-k3", ctx, undefined);
50
+ expect(result).toBeNull();
51
+ expect(completeMock).not.toHaveBeenCalled();
52
+ });
53
+ it("returns null when credentials are unavailable", async () => {
54
+ const { ctx, auth } = makeCtx();
55
+ auth.mockResolvedValue({ ok: false, error: "no key" });
56
+ const result = await captionImage(image, "tokenin/kimi-k3", ctx, undefined);
57
+ expect(result).toBeNull();
58
+ });
59
+ it("returns the caption text from the vision model", async () => {
60
+ completeMock.mockResolvedValue({
61
+ stopReason: "end_turn",
62
+ content: [{ type: "text", text: "A login form with a username field." }],
63
+ });
64
+ const { ctx, find } = makeCtx();
65
+ const result = await captionImage(image, "tokenin/kimi-k3", ctx, undefined);
66
+ expect(result).toBe("A login form with a username field.");
67
+ expect(find).toHaveBeenCalledWith("tokenin", "kimi-k3");
68
+ // The image block must be forwarded to the vision model.
69
+ const sentMessages = completeMock.mock.calls[0][1].messages;
70
+ expect(sentMessages[0].content).toContainEqual(image);
71
+ });
72
+ it("returns null and does not throw when the vision request errors", async () => {
73
+ completeMock.mockRejectedValue(new Error("boom"));
74
+ const { ctx } = makeCtx();
75
+ const result = await captionImage(image, "tokenin/kimi-k3", ctx, undefined);
76
+ expect(result).toBeNull();
77
+ });
78
+ });
79
+ describe("captionImageWithModel", () => {
80
+ beforeEach(() => {
81
+ completeMock.mockReset();
82
+ });
83
+ it("returns the caption text with a minimal context", async () => {
84
+ completeMock.mockResolvedValue({
85
+ stopReason: "end_turn",
86
+ content: [{ type: "text", text: "A red button labeled Save." }],
87
+ });
88
+ const result = await captionImageWithModel(visionModel(), image, {
89
+ apiKey: "sk-test",
90
+ });
91
+ expect(result).toBe("A red button labeled Save.");
92
+ const sentContext = completeMock.mock.calls[0][1];
93
+ // The caption request must not include any main-agent conversation history.
94
+ expect(sentContext.messages).toHaveLength(1);
95
+ expect(sentContext.messages[0].content).toContainEqual(image);
96
+ expect(completeMock.mock.calls[0][2]).toMatchObject({ apiKey: "sk-test" });
97
+ });
98
+ it("returns null on aborted stop reason", async () => {
99
+ completeMock.mockResolvedValue({ stopReason: "aborted", content: [] });
100
+ const result = await captionImageWithModel(visionModel(), image);
101
+ expect(result).toBeNull();
102
+ });
103
+ it("returns null on request failure", async () => {
104
+ completeMock.mockRejectedValue(new Error("network down"));
105
+ const result = await captionImageWithModel(visionModel(), image);
106
+ expect(result).toBeNull();
107
+ });
108
+ it("includes the user prompt and bounded context text", async () => {
109
+ completeMock.mockResolvedValue({
110
+ stopReason: "end_turn",
111
+ content: [{ type: "text", text: "A slider" }],
112
+ });
113
+ const result = await captionImageWithModel(visionModel(), image, {
114
+ userPrompt: "make this UI element bigger",
115
+ contextText: "User: we are fixing the settings panel",
116
+ });
117
+ expect(result).toBe("A slider");
118
+ const sentText = completeMock.mock.calls[0][1].messages[0].content[0].text;
119
+ expect(sentText).toContain("make this UI element bigger");
120
+ expect(sentText).toContain("we are fixing the settings panel");
121
+ });
122
+ });
@@ -1,6 +1,7 @@
1
1
  import type { AgentTool } from "@earendil-works/pi-agent-core";
2
+ import type { ImageContent } from "@earendil-works/pi-ai";
2
3
  import { type Static, Type } from "typebox";
3
- import type { ToolDefinition } from "../extensions/types.ts";
4
+ import type { ExtensionContext, ToolDefinition } from "../extensions/types.ts";
4
5
  import { type TruncationResult } from "./truncate.ts";
5
6
  declare const readSchema: Type.TObject<{
6
7
  path: Type.TString;
@@ -26,9 +27,20 @@ export interface ReadOperations {
26
27
  export interface ReadToolOptions {
27
28
  /** Whether to auto-resize images to 2000x2000 max. Default: true */
28
29
  autoResizeImages?: boolean;
30
+ /**
31
+ * Vision model ("provider/modelId") used to describe images when the active
32
+ * model cannot accept image input. The caption text is returned in place of
33
+ * the image block. Default: unset (images are omitted as before).
34
+ */
35
+ imageCaptionModel?: string;
29
36
  /** Custom operations for file reading. Default: local filesystem */
30
37
  operations?: ReadOperations;
31
38
  }
39
+ /**
40
+ * Ask a vision model to describe an image. Returns the caption text, or null if
41
+ * the caption model/credentials are unavailable or the request fails.
42
+ */
43
+ export declare function captionImage(image: ImageContent, captionModelId: string | undefined, ctx: ExtensionContext | undefined, signal: AbortSignal | undefined): Promise<string | null>;
32
44
  export declare function createReadToolDefinition(cwd: string, options?: ReadToolOptions): ToolDefinition<typeof readSchema, ReadToolDetails | undefined>;
33
45
  export declare function createReadTool(cwd: string, options?: ReadToolOptions): AgentTool<typeof readSchema>;
34
46
  export {};
@@ -10,6 +10,7 @@ import { processImage } from "../../utils/image-process.js";
10
10
  import { detectSupportedImageMimeTypeFromFile } from "../../utils/mime.js";
11
11
  import { formatPathRelativeToCwdOrAbsolute } from "../../utils/paths.js";
12
12
  import { getExperimentalToolSampling } from "../experimental.js";
13
+ import { captionImageWithModel } from "../vision-caption.js";
13
14
  import { resolveReadPathAsync, resolveToCwd } from "./path-utils.js";
14
15
  import { getTextOutput, renderToolPath, replaceTabs, str } from "./render-utils.js";
15
16
  import { wrapToolDefinition } from "./tool-definition-wrapper.js";
@@ -49,6 +50,36 @@ function getNonVisionImageNote(model) {
49
50
  }
50
51
  return "[Current model does not support images. The image will be omitted from this request.]";
51
52
  }
53
+ /**
54
+ * Ask a vision model to describe an image. Returns the caption text, or null if
55
+ * the caption model/credentials are unavailable or the request fails.
56
+ */
57
+ export async function captionImage(image, captionModelId, ctx, signal) {
58
+ if (!captionModelId || !ctx?.modelRegistry || !ctx.model)
59
+ return null;
60
+ const slash = captionModelId.indexOf("/");
61
+ if (slash <= 0 || slash === captionModelId.length - 1)
62
+ return null;
63
+ const provider = captionModelId.slice(0, slash);
64
+ const modelId = captionModelId.slice(slash + 1);
65
+ const captionModel = ctx.modelRegistry.find(provider, modelId);
66
+ if (!captionModel || !captionModel.input.includes("image"))
67
+ return null;
68
+ let auth;
69
+ try {
70
+ auth = await ctx.modelRegistry.getApiKeyAndHeaders(captionModel);
71
+ }
72
+ catch {
73
+ return null;
74
+ }
75
+ if (!auth.ok)
76
+ return null;
77
+ return captionImageWithModel(captionModel, image, {
78
+ apiKey: auth.apiKey,
79
+ headers: auth.headers,
80
+ signal,
81
+ });
82
+ }
52
83
  function toPosixPath(filePath) {
53
84
  return filePath.split(sep).join("/");
54
85
  }
@@ -130,6 +161,7 @@ function formatReadResult(args, result, options, theme, showImages, _cwd, isErro
130
161
  }
131
162
  export function createReadToolDefinition(cwd, options) {
132
163
  const autoResizeImages = options?.autoResizeImages ?? true;
164
+ const imageCaptionModel = options?.imageCaptionModel;
133
165
  const ops = options?.operations ?? defaultReadOperations;
134
166
  return {
135
167
  name: "read",
@@ -178,11 +210,29 @@ export function createReadToolDefinition(cwd, options) {
178
210
  let textNote = `Read image file [${processed.mimeType}]`;
179
211
  if (processed.hints.length > 0)
180
212
  textNote += `\n${processed.hints.join("\n")}`;
213
+ const image = { type: "image", data: processed.data, mimeType: processed.mimeType };
214
+ // Vision relay: when the active model cannot accept images, ask a
215
+ // configured vision model to describe the image and use that text instead.
216
+ if (nonVisionImageNote) {
217
+ const caption = await captionImage(image, imageCaptionModel, ctx, signal);
218
+ if (caption) {
219
+ content = [
220
+ {
221
+ type: "text",
222
+ text: `Read image file [${processed.mimeType}] (described by vision model)\n${caption}`,
223
+ },
224
+ ];
225
+ if (aborted)
226
+ return;
227
+ resolve({ content, details });
228
+ return;
229
+ }
230
+ }
181
231
  if (nonVisionImageNote)
182
232
  textNote += `\n${nonVisionImageNote}`;
183
233
  content = [
184
234
  { type: "text", text: textNote },
185
- { type: "image", data: processed.data, mimeType: processed.mimeType },
235
+ image,
186
236
  ];
187
237
  }
188
238
  }
@@ -0,0 +1,30 @@
1
+ /**
2
+ * Shared vision-caption relay: ask a vision-capable model to describe an image
3
+ * as plain text, so a text-only main model can still work from it.
4
+ */
5
+ import type { Api, ImageContent, Model, ProviderHeaders } from "@earendil-works/pi-ai";
6
+ export declare const VISION_CAPTION_SYSTEM_PROMPT: string;
7
+ export interface VisionCaptionRequestOptions {
8
+ apiKey?: string;
9
+ headers?: ProviderHeaders;
10
+ signal?: AbortSignal;
11
+ /** Called with the underlying error message when the caption request fails. */
12
+ onError?: (message: string) => void;
13
+ /**
14
+ * The user's current instruction (e.g. "make this UI element bigger").
15
+ * Included so the caption is targeted at the task, not generic.
16
+ */
17
+ userPrompt?: string;
18
+ /**
19
+ * A small, already-bounded slice of recent conversation text (or other relevant
20
+ * context) to help the caption model understand intent. Must be small enough to
21
+ * stay safely under the caption model's context window. Omit for none.
22
+ */
23
+ contextText?: string;
24
+ }
25
+ /**
26
+ * Ask a vision model to describe an image. Returns the caption text, or null if
27
+ * the request fails or is aborted. The request context is minimal (prompt +
28
+ * image only), so it is independent of any main-agent context usage.
29
+ */
30
+ export declare function captionImageWithModel(captionModel: Model<Api>, image: ImageContent, options?: VisionCaptionRequestOptions): Promise<string | null>;
@@ -0,0 +1,62 @@
1
+ /**
2
+ * Shared vision-caption relay: ask a vision-capable model to describe an image
3
+ * as plain text, so a text-only main model can still work from it.
4
+ */
5
+ import { complete } from "@earendil-works/pi-ai/compat";
6
+ export const VISION_CAPTION_SYSTEM_PROMPT = "You are an image captioner for a coding agent whose main model cannot see images. " +
7
+ "Describe the image as completely and precisely as possible in plain text so that a programmer " +
8
+ "can work from your description alone, as if looking at the image themselves. " +
9
+ "Read any visible text, code, UI labels, error messages, terminal output, or diagrams verbatim. " +
10
+ "Cover layout, colors, spatial relationships, relative positions and sizes of elements, " +
11
+ "window or panel structure, and anything relevant to debugging or implementing. " +
12
+ "When a simple ASCII sketch clarifies the layout, include it.";
13
+ function buildUserPrompt(userPrompt, contextText) {
14
+ const lines = [];
15
+ if (userPrompt) {
16
+ lines.push(`The user said:\n${userPrompt}`);
17
+ }
18
+ if (contextText) {
19
+ lines.push(`Relevant context (recent conversation):\n${contextText}`);
20
+ }
21
+ lines.push("Describe this image in detail:");
22
+ return lines.join("\n\n");
23
+ }
24
+ /**
25
+ * Ask a vision model to describe an image. Returns the caption text, or null if
26
+ * the request fails or is aborted. The request context is minimal (prompt +
27
+ * image only), so it is independent of any main-agent context usage.
28
+ */
29
+ export async function captionImageWithModel(captionModel, image, options) {
30
+ const aiContext = {
31
+ systemPrompt: VISION_CAPTION_SYSTEM_PROMPT,
32
+ messages: [
33
+ {
34
+ role: "user",
35
+ timestamp: Date.now(),
36
+ content: [
37
+ { type: "text", text: buildUserPrompt(options?.userPrompt, options?.contextText) },
38
+ image,
39
+ ],
40
+ },
41
+ ],
42
+ };
43
+ try {
44
+ const response = await complete(captionModel, aiContext, {
45
+ apiKey: options?.apiKey,
46
+ headers: options?.headers,
47
+ signal: options?.signal,
48
+ });
49
+ if (response.stopReason === "aborted")
50
+ return null;
51
+ const text = response.content
52
+ .filter((c) => c.type === "text")
53
+ .map((c) => c.text)
54
+ .join("\n")
55
+ .trim();
56
+ return text.length > 0 ? text : null;
57
+ }
58
+ catch (error) {
59
+ options?.onError?.(error instanceof Error ? error.message : String(error));
60
+ return null;
61
+ }
62
+ }
@@ -24,11 +24,22 @@
24
24
  "high": null,
25
25
  "xhigh": null,
26
26
  "max": "max"
27
- },
28
- "compat": {
29
- "supportsDeveloperRole": false,
30
- "requiresReasoningContentOnAssistantMessages": true,
31
- "thinkingFormat": "deepseek"
27
+ }
28
+ },
29
+ {
30
+ "id": "auto-premium",
31
+ "name": "Auto Premium",
32
+ "reasoning": true,
33
+ "input": ["text"],
34
+ "contextWindow": 512000,
35
+ "maxTokens": 64000,
36
+ "thinkingLevelMap": {
37
+ "minimal": null,
38
+ "low": null,
39
+ "medium": null,
40
+ "high": null,
41
+ "xhigh": null,
42
+ "max": "max"
32
43
  }
33
44
  },
34
45
  {
@@ -132,6 +143,18 @@
132
143
  "supportsDeveloperRole": false,
133
144
  "supportsReasoningEffort": true
134
145
  }
146
+ },
147
+ {
148
+ "id": "gemma-4",
149
+ "name": "Gemma 4 31B (Vision)",
150
+ "reasoning": false,
151
+ "input": ["text", "image"],
152
+ "contextWindow": 256000,
153
+ "maxTokens": 8192,
154
+ "compat": {
155
+ "supportsDeveloperRole": false,
156
+ "supportsReasoningEffort": false
157
+ }
135
158
  }
136
159
  ]
137
160
  }
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "rules": [
3
3
  {
4
- "match": ["deepseek-v4-pro"],
4
+ "match": ["deepseek-v4-*"],
5
5
  "mode": "prepend",
6
6
  "prompt": "text: You are a helpful software engineer assistant.\n\ndescription: Bootstrap with shell/read, then expose the full Standard tool catalog after the first durable tool call.\n\nwhen you thought, start with `we need...`",
7
7
  "enabled": true
@@ -3,6 +3,13 @@ import type { Transport } from "@earendil-works/pi-ai";
3
3
  import { Container, type ScrollViewScrollbar, SettingsList } from "@earendil-works/pi-tui";
4
4
  import type { DefaultProjectTrust, FullscreenExitOutput, TuiMode, WarningSettings } from "../../../core/settings-manager.ts";
5
5
  import { type TerminalTheme } from "../theme/theme.ts";
6
+ export interface SkillToggleItem {
7
+ name: string;
8
+ enabled: boolean;
9
+ pattern: string;
10
+ scope: "user" | "project";
11
+ id: string;
12
+ }
6
13
  export interface SettingsConfig {
7
14
  autoCompact: boolean;
8
15
  autoHandoffEnabled: boolean;
@@ -11,6 +18,9 @@ export interface SettingsConfig {
11
18
  imageWidthCells: number;
12
19
  autoResizeImages: boolean;
13
20
  blockImages: boolean;
21
+ imageCaptionModel: string | undefined;
22
+ imageCaptionContextTokens: number;
23
+ visionModels: string[];
14
24
  steeringMode: "all" | "one-at-a-time";
15
25
  followUpMode: "all" | "one-at-a-time";
16
26
  transport: Transport;
@@ -38,6 +48,7 @@ export interface SettingsConfig {
38
48
  fullscreenExitOutput: FullscreenExitOutput;
39
49
  fullscreenScrollbar: ScrollViewScrollbar;
40
50
  warnings: WarningSettings;
51
+ skills: SkillToggleItem[];
41
52
  }
42
53
  export interface SettingsCallbacks {
43
54
  onAutoCompactChange: (enabled: boolean) => void;
@@ -47,6 +58,8 @@ export interface SettingsCallbacks {
47
58
  onImageWidthCellsChange: (width: number) => void;
48
59
  onAutoResizeImagesChange: (enabled: boolean) => void;
49
60
  onBlockImagesChange: (blocked: boolean) => void;
61
+ onImageCaptionModelChange: (modelId: string | undefined) => void;
62
+ onImageCaptionContextTokensChange: (tokens: number) => void;
50
63
  onSteeringModeChange: (mode: "all" | "one-at-a-time") => void;
51
64
  onFollowUpModeChange: (mode: "all" | "one-at-a-time") => void;
52
65
  onTransportChange: (transport: Transport) => void;
@@ -72,6 +85,8 @@ export interface SettingsCallbacks {
72
85
  onFullscreenExitOutputChange: (output: FullscreenExitOutput) => void;
73
86
  onFullscreenScrollbarChange: (mode: ScrollViewScrollbar) => void;
74
87
  onWarningsChange: (warnings: WarningSettings) => void;
88
+ onSkillsToggle: (item: SkillToggleItem, enabled: boolean) => void;
89
+ onSkillsReload: () => void;
75
90
  onCancel: () => void;
76
91
  }
77
92
  /**
@@ -22,6 +22,46 @@ const DEFAULT_PROJECT_TRUST_LABELS = {
22
22
  never: "Never trust",
23
23
  };
24
24
  const DEFAULT_PROJECT_TRUST_BY_LABEL = new Map(Object.entries(DEFAULT_PROJECT_TRUST_LABELS).map(([value, label]) => [label, value]));
25
+ class SkillsSubmenu extends Container {
26
+ settingsList;
27
+ constructor(skills, onToggle, onCancel) {
28
+ super();
29
+ const initial = skills.map((s) => ({ ...s }));
30
+ const nameCounts = new Map();
31
+ for (const skill of initial) {
32
+ nameCounts.set(skill.name, (nameCounts.get(skill.name) ?? 0) + 1);
33
+ }
34
+ const items = initial.map((skill) => ({
35
+ id: skill.id,
36
+ label: (nameCounts.get(skill.name) ?? 0) > 1 ? `${skill.name} (${skill.pattern})` : skill.name,
37
+ description: `Scope: ${skill.scope}${(nameCounts.get(skill.name) ?? 0) > 1 ? ` — ${skill.pattern}` : ""}`,
38
+ currentValue: skill.enabled ? "true" : "false",
39
+ values: ["true", "false"],
40
+ }));
41
+ if (items.length === 0) {
42
+ items.push({
43
+ id: "no-skills",
44
+ label: "No skills discovered",
45
+ currentValue: "",
46
+ });
47
+ }
48
+ this.settingsList = new SettingsList(items, Math.min(items.length, 10), getSettingsListTheme(), (id, newValue) => {
49
+ const skill = initial.find((s) => s.id === id);
50
+ if (skill) {
51
+ const enabled = newValue === "true";
52
+ skill.enabled = enabled;
53
+ const source = skills.find((s) => s.id === id);
54
+ if (source)
55
+ source.enabled = enabled;
56
+ onToggle(skill, enabled);
57
+ }
58
+ }, onCancel);
59
+ this.addChild(this.settingsList);
60
+ }
61
+ handleInput(data) {
62
+ this.settingsList.handleInput(data);
63
+ }
64
+ }
25
65
  /**
26
66
  * A submenu component for selecting from a list of options.
27
67
  */
@@ -272,6 +312,7 @@ export class SettingsSelectorComponent extends Container {
272
312
  const supportsImages = getCapabilities().images;
273
313
  const followUpKey = keyDisplayText("app.message.followUp");
274
314
  let currentWarnings = { ...config.warnings };
315
+ let skillsChanged = false;
275
316
  const items = [
276
317
  {
277
318
  id: "autocompact",
@@ -378,6 +419,16 @@ export class SettingsSelectorComponent extends Container {
378
419
  currentValue: config.treeFilterMode,
379
420
  values: ["default", "no-tools", "user-only", "labeled-only", "all"],
380
421
  },
422
+ {
423
+ id: "skills",
424
+ label: "Skills",
425
+ description: "Choose which skills are loaded (opt-in). Skills apply and reload when you close this menu.",
426
+ currentValue: `${config.skills.filter((s) => s.enabled).length}/${config.skills.length} enabled`,
427
+ submenu: (_currentValue, done) => new SkillsSubmenu(config.skills, (item, enabled) => {
428
+ callbacks.onSkillsToggle(item, enabled);
429
+ skillsChanged = true;
430
+ }, () => done()),
431
+ },
381
432
  {
382
433
  id: "warnings",
383
434
  label: "Warnings",
@@ -466,9 +517,30 @@ export class SettingsSelectorComponent extends Container {
466
517
  currentValue: config.blockImages ? "true" : "false",
467
518
  values: ["true", "false"],
468
519
  });
469
- // Hardware cursor toggle (insert after block-images)
520
+ // Vision caption model picker (image -> text relay for text-only main models).
521
+ // Values are the available vision-capable models plus "off" for none.
522
+ const visionModelValues = ["off", ...config.visionModels];
523
+ const currentCaptionModel = config.imageCaptionModel ?? "off";
470
524
  const blockImagesIndex = items.findIndex((item) => item.id === "block-images");
471
525
  items.splice(blockImagesIndex + 1, 0, {
526
+ id: "image-caption-model",
527
+ label: "Vision caption model",
528
+ description: "Model used to describe images when the main model can't see them (off disables)",
529
+ currentValue: currentCaptionModel,
530
+ values: visionModelValues,
531
+ });
532
+ // Vision caption context-token budget
533
+ const captionModelIndex = items.findIndex((item) => item.id === "image-caption-model");
534
+ items.splice(captionModelIndex + 1, 0, {
535
+ id: "image-caption-context-tokens",
536
+ label: "Vision context tokens",
537
+ description: "Budget (approx) for recent-conversation context sent to the vision model",
538
+ currentValue: String(config.imageCaptionContextTokens),
539
+ values: ["0", "4096", "16384", "32768", "65536"],
540
+ });
541
+ // Hardware cursor toggle (insert after the image-caption settings)
542
+ const visionSettingsIndex = items.findIndex((item) => item.id === "image-caption-context-tokens");
543
+ items.splice(visionSettingsIndex + 1, 0, {
472
544
  id: "show-hardware-cursor",
473
545
  label: "Show hardware cursor",
474
546
  description: "Show the terminal cursor while still positioning it for IME support",
@@ -545,6 +617,12 @@ export class SettingsSelectorComponent extends Container {
545
617
  case "block-images":
546
618
  callbacks.onBlockImagesChange(newValue === "true");
547
619
  break;
620
+ case "image-caption-model":
621
+ callbacks.onImageCaptionModelChange(newValue === "off" ? undefined : newValue);
622
+ break;
623
+ case "image-caption-context-tokens":
624
+ callbacks.onImageCaptionContextTokensChange(parseInt(newValue, 10) || 0);
625
+ break;
548
626
  case "steering-mode":
549
627
  callbacks.onSteeringModeChange(newValue);
550
628
  break;
@@ -620,7 +698,12 @@ export class SettingsSelectorComponent extends Container {
620
698
  callbacks.onThemeChange(newValue);
621
699
  break;
622
700
  }
623
- }, callbacks.onCancel, { enableSearch: true });
701
+ }, () => {
702
+ if (skillsChanged) {
703
+ callbacks.onSkillsReload();
704
+ }
705
+ callbacks.onCancel();
706
+ }, { enableSearch: true });
624
707
  this.addChild(this.settingsList);
625
708
  this.addChild(new DynamicBorder());
626
709
  }
@@ -1,6 +1,6 @@
1
1
  import { type Component, Loader, type TUI } from "@earendil-works/pi-tui";
2
2
  import type { WorkingIndicatorOptions } from "../../../core/extensions/index.ts";
3
- export type StatusIndicatorKind = "working" | "retry" | "compaction" | "branchSummary";
3
+ export type StatusIndicatorKind = "working" | "retry" | "compaction" | "branchSummary" | "imageCaptioning";
4
4
  export declare class StatusIndicator extends Loader {
5
5
  readonly kind: StatusIndicatorKind;
6
6
  constructor(kind: StatusIndicatorKind, ui: TUI, spinnerColorFn: (str: string) => string, messageColorFn: (str: string) => string, message: string, indicator?: WorkingIndicatorOptions);
@@ -9,6 +9,9 @@ export declare class StatusIndicator extends Loader {
9
9
  export declare class WorkingStatusIndicator extends StatusIndicator {
10
10
  constructor(ui: TUI, message: string, indicator?: WorkingIndicatorOptions);
11
11
  }
12
+ export declare class ImageCaptioningStatusIndicator extends StatusIndicator {
13
+ constructor(ui: TUI);
14
+ }
12
15
  export declare class RetryStatusIndicator extends StatusIndicator {
13
16
  private countdown;
14
17
  constructor(ui: TUI, attempt: number, maxAttempts: number, delayMs: number);
@@ -17,6 +17,11 @@ export class WorkingStatusIndicator extends StatusIndicator {
17
17
  super("working", ui, (spinner) => theme.fg("accent", spinner), (text) => theme.fg("muted", text), message, indicator);
18
18
  }
19
19
  }
20
+ export class ImageCaptioningStatusIndicator extends StatusIndicator {
21
+ constructor(ui) {
22
+ super("imageCaptioning", ui, (spinner) => theme.fg("accent", spinner), (text) => theme.fg("muted", text), "Reading image with vision model...");
23
+ }
24
+ }
20
25
  export class RetryStatusIndicator extends StatusIndicator {
21
26
  countdown;
22
27
  constructor(ui, attempt, maxAttempts, delayMs) {
@@ -373,6 +373,10 @@ export declare class InteractiveMode {
373
373
  */
374
374
  private showSelector;
375
375
  private showSettingsSelector;
376
+ private getVisionModels;
377
+ private buildSkillToggleItems;
378
+ private toggleSkillSetting;
379
+ private reloadAfterSkillChange;
376
380
  private handleModelCommand;
377
381
  private findExactModelMatch;
378
382
  private getModelCandidates;