@bismawy/pi-vision-watcher 1.0.7 â 1.0.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +82 -82
- package/package.json +2 -2
- package/src/image.ts +3 -3
- package/src/index.ts +78 -0
- package/vision-watcher.ts +109 -2
package/README.md
CHANGED
|
@@ -1,79 +1,67 @@
|
|
|
1
|
-
# đī¸ pi-vision-watcher
|
|
1
|
+
# đī¸ @bismawy/pi-vision-watcher
|
|
2
2
|
|
|
3
|
-
**Give text-only [pi](https://github.com/earendil-works/pi-coding-agent) models vision
|
|
3
|
+
**Give text-only [pi](https://github.com/earendil-works/pi-coding-agent) models vision capabilities.**
|
|
4
4
|
|
|
5
|
-
|
|
5
|
+
Seamlessly inspect, describe, and convert visual inputs (screenshots, mockups, terminal errors, clipboard pastes) into structured descriptions using your preferred vision model, and hand them off to text-only coding models without interrupting your workflow.
|
|
6
6
|
|
|
7
7
|
[](https://github.com/earendil-works/pi-coding-agent)
|
|
8
8
|
[](https://www.npmjs.com/package/@bismawy/pi-vision-watcher)
|
|
9
9
|
[](./LICENSE)
|
|
10
10
|
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
## The Problem
|
|
14
|
-
|
|
15
|
-
Some of the best coding models are text-only. When you attach a screenshot, diagram, or UI mock, they either ignore it or fail the request entirely. Switching models just to read an image interrupts your workflow.
|
|
16
|
-
|
|
17
|
-
## The Solution
|
|
18
|
-
|
|
19
|
-
`pi-vision-watcher` bridges this gap automatically:
|
|
20
|
-
- **Interactive Picker:** Pick any vision model from your authenticated providers with `/vision-watcher`.
|
|
21
|
-
- **Automatic Handoff:** Whenever a non-vision model receives an image (via paste, attachment, or the `read` tool), the image is described behind the scenes and swapped for rich descriptive text before reaching the model.
|
|
22
|
-
- **Batched & Cached:** Uses a DataLoader pattern so multiple images in a turn coalesce into a **single batched vision request**, cached by SHA-256 hash.
|
|
11
|
+

|
|
23
12
|
|
|
24
13
|
---
|
|
25
14
|
|
|
26
|
-
##
|
|
27
|
-
|
|
28
|
-
- đ¯ **Connected-Only Model Picker** â `/vision-watcher` filters out unconfigured providers, showing only models you actually have credentials for (`/login`, `models.json`, or environment variables). Vision-capable models are highlighted with đī¸.
|
|
29
|
-
- ⥠**DataLoader Batching** â Multiple images from parallel `read` calls or multi-image attachments merge into ONE batched vision call during the tool-result phase, eliminating latency bottlenecks.
|
|
30
|
-
- đ§ **Thinking & Reasoning Support** â Configure reasoning effort (`/vision-watcher thinking <level>`) for reasoning-capable vision models (e.g. OpenAI o-series, Claude, DeepSeek).
|
|
31
|
-
- đ **Fallback Chains** â Specify backup vision models that automatically take over if your primary describer is unavailable or encounters rate limits.
|
|
32
|
-
- đ **Paste-Time Prewarm (Opt-in)** â Describe pasted images the moment the path lands in the prompt editor before you even press Enter.
|
|
33
|
-
- đŦ **Async Clipboard Fallback (Opt-in)** â Races direct reads against asynchronous description delivery to prevent stalling.
|
|
34
|
-
- đž **LRU Hash Caching** â Prevents duplicate calls for identical images across conversation turns.
|
|
35
|
-
- đĄī¸ **Graceful Degradation** â Never crashes your agent turn. If a description fails, a clean `[Image: description unavailable]` placeholder is provided and logged to `~/.pi/agent/logs/pi-vision-watcher/errors.log`.
|
|
36
|
-
|
|
37
|
-
---
|
|
38
|
-
|
|
39
|
-
## đĻ Install
|
|
15
|
+
## ⥠Quick Start
|
|
40
16
|
|
|
17
|
+
### 1. Installation
|
|
41
18
|
```bash
|
|
42
19
|
pi install npm:@bismawy/pi-vision-watcher
|
|
43
20
|
```
|
|
21
|
+
*(Or install directly from Git: `pi install git:github.com/bismawy/pi-vision-watcher`)*
|
|
44
22
|
|
|
45
|
-
|
|
46
|
-
|
|
23
|
+
### 2. Select Vision Model
|
|
24
|
+
Open the interactive TUI selector to choose your vision describer model from your connected providers:
|
|
47
25
|
```bash
|
|
48
|
-
|
|
26
|
+
/vision-watcher
|
|
49
27
|
```
|
|
28
|
+
*(You can also set it directly: `/vision-watcher model openai/gpt-4o`)*
|
|
50
29
|
|
|
51
|
-
|
|
30
|
+
### 3. Work Seamlessly
|
|
31
|
+
Switch to any text-only model in Pi (e.g. DeepSeek, Claude text-only, local models). Whenever you paste an image, attach a file, or the agent runs `read` on an image, `pi-vision-watcher` describes it automatically in the background.
|
|
52
32
|
|
|
53
33
|
---
|
|
54
34
|
|
|
55
|
-
##
|
|
35
|
+
## đ Key Capabilities
|
|
36
|
+
|
|
37
|
+
- đ¯ **Connected-Only Interactive Picker:** Shows only vision-capable models from providers where you actually have active credentials (`/login`, `models.json`, or environment variables).
|
|
38
|
+
- ⥠**DataLoader Batching & SHA-256 Cache:** Automatically groups multiple images across parallel tool calls or multi-file prompts into a single batched describer request. Cached images are never re-described.
|
|
39
|
+
- đĄī¸ **Proactive False-Vision Healing:** Aggregator providers often mistakenly flag models (like DeepSeek V4) as multimodal, causing HTTP 400 errors (`This model does not support image`). `pi-vision-watcher` proactively forces handoff for these models and auto-heals `models.json` `modelOverrides` in-process.
|
|
40
|
+
- đ§ **Thinking & Reasoning Controls:** Adjust reasoning levels (`off`, `minimal`, `low`, `medium`, `high`, `xhigh`, `max`) for reasoning-capable vision models (o-series, Claude, DeepSeek).
|
|
41
|
+
- đ **Multi-Model Fallback Chains:** Automatically falls back to backup vision models if your primary provider is rate-limited or unavailable.
|
|
56
42
|
|
|
57
|
-
|
|
43
|
+
---
|
|
44
|
+
|
|
45
|
+
## đšī¸ Command Reference
|
|
58
46
|
|
|
59
|
-
| Command |
|
|
47
|
+
| Command | Action |
|
|
60
48
|
|---|---|
|
|
61
|
-
| `/vision-watcher` | Open
|
|
62
|
-
| `/vision-watcher model <provider/id>` | Set
|
|
63
|
-
| `/vision-watcher status` | View
|
|
64
|
-
| `/vision-watcher
|
|
65
|
-
| `/vision-watcher
|
|
66
|
-
| `/vision-watcher add <provider/id>` | Force handoff for a specific model (e.g. weak vision models) |
|
|
49
|
+
| `/vision-watcher` | Open interactive TUI picker for connected vision models |
|
|
50
|
+
| `/vision-watcher model <provider/id>` | Set primary vision describer directly |
|
|
51
|
+
| `/vision-watcher status` | View current configuration and active model status |
|
|
52
|
+
| `/vision-watcher auto <on\|off>` | Toggle automatic handoff for non-vision models (default: `on`) |
|
|
53
|
+
| `/vision-watcher add <provider/id>` | Force handoff on a specific model |
|
|
67
54
|
| `/vision-watcher remove <provider/id>` | Remove model from forced handoff list |
|
|
68
|
-
| `/vision-watcher thinking <level>` |
|
|
69
|
-
| `/vision-watcher
|
|
70
|
-
| `/vision-watcher
|
|
71
|
-
| `/vision-watcher clear` | Clear configured vision model |
|
|
72
|
-
| `/vision-watcher help` | Display full command reference |
|
|
55
|
+
| `/vision-watcher thinking <level>` | Configure reasoning effort for vision models |
|
|
56
|
+
| `/vision-watcher enable` / `disable` | Toggle extension active state |
|
|
57
|
+
| `/vision-watcher help` | Show in-CLI command documentation |
|
|
73
58
|
|
|
74
59
|
---
|
|
75
60
|
|
|
76
|
-
##
|
|
61
|
+
## đ Deep Dive & Advanced Configuration
|
|
62
|
+
|
|
63
|
+
<details>
|
|
64
|
+
<summary><b>âī¸ Configuration File Schema (<code>pi-vision-watcher.json</code>)</b></summary>
|
|
77
65
|
|
|
78
66
|
Configuration is stored at `~/.pi/agent/extensions/pi-vision-watcher.json`:
|
|
79
67
|
|
|
@@ -86,6 +74,7 @@ Configuration is stored at `~/.pi/agent/extensions/pi-vision-watcher.json`:
|
|
|
86
74
|
"handoffModels": [],
|
|
87
75
|
"thinking": false,
|
|
88
76
|
"thinkingLevel": "medium",
|
|
77
|
+
"describeTimeoutMs": 45000,
|
|
89
78
|
"prewarmPastedImages": false,
|
|
90
79
|
"asyncClipboardHandoff": false,
|
|
91
80
|
"maxTokens": null,
|
|
@@ -94,47 +83,58 @@ Configuration is stored at `~/.pi/agent/extensions/pi-vision-watcher.json`:
|
|
|
94
83
|
}
|
|
95
84
|
```
|
|
96
85
|
|
|
97
|
-
| Field | Default | Description |
|
|
98
|
-
|
|
99
|
-
| `enabled` | `true` | Master switch for
|
|
100
|
-
| `visionModel` | `null` | Primary describer
|
|
101
|
-
| `fallbackModels` | `[]` |
|
|
102
|
-
| `autoHandoff` | `true` | Automatically
|
|
103
|
-
| `handoffModels` | `[]` |
|
|
104
|
-
| `thinking`
|
|
105
|
-
| `
|
|
106
|
-
| `
|
|
107
|
-
| `
|
|
108
|
-
| `
|
|
109
|
-
| `
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
86
|
+
| Field | Type | Default | Description |
|
|
87
|
+
|---|---|---|---|
|
|
88
|
+
| `enabled` | `boolean` | `true` | Master switch for handoff processing. |
|
|
89
|
+
| `visionModel` | `string \| null` | `null` | Primary describer model ref (`provider/id`). |
|
|
90
|
+
| `fallbackModels` | `string[]` | `[]` | Ordered backup models if the primary model fails. |
|
|
91
|
+
| `autoHandoff` | `boolean` | `true` | Automatically describe images for models lacking native vision. |
|
|
92
|
+
| `handoffModels` | `string[]` | `[]` | Specific model IDs forced to receive descriptions. |
|
|
93
|
+
| `thinking` | `boolean` | `false` | Enable reasoning tokens for vision model. |
|
|
94
|
+
| `thinkingLevel` | `string` | `"medium"` | Reasoning effort (`minimal`, `low`, `medium`, `high`, `xhigh`, `max`). |
|
|
95
|
+
| `describeTimeoutMs` | `number` | `45000` | Per-batch timeout before aborting or triggering fallbacks. |
|
|
96
|
+
| `prewarmPastedImages` | `boolean` | `false` | Start describing clipboard images immediately upon pasting in prompt. |
|
|
97
|
+
| `asyncClipboardHandoff` | `boolean` | `false` | Async clipboard injection fallback mechanism. |
|
|
98
|
+
| `maxTokens` | `number \| null` | `null` | Max output tokens for descriptions (`null` = model default). |
|
|
99
|
+
| `cacheMax` | `number` | `50` | Maximum cached image hashes per session. |
|
|
100
|
+
| `maxDescriptionLines` | `number` | `0` | Truncate lines in description block (`0` = full description). |
|
|
101
|
+
|
|
102
|
+
</details>
|
|
103
|
+
|
|
104
|
+
<details>
|
|
105
|
+
<summary><b>đ Troubleshooting, Recovery & Diagnostics</b></summary>
|
|
106
|
+
|
|
107
|
+
### Structured Error Logging
|
|
108
|
+
If a vision call fails, errors are appended with stack traces and request metadata to:
|
|
109
|
+
```text
|
|
110
|
+
~/.pi/agent/logs/pi-vision-watcher/errors.log
|
|
111
|
+
```
|
|
112
|
+
Failures degrade gracefully to `[Image: description unavailable]` without breaking the agent turn.
|
|
114
113
|
|
|
115
|
-
-
|
|
116
|
-
|
|
114
|
+
### False-Vision Auto-Recovery
|
|
115
|
+
When a model falsely advertises image capability and returns an HTTP 400 rejection:
|
|
116
|
+
1. `pi-vision-watcher` captures the error in the `message_end` event.
|
|
117
|
+
2. It automatically updates `~/.pi/agent/models.json` under `providers.<name>.modelOverrides.<model>.input = ["text"]`.
|
|
118
|
+
3. It triggers an in-process registry refresh so subsequent turns use handoff naturally.
|
|
117
119
|
|
|
118
|
-
|
|
120
|
+
</details>
|
|
119
121
|
|
|
120
|
-
|
|
122
|
+
<details>
|
|
123
|
+
<summary><b>đ ī¸ Development & Testing</b></summary>
|
|
121
124
|
|
|
122
125
|
```bash
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
+
bun install
|
|
127
|
+
bun run test # Run Vitest test suite (240+ unit tests)
|
|
128
|
+
bun run typecheck # Run TypeScript compiler check
|
|
129
|
+
bun run lint:dead # Scan for unused exports with Knip
|
|
126
130
|
```
|
|
127
131
|
|
|
128
|
-
|
|
132
|
+
</details>
|
|
129
133
|
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
`pi-vision-watcher` is inspired by and forked from [`pi-vision-handoff`](https://github.com/monotykamary/pi-vision-handoff) by [Tom X Nguyen](https://github.com/monotykamary) (originating from the concept in `pi-umans-provider`).
|
|
134
|
+
---
|
|
133
135
|
|
|
134
|
-
|
|
135
|
-
- Filters picker to only authenticated/connected models.
|
|
136
|
-
- Added thinking & reasoning controls for modern reasoning vision models.
|
|
137
|
-
- Multi-model fallback chain support.
|
|
138
|
-
- Streamlined settings and UI badging.
|
|
136
|
+
## đ License & Acknowledgments
|
|
139
137
|
|
|
140
|
-
|
|
138
|
+
- Built for the **[pi coding agent](https://github.com/earendil-works/pi-coding-agent)** ecosystem.
|
|
139
|
+
- Evolved from concepts in `pi-vision-handoff` by Tom X Nguyen and `pi-umans-provider`.
|
|
140
|
+
- Distributed under the **[MIT License](./LICENSE)**.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@bismawy/pi-vision-watcher",
|
|
3
|
-
"version": "1.0.
|
|
3
|
+
"version": "1.0.9",
|
|
4
4
|
"description": "Give text-only pi models vision â describe images with a vision model you pick via an interactive picker, then hand off the text description to non-vision models",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"author": "bismawy",
|
|
@@ -57,7 +57,7 @@
|
|
|
57
57
|
"extensions": [
|
|
58
58
|
"./vision-watcher.ts"
|
|
59
59
|
],
|
|
60
|
-
"image": "https://raw.githubusercontent.com/bismawy/pi-vision-watcher/main/assets/screenshot.
|
|
60
|
+
"image": "https://raw.githubusercontent.com/bismawy/pi-vision-watcher/main/assets/screenshot.webp"
|
|
61
61
|
},
|
|
62
62
|
"peerDependencies": {
|
|
63
63
|
"@earendil-works/pi-ai": "*",
|
package/src/image.ts
CHANGED
|
@@ -20,7 +20,7 @@
|
|
|
20
20
|
|
|
21
21
|
import { readFileSync, statSync } from "node:fs";
|
|
22
22
|
import { tmpdir } from "node:os";
|
|
23
|
-
import { isAbsolute, join, sep } from "node:path";
|
|
23
|
+
import { isAbsolute, join, normalize, sep } from "node:path";
|
|
24
24
|
import crypto from "node:crypto";
|
|
25
25
|
import type { ExtractedImage } from "./index.js";
|
|
26
26
|
|
|
@@ -299,14 +299,14 @@ const LEADING_WRAP_RE = new RegExp('^[^A-Za-z0-9_/@.~-]+');
|
|
|
299
299
|
* dir and mistaken for local files; leading wrapping chars (parentheses,
|
|
300
300
|
* quotes) are stripped so quoted/parenthesized paths still resolve. */
|
|
301
301
|
export function findPastedImagePaths(prompt: string): string[] {
|
|
302
|
-
const tmp = tmpdir();
|
|
302
|
+
const tmp = normalize(tmpdir());
|
|
303
303
|
const paths = new Set<string>();
|
|
304
304
|
for (const m of prompt.matchAll(PASTED_IMAGE_PATH_RE)) {
|
|
305
305
|
const token = m[1];
|
|
306
306
|
if (!token) continue;
|
|
307
307
|
if (URL_RE.test(token)) continue;
|
|
308
308
|
const p = token.replace(LEADING_WRAP_RE, "");
|
|
309
|
-
const abs = isAbsolute(p) ? p : join(tmp, p);
|
|
309
|
+
const abs = normalize(isAbsolute(p) ? p : join(tmp, p));
|
|
310
310
|
// Ensure the resolved candidate stays inside the temp directory.
|
|
311
311
|
if (abs.startsWith(tmp + sep) || abs === tmp) paths.add(abs);
|
|
312
312
|
}
|
package/src/index.ts
CHANGED
|
@@ -235,6 +235,84 @@ export function formatModelRef(provider: string, id: string): string {
|
|
|
235
235
|
return `${provider}/${id}`;
|
|
236
236
|
}
|
|
237
237
|
|
|
238
|
+
/** Match provider errors meaning "this model can't take image input" â e.g.
|
|
239
|
+
* AgentRouter's 400 `"This model does not support image"`, OpenRouter's
|
|
240
|
+
* "does not support image inputs". Used by the message_end auto-recovery:
|
|
241
|
+
* a model whose registry entry FALSELY declares `input: ["text","image"]`
|
|
242
|
+
* passes the autoHandoff vision check, so the raw image reaches the provider
|
|
243
|
+
* and 400s â we learn from that error and force handoff for the model. */
|
|
244
|
+
export function isImageNotSupportedError(text: string): boolean {
|
|
245
|
+
return /not support (?:the )?image|image[s]? (?:input[s]? )?(?:is |are )?not supported|does not accept image|unsupported (?:image|multimodal)/i.test(
|
|
246
|
+
text,
|
|
247
|
+
);
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
/** Models whose registry entries commonly declare image input while the
|
|
251
|
+
* backend rejects images (DeepSeek V4 via aggregator proxies). Conservative:
|
|
252
|
+
* VL / Vision / Janus variants keep image input. Used so autoHandoff covers
|
|
253
|
+
* them on the FIRST send â waiting for a 400 is too late. */
|
|
254
|
+
export function isKnownTextOnlyFalselyVision(id: string | undefined | null): boolean {
|
|
255
|
+
if (!id) return false;
|
|
256
|
+
const n = id.toLowerCase();
|
|
257
|
+
if (n.includes("vl") || n.includes("vision") || n.includes("janus")) return false;
|
|
258
|
+
return /deepseek[-_./]*v4/.test(n);
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
/** Extract the provider error string from a finalized assistant message.
|
|
262
|
+
* Pi stores it on `errorMessage`; some providers also put the 400 body in a
|
|
263
|
+
* text content block. */
|
|
264
|
+
export function assistantErrorText(msg: {
|
|
265
|
+
errorMessage?: unknown;
|
|
266
|
+
content?: unknown;
|
|
267
|
+
} | null | undefined): string {
|
|
268
|
+
if (!msg) return "";
|
|
269
|
+
if (typeof msg.errorMessage === "string" && msg.errorMessage) return msg.errorMessage;
|
|
270
|
+
if (typeof msg.content === "string") return msg.content;
|
|
271
|
+
if (!Array.isArray(msg.content)) return "";
|
|
272
|
+
return msg.content
|
|
273
|
+
.filter((b): b is { type: string; text: string } => !!b && typeof b === "object" && typeof (b as { text?: unknown }).text === "string")
|
|
274
|
+
.map((b) => b.text)
|
|
275
|
+
.join("\n");
|
|
276
|
+
}
|
|
277
|
+
|
|
278
|
+
/** Set (or add) a `modelOverrides` entry forcing a model's input to
|
|
279
|
+
* `["text"]` in a PARSED models.json config object. Pure: returns a new
|
|
280
|
+
* object, never mutates the input. Preserves every other field (credentials,
|
|
281
|
+
* compat, sibling models) â only the one model's `input` override is touched.
|
|
282
|
+
* Used by the message_end auto-recovery to fix models whose registry entry
|
|
283
|
+
* falsely declares image input the backend rejects with a 400. */
|
|
284
|
+
export function setTextOnlyInputOverride(
|
|
285
|
+
cfg: unknown,
|
|
286
|
+
providerId: string,
|
|
287
|
+
modelId: string,
|
|
288
|
+
): Record<string, unknown> {
|
|
289
|
+
const root =
|
|
290
|
+
cfg && typeof cfg === "object" && !Array.isArray(cfg)
|
|
291
|
+
? { ...(cfg as Record<string, unknown>) }
|
|
292
|
+
: {};
|
|
293
|
+
const providers =
|
|
294
|
+
root.providers && typeof root.providers === "object" && !Array.isArray(root.providers)
|
|
295
|
+
? { ...(root.providers as Record<string, unknown>) }
|
|
296
|
+
: {};
|
|
297
|
+
const provider =
|
|
298
|
+
providers[providerId] && typeof providers[providerId] === "object" && !Array.isArray(providers[providerId])
|
|
299
|
+
? { ...(providers[providerId] as Record<string, unknown>) }
|
|
300
|
+
: {};
|
|
301
|
+
const overrides =
|
|
302
|
+
provider.modelOverrides && typeof provider.modelOverrides === "object" && !Array.isArray(provider.modelOverrides)
|
|
303
|
+
? { ...(provider.modelOverrides as Record<string, unknown>) }
|
|
304
|
+
: {};
|
|
305
|
+
const existing =
|
|
306
|
+
overrides[modelId] && typeof overrides[modelId] === "object" && !Array.isArray(overrides[modelId])
|
|
307
|
+
? { ...(overrides[modelId] as Record<string, unknown>) }
|
|
308
|
+
: {};
|
|
309
|
+
overrides[modelId] = { ...existing, input: ["text"] };
|
|
310
|
+
provider.modelOverrides = overrides;
|
|
311
|
+
providers[providerId] = provider;
|
|
312
|
+
root.providers = providers;
|
|
313
|
+
return root;
|
|
314
|
+
}
|
|
315
|
+
|
|
238
316
|
/** Whether a model declares image input. */
|
|
239
317
|
export function isVisionModel(model: { input?: ("text" | "image")[] } | undefined | null): boolean {
|
|
240
318
|
return !!model && Array.isArray(model.input) && model.input.includes("image");
|
package/vision-watcher.ts
CHANGED
|
@@ -31,17 +31,24 @@
|
|
|
31
31
|
*/
|
|
32
32
|
|
|
33
33
|
import type { ExtensionAPI, ExtensionCommandContext, ExtensionContext, ModelRegistry } from "@earendil-works/pi-coding-agent";
|
|
34
|
+
import { getAgentDir, resizeImage } from "@earendil-works/pi-coding-agent";
|
|
34
35
|
import type { Api, ImageContent, Model, TextContent } from "@earendil-works/pi-ai";
|
|
35
36
|
import { isAbsolute, resolve } from "node:path";
|
|
37
|
+
import { existsSync, readFileSync, writeFileSync } from "node:fs";
|
|
38
|
+
import { join } from "node:path";
|
|
36
39
|
import {
|
|
37
40
|
DESCRIBE_TIMEOUT_MS,
|
|
38
41
|
extractImageFromBlock,
|
|
39
42
|
formatModelRef,
|
|
40
43
|
isThinkingLevel,
|
|
44
|
+
assistantErrorText,
|
|
45
|
+
isImageNotSupportedError,
|
|
46
|
+
isKnownTextOnlyFalselyVision,
|
|
41
47
|
isVisionModel,
|
|
42
48
|
NON_VISION_IMAGE_NOTE,
|
|
43
49
|
parseModelRef,
|
|
44
50
|
readConfig,
|
|
51
|
+
setTextOnlyInputOverride,
|
|
45
52
|
stripNonVisionImageNote,
|
|
46
53
|
writeConfig,
|
|
47
54
|
HANDOFF_COMMAND_DESCRIPTION,
|
|
@@ -56,7 +63,6 @@ import {
|
|
|
56
63
|
import { DescriptionLoader, UNAVAILABLE, type LoaderDeps } from "./src/dataloader.js";
|
|
57
64
|
import { imageHash, findPastedImagePaths, readImageBuffer, readImageBufferBounded, resolvePrewarmImage, isOmittedImageNote } from "./src/image.js";
|
|
58
65
|
import { appendVisionError } from "./src/error-log.js";
|
|
59
|
-
import { resizeImage } from "@earendil-works/pi-coding-agent";
|
|
60
66
|
import { VisionModelSelectorComponent, type VisionModelSelectorResult } from "./src/vision-model-selector.js";
|
|
61
67
|
import { PrewarmEditor } from "./src/prewarm-editor.js";
|
|
62
68
|
import { Text } from "@earendil-works/pi-tui";
|
|
@@ -238,6 +244,10 @@ function isHandoffTarget(
|
|
|
238
244
|
if (!model || !model.provider || !model.id) return false;
|
|
239
245
|
const ref = formatModelRef(model.provider, model.id);
|
|
240
246
|
if (cfg.handoffModels.includes(ref)) return true;
|
|
247
|
+
// Aggregators often mark DeepSeek V4 as vision; the backend 400s on images.
|
|
248
|
+
// Treat as text-only so the FIRST send is handed off â waiting for the 400
|
|
249
|
+
// is too late. Persist the models.json override separately so /model agrees.
|
|
250
|
+
if (cfg.autoHandoff && isKnownTextOnlyFalselyVision(model.id)) return true;
|
|
241
251
|
if (cfg.autoHandoff && !isVisionModel(model)) return true;
|
|
242
252
|
return false;
|
|
243
253
|
}
|
|
@@ -303,6 +313,46 @@ function installPrewarmEditor(ctx: ExtensionContext): void {
|
|
|
303
313
|
);
|
|
304
314
|
}
|
|
305
315
|
|
|
316
|
+
/** Write a `modelOverrides.<modelId>.input = ["text"]` fix to models.json and
|
|
317
|
+
* refresh the registry in-process so it applies without /reload (the same
|
|
318
|
+
* write+refresh pattern pi-auto-compat uses). Never touches credentials or
|
|
319
|
+
* other fields â the merge is done by the pure {@link setTextOnlyInputOverride}.
|
|
320
|
+
* Returns true when the override was written; false when models.json is
|
|
321
|
+
* unreadable/corrupt or the write fails (callers fall back to handoffModels). */
|
|
322
|
+
function fixModelInputTextOnly(ctx: ExtensionContext, providerId: string, modelId: string): boolean {
|
|
323
|
+
const path = join(getAgentDir(), "models.json");
|
|
324
|
+
let cfg: unknown;
|
|
325
|
+
try {
|
|
326
|
+
cfg = existsSync(path) ? JSON.parse(readFileSync(path, "utf8")) : {};
|
|
327
|
+
} catch {
|
|
328
|
+
return false;
|
|
329
|
+
}
|
|
330
|
+
try {
|
|
331
|
+
writeFileSync(path, JSON.stringify(setTextOnlyInputOverride(cfg, providerId, modelId), null, 2) + "\n", "utf8");
|
|
332
|
+
} catch {
|
|
333
|
+
return false;
|
|
334
|
+
}
|
|
335
|
+
// Refresh the merged registry in-process so the corrected input takes
|
|
336
|
+
// effect this session. Fire-and-forget: a refresh failure only delays the
|
|
337
|
+
// fix to the next session start (models.json is already corrected on disk).
|
|
338
|
+
void ctx.modelRegistry
|
|
339
|
+
?.refresh?.({ allowNetwork: false })
|
|
340
|
+
?.catch?.(() => {});
|
|
341
|
+
return true;
|
|
342
|
+
}
|
|
343
|
+
|
|
344
|
+
/** Persist `input: ["text"]` for a model that falsely declares image input.
|
|
345
|
+
* No-op if the model is already text-only or isn't a known false-vision id. */
|
|
346
|
+
function persistFalseVisionOverride(
|
|
347
|
+
ctx: ExtensionContext,
|
|
348
|
+
model: { provider?: string; id?: string; input?: ("text" | "image")[] } | undefined | null,
|
|
349
|
+
): boolean {
|
|
350
|
+
if (!model?.provider || !model.id) return false;
|
|
351
|
+
if (!isKnownTextOnlyFalselyVision(model.id)) return false;
|
|
352
|
+
if (!isVisionModel(model)) return false;
|
|
353
|
+
return fixModelInputTextOnly(ctx, model.provider, model.id);
|
|
354
|
+
}
|
|
355
|
+
|
|
306
356
|
function notifyUnresolvedVisionModel(ctx: ExtensionContext, ref: string): void {
|
|
307
357
|
if (visionModelUnresolvedRef === ref) return;
|
|
308
358
|
visionModelUnresolvedRef = ref;
|
|
@@ -401,6 +451,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
401
451
|
warnedHashes.clear();
|
|
402
452
|
loader.reset();
|
|
403
453
|
currentModel = ctx.model;
|
|
454
|
+
persistFalseVisionOverride(ctx, ctx.model);
|
|
404
455
|
installPrewarmEditor(ctx);
|
|
405
456
|
});
|
|
406
457
|
|
|
@@ -719,13 +770,69 @@ export default function (pi: ExtensionAPI) {
|
|
|
719
770
|
if (changed) return { messages: event.messages };
|
|
720
771
|
});
|
|
721
772
|
|
|
773
|
+
// Auto-recovery from "model does not support image" 400s: a model whose
|
|
774
|
+
// registry entry falsely declares `input: ["text","image"]` (common on
|
|
775
|
+
// aggregator providers) passes the autoHandoff vision check, so handoff is
|
|
776
|
+
// skipped and the raw image block reaches the provider â which rejects it.
|
|
777
|
+
// PRIMARY fix: correct the metadata at its layer â write a `modelOverrides`
|
|
778
|
+
// entry forcing `input: ["text"]` to models.json and refresh the registry
|
|
779
|
+
// in-process (modelOverrides is Pi's topmost config layer, so it wins even
|
|
780
|
+
// over a regenerated models[] entry). The model then registers as text-only,
|
|
781
|
+
// autoHandoff covers it naturally, and /model shows it correctly. FALLBACK
|
|
782
|
+
// (write failed): force handoff via handoffModels. Either way, the failed
|
|
783
|
+
// turn's images are still in history; the context hook describes them on the
|
|
784
|
+
// retry instead of failing again.
|
|
785
|
+
pi.on("message_end", (event, ctx) => {
|
|
786
|
+
const msg = event.message as {
|
|
787
|
+
role?: string;
|
|
788
|
+
type?: string;
|
|
789
|
+
stopReason?: string;
|
|
790
|
+
errorMessage?: unknown;
|
|
791
|
+
content?: unknown;
|
|
792
|
+
};
|
|
793
|
+
// Pi assistant messages use `role`, not `type`. Checking `type` made this
|
|
794
|
+
// handler a no-op, so the 400 never taught us anything.
|
|
795
|
+
if (msg?.role !== "assistant" && msg?.type !== "assistant") return;
|
|
796
|
+
if (msg.stopReason !== "error") return;
|
|
797
|
+
if (!isImageNotSupportedError(assistantErrorText(msg))) return;
|
|
798
|
+
if (!isConfigured(config)) return;
|
|
799
|
+
const model = ctx.model;
|
|
800
|
+
if (!model?.provider || !model.id) return;
|
|
801
|
+
const ref = formatModelRef(model.provider, model.id);
|
|
802
|
+
|
|
803
|
+
const fixed = persistFalseVisionOverride(ctx, model) || fixModelInputTextOnly(ctx, model.provider, model.id);
|
|
804
|
+
if (fixed) {
|
|
805
|
+
if (!config.handoffModels.includes(ref)) {
|
|
806
|
+
config = { ...config, handoffModels: [...config.handoffModels, ref] };
|
|
807
|
+
writeConfig(config);
|
|
808
|
+
}
|
|
809
|
+
if (ctx.hasUI) {
|
|
810
|
+
ctx.ui.notify(
|
|
811
|
+
`pi-vision-watcher: ${ref} rejected image input â marked as text-only. Resend; images will be described by ${config.visionModel}.`,
|
|
812
|
+
"warning",
|
|
813
|
+
);
|
|
814
|
+
}
|
|
815
|
+
return;
|
|
816
|
+
}
|
|
817
|
+
|
|
818
|
+
if (config.handoffModels.includes(ref)) return;
|
|
819
|
+
config = { ...config, handoffModels: [...config.handoffModels, ref] };
|
|
820
|
+
writeConfig(config);
|
|
821
|
+
if (!ctx.hasUI) return;
|
|
822
|
+
ctx.ui.notify(
|
|
823
|
+
`pi-vision-watcher: ${ref} rejected image input â handoff forced for this model. Resend; images will be described by ${config.visionModel}.`,
|
|
824
|
+
"warning",
|
|
825
|
+
);
|
|
826
|
+
});
|
|
827
|
+
|
|
722
828
|
pi.on("model_select", (event, ctx) => {
|
|
723
829
|
currentModel = event.model;
|
|
830
|
+
persistFalseVisionOverride(ctx, event.model);
|
|
724
831
|
if (!ctx.hasUI) return;
|
|
725
832
|
if (!isConfigured(config)) return;
|
|
726
833
|
const model = event.model;
|
|
727
834
|
if (!model) return;
|
|
728
|
-
if (isHandoffTarget(model, config)
|
|
835
|
+
if (isHandoffTarget(model, config)) {
|
|
729
836
|
ctx.ui.notify(
|
|
730
837
|
`pi-vision-watcher: active â images will be described by ${config.visionModel}`,
|
|
731
838
|
"info",
|