@stabgan/openrouter-mcp-multimodal 4.5.1 → 4.5.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +367 -283
- package/dist/index.js +1 -1
- package/dist/model-cache.d.ts +22 -12
- package/dist/model-cache.js +58 -21
- package/dist/tool-descriptions.d.ts +19 -0
- package/dist/tool-descriptions.js +423 -0
- package/dist/tool-handlers/analyze-audio.js +5 -1
- package/dist/tool-handlers/analyze-image.js +6 -5
- package/dist/tool-handlers/analyze-video.js +6 -5
- package/dist/tool-handlers/audio-utils.js +4 -2
- package/dist/tool-handlers/chat-completion.js +1 -1
- package/dist/tool-handlers/fetch-utils.js +16 -2
- package/dist/tool-handlers/generate-audio.js +2 -4
- package/dist/tool-handlers/generate-image-input.d.ts +3 -0
- package/dist/tool-handlers/generate-image-input.js +38 -0
- package/dist/tool-handlers/generate-image.d.ts +13 -51
- package/dist/tool-handlers/generate-image.js +32 -119
- package/dist/tool-handlers/generate-video.js +28 -24
- package/dist/tool-handlers/image-utils.d.ts +1 -0
- package/dist/tool-handlers/image-utils.js +26 -16
- package/dist/tool-handlers/openrouter-errors.js +6 -2
- package/dist/tool-handlers/provider-routing.js +7 -2
- package/dist/tool-handlers/rerank.js +2 -5
- package/dist/tool-handlers/search-models.d.ts +2 -2
- package/dist/tool-handlers/search-models.js +2 -6
- package/dist/tool-handlers/structured-output.d.ts +8 -0
- package/dist/tool-handlers/structured-output.js +11 -0
- package/dist/tool-handlers/video-utils.js +6 -9
- package/dist/tool-handlers.js +25 -123
- package/dist/version.d.ts +1 -1
- package/dist/version.js +1 -1
- package/package.json +26 -14
package/dist/index.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
import { Readable } from 'node:stream';
|
|
3
3
|
import { config } from 'dotenv';
|
|
4
|
-
config(); // Load .env file if present
|
|
4
|
+
config({ quiet: true }); // Load .env file if present (quiet — stdio transport owns stdout)
|
|
5
5
|
import { Server } from '@modelcontextprotocol/sdk/server/index.js';
|
|
6
6
|
import { StdioServerTransport } from '@modelcontextprotocol/sdk/server/stdio.js';
|
|
7
7
|
import { ToolHandlers } from './tool-handlers.js';
|
package/dist/model-cache.d.ts
CHANGED
|
@@ -8,6 +8,19 @@ export interface OpenRouterModelRecord {
|
|
|
8
8
|
context_length?: number;
|
|
9
9
|
[key: string]: unknown;
|
|
10
10
|
}
|
|
11
|
+
export interface ModelSearchParams {
|
|
12
|
+
query?: string;
|
|
13
|
+
provider?: string;
|
|
14
|
+
capabilities?: {
|
|
15
|
+
vision?: boolean;
|
|
16
|
+
audio?: boolean;
|
|
17
|
+
video?: boolean;
|
|
18
|
+
};
|
|
19
|
+
limit?: number;
|
|
20
|
+
/** When true, return the full filtered set and ignore `limit`. Used by pagination. */
|
|
21
|
+
all?: boolean;
|
|
22
|
+
}
|
|
23
|
+
export declare const MAX_SEARCH_LIMIT = 50;
|
|
11
24
|
export declare class ModelCache {
|
|
12
25
|
private static instance;
|
|
13
26
|
private models;
|
|
@@ -40,16 +53,13 @@ export declare class ModelCache {
|
|
|
40
53
|
size(): number;
|
|
41
54
|
get(id: string): OpenRouterModelRecord | null;
|
|
42
55
|
has(id: string): boolean;
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
/** When true, return the full filtered set and ignore `limit`. Used by pagination. */
|
|
53
|
-
all?: boolean;
|
|
54
|
-
}): OpenRouterModelRecord[];
|
|
56
|
+
/**
|
|
57
|
+
* Single-pass paginated search: O(n) time, O(limit) extra space for the page.
|
|
58
|
+
* Avoids materializing the full filtered array when only one page is needed.
|
|
59
|
+
*/
|
|
60
|
+
searchPaginated(params: ModelSearchParams, offset: number, limit: number): {
|
|
61
|
+
page: OpenRouterModelRecord[];
|
|
62
|
+
total: number;
|
|
63
|
+
};
|
|
64
|
+
search(params: ModelSearchParams): OpenRouterModelRecord[];
|
|
55
65
|
}
|
package/dist/model-cache.js
CHANGED
|
@@ -5,7 +5,33 @@ function getCacheTtlMs() {
|
|
|
5
5
|
const n = parseInt(raw, 10);
|
|
6
6
|
return Number.isFinite(n) && n > 0 ? n : 3600000;
|
|
7
7
|
}
|
|
8
|
-
const MAX_SEARCH_LIMIT = 50;
|
|
8
|
+
export const MAX_SEARCH_LIMIT = 50;
|
|
9
|
+
function buildMatcher(params) {
|
|
10
|
+
const q = params.query?.toLowerCase();
|
|
11
|
+
const providerPrefix = params.provider?.toLowerCase();
|
|
12
|
+
const needVision = params.capabilities?.vision === true;
|
|
13
|
+
const needAudio = params.capabilities?.audio === true;
|
|
14
|
+
const needVideo = params.capabilities?.video === true;
|
|
15
|
+
return (m) => {
|
|
16
|
+
if (q) {
|
|
17
|
+
const id = m.id.toLowerCase();
|
|
18
|
+
const name = m.name?.toLowerCase() ?? '';
|
|
19
|
+
if (!id.includes(q) && !name.includes(q))
|
|
20
|
+
return false;
|
|
21
|
+
}
|
|
22
|
+
if (providerPrefix && !m.id.toLowerCase().startsWith(`${providerPrefix}/`)) {
|
|
23
|
+
return false;
|
|
24
|
+
}
|
|
25
|
+
const mods = m.architecture?.input_modalities;
|
|
26
|
+
if (needVision && !mods?.includes('image'))
|
|
27
|
+
return false;
|
|
28
|
+
if (needAudio && !mods?.includes('audio'))
|
|
29
|
+
return false;
|
|
30
|
+
if (needVideo && !mods?.includes('video'))
|
|
31
|
+
return false;
|
|
32
|
+
return true;
|
|
33
|
+
};
|
|
34
|
+
}
|
|
9
35
|
export class ModelCache {
|
|
10
36
|
static instance;
|
|
11
37
|
models = {};
|
|
@@ -75,28 +101,39 @@ export class ModelCache {
|
|
|
75
101
|
has(id) {
|
|
76
102
|
return id in this.models;
|
|
77
103
|
}
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
104
|
+
/**
|
|
105
|
+
* Single-pass paginated search: O(n) time, O(limit) extra space for the page.
|
|
106
|
+
* Avoids materializing the full filtered array when only one page is needed.
|
|
107
|
+
*/
|
|
108
|
+
searchPaginated(params, offset, limit) {
|
|
109
|
+
const matches = buildMatcher(params);
|
|
110
|
+
const safeOffset = Math.max(0, offset);
|
|
111
|
+
const safeLimit = Math.min(Math.max(1, limit), MAX_SEARCH_LIMIT);
|
|
112
|
+
const page = [];
|
|
113
|
+
let total = 0;
|
|
114
|
+
let matchIndex = 0;
|
|
115
|
+
for (const model of Object.values(this.models)) {
|
|
116
|
+
if (!matches(model))
|
|
117
|
+
continue;
|
|
118
|
+
if (matchIndex >= safeOffset && page.length < safeLimit) {
|
|
119
|
+
page.push(model);
|
|
120
|
+
}
|
|
121
|
+
matchIndex++;
|
|
96
122
|
}
|
|
97
|
-
|
|
123
|
+
total = matchIndex;
|
|
124
|
+
return { page, total };
|
|
125
|
+
}
|
|
126
|
+
search(params) {
|
|
127
|
+
if (params.all) {
|
|
128
|
+
const matches = buildMatcher(params);
|
|
129
|
+
const results = [];
|
|
130
|
+
for (const model of Object.values(this.models)) {
|
|
131
|
+
if (matches(model))
|
|
132
|
+
results.push(model);
|
|
133
|
+
}
|
|
98
134
|
return results;
|
|
135
|
+
}
|
|
99
136
|
const limit = Math.min(Math.max(1, params.limit ?? 10), MAX_SEARCH_LIMIT);
|
|
100
|
-
return
|
|
137
|
+
return this.searchPaginated(params, 0, limit).page;
|
|
101
138
|
}
|
|
102
139
|
}
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* MCP tool descriptions with explicit routing, examples, and failure modes.
|
|
3
|
+
* See docs/plans/tool-description-improvement.md for the authoring guide.
|
|
4
|
+
*/
|
|
5
|
+
export interface ToolDescriptionParts {
|
|
6
|
+
summary: string;
|
|
7
|
+
useWhen: string[];
|
|
8
|
+
notWhen: string[];
|
|
9
|
+
goodExamples: string[];
|
|
10
|
+
badExamples: string[];
|
|
11
|
+
failsWhen: string[];
|
|
12
|
+
worksWith: string[];
|
|
13
|
+
}
|
|
14
|
+
export declare function buildToolDescription(parts: ToolDescriptionParts): string;
|
|
15
|
+
/** Required sections every tool description must contain (regression-tested). */
|
|
16
|
+
export declare const REQUIRED_DESCRIPTION_SECTIONS: readonly ["Use when:", "Do NOT use when:", "Good examples:", "Bad examples:", "Fails when:", "Works with:"];
|
|
17
|
+
export declare const TOOL_NAMES: readonly ["chat_completion", "analyze_image", "analyze_audio", "analyze_video", "search_models", "get_model_info", "validate_model", "generate_image", "generate_audio", "generate_video", "generate_video_from_image", "get_video_status", "rerank_documents", "health_check"];
|
|
18
|
+
export type ToolName = (typeof TOOL_NAMES)[number];
|
|
19
|
+
export declare const TOOL_DESCRIPTIONS: Record<ToolName, string>;
|
|
@@ -0,0 +1,423 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* MCP tool descriptions with explicit routing, examples, and failure modes.
|
|
3
|
+
* See docs/plans/tool-description-improvement.md for the authoring guide.
|
|
4
|
+
*/
|
|
5
|
+
function formatBullets(items) {
|
|
6
|
+
return items.map((item) => `- ${item}`).join('\n');
|
|
7
|
+
}
|
|
8
|
+
export function buildToolDescription(parts) {
|
|
9
|
+
return (`${parts.summary}\n\n` +
|
|
10
|
+
`Use when:\n${formatBullets(parts.useWhen)}\n\n` +
|
|
11
|
+
`Do NOT use when:\n${formatBullets(parts.notWhen)}\n\n` +
|
|
12
|
+
`Good examples:\n${formatBullets(parts.goodExamples)}\n\n` +
|
|
13
|
+
`Bad examples:\n${formatBullets(parts.badExamples)}\n\n` +
|
|
14
|
+
`Fails when:\n${formatBullets(parts.failsWhen)}\n\n` +
|
|
15
|
+
`Works with: ${parts.worksWith.join(', ')}.`);
|
|
16
|
+
}
|
|
17
|
+
/** Required sections every tool description must contain (regression-tested). */
|
|
18
|
+
export const REQUIRED_DESCRIPTION_SECTIONS = [
|
|
19
|
+
'Use when:',
|
|
20
|
+
'Do NOT use when:',
|
|
21
|
+
'Good examples:',
|
|
22
|
+
'Bad examples:',
|
|
23
|
+
'Fails when:',
|
|
24
|
+
'Works with:',
|
|
25
|
+
];
|
|
26
|
+
export const TOOL_NAMES = [
|
|
27
|
+
'chat_completion',
|
|
28
|
+
'analyze_image',
|
|
29
|
+
'analyze_audio',
|
|
30
|
+
'analyze_video',
|
|
31
|
+
'search_models',
|
|
32
|
+
'get_model_info',
|
|
33
|
+
'validate_model',
|
|
34
|
+
'generate_image',
|
|
35
|
+
'generate_audio',
|
|
36
|
+
'generate_video',
|
|
37
|
+
'generate_video_from_image',
|
|
38
|
+
'get_video_status',
|
|
39
|
+
'rerank_documents',
|
|
40
|
+
'health_check',
|
|
41
|
+
];
|
|
42
|
+
export const TOOL_DESCRIPTIONS = {
|
|
43
|
+
chat_completion: buildToolDescription({
|
|
44
|
+
summary: 'Send messages to an OpenRouter chat model and get a text reply. Supports provider routing, ' +
|
|
45
|
+
'model suffixes (`:nitro` fastest, `:floor` cheapest, `:exacto` tool accuracy), reasoning ' +
|
|
46
|
+
'tokens, web search (`online: true`), and response caching.',
|
|
47
|
+
useWhen: [
|
|
48
|
+
'You need text generation, Q&A, summarization, or multi-turn dialogue',
|
|
49
|
+
'You want web-grounded answers (`online: true`)',
|
|
50
|
+
'You already know the model id (or rely on the server default)',
|
|
51
|
+
],
|
|
52
|
+
notWhen: [
|
|
53
|
+
'Input is an image/audio/video file → use analyze_image / analyze_audio / analyze_video',
|
|
54
|
+
'You need to create images, audio, or video → use generate_* tools',
|
|
55
|
+
'You only need to check if a model exists → use validate_model',
|
|
56
|
+
],
|
|
57
|
+
goodExamples: [
|
|
58
|
+
'`{ "messages": [{ "role": "user", "content": "Explain recursion in one paragraph." }] }`',
|
|
59
|
+
'`{ "model": "openai/gpt-4o:nitro", "messages": [...], "online": true }` for web search',
|
|
60
|
+
'`{ "messages": [...], "include_reasoning": true }` for chain-of-thought models',
|
|
61
|
+
],
|
|
62
|
+
badExamples: [
|
|
63
|
+
'`{ "messages": [] }` → INVALID_INPUT (empty array)',
|
|
64
|
+
'`{ "image_path": "photo.jpg" }` → wrong tool; use analyze_image',
|
|
65
|
+
'Putting file paths inside message content without a vision model configured',
|
|
66
|
+
],
|
|
67
|
+
failsWhen: [
|
|
68
|
+
'INVALID_INPUT: messages array is empty',
|
|
69
|
+
'UPSTREAM_REFUSED: credits, content policy, or rate limit',
|
|
70
|
+
'UPSTREAM_TIMEOUT: upstream did not respond in time',
|
|
71
|
+
'MODEL_NOT_FOUND: model slug does not exist on OpenRouter',
|
|
72
|
+
],
|
|
73
|
+
worksWith: ['validate_model', 'search_models'],
|
|
74
|
+
}),
|
|
75
|
+
analyze_image: buildToolDescription({
|
|
76
|
+
summary: 'Analyze one image with a vision model. Accepts a sandboxed local path, https URL, or base64 data URL. ' +
|
|
77
|
+
'Output is model-generated and tagged `_meta.content_is_untrusted: true`.',
|
|
78
|
+
useWhen: [
|
|
79
|
+
'You have one image and need OCR, captioning, or visual Q&A',
|
|
80
|
+
'The image is a local file under the input sandbox, a public https URL, or a data URL',
|
|
81
|
+
],
|
|
82
|
+
notWhen: [
|
|
83
|
+
'You want to generate a new image → use generate_image',
|
|
84
|
+
'You need multi-file batch analysis in one call → not supported; call once per image',
|
|
85
|
+
'Pure text chat → use chat_completion with a vision-capable model instead (less ergonomic)',
|
|
86
|
+
],
|
|
87
|
+
goodExamples: [
|
|
88
|
+
'`{ "image_path": "diagram.png", "question": "List every label in this diagram." }`',
|
|
89
|
+
'`{ "image_path": "https://example.com/photo.jpg", "question": "Describe the scene." }`',
|
|
90
|
+
'`{ "model": "google/gemini-2.5-flash", "image_path": "scan.jpg", "question": "Extract text" }`',
|
|
91
|
+
],
|
|
92
|
+
badExamples: [
|
|
93
|
+
'`{ "url": "photo.jpg" }` → wrong key; use `image_path`',
|
|
94
|
+
'`{ "prompt": "describe" }` → wrong key; use `question` (optional, defaults to "What\'s in this image?")',
|
|
95
|
+
'`{ "image_path": "../../../etc/passwd" }` → UNSAFE_PATH (sandbox escape)',
|
|
96
|
+
],
|
|
97
|
+
failsWhen: [
|
|
98
|
+
'INVALID_INPUT: image_path missing or malformed',
|
|
99
|
+
'UNSAFE_PATH: local path escaped the input sandbox',
|
|
100
|
+
'RESOURCE_TOO_LARGE: image exceeded fetch size cap',
|
|
101
|
+
'UPSTREAM_REFUSED: SSRF block, bad URL, or content policy',
|
|
102
|
+
],
|
|
103
|
+
worksWith: ['search_models', 'generate_image'],
|
|
104
|
+
}),
|
|
105
|
+
analyze_audio: buildToolDescription({
|
|
106
|
+
summary: 'Transcribe or analyze one audio file (WAV, MP3, FLAC, OGG, etc.) with a multimodal model. ' +
|
|
107
|
+
'Output is tagged `_meta.content_is_untrusted: true`.',
|
|
108
|
+
useWhen: [
|
|
109
|
+
'You have a local audio file or URL and need transcription or audio understanding',
|
|
110
|
+
'Format is a common audio container the decoder recognizes',
|
|
111
|
+
],
|
|
112
|
+
notWhen: [
|
|
113
|
+
'You want text-to-speech → use generate_audio',
|
|
114
|
+
'Input is video → use analyze_video (or extract audio first)',
|
|
115
|
+
'Pure text chat → use chat_completion',
|
|
116
|
+
],
|
|
117
|
+
goodExamples: [
|
|
118
|
+
'`{ "audio_path": "meeting.wav", "question": "Transcribe verbatim." }`',
|
|
119
|
+
'`{ "audio_path": "https://example.com/podcast.mp3", "question": "Summarize topics." }`',
|
|
120
|
+
],
|
|
121
|
+
badExamples: [
|
|
122
|
+
'`{ "audio_path": "/etc/shadow" }` → UNSAFE_PATH',
|
|
123
|
+
'`{ "path": "song.mp3" }` → wrong key; use `audio_path`',
|
|
124
|
+
'Non-audio binary renamed to .mp3 → UNSUPPORTED_FORMAT',
|
|
125
|
+
],
|
|
126
|
+
failsWhen: [
|
|
127
|
+
'INVALID_INPUT: audio_path missing',
|
|
128
|
+
'UNSAFE_PATH: local path escaped the sandbox',
|
|
129
|
+
'UNSUPPORTED_FORMAT: file is not recognized as audio',
|
|
130
|
+
'RESOURCE_TOO_LARGE: exceeds size cap',
|
|
131
|
+
],
|
|
132
|
+
worksWith: ['generate_audio', 'search_models'],
|
|
133
|
+
}),
|
|
134
|
+
analyze_video: buildToolDescription({
|
|
135
|
+
summary: 'Describe or analyze one video file (mp4, mpeg, mov, webm). Default model: google/gemini-2.5-flash. ' +
|
|
136
|
+
'Output is tagged `_meta.content_is_untrusted: true`. Large files are fully buffered — prefer short clips.',
|
|
137
|
+
useWhen: [
|
|
138
|
+
'You need a summary, scene description, or Q&A over a video file',
|
|
139
|
+
'Video is within size limits and readable by the decoder',
|
|
140
|
+
],
|
|
141
|
+
notWhen: [
|
|
142
|
+
'You want to generate video → use generate_video',
|
|
143
|
+
'You only need audio → use analyze_audio',
|
|
144
|
+
'Video is very large → trim first or expect RESOURCE_TOO_LARGE',
|
|
145
|
+
],
|
|
146
|
+
goodExamples: [
|
|
147
|
+
'`{ "video_path": "clip.mp4", "question": "What happens in the first 30 seconds?" }`',
|
|
148
|
+
'`{ "video_path": "demo.webm", "question": "List on-screen text." }`',
|
|
149
|
+
],
|
|
150
|
+
badExamples: [
|
|
151
|
+
'`{ "video_path": "../secret.mp4" }` → UNSAFE_PATH',
|
|
152
|
+
'Expecting frame-by-frame timestamps without asking in the prompt',
|
|
153
|
+
'Using analyze_video for async generation status → use get_video_status',
|
|
154
|
+
],
|
|
155
|
+
failsWhen: [
|
|
156
|
+
'INVALID_INPUT: video_path missing',
|
|
157
|
+
'UNSAFE_PATH: path escaped the sandbox',
|
|
158
|
+
'UNSUPPORTED_FORMAT: unrecognized video container',
|
|
159
|
+
'RESOURCE_TOO_LARGE: exceeds fetch cap',
|
|
160
|
+
],
|
|
161
|
+
worksWith: ['generate_video', 'get_video_status', 'search_models'],
|
|
162
|
+
}),
|
|
163
|
+
search_models: buildToolDescription({
|
|
164
|
+
summary: 'Search the OpenRouter model catalog by name, provider, or capability. Returns a paginated slice; ' +
|
|
165
|
+
'use `offset`, `limit`, and `next_offset` to page through large result sets.',
|
|
166
|
+
useWhen: [
|
|
167
|
+
'You do not know which model id to use',
|
|
168
|
+
'You need vision/audio/video-capable models filtered by modality',
|
|
169
|
+
'You want models from a specific provider prefix (e.g. `google`)',
|
|
170
|
+
],
|
|
171
|
+
notWhen: [
|
|
172
|
+
'You already have a model id and only need existence check → validate_model',
|
|
173
|
+
'You need pricing/context details for one id → get_model_info',
|
|
174
|
+
'You expect all 400+ models in one response without paging',
|
|
175
|
+
],
|
|
176
|
+
goodExamples: [
|
|
177
|
+
'`{ "query": "gemini", "capabilities": { "vision": true }, "limit": 10, "offset": 0 }`',
|
|
178
|
+
'`{ "provider": "anthropic", "limit": 20 }`',
|
|
179
|
+
'Page 2: `{ "query": "llama", "offset": 20, "limit": 20 }` using prior `next_offset`',
|
|
180
|
+
],
|
|
181
|
+
badExamples: [
|
|
182
|
+
'Omitting pagination on broad queries → large payload; use limit/offset',
|
|
183
|
+
'Using search_models output as chat messages → use returned `id` in chat_completion',
|
|
184
|
+
'`{ "capability": "vision" }` → wrong shape; use `capabilities: { "vision": true }`',
|
|
185
|
+
],
|
|
186
|
+
failsWhen: ['UPSTREAM_HTTP: /models endpoint error', 'UPSTREAM_REFUSED: invalid API key'],
|
|
187
|
+
worksWith: ['validate_model', 'get_model_info'],
|
|
188
|
+
}),
|
|
189
|
+
get_model_info: buildToolDescription({
|
|
190
|
+
summary: 'Return pricing, context length, and modality architecture for one model id from the cached catalog.',
|
|
191
|
+
useWhen: [
|
|
192
|
+
'You have a model id and need context window, pricing, or input/output modalities',
|
|
193
|
+
'You are choosing between two known model slugs',
|
|
194
|
+
],
|
|
195
|
+
notWhen: [
|
|
196
|
+
'You only need true/false existence → validate_model (cheaper)',
|
|
197
|
+
'You are browsing unknown models → search_models first',
|
|
198
|
+
],
|
|
199
|
+
goodExamples: [
|
|
200
|
+
'`{ "model": "openai/gpt-4o" }`',
|
|
201
|
+
'`{ "model": "google/gemini-2.5-flash" }` before analyze_video',
|
|
202
|
+
],
|
|
203
|
+
badExamples: [
|
|
204
|
+
'`{ "model": "" }` → INVALID_INPUT',
|
|
205
|
+
'`{ "name": "gpt-4o" }` → wrong key; use `model` with full slug `openai/gpt-4o`',
|
|
206
|
+
'Calling repeatedly in a loop → cache is shared; call once per id',
|
|
207
|
+
],
|
|
208
|
+
failsWhen: [
|
|
209
|
+
'INVALID_INPUT: model not provided',
|
|
210
|
+
'MODEL_NOT_FOUND: slug not in catalog',
|
|
211
|
+
'UPSTREAM_HTTP: catalog refresh failed',
|
|
212
|
+
],
|
|
213
|
+
worksWith: ['search_models', 'validate_model'],
|
|
214
|
+
}),
|
|
215
|
+
validate_model: buildToolDescription({
|
|
216
|
+
summary: 'Cheap boolean check: does this model id exist in the OpenRouter catalog? Uses the shared cache.',
|
|
217
|
+
useWhen: [
|
|
218
|
+
'Pre-flight before chat_completion or generate_* to avoid MODEL_NOT_FOUND',
|
|
219
|
+
'You only need `{ valid: true|false }`, not pricing or modalities',
|
|
220
|
+
],
|
|
221
|
+
notWhen: [
|
|
222
|
+
'You need pricing or context length → get_model_info',
|
|
223
|
+
'You are discovering models → search_models',
|
|
224
|
+
],
|
|
225
|
+
goodExamples: [
|
|
226
|
+
'`{ "model": "anthropic/claude-sonnet-4" }` → `{ "valid": true, "model": "..." }`',
|
|
227
|
+
'`{ "model": "fake/model" }` → `{ "valid": false }` (not an error)',
|
|
228
|
+
],
|
|
229
|
+
badExamples: [
|
|
230
|
+
'Treating `valid: false` as a tool error — it is a successful response',
|
|
231
|
+
'Using validate_model to search partial names → use search_models with `query`',
|
|
232
|
+
],
|
|
233
|
+
failsWhen: ['INVALID_INPUT: model not provided', 'UPSTREAM_HTTP: catalog refresh failed'],
|
|
234
|
+
worksWith: ['get_model_info', 'chat_completion'],
|
|
235
|
+
}),
|
|
236
|
+
generate_image: buildToolDescription({
|
|
237
|
+
summary: 'Generate an image from a text prompt. Optional `input_images` condition style/identity. ' +
|
|
238
|
+
'Default model: google/gemini-2.5-flash-image.',
|
|
239
|
+
useWhen: [
|
|
240
|
+
'You need a new image from a text prompt',
|
|
241
|
+
'You have reference images for style or subject consistency',
|
|
242
|
+
],
|
|
243
|
+
notWhen: [
|
|
244
|
+
'You want to analyze an existing image → analyze_image',
|
|
245
|
+
'You want video → generate_video or generate_video_from_image',
|
|
246
|
+
'Prompt is empty or only whitespace',
|
|
247
|
+
],
|
|
248
|
+
goodExamples: [
|
|
249
|
+
'`{ "prompt": "A watercolor fox in autumn leaves" }`',
|
|
250
|
+
'`{ "prompt": "Same character", "input_images": ["ref.png"], "aspect_ratio": "16:9" }`',
|
|
251
|
+
'`{ "prompt": "Logo", "save_path": "out/logo.png" }` inside output sandbox',
|
|
252
|
+
],
|
|
253
|
+
badExamples: [
|
|
254
|
+
'`{ "prompt": "" }` → INVALID_INPUT',
|
|
255
|
+
'`{ "input_images": ["/etc/passwd"] }` → UNSAFE_PATH',
|
|
256
|
+
'`{ "aspect_ratio": "21:9" }` if not in allowed enum → INVALID_INPUT',
|
|
257
|
+
],
|
|
258
|
+
failsWhen: [
|
|
259
|
+
'INVALID_INPUT: empty prompt, bad aspect_ratio/image_size, unreadable reference',
|
|
260
|
+
'UNSAFE_PATH: save_path or input_images escaped sandbox',
|
|
261
|
+
'UPSTREAM_REFUSED: content policy or insufficient credits',
|
|
262
|
+
'MODEL_NOT_FOUND: invalid model slug',
|
|
263
|
+
],
|
|
264
|
+
worksWith: ['analyze_image', 'generate_video_from_image'],
|
|
265
|
+
}),
|
|
266
|
+
generate_audio: buildToolDescription({
|
|
267
|
+
summary: 'Generate speech or music from a text prompt. Output format is auto-detected; file extension auto-corrected on save.',
|
|
268
|
+
useWhen: [
|
|
269
|
+
'You need TTS or audio generation from text',
|
|
270
|
+
'Optional save_path is inside the output sandbox',
|
|
271
|
+
],
|
|
272
|
+
notWhen: ['You want to transcribe existing audio → analyze_audio', 'Prompt is empty'],
|
|
273
|
+
goodExamples: [
|
|
274
|
+
'`{ "prompt": "Say hello world in a calm voice." }`',
|
|
275
|
+
'`{ "prompt": "Upbeat jingle", "save_path": "out/jingle.mp3" }`',
|
|
276
|
+
],
|
|
277
|
+
badExamples: [
|
|
278
|
+
'`{ "text": "hello" }` → wrong key; use `prompt`',
|
|
279
|
+
'`{ "save_path": "../../../tmp/out.wav" }` → UNSAFE_PATH',
|
|
280
|
+
],
|
|
281
|
+
failsWhen: [
|
|
282
|
+
'INVALID_INPUT: prompt empty',
|
|
283
|
+
'UNSAFE_PATH: save_path escaped sandbox',
|
|
284
|
+
'UPSTREAM_REFUSED: content policy or credits',
|
|
285
|
+
],
|
|
286
|
+
worksWith: ['analyze_audio'],
|
|
287
|
+
}),
|
|
288
|
+
generate_video: buildToolDescription({
|
|
289
|
+
summary: 'Generate video from a text prompt (optional first/last frame or reference images). Submits an async job, ' +
|
|
290
|
+
'polls until `max_wait_ms`, downloads on completion. Emits MCP progress when client sends `progressToken`. ' +
|
|
291
|
+
'Default model: google/veo-3.1.',
|
|
292
|
+
useWhen: [
|
|
293
|
+
'You need text-to-video or frame-conditioned video',
|
|
294
|
+
'You can wait for polling or resume later with get_video_status',
|
|
295
|
+
'You need last_frame or multiple reference_images (not available on generate_video_from_image)',
|
|
296
|
+
],
|
|
297
|
+
notWhen: [
|
|
298
|
+
'You only have one image and simple image-to-video → generate_video_from_image (fewer params)',
|
|
299
|
+
'Job already submitted → get_video_status with `video_id`',
|
|
300
|
+
'You want to analyze existing video → analyze_video',
|
|
301
|
+
],
|
|
302
|
+
goodExamples: [
|
|
303
|
+
'`{ "prompt": "Ocean waves at sunset, cinematic" }`',
|
|
304
|
+
'Timeout resume: response has `_meta.code: JOB_STILL_RUNNING` and `_meta.video_id` → call `get_video_status`',
|
|
305
|
+
'`{ "prompt": "Morph", "first_frame_image": "a.jpg", "last_frame_image": "b.jpg" }`',
|
|
306
|
+
],
|
|
307
|
+
badExamples: [
|
|
308
|
+
'Treating JOB_STILL_RUNNING as failure — it is success with resume metadata',
|
|
309
|
+
'`{ "prompt": " " }` → INVALID_INPUT',
|
|
310
|
+
'Polling get_video_status in the same turn without waiting → expect JOB_STILL_RUNNING again',
|
|
311
|
+
],
|
|
312
|
+
failsWhen: [
|
|
313
|
+
'INVALID_INPUT: empty prompt',
|
|
314
|
+
'UNSAFE_PATH: save_path or image paths escaped sandbox',
|
|
315
|
+
'UPSTREAM_REFUSED: policy, credits, or bad request',
|
|
316
|
+
'JOB_FAILED: provider marked job failed',
|
|
317
|
+
],
|
|
318
|
+
worksWith: ['get_video_status', 'generate_video_from_image'],
|
|
319
|
+
}),
|
|
320
|
+
generate_video_from_image: buildToolDescription({
|
|
321
|
+
summary: 'Narrow image-to-video wrapper: one `image` (first frame) + `prompt`. Fewer parameters → higher tool-call accuracy. ' +
|
|
322
|
+
'For last-frame or reference images use generate_video.',
|
|
323
|
+
useWhen: [
|
|
324
|
+
'Single reference image + motion prompt is enough',
|
|
325
|
+
'You want the smallest argument surface for image-to-video',
|
|
326
|
+
],
|
|
327
|
+
notWhen: [
|
|
328
|
+
'You need last_frame_image or reference_images[] → generate_video',
|
|
329
|
+
'Checking job status → get_video_status',
|
|
330
|
+
],
|
|
331
|
+
goodExamples: [
|
|
332
|
+
'`{ "image": "start.png", "prompt": "Camera slowly zooms in" }`',
|
|
333
|
+
'On timeout: same JOB_STILL_RUNNING + video_id resume as generate_video',
|
|
334
|
+
],
|
|
335
|
+
badExamples: [
|
|
336
|
+
'`{ "first_frame_image": "x.png" }` → wrong key; use `image`',
|
|
337
|
+
'Passing video_id here → use get_video_status',
|
|
338
|
+
],
|
|
339
|
+
failsWhen: [
|
|
340
|
+
'INVALID_INPUT: image or prompt missing',
|
|
341
|
+
'UNSAFE_PATH: image path escaped sandbox',
|
|
342
|
+
'UPSTREAM_REFUSED / JOB_FAILED: same as generate_video',
|
|
343
|
+
],
|
|
344
|
+
worksWith: ['generate_video', 'get_video_status'],
|
|
345
|
+
}),
|
|
346
|
+
get_video_status: buildToolDescription({
|
|
347
|
+
summary: 'Poll an async video job by id. Downloads and optionally saves when complete. ' +
|
|
348
|
+
'Still running → success with `_meta.code: JOB_STILL_RUNNING` (not an error).',
|
|
349
|
+
useWhen: [
|
|
350
|
+
'generate_video returned JOB_STILL_RUNNING or you have a video_id from a prior call',
|
|
351
|
+
'You want to check progress without resubmitting',
|
|
352
|
+
],
|
|
353
|
+
notWhen: [
|
|
354
|
+
'Starting a new generation → generate_video or generate_video_from_image',
|
|
355
|
+
'You do not have a video_id yet',
|
|
356
|
+
],
|
|
357
|
+
goodExamples: [
|
|
358
|
+
'`{ "video_id": "vid_abc123" }`',
|
|
359
|
+
'`{ "video_id": "vid_abc123", "save_path": "out/clip.mp4" }`',
|
|
360
|
+
'Repeat until status completes or you accept partial progress from `_meta.progress`',
|
|
361
|
+
],
|
|
362
|
+
badExamples: [
|
|
363
|
+
'`{ "id": "vid_abc" }` → wrong key; use `video_id`',
|
|
364
|
+
'Expecting instant completion on first poll for long jobs',
|
|
365
|
+
'Treating JOB_STILL_RUNNING as tool failure',
|
|
366
|
+
],
|
|
367
|
+
failsWhen: [
|
|
368
|
+
'INVALID_INPUT: video_id missing',
|
|
369
|
+
'UNSAFE_PATH: save_path escaped sandbox',
|
|
370
|
+
'JOB_FAILED: provider marked job failed',
|
|
371
|
+
],
|
|
372
|
+
worksWith: ['generate_video', 'generate_video_from_image'],
|
|
373
|
+
}),
|
|
374
|
+
rerank_documents: buildToolDescription({
|
|
375
|
+
summary: 'Re-order documents by relevance to a query using an OpenRouter reranker. Default: cohere/rerank-english-v3.0.',
|
|
376
|
+
useWhen: [
|
|
377
|
+
'You have a query and a list of text snippets to sort by relevance',
|
|
378
|
+
'You will feed top results into chat_completion for grounded answers',
|
|
379
|
+
],
|
|
380
|
+
notWhen: [
|
|
381
|
+
'You need to fetch documents from the web → chat_completion with online or external retrieval first',
|
|
382
|
+
'documents is empty or contains non-strings',
|
|
383
|
+
],
|
|
384
|
+
goodExamples: [
|
|
385
|
+
'`{ "query": "battery life", "documents": ["Doc A text...", "Doc B text..."] }`',
|
|
386
|
+
'`{ "query": "...", "documents": [...], "model": "cohere/rerank-english-v3.0" }`',
|
|
387
|
+
],
|
|
388
|
+
badExamples: [
|
|
389
|
+
'`{ "documents": [] }` → INVALID_INPUT',
|
|
390
|
+
'`{ "query": "x", "documents": [{ "text": "y" }] }` → elements must be strings',
|
|
391
|
+
'Using rerank output as model messages without extracting text fields',
|
|
392
|
+
],
|
|
393
|
+
failsWhen: [
|
|
394
|
+
'INVALID_INPUT: query missing, documents empty, or non-string elements',
|
|
395
|
+
'MODEL_NOT_FOUND: reranker slug invalid',
|
|
396
|
+
'UPSTREAM_HTTP: provider error',
|
|
397
|
+
],
|
|
398
|
+
worksWith: ['search_models', 'chat_completion'],
|
|
399
|
+
}),
|
|
400
|
+
health_check: buildToolDescription({
|
|
401
|
+
summary: 'Verify API key, OpenRouter reachability, cached model count, and server/protocol versions. No arguments.',
|
|
402
|
+
useWhen: [
|
|
403
|
+
'Startup / ops probe before other tools',
|
|
404
|
+
'You need `{ ok, api_key_valid }` without triggering generation costs',
|
|
405
|
+
],
|
|
406
|
+
notWhen: [
|
|
407
|
+
'You need to test a specific model quality → use chat_completion with a tiny prompt',
|
|
408
|
+
'You expect isError on bad API key — this tool always returns structured payload',
|
|
409
|
+
],
|
|
410
|
+
goodExamples: [
|
|
411
|
+
'`{}` — empty args',
|
|
412
|
+
'Branch on `structuredContent.api_key_valid === false` to prompt re-auth',
|
|
413
|
+
],
|
|
414
|
+
badExamples: [
|
|
415
|
+
'Passing model or prompt — ignored; not a chat tool',
|
|
416
|
+
'Expecting isError: true on failure — check `ok` field instead',
|
|
417
|
+
],
|
|
418
|
+
failsWhen: [
|
|
419
|
+
'Never returns isError — always `{ ok, api_key_valid, ... }` for programmatic branching',
|
|
420
|
+
],
|
|
421
|
+
worksWith: ['every other tool (run once at startup)'],
|
|
422
|
+
}),
|
|
423
|
+
};
|
|
@@ -1,9 +1,10 @@
|
|
|
1
1
|
import { prepareAudioData } from './audio-utils.js';
|
|
2
|
+
import { UnsafeOutputPathError } from './path-safety.js';
|
|
2
3
|
import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
|
|
3
4
|
import { SERVER_VERSION } from '../version.js';
|
|
4
5
|
import { classifyUpstreamError } from './openrouter-errors.js';
|
|
5
6
|
import { extractCompletionText, detectReasoningCutoff, buildCompletionMeta, } from './completion-utils.js';
|
|
6
|
-
import { buildCacheHeaders, extractCacheMeta
|
|
7
|
+
import { buildCacheHeaders, extractCacheMeta } from './cache.js';
|
|
7
8
|
import { awaitCompletionWithHeaders } from './openai-withresponse.js';
|
|
8
9
|
const DEFAULT_MODEL = 'google/gemini-2.5-flash';
|
|
9
10
|
export async function handleAnalyzeAudio(request, openai, defaultModel) {
|
|
@@ -17,6 +18,9 @@ export async function handleAnalyzeAudio(request, openai, defaultModel) {
|
|
|
17
18
|
audioData = await prepareAudioData(audio_path);
|
|
18
19
|
}
|
|
19
20
|
catch (err) {
|
|
21
|
+
if (err instanceof UnsafeOutputPathError) {
|
|
22
|
+
return toolErrorFrom(ErrorCode.UNSAFE_PATH, err);
|
|
23
|
+
}
|
|
20
24
|
const msg = err instanceof Error ? err.message : String(err);
|
|
21
25
|
if (msg.includes('Blocked host'))
|
|
22
26
|
return toolErrorFrom(ErrorCode.UPSTREAM_REFUSED, err);
|
|
@@ -1,9 +1,10 @@
|
|
|
1
1
|
import { prepareImageUrl } from './image-utils.js';
|
|
2
|
+
import { UnsafeOutputPathError } from './path-safety.js';
|
|
2
3
|
import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
|
|
3
4
|
import { SERVER_VERSION } from '../version.js';
|
|
4
5
|
import { classifyUpstreamError } from './openrouter-errors.js';
|
|
5
6
|
import { extractCompletionText, detectReasoningCutoff, buildCompletionMeta, } from './completion-utils.js';
|
|
6
|
-
import { buildCacheHeaders, extractCacheMeta
|
|
7
|
+
import { buildCacheHeaders, extractCacheMeta } from './cache.js';
|
|
7
8
|
import { awaitCompletionWithHeaders } from './openai-withresponse.js';
|
|
8
9
|
const DEFAULT_MODEL = 'nvidia/nemotron-nano-12b-v2-vl:free';
|
|
9
10
|
export async function handleAnalyzeImage(request, openai, defaultModel) {
|
|
@@ -17,6 +18,9 @@ export async function handleAnalyzeImage(request, openai, defaultModel) {
|
|
|
17
18
|
imageUrl = await prepareImageUrl(image_path);
|
|
18
19
|
}
|
|
19
20
|
catch (err) {
|
|
21
|
+
if (err instanceof UnsafeOutputPathError) {
|
|
22
|
+
return toolErrorFrom(ErrorCode.UNSAFE_PATH, err);
|
|
23
|
+
}
|
|
20
24
|
const msg = err instanceof Error ? err.message : String(err);
|
|
21
25
|
if (msg.includes('Blocked host'))
|
|
22
26
|
return toolErrorFrom(ErrorCode.UPSTREAM_REFUSED, err);
|
|
@@ -44,10 +48,7 @@ export async function handleAnalyzeImage(request, openai, defaultModel) {
|
|
|
44
48
|
messages: [
|
|
45
49
|
{
|
|
46
50
|
role: 'user',
|
|
47
|
-
content: [
|
|
48
|
-
{ type: 'text', text: question || "What's in this image?" },
|
|
49
|
-
imageBlock,
|
|
50
|
-
],
|
|
51
|
+
content: [{ type: 'text', text: question || "What's in this image?" }, imageBlock],
|
|
51
52
|
},
|
|
52
53
|
],
|
|
53
54
|
}, requestOpts);
|