@stabgan/openrouter-mcp-multimodal 4.0.1 → 4.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +95 -8
- package/dist/errors.d.ts +26 -8
- package/dist/errors.js +17 -12
- package/dist/index.js +21 -5
- package/dist/logger.d.ts +11 -0
- package/dist/logger.js +26 -0
- package/dist/model-cache.d.ts +17 -0
- package/dist/model-cache.js +27 -1
- package/dist/openrouter-api.d.ts +22 -2
- package/dist/openrouter-api.js +21 -2
- package/dist/tool-handlers/analyze-audio.d.ts +9 -4
- package/dist/tool-handlers/analyze-audio.js +29 -15
- package/dist/tool-handlers/analyze-image.d.ts +10 -4
- package/dist/tool-handlers/analyze-image.js +37 -9
- package/dist/tool-handlers/analyze-video.d.ts +9 -4
- package/dist/tool-handlers/analyze-video.js +31 -19
- package/dist/tool-handlers/cache.d.ts +33 -0
- package/dist/tool-handlers/cache.js +54 -0
- package/dist/tool-handlers/chat-completion.d.ts +18 -4
- package/dist/tool-handlers/chat-completion.js +39 -11
- package/dist/tool-handlers/completion-utils.d.ts +29 -0
- package/dist/tool-handlers/completion-utils.js +76 -17
- package/dist/tool-handlers/generate-audio.d.ts +2 -0
- package/dist/tool-handlers/generate-audio.js +18 -1
- package/dist/tool-handlers/generate-image.d.ts +2 -0
- package/dist/tool-handlers/generate-image.js +15 -0
- package/dist/tool-handlers/generate-video.d.ts +43 -0
- package/dist/tool-handlers/generate-video.js +46 -3
- package/dist/tool-handlers/get-model-info.d.ts +1 -6
- package/dist/tool-handlers/get-model-info.js +2 -1
- package/dist/tool-handlers/health-check.d.ts +23 -0
- package/dist/tool-handlers/health-check.js +35 -0
- package/dist/tool-handlers/openai-withresponse.d.ts +16 -0
- package/dist/tool-handlers/openai-withresponse.js +16 -0
- package/dist/tool-handlers/openrouter-errors.d.ts +5 -1
- package/dist/tool-handlers/openrouter-errors.js +72 -11
- package/dist/tool-handlers/rerank.d.ts +17 -0
- package/dist/tool-handlers/rerank.js +52 -0
- package/dist/tool-handlers/search-models.d.ts +18 -7
- package/dist/tool-handlers/search-models.js +25 -2
- package/dist/tool-handlers/structured-output.d.ts +13 -0
- package/dist/tool-handlers/structured-output.js +24 -0
- package/dist/tool-handlers/validate-model.d.ts +4 -6
- package/dist/tool-handlers/validate-model.js +3 -8
- package/dist/tool-handlers.d.ts +1 -0
- package/dist/tool-handlers.js +435 -165
- package/dist/version.d.ts +16 -0
- package/dist/version.js +16 -0
- package/package.json +1 -1
|
@@ -1,8 +1,16 @@
|
|
|
1
1
|
import OpenAI from 'openai';
|
|
2
|
-
|
|
2
|
+
import { type CacheOptions } from './cache.js';
|
|
3
|
+
export interface AnalyzeImageToolRequest extends CacheOptions {
|
|
3
4
|
image_path: string;
|
|
4
5
|
question?: string;
|
|
5
6
|
model?: string;
|
|
7
|
+
/**
|
|
8
|
+
* When true, attach Anthropic-style `cache_control: {type: 'ephemeral'}`
|
|
9
|
+
* to the image block so Claude / Gemini 2.5+ prompt-caches it. Repeat
|
|
10
|
+
* questions about the same image then cost ~0.1x on Anthropic and
|
|
11
|
+
* ~0.25x on Gemini for the image input.
|
|
12
|
+
*/
|
|
13
|
+
cache_input?: boolean;
|
|
6
14
|
}
|
|
7
15
|
export declare function handleAnalyzeImage(request: {
|
|
8
16
|
params: {
|
|
@@ -13,7 +21,5 @@ export declare function handleAnalyzeImage(request: {
|
|
|
13
21
|
type: "text";
|
|
14
22
|
text: string;
|
|
15
23
|
}[];
|
|
16
|
-
_meta:
|
|
17
|
-
finish_reason: "length" | "stop" | "tool_calls" | "content_filter" | "function_call" | undefined;
|
|
18
|
-
};
|
|
24
|
+
_meta: Record<string, unknown>;
|
|
19
25
|
}>;
|
|
@@ -1,10 +1,14 @@
|
|
|
1
1
|
import { prepareImageUrl } from './image-utils.js';
|
|
2
2
|
import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
|
|
3
|
+
import { SERVER_VERSION } from '../version.js';
|
|
3
4
|
import { classifyUpstreamError } from './openrouter-errors.js';
|
|
4
|
-
import { extractCompletionText, detectReasoningCutoff,
|
|
5
|
+
import { extractCompletionText, detectReasoningCutoff, buildCompletionMeta, } from './completion-utils.js';
|
|
6
|
+
import { buildCacheHeaders, extractCacheMeta, } from './cache.js';
|
|
7
|
+
import { awaitCompletionWithHeaders } from './openai-withresponse.js';
|
|
5
8
|
const DEFAULT_MODEL = 'nvidia/nemotron-nano-12b-v2-vl:free';
|
|
6
9
|
export async function handleAnalyzeImage(request, openai, defaultModel) {
|
|
7
|
-
const
|
|
10
|
+
const args = request.params.arguments ?? { image_path: '' };
|
|
11
|
+
const { image_path, question, model, cache_input, cache, cache_ttl, cache_clear } = args;
|
|
8
12
|
if (!image_path) {
|
|
9
13
|
return toolError(ErrorCode.INVALID_INPUT, 'image_path is required.');
|
|
10
14
|
}
|
|
@@ -21,20 +25,35 @@ export async function handleAnalyzeImage(request, openai, defaultModel) {
|
|
|
21
25
|
}
|
|
22
26
|
return toolErrorFrom(ErrorCode.INVALID_INPUT, err);
|
|
23
27
|
}
|
|
28
|
+
// Attach `cache_control` to the image block when requested. The openai
|
|
29
|
+
// SDK doesn't type this field but passes it through to the server,
|
|
30
|
+
// which forwards it to providers that support prompt caching.
|
|
31
|
+
const imageBlock = {
|
|
32
|
+
type: 'image_url',
|
|
33
|
+
image_url: { url: imageUrl },
|
|
34
|
+
};
|
|
35
|
+
if (cache_input)
|
|
36
|
+
imageBlock.cache_control = { type: 'ephemeral' };
|
|
37
|
+
const headers = buildCacheHeaders({ cache, cache_ttl, cache_clear });
|
|
38
|
+
const requestOpts = Object.keys(headers).length > 0 ? { headers } : undefined;
|
|
24
39
|
let completion;
|
|
40
|
+
let responseHeaders;
|
|
25
41
|
try {
|
|
26
|
-
|
|
42
|
+
const call = openai.chat.completions.create({
|
|
27
43
|
model: model || defaultModel || DEFAULT_MODEL,
|
|
28
44
|
messages: [
|
|
29
45
|
{
|
|
30
46
|
role: 'user',
|
|
31
47
|
content: [
|
|
32
48
|
{ type: 'text', text: question || "What's in this image?" },
|
|
33
|
-
|
|
49
|
+
imageBlock,
|
|
34
50
|
],
|
|
35
51
|
},
|
|
36
52
|
],
|
|
37
|
-
});
|
|
53
|
+
}, requestOpts);
|
|
54
|
+
const { data, response } = await awaitCompletionWithHeaders(call);
|
|
55
|
+
completion = data;
|
|
56
|
+
responseHeaders = response?.headers;
|
|
38
57
|
}
|
|
39
58
|
catch (err) {
|
|
40
59
|
return classifyUpstreamError(err);
|
|
@@ -48,11 +67,20 @@ export async function handleAnalyzeImage(request, openai, defaultModel) {
|
|
|
48
67
|
finish_reason: extracted.finishReason,
|
|
49
68
|
});
|
|
50
69
|
}
|
|
70
|
+
const cacheMeta = extractCacheMeta(responseHeaders);
|
|
71
|
+
// Output originates from model interpretation of potentially
|
|
72
|
+
// attacker-controlled image content (typography attacks, QR codes,
|
|
73
|
+
// adversarial watermarks). Flag it so downstream agents know to treat
|
|
74
|
+
// this text as data, not instructions. Inspired by ClawGuard (arxiv
|
|
75
|
+
// 2604.11790) and tool-result-parsing defenses (2601.04795).
|
|
76
|
+
const extra = {
|
|
77
|
+
server_version: SERVER_VERSION,
|
|
78
|
+
content_is_untrusted: true,
|
|
79
|
+
};
|
|
80
|
+
if (cacheMeta)
|
|
81
|
+
extra.cache = cacheMeta;
|
|
51
82
|
return {
|
|
52
83
|
content: [{ type: 'text', text: extracted.text }],
|
|
53
|
-
_meta: {
|
|
54
|
-
finish_reason: extracted.finishReason,
|
|
55
|
-
...(toUsageMeta(extracted.usage) ?? {}),
|
|
56
|
-
},
|
|
84
|
+
_meta: buildCompletionMeta(extracted, { extra }),
|
|
57
85
|
};
|
|
58
86
|
}
|
|
@@ -1,8 +1,15 @@
|
|
|
1
1
|
import OpenAI from 'openai';
|
|
2
|
-
|
|
2
|
+
import { type CacheOptions } from './cache.js';
|
|
3
|
+
export interface AnalyzeVideoToolRequest extends CacheOptions {
|
|
3
4
|
video_path: string;
|
|
4
5
|
question?: string;
|
|
5
6
|
model?: string;
|
|
7
|
+
/**
|
|
8
|
+
* Attach `cache_control: {type: 'ephemeral'}` to the video block so
|
|
9
|
+
* Claude / Gemini 2.5+ prompt-caches it. Very valuable for large
|
|
10
|
+
* videos where repeat questions save 10x on Anthropic pricing.
|
|
11
|
+
*/
|
|
12
|
+
cache_input?: boolean;
|
|
6
13
|
}
|
|
7
14
|
export declare function handleAnalyzeVideo(request: {
|
|
8
15
|
params: {
|
|
@@ -13,7 +20,5 @@ export declare function handleAnalyzeVideo(request: {
|
|
|
13
20
|
type: "text";
|
|
14
21
|
text: string;
|
|
15
22
|
}[];
|
|
16
|
-
_meta:
|
|
17
|
-
finish_reason: "length" | "stop" | "tool_calls" | "content_filter" | "function_call" | undefined;
|
|
18
|
-
};
|
|
23
|
+
_meta: Record<string, unknown>;
|
|
19
24
|
}>;
|
|
@@ -1,8 +1,11 @@
|
|
|
1
1
|
import { prepareVideoData } from './video-utils.js';
|
|
2
2
|
import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
|
|
3
|
+
import { SERVER_VERSION } from '../version.js';
|
|
3
4
|
import { logger } from '../logger.js';
|
|
4
5
|
import { classifyUpstreamError } from './openrouter-errors.js';
|
|
5
|
-
import { extractCompletionText, detectReasoningCutoff,
|
|
6
|
+
import { extractCompletionText, detectReasoningCutoff, buildCompletionMeta, } from './completion-utils.js';
|
|
7
|
+
import { buildCacheHeaders, extractCacheMeta, } from './cache.js';
|
|
8
|
+
import { awaitCompletionWithHeaders } from './openai-withresponse.js';
|
|
6
9
|
/**
|
|
7
10
|
* Default model — `google/gemini-2.5-flash` has the widest video-input
|
|
8
11
|
* support on OpenRouter at time of writing. Override via env
|
|
@@ -10,9 +13,8 @@ import { extractCompletionText, detectReasoningCutoff, toUsageMeta, } from './co
|
|
|
10
13
|
*/
|
|
11
14
|
const FALLBACK_DEFAULT_MODEL = 'google/gemini-2.5-flash';
|
|
12
15
|
export async function handleAnalyzeVideo(request, openai, defaultModel) {
|
|
13
|
-
const
|
|
14
|
-
|
|
15
|
-
};
|
|
16
|
+
const args = request.params.arguments ?? { video_path: '' };
|
|
17
|
+
const { video_path, question, model, cache_input, cache, cache_ttl, cache_clear } = args;
|
|
16
18
|
if (!video_path) {
|
|
17
19
|
return toolError(ErrorCode.INVALID_INPUT, 'video_path is required.');
|
|
18
20
|
}
|
|
@@ -37,14 +39,25 @@ export async function handleAnalyzeVideo(request, openai, defaultModel) {
|
|
|
37
39
|
}
|
|
38
40
|
return toolErrorFrom(ErrorCode.INVALID_INPUT, err);
|
|
39
41
|
}
|
|
42
|
+
const videoBlock = {
|
|
43
|
+
// The `video_url` content type is an OpenRouter extension; the OpenAI
|
|
44
|
+
// SDK's typings don't know about it yet.
|
|
45
|
+
type: 'video_url',
|
|
46
|
+
video_url: { url: `data:${videoData.mediaType};base64,${videoData.data}` },
|
|
47
|
+
};
|
|
48
|
+
if (cache_input)
|
|
49
|
+
videoBlock.cache_control = { type: 'ephemeral' };
|
|
50
|
+
const headers = buildCacheHeaders({ cache, cache_ttl, cache_clear });
|
|
51
|
+
const requestOpts = Object.keys(headers).length > 0 ? { headers } : undefined;
|
|
40
52
|
let completion;
|
|
53
|
+
let responseHeaders;
|
|
41
54
|
try {
|
|
42
55
|
logger.debug('analyze_video.submit', {
|
|
43
56
|
model: pickedModel,
|
|
44
57
|
format: videoData.format,
|
|
45
58
|
size_bytes: videoData.sizeBytes,
|
|
46
59
|
});
|
|
47
|
-
|
|
60
|
+
const call = openai.chat.completions.create({
|
|
48
61
|
model: pickedModel,
|
|
49
62
|
messages: [
|
|
50
63
|
{
|
|
@@ -54,19 +67,14 @@ export async function handleAnalyzeVideo(request, openai, defaultModel) {
|
|
|
54
67
|
type: 'text',
|
|
55
68
|
text: question || 'Describe what happens in this video, step by step.',
|
|
56
69
|
},
|
|
57
|
-
|
|
58
|
-
// The `video_url` content type is an OpenRouter extension; the
|
|
59
|
-
// OpenAI SDK's typings don't know about it yet. See:
|
|
60
|
-
// https://openrouter.ai/docs/guides/overview/multimodal/videos
|
|
61
|
-
type: 'video_url',
|
|
62
|
-
video_url: {
|
|
63
|
-
url: `data:${videoData.mediaType};base64,${videoData.data}`,
|
|
64
|
-
},
|
|
65
|
-
},
|
|
70
|
+
videoBlock,
|
|
66
71
|
],
|
|
67
72
|
},
|
|
68
73
|
],
|
|
69
|
-
});
|
|
74
|
+
}, requestOpts);
|
|
75
|
+
const { data, response } = await awaitCompletionWithHeaders(call);
|
|
76
|
+
completion = data;
|
|
77
|
+
responseHeaders = response?.headers;
|
|
70
78
|
}
|
|
71
79
|
catch (err) {
|
|
72
80
|
logger.warn('analyze_video.error', {
|
|
@@ -83,11 +91,15 @@ export async function handleAnalyzeVideo(request, openai, defaultModel) {
|
|
|
83
91
|
finish_reason: extracted.finishReason,
|
|
84
92
|
});
|
|
85
93
|
}
|
|
94
|
+
const cacheMeta = extractCacheMeta(responseHeaders);
|
|
95
|
+
const extra = {
|
|
96
|
+
server_version: SERVER_VERSION,
|
|
97
|
+
content_is_untrusted: true,
|
|
98
|
+
};
|
|
99
|
+
if (cacheMeta)
|
|
100
|
+
extra.cache = cacheMeta;
|
|
86
101
|
return {
|
|
87
102
|
content: [{ type: 'text', text: extracted.text }],
|
|
88
|
-
_meta: {
|
|
89
|
-
finish_reason: extracted.finishReason,
|
|
90
|
-
...(toUsageMeta(extracted.usage) ?? {}),
|
|
91
|
-
},
|
|
103
|
+
_meta: buildCompletionMeta(extracted, { extra }),
|
|
92
104
|
};
|
|
93
105
|
}
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Response caching helpers for OpenRouter's X-OpenRouter-Cache header family.
|
|
3
|
+
* See https://openrouter.ai/docs/guides/features/response-caching
|
|
4
|
+
*
|
|
5
|
+
* Three caller inputs:
|
|
6
|
+
* - cache: enable caching for this request
|
|
7
|
+
* - cache_ttl: TTL string (1s .. 24h), e.g. "5m", "1h"; pass-through
|
|
8
|
+
* - cache_clear: bust the cache entry for this request
|
|
9
|
+
*
|
|
10
|
+
* Server-wide default: OPENROUTER_CACHE_RESPONSES=1 enables cache on every
|
|
11
|
+
* request unless the caller explicitly passes cache=false.
|
|
12
|
+
*/
|
|
13
|
+
export interface CacheOptions {
|
|
14
|
+
cache?: boolean;
|
|
15
|
+
cache_ttl?: string;
|
|
16
|
+
cache_clear?: boolean;
|
|
17
|
+
}
|
|
18
|
+
/** Parse the env-default and return `true` when caching should be on by default. */
|
|
19
|
+
export declare function readCacheDefault(): boolean;
|
|
20
|
+
/**
|
|
21
|
+
* Build the headers object to pass as the second argument to
|
|
22
|
+
* `openai.chat.completions.create(body, { headers })`. Returns an empty
|
|
23
|
+
* object when nothing should be sent, so the caller can always spread the
|
|
24
|
+
* result without a conditional.
|
|
25
|
+
*/
|
|
26
|
+
export declare function buildCacheHeaders(opts: CacheOptions | undefined): Record<string, string>;
|
|
27
|
+
/** Extract cache metadata from response headers, null when not present. */
|
|
28
|
+
export interface CacheMeta {
|
|
29
|
+
status: 'HIT' | 'MISS' | string;
|
|
30
|
+
age?: number;
|
|
31
|
+
ttl?: string;
|
|
32
|
+
}
|
|
33
|
+
export declare function extractCacheMeta(headers: Headers | undefined): CacheMeta | null;
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Response caching helpers for OpenRouter's X-OpenRouter-Cache header family.
|
|
3
|
+
* See https://openrouter.ai/docs/guides/features/response-caching
|
|
4
|
+
*
|
|
5
|
+
* Three caller inputs:
|
|
6
|
+
* - cache: enable caching for this request
|
|
7
|
+
* - cache_ttl: TTL string (1s .. 24h), e.g. "5m", "1h"; pass-through
|
|
8
|
+
* - cache_clear: bust the cache entry for this request
|
|
9
|
+
*
|
|
10
|
+
* Server-wide default: OPENROUTER_CACHE_RESPONSES=1 enables cache on every
|
|
11
|
+
* request unless the caller explicitly passes cache=false.
|
|
12
|
+
*/
|
|
13
|
+
/** Parse the env-default and return `true` when caching should be on by default. */
|
|
14
|
+
export function readCacheDefault() {
|
|
15
|
+
const raw = (process.env.OPENROUTER_CACHE_RESPONSES ?? '').trim().toLowerCase();
|
|
16
|
+
return raw === '1' || raw === 'true' || raw === 'yes';
|
|
17
|
+
}
|
|
18
|
+
/**
|
|
19
|
+
* Build the headers object to pass as the second argument to
|
|
20
|
+
* `openai.chat.completions.create(body, { headers })`. Returns an empty
|
|
21
|
+
* object when nothing should be sent, so the caller can always spread the
|
|
22
|
+
* result without a conditional.
|
|
23
|
+
*/
|
|
24
|
+
export function buildCacheHeaders(opts) {
|
|
25
|
+
const headers = {};
|
|
26
|
+
const defaultOn = readCacheDefault();
|
|
27
|
+
// Caller-explicit `cache` wins. If unset, fall back to env default.
|
|
28
|
+
const enabled = opts?.cache ?? defaultOn;
|
|
29
|
+
if (enabled)
|
|
30
|
+
headers['X-OpenRouter-Cache'] = 'true';
|
|
31
|
+
if (opts?.cache_ttl)
|
|
32
|
+
headers['X-OpenRouter-Cache-TTL'] = opts.cache_ttl;
|
|
33
|
+
if (opts?.cache_clear)
|
|
34
|
+
headers['X-OpenRouter-Cache-Clear'] = 'true';
|
|
35
|
+
return headers;
|
|
36
|
+
}
|
|
37
|
+
export function extractCacheMeta(headers) {
|
|
38
|
+
if (!headers)
|
|
39
|
+
return null;
|
|
40
|
+
const status = headers.get('x-openrouter-cache-status');
|
|
41
|
+
if (!status)
|
|
42
|
+
return null;
|
|
43
|
+
const ageStr = headers.get('x-openrouter-cache-age');
|
|
44
|
+
const ttl = headers.get('x-openrouter-cache-ttl') ?? undefined;
|
|
45
|
+
const meta = { status };
|
|
46
|
+
if (ageStr) {
|
|
47
|
+
const n = Number(ageStr);
|
|
48
|
+
if (Number.isFinite(n))
|
|
49
|
+
meta.age = n;
|
|
50
|
+
}
|
|
51
|
+
if (ttl)
|
|
52
|
+
meta.ttl = ttl;
|
|
53
|
+
return meta;
|
|
54
|
+
}
|
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
import OpenAI from 'openai';
|
|
2
2
|
import type { ChatCompletionMessageParam } from 'openai/resources/chat/completions.js';
|
|
3
3
|
import { type ProviderRoutingOptions } from './provider-routing.js';
|
|
4
|
-
|
|
4
|
+
import { type CacheOptions } from './cache.js';
|
|
5
|
+
export interface ChatCompletionToolRequest extends CacheOptions {
|
|
5
6
|
model?: string;
|
|
6
7
|
messages: ChatCompletionMessageParam[];
|
|
7
8
|
temperature?: number;
|
|
@@ -12,6 +13,21 @@ export interface ChatCompletionToolRequest {
|
|
|
12
13
|
* https://openrouter.ai/docs/features/provider-routing
|
|
13
14
|
*/
|
|
14
15
|
provider?: ProviderRoutingOptions;
|
|
16
|
+
/**
|
|
17
|
+
* Surface the model's chain-of-thought trace on `_meta.reasoning` when
|
|
18
|
+
* the upstream response carries one (DeepSeek R1, Gemini Thinking,
|
|
19
|
+
* Claude Opus 4.7). Defaults to `false` or the value of
|
|
20
|
+
* `OPENROUTER_INCLUDE_REASONING`.
|
|
21
|
+
*/
|
|
22
|
+
include_reasoning?: boolean;
|
|
23
|
+
/**
|
|
24
|
+
* Enable OpenRouter's web-search plugin (Exa-backed). When true, the
|
|
25
|
+
* plugin fetches current web results and merges them into the prompt.
|
|
26
|
+
* Billed at $4 / 1000 results.
|
|
27
|
+
*/
|
|
28
|
+
online?: boolean;
|
|
29
|
+
/** Max web-search results when `online: true`. Default 5. */
|
|
30
|
+
web_max_results?: number;
|
|
15
31
|
}
|
|
16
32
|
export declare function handleChatCompletion(request: {
|
|
17
33
|
params: {
|
|
@@ -22,7 +38,5 @@ export declare function handleChatCompletion(request: {
|
|
|
22
38
|
type: "text";
|
|
23
39
|
text: string;
|
|
24
40
|
}[];
|
|
25
|
-
_meta:
|
|
26
|
-
finish_reason: "length" | "stop" | "tool_calls" | "content_filter" | "function_call" | undefined;
|
|
27
|
-
};
|
|
41
|
+
_meta: Record<string, unknown>;
|
|
28
42
|
}>;
|
|
@@ -1,20 +1,28 @@
|
|
|
1
1
|
import { ErrorCode, toolError } from '../errors.js';
|
|
2
|
+
import { SERVER_VERSION } from '../version.js';
|
|
2
3
|
import { classifyUpstreamError } from './openrouter-errors.js';
|
|
3
|
-
import { extractCompletionText, detectReasoningCutoff,
|
|
4
|
+
import { extractCompletionText, detectReasoningCutoff, buildCompletionMeta, } from './completion-utils.js';
|
|
4
5
|
import { readProviderDefaults, mergeProviderOptions, buildProviderBody, resolveMaxTokens, } from './provider-routing.js';
|
|
6
|
+
import { buildCacheHeaders, extractCacheMeta, } from './cache.js';
|
|
7
|
+
import { awaitCompletionWithHeaders } from './openai-withresponse.js';
|
|
5
8
|
const DEFAULT_MODEL = 'nvidia/nemotron-nano-12b-v2-vl:free';
|
|
9
|
+
function readIncludeReasoningDefault() {
|
|
10
|
+
const raw = (process.env.OPENROUTER_INCLUDE_REASONING ?? '').trim().toLowerCase();
|
|
11
|
+
return raw === '1' || raw === 'true' || raw === 'yes';
|
|
12
|
+
}
|
|
6
13
|
export async function handleChatCompletion(request, openai, defaultModel) {
|
|
7
|
-
const
|
|
8
|
-
|
|
9
|
-
};
|
|
14
|
+
const args = request.params.arguments ?? { messages: [] };
|
|
15
|
+
const { messages, model, temperature, max_tokens, provider, include_reasoning, online, web_max_results, cache, cache_ttl, cache_clear, } = args;
|
|
10
16
|
if (!messages?.length) {
|
|
11
17
|
return toolError(ErrorCode.INVALID_INPUT, 'Messages array cannot be empty.');
|
|
12
18
|
}
|
|
13
19
|
const providerOptions = mergeProviderOptions(readProviderDefaults(), provider);
|
|
14
20
|
const providerBody = buildProviderBody(providerOptions);
|
|
15
21
|
const effectiveMaxTokens = resolveMaxTokens(max_tokens);
|
|
16
|
-
|
|
17
|
-
// the
|
|
22
|
+
const wantsReasoning = include_reasoning ?? readIncludeReasoningDefault();
|
|
23
|
+
// Build the request body. Several OpenRouter extensions aren't in the
|
|
24
|
+
// OpenAI SDK types, so we assemble as `Record<string, unknown>` and cast
|
|
25
|
+
// at the call site.
|
|
18
26
|
const body = {
|
|
19
27
|
model: model || defaultModel || DEFAULT_MODEL,
|
|
20
28
|
messages,
|
|
@@ -24,9 +32,24 @@ export async function handleChatCompletion(request, openai, defaultModel) {
|
|
|
24
32
|
body.max_tokens = effectiveMaxTokens;
|
|
25
33
|
if (providerBody)
|
|
26
34
|
body.provider = providerBody;
|
|
35
|
+
if (wantsReasoning)
|
|
36
|
+
body.include_reasoning = true;
|
|
37
|
+
if (online) {
|
|
38
|
+
const plugin = { id: 'web' };
|
|
39
|
+
if (typeof web_max_results === 'number' && web_max_results > 0) {
|
|
40
|
+
plugin.max_results = web_max_results;
|
|
41
|
+
}
|
|
42
|
+
body.plugins = [plugin];
|
|
43
|
+
}
|
|
44
|
+
const headers = buildCacheHeaders({ cache, cache_ttl, cache_clear });
|
|
45
|
+
const requestOpts = Object.keys(headers).length > 0 ? { headers } : undefined;
|
|
27
46
|
let completion;
|
|
47
|
+
let responseHeaders;
|
|
28
48
|
try {
|
|
29
|
-
|
|
49
|
+
const call = openai.chat.completions.create(body, requestOpts);
|
|
50
|
+
const { data, response } = await awaitCompletionWithHeaders(call);
|
|
51
|
+
completion = data;
|
|
52
|
+
responseHeaders = response?.headers;
|
|
30
53
|
}
|
|
31
54
|
catch (err) {
|
|
32
55
|
return classifyUpstreamError(err);
|
|
@@ -38,13 +61,18 @@ export async function handleChatCompletion(request, openai, defaultModel) {
|
|
|
38
61
|
if (!extracted.text) {
|
|
39
62
|
return toolError(ErrorCode.INTERNAL, 'Model returned no textual content.', {
|
|
40
63
|
finish_reason: extracted.finishReason,
|
|
64
|
+
native_finish_reason: extracted.nativeFinishReason,
|
|
41
65
|
});
|
|
42
66
|
}
|
|
67
|
+
const cacheMeta = extractCacheMeta(responseHeaders);
|
|
68
|
+
const extra = { server_version: SERVER_VERSION };
|
|
69
|
+
if (cacheMeta)
|
|
70
|
+
extra.cache = cacheMeta;
|
|
43
71
|
return {
|
|
44
72
|
content: [{ type: 'text', text: extracted.text }],
|
|
45
|
-
_meta: {
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
},
|
|
73
|
+
_meta: buildCompletionMeta(extracted, {
|
|
74
|
+
includeReasoning: wantsReasoning,
|
|
75
|
+
extra,
|
|
76
|
+
}),
|
|
49
77
|
};
|
|
50
78
|
}
|
|
@@ -14,6 +14,19 @@ export interface ExtractedText {
|
|
|
14
14
|
/** True when `text` came from the reasoning trace (not a final answer). */
|
|
15
15
|
reasonedOnly: boolean;
|
|
16
16
|
finishReason: ChatCompletion.Choice['finish_reason'] | undefined;
|
|
17
|
+
/**
|
|
18
|
+
* OpenRouter's `native_finish_reason`, when present. Carries the
|
|
19
|
+
* provider-native value before OpenRouter normalizes it. Surfaced in
|
|
20
|
+
* `_meta.native_finish_reason` for debuggability.
|
|
21
|
+
*/
|
|
22
|
+
nativeFinishReason: string | undefined;
|
|
23
|
+
/**
|
|
24
|
+
* Raw reasoning trace (content of `reasoning` or joined `reasoning_details`).
|
|
25
|
+
* Populated whenever the upstream response carried one, even when the
|
|
26
|
+
* assistant also produced a final `content` answer. Surfaced to callers
|
|
27
|
+
* via `_meta.reasoning` when they opt in with `include_reasoning: true`.
|
|
28
|
+
*/
|
|
29
|
+
reasoning?: string;
|
|
17
30
|
usage?: ChatCompletion['usage'];
|
|
18
31
|
}
|
|
19
32
|
export declare function extractCompletionText(completion: ChatCompletion): ExtractedText;
|
|
@@ -25,3 +38,19 @@ export declare function extractCompletionText(completion: ChatCompletion): Extra
|
|
|
25
38
|
*/
|
|
26
39
|
export declare function detectReasoningCutoff(extracted: ExtractedText): ToolErrorResult | null;
|
|
27
40
|
export declare function toUsageMeta(usage: ChatCompletion['usage'] | undefined): Record<string, unknown> | undefined;
|
|
41
|
+
/**
|
|
42
|
+
* Build the common `_meta` shape for chat-completion-derived tools.
|
|
43
|
+
* Folds in:
|
|
44
|
+
* - normalized and native finish reasons (from the choice)
|
|
45
|
+
* - optional `reasoning` trace (when the caller opted in)
|
|
46
|
+
* - token usage (prompt / completion / total)
|
|
47
|
+
* - server version stamp
|
|
48
|
+
*
|
|
49
|
+
* Caller can pass `extra` to merge additional keys (cache metadata,
|
|
50
|
+
* content_is_untrusted, etc.) without repeating this boilerplate.
|
|
51
|
+
*/
|
|
52
|
+
export interface BuildMetaOptions {
|
|
53
|
+
includeReasoning?: boolean;
|
|
54
|
+
extra?: Record<string, unknown>;
|
|
55
|
+
}
|
|
56
|
+
export declare function buildCompletionMeta(extracted: ExtractedText, opts?: BuildMetaOptions): Record<string, unknown>;
|
|
@@ -1,14 +1,45 @@
|
|
|
1
1
|
import { ErrorCode, toolError } from '../errors.js';
|
|
2
|
+
function extractReasoning(msg) {
|
|
3
|
+
if (typeof msg.reasoning === 'string' && msg.reasoning.length > 0)
|
|
4
|
+
return msg.reasoning;
|
|
5
|
+
if (Array.isArray(msg.reasoning_details) && msg.reasoning_details.length > 0) {
|
|
6
|
+
const joined = msg.reasoning_details
|
|
7
|
+
.filter((d) => typeof d.text === 'string')
|
|
8
|
+
.map((d) => d.text)
|
|
9
|
+
.join('\n');
|
|
10
|
+
if (joined.length > 0)
|
|
11
|
+
return joined;
|
|
12
|
+
}
|
|
13
|
+
return undefined;
|
|
14
|
+
}
|
|
2
15
|
export function extractCompletionText(completion) {
|
|
3
16
|
const choice = completion.choices?.[0];
|
|
4
17
|
const msg = choice?.message;
|
|
5
18
|
const finishReason = choice?.finish_reason;
|
|
19
|
+
// `native_finish_reason` is an OpenRouter extension, not in the OpenAI
|
|
20
|
+
// SDK types — read it via an unknown-cast.
|
|
21
|
+
const nativeFinishReason = choice?.native_finish_reason ?? undefined;
|
|
6
22
|
const usage = completion.usage ?? undefined;
|
|
7
|
-
if (!msg)
|
|
8
|
-
return {
|
|
9
|
-
|
|
23
|
+
if (!msg) {
|
|
24
|
+
return {
|
|
25
|
+
text: '',
|
|
26
|
+
reasonedOnly: false,
|
|
27
|
+
finishReason,
|
|
28
|
+
nativeFinishReason: nativeFinishReason ?? undefined,
|
|
29
|
+
usage,
|
|
30
|
+
};
|
|
31
|
+
}
|
|
32
|
+
const { content } = msg;
|
|
33
|
+
const reasoning = extractReasoning(msg);
|
|
10
34
|
if (typeof content === 'string' && content.length > 0) {
|
|
11
|
-
return {
|
|
35
|
+
return {
|
|
36
|
+
text: content,
|
|
37
|
+
reasonedOnly: false,
|
|
38
|
+
finishReason,
|
|
39
|
+
nativeFinishReason: nativeFinishReason ?? undefined,
|
|
40
|
+
reasoning,
|
|
41
|
+
usage,
|
|
42
|
+
};
|
|
12
43
|
}
|
|
13
44
|
if (Array.isArray(content)) {
|
|
14
45
|
const parts = content
|
|
@@ -16,22 +47,33 @@ export function extractCompletionText(completion) {
|
|
|
16
47
|
.map((p) => p.text ?? '');
|
|
17
48
|
const joined = parts.join('');
|
|
18
49
|
if (joined.length > 0) {
|
|
19
|
-
return {
|
|
50
|
+
return {
|
|
51
|
+
text: joined,
|
|
52
|
+
reasonedOnly: false,
|
|
53
|
+
finishReason,
|
|
54
|
+
nativeFinishReason: nativeFinishReason ?? undefined,
|
|
55
|
+
reasoning,
|
|
56
|
+
usage,
|
|
57
|
+
};
|
|
20
58
|
}
|
|
21
59
|
}
|
|
22
|
-
if (
|
|
23
|
-
return {
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
return { text: joined, reasonedOnly: true, finishReason, usage };
|
|
32
|
-
}
|
|
60
|
+
if (reasoning && reasoning.length > 0) {
|
|
61
|
+
return {
|
|
62
|
+
text: reasoning,
|
|
63
|
+
reasonedOnly: true,
|
|
64
|
+
finishReason,
|
|
65
|
+
nativeFinishReason: nativeFinishReason ?? undefined,
|
|
66
|
+
reasoning,
|
|
67
|
+
usage,
|
|
68
|
+
};
|
|
33
69
|
}
|
|
34
|
-
return {
|
|
70
|
+
return {
|
|
71
|
+
text: '',
|
|
72
|
+
reasonedOnly: false,
|
|
73
|
+
finishReason,
|
|
74
|
+
nativeFinishReason: nativeFinishReason ?? undefined,
|
|
75
|
+
usage,
|
|
76
|
+
};
|
|
35
77
|
}
|
|
36
78
|
/**
|
|
37
79
|
* If the extracted response is reasoning-only and was cut off by
|
|
@@ -67,3 +109,20 @@ export function toUsageMeta(usage) {
|
|
|
67
109
|
},
|
|
68
110
|
};
|
|
69
111
|
}
|
|
112
|
+
export function buildCompletionMeta(extracted, opts = {}) {
|
|
113
|
+
const meta = {
|
|
114
|
+
finish_reason: extracted.finishReason,
|
|
115
|
+
};
|
|
116
|
+
if (extracted.nativeFinishReason) {
|
|
117
|
+
meta.native_finish_reason = extracted.nativeFinishReason;
|
|
118
|
+
}
|
|
119
|
+
if (opts.includeReasoning && extracted.reasoning && !extracted.reasonedOnly) {
|
|
120
|
+
meta.reasoning = extracted.reasoning;
|
|
121
|
+
}
|
|
122
|
+
const usageMeta = toUsageMeta(extracted.usage);
|
|
123
|
+
if (usageMeta)
|
|
124
|
+
Object.assign(meta, usageMeta);
|
|
125
|
+
if (opts.extra)
|
|
126
|
+
Object.assign(meta, opts.extra);
|
|
127
|
+
return meta;
|
|
128
|
+
}
|
|
@@ -41,6 +41,7 @@ export declare function handleGenerateAudio(request: {
|
|
|
41
41
|
text?: undefined;
|
|
42
42
|
})[];
|
|
43
43
|
_meta: {
|
|
44
|
+
server_version: string;
|
|
44
45
|
save_path: string;
|
|
45
46
|
mime: string;
|
|
46
47
|
size_bytes: number;
|
|
@@ -58,6 +59,7 @@ export declare function handleGenerateAudio(request: {
|
|
|
58
59
|
text?: undefined;
|
|
59
60
|
})[];
|
|
60
61
|
_meta: {
|
|
62
|
+
server_version: string;
|
|
61
63
|
mime: string;
|
|
62
64
|
size_bytes: number;
|
|
63
65
|
save_path?: undefined;
|
|
@@ -2,6 +2,8 @@ import { promises as fs } from 'fs';
|
|
|
2
2
|
import { extname } from 'path';
|
|
3
3
|
import { resolveSafeOutputPath, UnsafeOutputPathError } from './path-safety.js';
|
|
4
4
|
import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
|
|
5
|
+
import { SERVER_VERSION } from '../version.js';
|
|
6
|
+
import { logger } from '../logger.js';
|
|
5
7
|
import { classifyUpstreamError } from './openrouter-errors.js';
|
|
6
8
|
const DEFAULT_MODEL = 'openai/gpt-audio';
|
|
7
9
|
const DEFAULT_VOICE = 'alloy';
|
|
@@ -89,6 +91,16 @@ export async function handleGenerateAudio(request, openai) {
|
|
|
89
91
|
if (!prompt?.trim()) {
|
|
90
92
|
return toolError(ErrorCode.INVALID_INPUT, 'prompt is required.');
|
|
91
93
|
}
|
|
94
|
+
// Audit entry. See generate_image for rationale.
|
|
95
|
+
logger.audit('generate_audio.start', {
|
|
96
|
+
model: model || DEFAULT_MODEL,
|
|
97
|
+
voice: voice?.trim() || DEFAULT_VOICE,
|
|
98
|
+
format: VALID_FORMATS.includes(format ?? '')
|
|
99
|
+
? format
|
|
100
|
+
: DEFAULT_FORMAT,
|
|
101
|
+
prompt_preview: prompt.slice(0, 80),
|
|
102
|
+
save_path: save_path ? 'provided' : 'none',
|
|
103
|
+
});
|
|
92
104
|
// Fail-fast on unsafe paths BEFORE spending tokens.
|
|
93
105
|
let safeBase = null;
|
|
94
106
|
if (save_path) {
|
|
@@ -165,6 +177,7 @@ export async function handleGenerateAudio(request, openai) {
|
|
|
165
177
|
{ type: 'audio', mimeType: detected.mimeType, data: returnBase64 },
|
|
166
178
|
],
|
|
167
179
|
_meta: {
|
|
180
|
+
server_version: SERVER_VERSION,
|
|
168
181
|
save_path: actualSavePath,
|
|
169
182
|
mime: detected.mimeType,
|
|
170
183
|
size_bytes: audioBuffer.length,
|
|
@@ -176,7 +189,11 @@ export async function handleGenerateAudio(request, openai) {
|
|
|
176
189
|
{ type: 'text', text: transcript || 'Audio generated successfully.' },
|
|
177
190
|
{ type: 'audio', mimeType: detected.mimeType, data: returnBase64 },
|
|
178
191
|
],
|
|
179
|
-
_meta: {
|
|
192
|
+
_meta: {
|
|
193
|
+
server_version: SERVER_VERSION,
|
|
194
|
+
mime: detected.mimeType,
|
|
195
|
+
size_bytes: audioBuffer.length,
|
|
196
|
+
},
|
|
180
197
|
};
|
|
181
198
|
}
|
|
182
199
|
catch (err) {
|