@stabgan/openrouter-mcp-multimodal 4.0.1 → 4.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/README.md +95 -8
  2. package/dist/errors.d.ts +26 -8
  3. package/dist/errors.js +17 -12
  4. package/dist/index.js +21 -5
  5. package/dist/logger.d.ts +11 -0
  6. package/dist/logger.js +26 -0
  7. package/dist/model-cache.d.ts +17 -0
  8. package/dist/model-cache.js +27 -1
  9. package/dist/openrouter-api.d.ts +22 -2
  10. package/dist/openrouter-api.js +21 -2
  11. package/dist/tool-handlers/analyze-audio.d.ts +9 -4
  12. package/dist/tool-handlers/analyze-audio.js +29 -15
  13. package/dist/tool-handlers/analyze-image.d.ts +10 -4
  14. package/dist/tool-handlers/analyze-image.js +37 -9
  15. package/dist/tool-handlers/analyze-video.d.ts +9 -4
  16. package/dist/tool-handlers/analyze-video.js +31 -19
  17. package/dist/tool-handlers/cache.d.ts +33 -0
  18. package/dist/tool-handlers/cache.js +54 -0
  19. package/dist/tool-handlers/chat-completion.d.ts +18 -4
  20. package/dist/tool-handlers/chat-completion.js +39 -11
  21. package/dist/tool-handlers/completion-utils.d.ts +29 -0
  22. package/dist/tool-handlers/completion-utils.js +76 -17
  23. package/dist/tool-handlers/generate-audio.d.ts +2 -0
  24. package/dist/tool-handlers/generate-audio.js +18 -1
  25. package/dist/tool-handlers/generate-image.d.ts +2 -0
  26. package/dist/tool-handlers/generate-image.js +15 -0
  27. package/dist/tool-handlers/generate-video.d.ts +43 -0
  28. package/dist/tool-handlers/generate-video.js +46 -3
  29. package/dist/tool-handlers/get-model-info.d.ts +1 -6
  30. package/dist/tool-handlers/get-model-info.js +2 -1
  31. package/dist/tool-handlers/health-check.d.ts +23 -0
  32. package/dist/tool-handlers/health-check.js +35 -0
  33. package/dist/tool-handlers/openai-withresponse.d.ts +16 -0
  34. package/dist/tool-handlers/openai-withresponse.js +16 -0
  35. package/dist/tool-handlers/openrouter-errors.d.ts +5 -1
  36. package/dist/tool-handlers/openrouter-errors.js +72 -11
  37. package/dist/tool-handlers/rerank.d.ts +17 -0
  38. package/dist/tool-handlers/rerank.js +52 -0
  39. package/dist/tool-handlers/search-models.d.ts +18 -7
  40. package/dist/tool-handlers/search-models.js +25 -2
  41. package/dist/tool-handlers/structured-output.d.ts +13 -0
  42. package/dist/tool-handlers/structured-output.js +24 -0
  43. package/dist/tool-handlers/validate-model.d.ts +4 -6
  44. package/dist/tool-handlers/validate-model.js +3 -8
  45. package/dist/tool-handlers.d.ts +1 -0
  46. package/dist/tool-handlers.js +435 -165
  47. package/dist/version.d.ts +16 -0
  48. package/dist/version.js +16 -0
  49. package/package.json +1 -1
@@ -1,8 +1,16 @@
1
1
  import OpenAI from 'openai';
2
- export interface AnalyzeImageToolRequest {
2
+ import { type CacheOptions } from './cache.js';
3
+ export interface AnalyzeImageToolRequest extends CacheOptions {
3
4
  image_path: string;
4
5
  question?: string;
5
6
  model?: string;
7
+ /**
8
+ * When true, attach Anthropic-style `cache_control: {type: 'ephemeral'}`
9
+ * to the image block so Claude / Gemini 2.5+ prompt-caches it. Repeat
10
+ * questions about the same image then cost ~0.1x on Anthropic and
11
+ * ~0.25x on Gemini for the image input.
12
+ */
13
+ cache_input?: boolean;
6
14
  }
7
15
  export declare function handleAnalyzeImage(request: {
8
16
  params: {
@@ -13,7 +21,5 @@ export declare function handleAnalyzeImage(request: {
13
21
  type: "text";
14
22
  text: string;
15
23
  }[];
16
- _meta: {
17
- finish_reason: "length" | "stop" | "tool_calls" | "content_filter" | "function_call" | undefined;
18
- };
24
+ _meta: Record<string, unknown>;
19
25
  }>;
@@ -1,10 +1,14 @@
1
1
  import { prepareImageUrl } from './image-utils.js';
2
2
  import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
3
+ import { SERVER_VERSION } from '../version.js';
3
4
  import { classifyUpstreamError } from './openrouter-errors.js';
4
- import { extractCompletionText, detectReasoningCutoff, toUsageMeta, } from './completion-utils.js';
5
+ import { extractCompletionText, detectReasoningCutoff, buildCompletionMeta, } from './completion-utils.js';
6
+ import { buildCacheHeaders, extractCacheMeta, } from './cache.js';
7
+ import { awaitCompletionWithHeaders } from './openai-withresponse.js';
5
8
  const DEFAULT_MODEL = 'nvidia/nemotron-nano-12b-v2-vl:free';
6
9
  export async function handleAnalyzeImage(request, openai, defaultModel) {
7
- const { image_path, question, model } = request.params.arguments ?? { image_path: '' };
10
+ const args = request.params.arguments ?? { image_path: '' };
11
+ const { image_path, question, model, cache_input, cache, cache_ttl, cache_clear } = args;
8
12
  if (!image_path) {
9
13
  return toolError(ErrorCode.INVALID_INPUT, 'image_path is required.');
10
14
  }
@@ -21,20 +25,35 @@ export async function handleAnalyzeImage(request, openai, defaultModel) {
21
25
  }
22
26
  return toolErrorFrom(ErrorCode.INVALID_INPUT, err);
23
27
  }
28
+ // Attach `cache_control` to the image block when requested. The openai
29
+ // SDK doesn't type this field but passes it through to the server,
30
+ // which forwards it to providers that support prompt caching.
31
+ const imageBlock = {
32
+ type: 'image_url',
33
+ image_url: { url: imageUrl },
34
+ };
35
+ if (cache_input)
36
+ imageBlock.cache_control = { type: 'ephemeral' };
37
+ const headers = buildCacheHeaders({ cache, cache_ttl, cache_clear });
38
+ const requestOpts = Object.keys(headers).length > 0 ? { headers } : undefined;
24
39
  let completion;
40
+ let responseHeaders;
25
41
  try {
26
- completion = await openai.chat.completions.create({
42
+ const call = openai.chat.completions.create({
27
43
  model: model || defaultModel || DEFAULT_MODEL,
28
44
  messages: [
29
45
  {
30
46
  role: 'user',
31
47
  content: [
32
48
  { type: 'text', text: question || "What's in this image?" },
33
- { type: 'image_url', image_url: { url: imageUrl } },
49
+ imageBlock,
34
50
  ],
35
51
  },
36
52
  ],
37
- });
53
+ }, requestOpts);
54
+ const { data, response } = await awaitCompletionWithHeaders(call);
55
+ completion = data;
56
+ responseHeaders = response?.headers;
38
57
  }
39
58
  catch (err) {
40
59
  return classifyUpstreamError(err);
@@ -48,11 +67,20 @@ export async function handleAnalyzeImage(request, openai, defaultModel) {
48
67
  finish_reason: extracted.finishReason,
49
68
  });
50
69
  }
70
+ const cacheMeta = extractCacheMeta(responseHeaders);
71
+ // Output originates from model interpretation of potentially
72
+ // attacker-controlled image content (typography attacks, QR codes,
73
+ // adversarial watermarks). Flag it so downstream agents know to treat
74
+ // this text as data, not instructions. Inspired by ClawGuard (arxiv
75
+ // 2604.11790) and tool-result-parsing defenses (2601.04795).
76
+ const extra = {
77
+ server_version: SERVER_VERSION,
78
+ content_is_untrusted: true,
79
+ };
80
+ if (cacheMeta)
81
+ extra.cache = cacheMeta;
51
82
  return {
52
83
  content: [{ type: 'text', text: extracted.text }],
53
- _meta: {
54
- finish_reason: extracted.finishReason,
55
- ...(toUsageMeta(extracted.usage) ?? {}),
56
- },
84
+ _meta: buildCompletionMeta(extracted, { extra }),
57
85
  };
58
86
  }
@@ -1,8 +1,15 @@
1
1
  import OpenAI from 'openai';
2
- export interface AnalyzeVideoToolRequest {
2
+ import { type CacheOptions } from './cache.js';
3
+ export interface AnalyzeVideoToolRequest extends CacheOptions {
3
4
  video_path: string;
4
5
  question?: string;
5
6
  model?: string;
7
+ /**
8
+ * Attach `cache_control: {type: 'ephemeral'}` to the video block so
9
+ * Claude / Gemini 2.5+ prompt-caches it. Very valuable for large
10
+ * videos where repeat questions save 10x on Anthropic pricing.
11
+ */
12
+ cache_input?: boolean;
6
13
  }
7
14
  export declare function handleAnalyzeVideo(request: {
8
15
  params: {
@@ -13,7 +20,5 @@ export declare function handleAnalyzeVideo(request: {
13
20
  type: "text";
14
21
  text: string;
15
22
  }[];
16
- _meta: {
17
- finish_reason: "length" | "stop" | "tool_calls" | "content_filter" | "function_call" | undefined;
18
- };
23
+ _meta: Record<string, unknown>;
19
24
  }>;
@@ -1,8 +1,11 @@
1
1
  import { prepareVideoData } from './video-utils.js';
2
2
  import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
3
+ import { SERVER_VERSION } from '../version.js';
3
4
  import { logger } from '../logger.js';
4
5
  import { classifyUpstreamError } from './openrouter-errors.js';
5
- import { extractCompletionText, detectReasoningCutoff, toUsageMeta, } from './completion-utils.js';
6
+ import { extractCompletionText, detectReasoningCutoff, buildCompletionMeta, } from './completion-utils.js';
7
+ import { buildCacheHeaders, extractCacheMeta, } from './cache.js';
8
+ import { awaitCompletionWithHeaders } from './openai-withresponse.js';
6
9
  /**
7
10
  * Default model — `google/gemini-2.5-flash` has the widest video-input
8
11
  * support on OpenRouter at time of writing. Override via env
@@ -10,9 +13,8 @@ import { extractCompletionText, detectReasoningCutoff, toUsageMeta, } from './co
10
13
  */
11
14
  const FALLBACK_DEFAULT_MODEL = 'google/gemini-2.5-flash';
12
15
  export async function handleAnalyzeVideo(request, openai, defaultModel) {
13
- const { video_path, question, model } = request.params.arguments ?? {
14
- video_path: '',
15
- };
16
+ const args = request.params.arguments ?? { video_path: '' };
17
+ const { video_path, question, model, cache_input, cache, cache_ttl, cache_clear } = args;
16
18
  if (!video_path) {
17
19
  return toolError(ErrorCode.INVALID_INPUT, 'video_path is required.');
18
20
  }
@@ -37,14 +39,25 @@ export async function handleAnalyzeVideo(request, openai, defaultModel) {
37
39
  }
38
40
  return toolErrorFrom(ErrorCode.INVALID_INPUT, err);
39
41
  }
42
+ const videoBlock = {
43
+ // The `video_url` content type is an OpenRouter extension; the OpenAI
44
+ // SDK's typings don't know about it yet.
45
+ type: 'video_url',
46
+ video_url: { url: `data:${videoData.mediaType};base64,${videoData.data}` },
47
+ };
48
+ if (cache_input)
49
+ videoBlock.cache_control = { type: 'ephemeral' };
50
+ const headers = buildCacheHeaders({ cache, cache_ttl, cache_clear });
51
+ const requestOpts = Object.keys(headers).length > 0 ? { headers } : undefined;
40
52
  let completion;
53
+ let responseHeaders;
41
54
  try {
42
55
  logger.debug('analyze_video.submit', {
43
56
  model: pickedModel,
44
57
  format: videoData.format,
45
58
  size_bytes: videoData.sizeBytes,
46
59
  });
47
- completion = await openai.chat.completions.create({
60
+ const call = openai.chat.completions.create({
48
61
  model: pickedModel,
49
62
  messages: [
50
63
  {
@@ -54,19 +67,14 @@ export async function handleAnalyzeVideo(request, openai, defaultModel) {
54
67
  type: 'text',
55
68
  text: question || 'Describe what happens in this video, step by step.',
56
69
  },
57
- {
58
- // The `video_url` content type is an OpenRouter extension; the
59
- // OpenAI SDK's typings don't know about it yet. See:
60
- // https://openrouter.ai/docs/guides/overview/multimodal/videos
61
- type: 'video_url',
62
- video_url: {
63
- url: `data:${videoData.mediaType};base64,${videoData.data}`,
64
- },
65
- },
70
+ videoBlock,
66
71
  ],
67
72
  },
68
73
  ],
69
- });
74
+ }, requestOpts);
75
+ const { data, response } = await awaitCompletionWithHeaders(call);
76
+ completion = data;
77
+ responseHeaders = response?.headers;
70
78
  }
71
79
  catch (err) {
72
80
  logger.warn('analyze_video.error', {
@@ -83,11 +91,15 @@ export async function handleAnalyzeVideo(request, openai, defaultModel) {
83
91
  finish_reason: extracted.finishReason,
84
92
  });
85
93
  }
94
+ const cacheMeta = extractCacheMeta(responseHeaders);
95
+ const extra = {
96
+ server_version: SERVER_VERSION,
97
+ content_is_untrusted: true,
98
+ };
99
+ if (cacheMeta)
100
+ extra.cache = cacheMeta;
86
101
  return {
87
102
  content: [{ type: 'text', text: extracted.text }],
88
- _meta: {
89
- finish_reason: extracted.finishReason,
90
- ...(toUsageMeta(extracted.usage) ?? {}),
91
- },
103
+ _meta: buildCompletionMeta(extracted, { extra }),
92
104
  };
93
105
  }
@@ -0,0 +1,33 @@
1
+ /**
2
+ * Response caching helpers for OpenRouter's X-OpenRouter-Cache header family.
3
+ * See https://openrouter.ai/docs/guides/features/response-caching
4
+ *
5
+ * Three caller inputs:
6
+ * - cache: enable caching for this request
7
+ * - cache_ttl: TTL string (1s .. 24h), e.g. "5m", "1h"; pass-through
8
+ * - cache_clear: bust the cache entry for this request
9
+ *
10
+ * Server-wide default: OPENROUTER_CACHE_RESPONSES=1 enables cache on every
11
+ * request unless the caller explicitly passes cache=false.
12
+ */
13
+ export interface CacheOptions {
14
+ cache?: boolean;
15
+ cache_ttl?: string;
16
+ cache_clear?: boolean;
17
+ }
18
+ /** Parse the env-default and return `true` when caching should be on by default. */
19
+ export declare function readCacheDefault(): boolean;
20
+ /**
21
+ * Build the headers object to pass as the second argument to
22
+ * `openai.chat.completions.create(body, { headers })`. Returns an empty
23
+ * object when nothing should be sent, so the caller can always spread the
24
+ * result without a conditional.
25
+ */
26
+ export declare function buildCacheHeaders(opts: CacheOptions | undefined): Record<string, string>;
27
+ /** Extract cache metadata from response headers, null when not present. */
28
+ export interface CacheMeta {
29
+ status: 'HIT' | 'MISS' | string;
30
+ age?: number;
31
+ ttl?: string;
32
+ }
33
+ export declare function extractCacheMeta(headers: Headers | undefined): CacheMeta | null;
@@ -0,0 +1,54 @@
1
+ /**
2
+ * Response caching helpers for OpenRouter's X-OpenRouter-Cache header family.
3
+ * See https://openrouter.ai/docs/guides/features/response-caching
4
+ *
5
+ * Three caller inputs:
6
+ * - cache: enable caching for this request
7
+ * - cache_ttl: TTL string (1s .. 24h), e.g. "5m", "1h"; pass-through
8
+ * - cache_clear: bust the cache entry for this request
9
+ *
10
+ * Server-wide default: OPENROUTER_CACHE_RESPONSES=1 enables cache on every
11
+ * request unless the caller explicitly passes cache=false.
12
+ */
13
+ /** Parse the env-default and return `true` when caching should be on by default. */
14
+ export function readCacheDefault() {
15
+ const raw = (process.env.OPENROUTER_CACHE_RESPONSES ?? '').trim().toLowerCase();
16
+ return raw === '1' || raw === 'true' || raw === 'yes';
17
+ }
18
+ /**
19
+ * Build the headers object to pass as the second argument to
20
+ * `openai.chat.completions.create(body, { headers })`. Returns an empty
21
+ * object when nothing should be sent, so the caller can always spread the
22
+ * result without a conditional.
23
+ */
24
+ export function buildCacheHeaders(opts) {
25
+ const headers = {};
26
+ const defaultOn = readCacheDefault();
27
+ // Caller-explicit `cache` wins. If unset, fall back to env default.
28
+ const enabled = opts?.cache ?? defaultOn;
29
+ if (enabled)
30
+ headers['X-OpenRouter-Cache'] = 'true';
31
+ if (opts?.cache_ttl)
32
+ headers['X-OpenRouter-Cache-TTL'] = opts.cache_ttl;
33
+ if (opts?.cache_clear)
34
+ headers['X-OpenRouter-Cache-Clear'] = 'true';
35
+ return headers;
36
+ }
37
+ export function extractCacheMeta(headers) {
38
+ if (!headers)
39
+ return null;
40
+ const status = headers.get('x-openrouter-cache-status');
41
+ if (!status)
42
+ return null;
43
+ const ageStr = headers.get('x-openrouter-cache-age');
44
+ const ttl = headers.get('x-openrouter-cache-ttl') ?? undefined;
45
+ const meta = { status };
46
+ if (ageStr) {
47
+ const n = Number(ageStr);
48
+ if (Number.isFinite(n))
49
+ meta.age = n;
50
+ }
51
+ if (ttl)
52
+ meta.ttl = ttl;
53
+ return meta;
54
+ }
@@ -1,7 +1,8 @@
1
1
  import OpenAI from 'openai';
2
2
  import type { ChatCompletionMessageParam } from 'openai/resources/chat/completions.js';
3
3
  import { type ProviderRoutingOptions } from './provider-routing.js';
4
- export interface ChatCompletionToolRequest {
4
+ import { type CacheOptions } from './cache.js';
5
+ export interface ChatCompletionToolRequest extends CacheOptions {
5
6
  model?: string;
6
7
  messages: ChatCompletionMessageParam[];
7
8
  temperature?: number;
@@ -12,6 +13,21 @@ export interface ChatCompletionToolRequest {
12
13
  * https://openrouter.ai/docs/features/provider-routing
13
14
  */
14
15
  provider?: ProviderRoutingOptions;
16
+ /**
17
+ * Surface the model's chain-of-thought trace on `_meta.reasoning` when
18
+ * the upstream response carries one (DeepSeek R1, Gemini Thinking,
19
+ * Claude Opus 4.7). Defaults to `false` or the value of
20
+ * `OPENROUTER_INCLUDE_REASONING`.
21
+ */
22
+ include_reasoning?: boolean;
23
+ /**
24
+ * Enable OpenRouter's web-search plugin (Exa-backed). When true, the
25
+ * plugin fetches current web results and merges them into the prompt.
26
+ * Billed at $4 / 1000 results.
27
+ */
28
+ online?: boolean;
29
+ /** Max web-search results when `online: true`. Default 5. */
30
+ web_max_results?: number;
15
31
  }
16
32
  export declare function handleChatCompletion(request: {
17
33
  params: {
@@ -22,7 +38,5 @@ export declare function handleChatCompletion(request: {
22
38
  type: "text";
23
39
  text: string;
24
40
  }[];
25
- _meta: {
26
- finish_reason: "length" | "stop" | "tool_calls" | "content_filter" | "function_call" | undefined;
27
- };
41
+ _meta: Record<string, unknown>;
28
42
  }>;
@@ -1,20 +1,28 @@
1
1
  import { ErrorCode, toolError } from '../errors.js';
2
+ import { SERVER_VERSION } from '../version.js';
2
3
  import { classifyUpstreamError } from './openrouter-errors.js';
3
- import { extractCompletionText, detectReasoningCutoff, toUsageMeta, } from './completion-utils.js';
4
+ import { extractCompletionText, detectReasoningCutoff, buildCompletionMeta, } from './completion-utils.js';
4
5
  import { readProviderDefaults, mergeProviderOptions, buildProviderBody, resolveMaxTokens, } from './provider-routing.js';
6
+ import { buildCacheHeaders, extractCacheMeta, } from './cache.js';
7
+ import { awaitCompletionWithHeaders } from './openai-withresponse.js';
5
8
  const DEFAULT_MODEL = 'nvidia/nemotron-nano-12b-v2-vl:free';
9
+ function readIncludeReasoningDefault() {
10
+ const raw = (process.env.OPENROUTER_INCLUDE_REASONING ?? '').trim().toLowerCase();
11
+ return raw === '1' || raw === 'true' || raw === 'yes';
12
+ }
6
13
  export async function handleChatCompletion(request, openai, defaultModel) {
7
- const { messages, model, temperature, max_tokens, provider } = request.params.arguments ?? {
8
- messages: [],
9
- };
14
+ const args = request.params.arguments ?? { messages: [] };
15
+ const { messages, model, temperature, max_tokens, provider, include_reasoning, online, web_max_results, cache, cache_ttl, cache_clear, } = args;
10
16
  if (!messages?.length) {
11
17
  return toolError(ErrorCode.INVALID_INPUT, 'Messages array cannot be empty.');
12
18
  }
13
19
  const providerOptions = mergeProviderOptions(readProviderDefaults(), provider);
14
20
  const providerBody = buildProviderBody(providerOptions);
15
21
  const effectiveMaxTokens = resolveMaxTokens(max_tokens);
16
- // Build the request body. `provider` is an OpenRouter extension not in
17
- // the OpenAI SDK's types, so we cast to unknown to thread it through.
22
+ const wantsReasoning = include_reasoning ?? readIncludeReasoningDefault();
23
+ // Build the request body. Several OpenRouter extensions aren't in the
24
+ // OpenAI SDK types, so we assemble as `Record<string, unknown>` and cast
25
+ // at the call site.
18
26
  const body = {
19
27
  model: model || defaultModel || DEFAULT_MODEL,
20
28
  messages,
@@ -24,9 +32,24 @@ export async function handleChatCompletion(request, openai, defaultModel) {
24
32
  body.max_tokens = effectiveMaxTokens;
25
33
  if (providerBody)
26
34
  body.provider = providerBody;
35
+ if (wantsReasoning)
36
+ body.include_reasoning = true;
37
+ if (online) {
38
+ const plugin = { id: 'web' };
39
+ if (typeof web_max_results === 'number' && web_max_results > 0) {
40
+ plugin.max_results = web_max_results;
41
+ }
42
+ body.plugins = [plugin];
43
+ }
44
+ const headers = buildCacheHeaders({ cache, cache_ttl, cache_clear });
45
+ const requestOpts = Object.keys(headers).length > 0 ? { headers } : undefined;
27
46
  let completion;
47
+ let responseHeaders;
28
48
  try {
29
- completion = (await openai.chat.completions.create(body));
49
+ const call = openai.chat.completions.create(body, requestOpts);
50
+ const { data, response } = await awaitCompletionWithHeaders(call);
51
+ completion = data;
52
+ responseHeaders = response?.headers;
30
53
  }
31
54
  catch (err) {
32
55
  return classifyUpstreamError(err);
@@ -38,13 +61,18 @@ export async function handleChatCompletion(request, openai, defaultModel) {
38
61
  if (!extracted.text) {
39
62
  return toolError(ErrorCode.INTERNAL, 'Model returned no textual content.', {
40
63
  finish_reason: extracted.finishReason,
64
+ native_finish_reason: extracted.nativeFinishReason,
41
65
  });
42
66
  }
67
+ const cacheMeta = extractCacheMeta(responseHeaders);
68
+ const extra = { server_version: SERVER_VERSION };
69
+ if (cacheMeta)
70
+ extra.cache = cacheMeta;
43
71
  return {
44
72
  content: [{ type: 'text', text: extracted.text }],
45
- _meta: {
46
- finish_reason: extracted.finishReason,
47
- ...(toUsageMeta(extracted.usage) ?? {}),
48
- },
73
+ _meta: buildCompletionMeta(extracted, {
74
+ includeReasoning: wantsReasoning,
75
+ extra,
76
+ }),
49
77
  };
50
78
  }
@@ -14,6 +14,19 @@ export interface ExtractedText {
14
14
  /** True when `text` came from the reasoning trace (not a final answer). */
15
15
  reasonedOnly: boolean;
16
16
  finishReason: ChatCompletion.Choice['finish_reason'] | undefined;
17
+ /**
18
+ * OpenRouter's `native_finish_reason`, when present. Carries the
19
+ * provider-native value before OpenRouter normalizes it. Surfaced in
20
+ * `_meta.native_finish_reason` for debuggability.
21
+ */
22
+ nativeFinishReason: string | undefined;
23
+ /**
24
+ * Raw reasoning trace (content of `reasoning` or joined `reasoning_details`).
25
+ * Populated whenever the upstream response carried one, even when the
26
+ * assistant also produced a final `content` answer. Surfaced to callers
27
+ * via `_meta.reasoning` when they opt in with `include_reasoning: true`.
28
+ */
29
+ reasoning?: string;
17
30
  usage?: ChatCompletion['usage'];
18
31
  }
19
32
  export declare function extractCompletionText(completion: ChatCompletion): ExtractedText;
@@ -25,3 +38,19 @@ export declare function extractCompletionText(completion: ChatCompletion): Extra
25
38
  */
26
39
  export declare function detectReasoningCutoff(extracted: ExtractedText): ToolErrorResult | null;
27
40
  export declare function toUsageMeta(usage: ChatCompletion['usage'] | undefined): Record<string, unknown> | undefined;
41
+ /**
42
+ * Build the common `_meta` shape for chat-completion-derived tools.
43
+ * Folds in:
44
+ * - normalized and native finish reasons (from the choice)
45
+ * - optional `reasoning` trace (when the caller opted in)
46
+ * - token usage (prompt / completion / total)
47
+ * - server version stamp
48
+ *
49
+ * Caller can pass `extra` to merge additional keys (cache metadata,
50
+ * content_is_untrusted, etc.) without repeating this boilerplate.
51
+ */
52
+ export interface BuildMetaOptions {
53
+ includeReasoning?: boolean;
54
+ extra?: Record<string, unknown>;
55
+ }
56
+ export declare function buildCompletionMeta(extracted: ExtractedText, opts?: BuildMetaOptions): Record<string, unknown>;
@@ -1,14 +1,45 @@
1
1
  import { ErrorCode, toolError } from '../errors.js';
2
+ function extractReasoning(msg) {
3
+ if (typeof msg.reasoning === 'string' && msg.reasoning.length > 0)
4
+ return msg.reasoning;
5
+ if (Array.isArray(msg.reasoning_details) && msg.reasoning_details.length > 0) {
6
+ const joined = msg.reasoning_details
7
+ .filter((d) => typeof d.text === 'string')
8
+ .map((d) => d.text)
9
+ .join('\n');
10
+ if (joined.length > 0)
11
+ return joined;
12
+ }
13
+ return undefined;
14
+ }
2
15
  export function extractCompletionText(completion) {
3
16
  const choice = completion.choices?.[0];
4
17
  const msg = choice?.message;
5
18
  const finishReason = choice?.finish_reason;
19
+ // `native_finish_reason` is an OpenRouter extension, not in the OpenAI
20
+ // SDK types — read it via an unknown-cast.
21
+ const nativeFinishReason = choice?.native_finish_reason ?? undefined;
6
22
  const usage = completion.usage ?? undefined;
7
- if (!msg)
8
- return { text: '', reasonedOnly: false, finishReason, usage };
9
- const { content, reasoning, reasoning_details } = msg;
23
+ if (!msg) {
24
+ return {
25
+ text: '',
26
+ reasonedOnly: false,
27
+ finishReason,
28
+ nativeFinishReason: nativeFinishReason ?? undefined,
29
+ usage,
30
+ };
31
+ }
32
+ const { content } = msg;
33
+ const reasoning = extractReasoning(msg);
10
34
  if (typeof content === 'string' && content.length > 0) {
11
- return { text: content, reasonedOnly: false, finishReason, usage };
35
+ return {
36
+ text: content,
37
+ reasonedOnly: false,
38
+ finishReason,
39
+ nativeFinishReason: nativeFinishReason ?? undefined,
40
+ reasoning,
41
+ usage,
42
+ };
12
43
  }
13
44
  if (Array.isArray(content)) {
14
45
  const parts = content
@@ -16,22 +47,33 @@ export function extractCompletionText(completion) {
16
47
  .map((p) => p.text ?? '');
17
48
  const joined = parts.join('');
18
49
  if (joined.length > 0) {
19
- return { text: joined, reasonedOnly: false, finishReason, usage };
50
+ return {
51
+ text: joined,
52
+ reasonedOnly: false,
53
+ finishReason,
54
+ nativeFinishReason: nativeFinishReason ?? undefined,
55
+ reasoning,
56
+ usage,
57
+ };
20
58
  }
21
59
  }
22
- if (typeof reasoning === 'string' && reasoning.length > 0) {
23
- return { text: reasoning, reasonedOnly: true, finishReason, usage };
24
- }
25
- if (Array.isArray(reasoning_details) && reasoning_details.length > 0) {
26
- const joined = reasoning_details
27
- .filter((d) => typeof d.text === 'string')
28
- .map((d) => d.text)
29
- .join('\n');
30
- if (joined.length > 0) {
31
- return { text: joined, reasonedOnly: true, finishReason, usage };
32
- }
60
+ if (reasoning && reasoning.length > 0) {
61
+ return {
62
+ text: reasoning,
63
+ reasonedOnly: true,
64
+ finishReason,
65
+ nativeFinishReason: nativeFinishReason ?? undefined,
66
+ reasoning,
67
+ usage,
68
+ };
33
69
  }
34
- return { text: '', reasonedOnly: false, finishReason, usage };
70
+ return {
71
+ text: '',
72
+ reasonedOnly: false,
73
+ finishReason,
74
+ nativeFinishReason: nativeFinishReason ?? undefined,
75
+ usage,
76
+ };
35
77
  }
36
78
  /**
37
79
  * If the extracted response is reasoning-only and was cut off by
@@ -67,3 +109,20 @@ export function toUsageMeta(usage) {
67
109
  },
68
110
  };
69
111
  }
112
+ export function buildCompletionMeta(extracted, opts = {}) {
113
+ const meta = {
114
+ finish_reason: extracted.finishReason,
115
+ };
116
+ if (extracted.nativeFinishReason) {
117
+ meta.native_finish_reason = extracted.nativeFinishReason;
118
+ }
119
+ if (opts.includeReasoning && extracted.reasoning && !extracted.reasonedOnly) {
120
+ meta.reasoning = extracted.reasoning;
121
+ }
122
+ const usageMeta = toUsageMeta(extracted.usage);
123
+ if (usageMeta)
124
+ Object.assign(meta, usageMeta);
125
+ if (opts.extra)
126
+ Object.assign(meta, opts.extra);
127
+ return meta;
128
+ }
@@ -41,6 +41,7 @@ export declare function handleGenerateAudio(request: {
41
41
  text?: undefined;
42
42
  })[];
43
43
  _meta: {
44
+ server_version: string;
44
45
  save_path: string;
45
46
  mime: string;
46
47
  size_bytes: number;
@@ -58,6 +59,7 @@ export declare function handleGenerateAudio(request: {
58
59
  text?: undefined;
59
60
  })[];
60
61
  _meta: {
62
+ server_version: string;
61
63
  mime: string;
62
64
  size_bytes: number;
63
65
  save_path?: undefined;
@@ -2,6 +2,8 @@ import { promises as fs } from 'fs';
2
2
  import { extname } from 'path';
3
3
  import { resolveSafeOutputPath, UnsafeOutputPathError } from './path-safety.js';
4
4
  import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
5
+ import { SERVER_VERSION } from '../version.js';
6
+ import { logger } from '../logger.js';
5
7
  import { classifyUpstreamError } from './openrouter-errors.js';
6
8
  const DEFAULT_MODEL = 'openai/gpt-audio';
7
9
  const DEFAULT_VOICE = 'alloy';
@@ -89,6 +91,16 @@ export async function handleGenerateAudio(request, openai) {
89
91
  if (!prompt?.trim()) {
90
92
  return toolError(ErrorCode.INVALID_INPUT, 'prompt is required.');
91
93
  }
94
+ // Audit entry. See generate_image for rationale.
95
+ logger.audit('generate_audio.start', {
96
+ model: model || DEFAULT_MODEL,
97
+ voice: voice?.trim() || DEFAULT_VOICE,
98
+ format: VALID_FORMATS.includes(format ?? '')
99
+ ? format
100
+ : DEFAULT_FORMAT,
101
+ prompt_preview: prompt.slice(0, 80),
102
+ save_path: save_path ? 'provided' : 'none',
103
+ });
92
104
  // Fail-fast on unsafe paths BEFORE spending tokens.
93
105
  let safeBase = null;
94
106
  if (save_path) {
@@ -165,6 +177,7 @@ export async function handleGenerateAudio(request, openai) {
165
177
  { type: 'audio', mimeType: detected.mimeType, data: returnBase64 },
166
178
  ],
167
179
  _meta: {
180
+ server_version: SERVER_VERSION,
168
181
  save_path: actualSavePath,
169
182
  mime: detected.mimeType,
170
183
  size_bytes: audioBuffer.length,
@@ -176,7 +189,11 @@ export async function handleGenerateAudio(request, openai) {
176
189
  { type: 'text', text: transcript || 'Audio generated successfully.' },
177
190
  { type: 'audio', mimeType: detected.mimeType, data: returnBase64 },
178
191
  ],
179
- _meta: { mime: detected.mimeType, size_bytes: audioBuffer.length },
192
+ _meta: {
193
+ server_version: SERVER_VERSION,
194
+ mime: detected.mimeType,
195
+ size_bytes: audioBuffer.length,
196
+ },
180
197
  };
181
198
  }
182
199
  catch (err) {