@stabgan/openrouter-mcp-multimodal 4.7.0 → 4.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. package/README.md +90 -36
  2. package/dist/errors.d.ts +5 -20
  3. package/dist/errors.js +1 -10
  4. package/dist/index.js +7 -1
  5. package/dist/logger.js +54 -24
  6. package/dist/model-cache.d.ts +13 -0
  7. package/dist/model-cache.js +61 -5
  8. package/dist/openrouter-api.d.ts +14 -15
  9. package/dist/openrouter-api.js +68 -22
  10. package/dist/tool-definitions.d.ts +24 -0
  11. package/dist/tool-definitions.js +276 -174
  12. package/dist/tool-descriptions.js +23 -15
  13. package/dist/tool-handlers/analyze-audio.js +4 -1
  14. package/dist/tool-handlers/analyze-image.js +10 -5
  15. package/dist/tool-handlers/analyze-video.js +9 -5
  16. package/dist/tool-handlers/async-chat.d.ts +17 -0
  17. package/dist/tool-handlers/async-chat.js +104 -30
  18. package/dist/tool-handlers/audio-utils.d.ts +19 -4
  19. package/dist/tool-handlers/audio-utils.js +170 -16
  20. package/dist/tool-handlers/cache.d.ts +3 -3
  21. package/dist/tool-handlers/cache.js +56 -4
  22. package/dist/tool-handlers/chat-completion.js +16 -7
  23. package/dist/tool-handlers/chat-request.d.ts +3 -0
  24. package/dist/tool-handlers/chat-request.js +28 -0
  25. package/dist/tool-handlers/completion-utils.d.ts +5 -11
  26. package/dist/tool-handlers/completion-utils.js +76 -47
  27. package/dist/tool-handlers/fetch-utils.d.ts +14 -0
  28. package/dist/tool-handlers/fetch-utils.js +331 -68
  29. package/dist/tool-handlers/generate-audio.d.ts +4 -15
  30. package/dist/tool-handlers/generate-audio.js +21 -53
  31. package/dist/tool-handlers/generate-image-dedicated.d.ts +1 -1
  32. package/dist/tool-handlers/generate-image-dedicated.js +50 -31
  33. package/dist/tool-handlers/generate-image.d.ts +1 -1
  34. package/dist/tool-handlers/generate-image.js +19 -22
  35. package/dist/tool-handlers/generate-video.d.ts +4 -3
  36. package/dist/tool-handlers/generate-video.js +42 -18
  37. package/dist/tool-handlers/get-model-info.js +1 -1
  38. package/dist/tool-handlers/health-check.js +39 -15
  39. package/dist/tool-handlers/image-utils.js +2 -2
  40. package/dist/tool-handlers/openrouter-errors.d.ts +2 -0
  41. package/dist/tool-handlers/openrouter-errors.js +138 -31
  42. package/dist/tool-handlers/path-safety.js +49 -17
  43. package/dist/tool-handlers/path-utils.d.ts +2 -0
  44. package/dist/tool-handlers/path-utils.js +13 -0
  45. package/dist/tool-handlers/provider-routing.d.ts +2 -0
  46. package/dist/tool-handlers/provider-routing.js +10 -0
  47. package/dist/tool-handlers/rerank.d.ts +1 -4
  48. package/dist/tool-handlers/rerank.js +43 -14
  49. package/dist/tool-handlers/search-models.js +3 -3
  50. package/dist/tool-handlers/speech-to-text.d.ts +1 -0
  51. package/dist/tool-handlers/speech-to-text.js +23 -56
  52. package/dist/tool-handlers/text-to-speech.d.ts +1 -1
  53. package/dist/tool-handlers/text-to-speech.js +18 -8
  54. package/dist/tool-handlers/tool-result-payload.js +11 -9
  55. package/dist/tool-handlers/validate-model.js +1 -1
  56. package/dist/tool-handlers.d.ts +9 -0
  57. package/dist/tool-handlers.js +17 -5
  58. package/dist/version.d.ts +1 -1
  59. package/dist/version.js +1 -1
  60. package/package.json +1 -3
@@ -55,8 +55,8 @@ export const TOOL_DESCRIPTIONS = {
55
55
  'You already know the model id (or rely on the server default)',
56
56
  ],
57
57
  notWhen: [
58
- 'Input is an image/audio/video file → use analyze_image / analyze_audio / analyze_video',
59
- 'You need to create images, audio, or video → use generate_* tools',
58
+ 'Input is a single image/audio/video file → use analyze_image / analyze_audio / analyze_video (dedicated wrappers)',
59
+ 'You need to create images, audio, or video → use generate_* / text_to_speech tools',
60
60
  'You only need to check if a model exists → use validate_model',
61
61
  ],
62
62
  goodExamples: [
@@ -136,14 +136,14 @@ export const TOOL_DESCRIPTIONS = {
136
136
  'The image is a local file under the input sandbox, a public https URL, or a data URL',
137
137
  ],
138
138
  notWhen: [
139
- 'You want to generate a new image → use generate_image',
139
+ 'You want to generate a new image → use generate_image or generate_image_dedicated',
140
140
  'You need multi-file batch analysis in one call → not supported; call once per image',
141
- 'Pure text chat → use chat_completion with a vision-capable model instead (less ergonomic)',
141
+ 'Multi-turn vision chat with several images → use chat_completion with a vision model (analyze_image is single-image only)',
142
142
  ],
143
143
  goodExamples: [
144
144
  '`{ "image_path": "diagram.png", "question": "List every label in this diagram." }`',
145
145
  '`{ "image_path": "https://example.com/photo.jpg", "question": "Describe the scene." }`',
146
- '`{ "model": "google/gemini-2.5-flash", "image_path": "scan.jpg", "question": "Extract text" }`',
146
+ '`{ "image_path": "scan.jpg", "question": "Extract text", "model": "google/gemini-2.5-flash" }` (optional model override)',
147
147
  ],
148
148
  badExamples: [
149
149
  '`{ "url": "photo.jpg" }` → wrong key; use `image_path`',
@@ -166,9 +166,9 @@ export const TOOL_DESCRIPTIONS = {
166
166
  'Format is a common audio container the decoder recognizes',
167
167
  ],
168
168
  notWhen: [
169
- 'You want text-to-speech → use generate_audio',
169
+ 'You want text-to-speech → use text_to_speech (dedicated, faster) or generate_audio (chat route, music/SFX)',
170
+ 'You want pure transcription without Q&A → use speech_to_text',
170
171
  'Input is video → use analyze_video (or extract audio first)',
171
- 'Pure text chat → use chat_completion',
172
172
  ],
173
173
  goodExamples: [
174
174
  '`{ "audio_path": "meeting.wav", "question": "Transcribe verbatim." }`',
@@ -290,13 +290,15 @@ export const TOOL_DESCRIPTIONS = {
290
290
  worksWith: ['get_model_info', 'chat_completion'],
291
291
  }),
292
292
  generate_image: buildToolDescription({
293
- summary: 'Generate an image from a text prompt. Optional `input_images` condition style/identity. ' +
294
- 'Default model: google/gemini-2.5-flash-image.',
293
+ summary: 'Generate an image via chat completions (modalities route). Optional `input_images` for style/identity. ' +
294
+ 'Default model: google/gemini-2.5-flash-image. Use generate_image_dedicated for resolution/quality/format control or newer Image API models.',
295
295
  useWhen: [
296
- 'You need a new image from a text prompt',
297
- 'You have reference images for style or subject consistency',
296
+ 'You need a new image from a text prompt via the chat-completions route',
297
+ 'You have reference images for style or subject consistency (`input_images`)',
298
+ 'Simple text-to-image without dedicated API knobs',
298
299
  ],
299
300
  notWhen: [
301
+ 'You need resolution tiers, quality, output_format, or Image API-only models → generate_image_dedicated',
300
302
  'You want to analyze an existing image → analyze_image',
301
303
  'You want video → generate_video or generate_video_from_image',
302
304
  'Prompt is empty or only whitespace',
@@ -329,6 +331,7 @@ export const TOOL_DESCRIPTIONS = {
329
331
  'You need image-to-image with `input_references`',
330
332
  ],
331
333
  notWhen: [
334
+ 'Simple text-to-image without format/resolution control → generate_image (fewer params)',
332
335
  'You want to analyze an existing image → analyze_image',
333
336
  'You want video → generate_video or generate_video_from_image',
334
337
  'Prompt is empty or only whitespace',
@@ -352,12 +355,17 @@ export const TOOL_DESCRIPTIONS = {
352
355
  worksWith: ['analyze_image', 'generate_video_from_image'],
353
356
  }),
354
357
  generate_audio: buildToolDescription({
355
- summary: 'Generate speech or music from a text prompt. Output format is auto-detected; file extension auto-corrected on save.',
358
+ summary: 'Generate speech or music via chat completions (modalities route). Formats: wav, mp3, flac, opus, pcm16. ' +
359
+ 'Default model: openai/gpt-audio, voice alloy. Use text_to_speech for dedicated TTS with speed/instructions.',
356
360
  useWhen: [
357
- 'You need TTS or audio generation from text',
358
- 'Optional save_path is inside the output sandbox',
361
+ 'You need music, sound effects, or expressive speech via the chat route',
362
+ 'You want pcm16/wav output from the gpt-audio family',
363
+ ],
364
+ notWhen: [
365
+ 'You want fast dedicated TTS with speed/instructions → text_to_speech',
366
+ 'You want to transcribe existing audio → speech_to_text or analyze_audio',
367
+ 'Prompt is empty',
359
368
  ],
360
- notWhen: ['You want to transcribe existing audio → analyze_audio', 'Prompt is empty'],
361
369
  goodExamples: [
362
370
  '`{ "prompt": "Say hello world in a calm voice." }`',
363
371
  '`{ "prompt": "Upbeat jingle", "save_path": "out/jingle.mp3" }`',
@@ -4,7 +4,7 @@ import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
4
4
  import { SERVER_VERSION } from '../version.js';
5
5
  import { classifyUpstreamError } from './openrouter-errors.js';
6
6
  import { extractCompletionText, detectReasoningCutoff, buildCompletionMeta, } from './completion-utils.js';
7
- import { buildCacheHeaders, extractCacheMeta } from './cache.js';
7
+ import { buildCacheHeaders, extractCacheMeta, validateCacheOptions, } from './cache.js';
8
8
  import { awaitCompletionWithHeaders } from './openai-withresponse.js';
9
9
  const DEFAULT_MODEL = 'google/gemini-2.5-flash';
10
10
  export async function handleAnalyzeAudio(request, openai, defaultModel) {
@@ -13,6 +13,9 @@ export async function handleAnalyzeAudio(request, openai, defaultModel) {
13
13
  if (!audio_path) {
14
14
  return toolError(ErrorCode.INVALID_INPUT, 'audio_path is required.');
15
15
  }
16
+ const cacheError = validateCacheOptions({ cache, cache_ttl, cache_clear });
17
+ if (cacheError)
18
+ return cacheError;
16
19
  let audioData;
17
20
  try {
18
21
  audioData = await prepareAudioData(audio_path);
@@ -4,7 +4,7 @@ import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
4
4
  import { SERVER_VERSION } from '../version.js';
5
5
  import { classifyUpstreamError } from './openrouter-errors.js';
6
6
  import { extractCompletionText, detectReasoningCutoff, buildCompletionMeta, } from './completion-utils.js';
7
- import { buildCacheHeaders, extractCacheMeta } from './cache.js';
7
+ import { buildCacheHeaders, extractCacheMeta, validateCacheOptions, } from './cache.js';
8
8
  import { awaitCompletionWithHeaders } from './openai-withresponse.js';
9
9
  const DEFAULT_MODEL = 'nvidia/nemotron-nano-12b-v2-vl:free';
10
10
  export async function handleAnalyzeImage(request, openai, defaultModel) {
@@ -13,6 +13,9 @@ export async function handleAnalyzeImage(request, openai, defaultModel) {
13
13
  if (!image_path) {
14
14
  return toolError(ErrorCode.INVALID_INPUT, 'image_path is required.');
15
15
  }
16
+ const cacheError = validateCacheOptions({ cache, cache_ttl, cache_clear });
17
+ if (cacheError)
18
+ return cacheError;
16
19
  let imageUrl;
17
20
  try {
18
21
  imageUrl = await prepareImageUrl(image_path);
@@ -22,12 +25,14 @@ export async function handleAnalyzeImage(request, openai, defaultModel) {
22
25
  return toolErrorFrom(ErrorCode.UNSAFE_PATH, err);
23
26
  }
24
27
  const msg = err instanceof Error ? err.message : String(err);
25
- if (msg.includes('Blocked host'))
26
- return toolErrorFrom(ErrorCode.UPSTREAM_REFUSED, err);
28
+ const detail = `image_path "${image_path}": ${msg}`;
29
+ if (msg.includes('Blocked host')) {
30
+ return toolError(ErrorCode.UPSTREAM_REFUSED, detail);
31
+ }
27
32
  if (msg.toLowerCase().includes('too large')) {
28
- return toolErrorFrom(ErrorCode.RESOURCE_TOO_LARGE, err);
33
+ return toolError(ErrorCode.RESOURCE_TOO_LARGE, detail);
29
34
  }
30
- return toolErrorFrom(ErrorCode.INVALID_INPUT, err);
35
+ return toolError(ErrorCode.INVALID_INPUT, detail);
31
36
  }
32
37
  const imageBlock = {
33
38
  type: 'image_url',
@@ -5,7 +5,7 @@ import { SERVER_VERSION } from '../version.js';
5
5
  import { logger } from '../logger.js';
6
6
  import { classifyUpstreamError } from './openrouter-errors.js';
7
7
  import { extractCompletionText, detectReasoningCutoff, buildCompletionMeta, } from './completion-utils.js';
8
- import { buildCacheHeaders, extractCacheMeta } from './cache.js';
8
+ import { buildCacheHeaders, extractCacheMeta, validateCacheOptions, } from './cache.js';
9
9
  import { awaitCompletionWithHeaders } from './openai-withresponse.js';
10
10
  const FALLBACK_DEFAULT_MODEL = 'google/gemini-2.5-flash';
11
11
  export async function handleAnalyzeVideo(request, openai, defaultModel) {
@@ -14,6 +14,9 @@ export async function handleAnalyzeVideo(request, openai, defaultModel) {
14
14
  if (!video_path) {
15
15
  return toolError(ErrorCode.INVALID_INPUT, 'video_path is required.');
16
16
  }
17
+ const cacheError = validateCacheOptions({ cache, cache_ttl, cache_clear });
18
+ if (cacheError)
19
+ return cacheError;
17
20
  const pickedModel = model || process.env.OPENROUTER_DEFAULT_VIDEO_MODEL || defaultModel || FALLBACK_DEFAULT_MODEL;
18
21
  let videoData;
19
22
  try {
@@ -24,16 +27,17 @@ export async function handleAnalyzeVideo(request, openai, defaultModel) {
24
27
  return toolErrorFrom(ErrorCode.UNSAFE_PATH, err);
25
28
  }
26
29
  const msg = err instanceof Error ? err.message : String(err);
30
+ const detail = `video_path "${video_path}": ${msg}`;
27
31
  if (msg.includes('Blocked host')) {
28
- return toolErrorFrom(ErrorCode.UPSTREAM_REFUSED, err);
32
+ return toolError(ErrorCode.UPSTREAM_REFUSED, detail);
29
33
  }
30
34
  if (msg.includes('too large')) {
31
- return toolErrorFrom(ErrorCode.RESOURCE_TOO_LARGE, err);
35
+ return toolError(ErrorCode.RESOURCE_TOO_LARGE, detail);
32
36
  }
33
37
  if (msg.includes('Unsupported') || msg.includes('not a video')) {
34
- return toolErrorFrom(ErrorCode.UNSUPPORTED_FORMAT, err);
38
+ return toolError(ErrorCode.UNSUPPORTED_FORMAT, detail);
35
39
  }
36
- return toolErrorFrom(ErrorCode.INVALID_INPUT, err);
40
+ return toolError(ErrorCode.INVALID_INPUT, detail);
37
41
  }
38
42
  const videoBlock = {
39
43
  type: 'video_url',
@@ -18,6 +18,10 @@ export interface AsyncJob {
18
18
  error?: string;
19
19
  error_code?: ErrorCode;
20
20
  }
21
+ /** Test-only reset for module-level job state. */
22
+ export declare function resetAsyncJobStateForTests(): void;
23
+ /** Exported for tests — produces ids accepted by `isValidJobId`. */
24
+ export declare function generateJobId(): string;
21
25
  /** Load a persisted job from disk (exported for tests). */
22
26
  export declare function loadJobFromDisk(jobId: string): Promise<AsyncJob | null>;
23
27
  export declare function handleStartChatCompletion(request: {
@@ -41,6 +45,19 @@ export declare function handleGetChatCompletionStatus(request: {
41
45
  arguments: GetChatCompletionStatusRequest;
42
46
  };
43
47
  }): Promise<import("../errors.js").ToolErrorResult | {
48
+ content: {
49
+ type: "text";
50
+ text: string;
51
+ }[];
52
+ _meta: {
53
+ result_truncated?: boolean | undefined;
54
+ server_version: string;
55
+ job_id: string;
56
+ status: "completed";
57
+ model: string;
58
+ created_at: string;
59
+ };
60
+ } | {
44
61
  content: {
45
62
  type: "text";
46
63
  text: string;
@@ -1,19 +1,54 @@
1
1
  /** Async chat completions — in-memory jobs, optionally persisted under OPENROUTER_OUTPUT_DIR/openrouter-jobs/. */
2
+ import { randomBytes } from 'node:crypto';
2
3
  import { promises as fs } from 'node:fs';
3
4
  import path from 'node:path';
4
5
  import { ErrorCode, toolError } from '../errors.js';
5
6
  import { SERVER_VERSION } from '../version.js';
6
7
  import { logger } from '../logger.js';
7
- import { extractCompletionText, buildCompletionMeta } from './completion-utils.js';
8
+ import { extractCompletionText, buildCompletionMeta, capResultText } from './completion-utils.js';
8
9
  import { resolveSafeJobStatusPath, isValidJobId } from './path-safety.js';
9
10
  import { classifyUpstreamError } from './openrouter-errors.js';
10
- import { DEFAULT_CHAT_MODEL, buildChatCompletionBody, buildChatCompletionRequestOpts, asOpenAIChatBody, readIncludeReasoningDefault, } from './chat-request.js';
11
+ import { validateCacheOptions } from './cache.js';
12
+ import { DEFAULT_CHAT_MODEL, buildChatCompletionBody, buildChatCompletionRequestOpts, asOpenAIChatBody, readIncludeReasoningDefault, validateChatMessages, validateMaxTokens, } from './chat-request.js';
11
13
  const jobs = new Map();
12
14
  let jobCounter = 0;
13
- function generateJobId() {
15
+ const DEFAULT_ASYNC_JOBS_MEMORY_MAX = 200;
16
+ function readAsyncJobsMemoryMax() {
17
+ const raw = process.env.OPENROUTER_ASYNC_JOBS_MEMORY_MAX;
18
+ if (raw === undefined || raw === '')
19
+ return DEFAULT_ASYNC_JOBS_MEMORY_MAX;
20
+ if (raw === '0')
21
+ return 0;
22
+ const n = parseInt(raw, 10);
23
+ return Number.isFinite(n) && n > 0 ? n : DEFAULT_ASYNC_JOBS_MEMORY_MAX;
24
+ }
25
+ function evictTerminalJobsIfNeeded() {
26
+ const max = readAsyncJobsMemoryMax();
27
+ if (max <= 0 || jobs.size < max)
28
+ return;
29
+ const terminal = [...jobs.entries()]
30
+ .filter(([, job]) => job.status === 'completed' || job.status === 'failed')
31
+ .sort((a, b) => a[1].createdAt.localeCompare(b[1].createdAt));
32
+ while (jobs.size > max && terminal.length > 0) {
33
+ const [id] = terminal.shift();
34
+ jobs.delete(id);
35
+ }
36
+ }
37
+ function rememberJob(job) {
38
+ evictTerminalJobsIfNeeded();
39
+ jobs.set(job.id, job);
40
+ }
41
+ /** Test-only reset for module-level job state. */
42
+ export function resetAsyncJobStateForTests() {
43
+ jobs.clear();
44
+ jobCounter = 0;
45
+ }
46
+ /** Exported for tests — produces ids accepted by `isValidJobId`. */
47
+ export function generateJobId() {
14
48
  jobCounter += 1;
15
49
  const ts = new Date().toISOString().replace(/[-:T]/g, '').slice(0, 14);
16
- return `chat_${ts}_${String(jobCounter).padStart(3, '0')}`;
50
+ const entropy = randomBytes(4).toString('hex');
51
+ return `chat_${ts}_${String(jobCounter).padStart(3, '0')}_${entropy}`;
17
52
  }
18
53
  function getJobsDir() {
19
54
  const outputDir = process.env.OPENROUTER_OUTPUT_DIR;
@@ -62,7 +97,7 @@ async function resolveJob(jobId) {
62
97
  return inMemory;
63
98
  const fromDisk = await loadJobFromDisk(jobId);
64
99
  if (fromDisk) {
65
- jobs.set(jobId, fromDisk);
100
+ rememberJob(fromDisk);
66
101
  return fromDisk;
67
102
  }
68
103
  return undefined;
@@ -70,9 +105,15 @@ async function resolveJob(jobId) {
70
105
  export async function handleStartChatCompletion(request, openai, defaultModel) {
71
106
  const args = request.params.arguments ?? { messages: [] };
72
107
  const { messages, model, temperature, max_tokens, provider, include_reasoning, online, web_max_results, cache, cache_ttl, cache_clear, } = args;
73
- if (!messages?.length) {
74
- return toolError(ErrorCode.INVALID_INPUT, 'Messages array cannot be empty.');
75
- }
108
+ const messagesError = validateChatMessages(messages);
109
+ if (messagesError)
110
+ return messagesError;
111
+ const cacheError = validateCacheOptions({ cache, cache_ttl, cache_clear });
112
+ if (cacheError)
113
+ return cacheError;
114
+ const maxTokensError = validateMaxTokens(max_tokens);
115
+ if (maxTokensError)
116
+ return maxTokensError;
76
117
  const effectiveModel = model || defaultModel || DEFAULT_CHAT_MODEL;
77
118
  const jobId = generateJobId();
78
119
  const job = {
@@ -81,7 +122,7 @@ export async function handleStartChatCompletion(request, openai, defaultModel) {
81
122
  createdAt: new Date().toISOString(),
82
123
  model: effectiveModel,
83
124
  };
84
- jobs.set(jobId, job);
125
+ rememberJob(job);
85
126
  logger.audit('async_chat.start', {
86
127
  job_id: jobId,
87
128
  model: effectiveModel,
@@ -99,6 +140,16 @@ export async function handleStartChatCompletion(request, openai, defaultModel) {
99
140
  cache,
100
141
  cache_ttl,
101
142
  cache_clear,
143
+ }).catch((err) => {
144
+ logger.error('async_chat.unhandled', {
145
+ job_id: job.id,
146
+ err: err instanceof Error ? err.message : String(err),
147
+ });
148
+ if (job.status === 'running') {
149
+ job.status = 'failed';
150
+ job.error = 'Unexpected error during background completion.';
151
+ job.error_code = ErrorCode.INTERNAL;
152
+ }
102
153
  });
103
154
  return {
104
155
  content: [
@@ -116,35 +167,56 @@ export async function handleStartChatCompletion(request, openai, defaultModel) {
116
167
  };
117
168
  }
118
169
  async function runCompletionInBackground(job, openai, opts) {
119
- const wantsReasoning = opts.include_reasoning ?? readIncludeReasoningDefault();
120
- const body = buildChatCompletionBody(opts);
121
- const requestOpts = buildChatCompletionRequestOpts(opts);
122
170
  try {
123
- const completion = (await openai.chat.completions.create(asOpenAIChatBody(body), requestOpts));
124
- const extracted = extractCompletionText(completion);
125
- if (!extracted.text) {
171
+ const cacheError = validateCacheOptions(opts);
172
+ if (cacheError) {
126
173
  job.status = 'failed';
127
- job.error = 'Model returned no textual content.';
174
+ job.error = cacheError.content[0]?.text ?? 'Invalid cache options.';
175
+ job.error_code = cacheError._meta.code;
176
+ await persistJob(job);
177
+ evictTerminalJobsIfNeeded();
178
+ return;
179
+ }
180
+ const wantsReasoning = opts.include_reasoning ?? readIncludeReasoningDefault();
181
+ const body = buildChatCompletionBody(opts);
182
+ const requestOpts = buildChatCompletionRequestOpts(opts);
183
+ try {
184
+ const completion = (await openai.chat.completions.create(asOpenAIChatBody(body), requestOpts));
185
+ const extracted = extractCompletionText(completion);
186
+ if (!extracted.text) {
187
+ job.status = 'failed';
188
+ job.error = 'Model returned no textual content.';
189
+ }
190
+ else {
191
+ job.status = 'completed';
192
+ job.result = {
193
+ text: extracted.text,
194
+ meta: buildCompletionMeta(extracted, {
195
+ includeReasoning: wantsReasoning,
196
+ extra: { server_version: SERVER_VERSION },
197
+ }),
198
+ };
199
+ }
128
200
  }
129
- else {
130
- job.status = 'completed';
131
- job.result = {
132
- text: extracted.text,
133
- meta: buildCompletionMeta(extracted, {
134
- includeReasoning: wantsReasoning,
135
- extra: { server_version: SERVER_VERSION },
136
- }),
137
- };
201
+ catch (err) {
202
+ job.status = 'failed';
203
+ const classified = classifyUpstreamError(err);
204
+ job.error = classified.content[0]?.text ?? 'Job failed.';
205
+ job.error_code = classified._meta.code;
206
+ logger.warn('async_chat.failed', { job_id: job.id, error: job.error, code: job.error_code });
138
207
  }
139
208
  }
140
209
  catch (err) {
141
210
  job.status = 'failed';
142
- const classified = classifyUpstreamError(err);
143
- job.error = classified.content[0]?.text ?? 'Job failed.';
144
- job.error_code = classified._meta.code;
145
- logger.warn('async_chat.failed', { job_id: job.id, error: job.error, code: job.error_code });
211
+ job.error = 'Unexpected error during background completion.';
212
+ job.error_code = ErrorCode.INTERNAL;
213
+ logger.error('async_chat.background_error', {
214
+ job_id: job.id,
215
+ err: err instanceof Error ? err.message : String(err),
216
+ });
146
217
  }
147
218
  await persistJob(job);
219
+ evictTerminalJobsIfNeeded();
148
220
  }
149
221
  export async function handleGetChatCompletionStatus(request) {
150
222
  const args = request.params.arguments ?? {};
@@ -163,14 +235,16 @@ export async function handleGetChatCompletionStatus(request) {
163
235
  return toolError(ErrorCode.INVALID_INPUT, `No job found with id "${jobId}".${hint}`);
164
236
  }
165
237
  if (job.status === 'completed' && job.result) {
238
+ const capped = capResultText(job.result.text);
166
239
  return {
167
- content: [{ type: 'text', text: job.result.text }],
240
+ content: [{ type: 'text', text: capped.text }],
168
241
  _meta: {
169
242
  server_version: SERVER_VERSION,
170
243
  job_id: jobId,
171
244
  status: 'completed',
172
245
  model: job.model,
173
246
  created_at: job.createdAt,
247
+ ...(capped.truncated ? { result_truncated: true } : {}),
174
248
  ...job.result.meta,
175
249
  },
176
250
  };
@@ -1,16 +1,31 @@
1
1
  export { isBlockedIPv4, assertUrlSafeForFetch } from './fetch-utils.js';
2
- /** File-extension formats (matchable by .ext) */
2
+ /** Shared ceiling for speech_to_text local, HTTP, and data-URL inputs. */
3
+ export declare function getMaxAudioInputBytes(): number;
3
4
  declare const FILE_AUDIO_FORMATS: readonly ["wav", "mp3", "aiff", "aac", "ogg", "flac", "m4a"];
4
5
  export declare const SUPPORTED_AUDIO_FORMATS: readonly ["wav", "mp3", "aiff", "aac", "ogg", "flac", "m4a", "pcm16", "pcm24"];
5
6
  export type AudioFormat = (typeof SUPPORTED_AUDIO_FORMATS)[number];
6
7
  type FileAudioFormat = (typeof FILE_AUDIO_FORMATS)[number];
7
- /** Get audio format from file extension. Returns undefined for non-audio or API-only formats. */
8
+ export declare const STT_FILE_EXTENSIONS: readonly ["mp3", "mp4", "m4a", "wav", "flac", "ogg", "oga", "webm", "opus"];
9
+ export type SttFileExtension = (typeof STT_FILE_EXTENSIONS)[number];
10
+ /**
11
+ * Detect audio container format from magic bytes. MP3 detection is strict:
12
+ * ID3 tags or frame sync with valid MPEG header fields.
13
+ */
14
+ export declare function detectAudioFormat(data: Buffer): {
15
+ ext: string;
16
+ mimeType: string;
17
+ };
8
18
  export declare function getAudioFormat(filePath: string): FileAudioFormat | undefined;
9
- /** Get MIME type for an audio format. */
10
19
  export declare function getAudioMimeType(format: AudioFormat): string;
11
20
  export interface AudioData {
12
21
  data: string;
13
22
  format: AudioFormat;
14
23
  }
15
- /** Prepare audio from data URL, HTTP URL, or sandboxed local file. */
16
24
  export declare function prepareAudioData(source: string): Promise<AudioData>;
25
+ /** Map a speech_to_text file extension to the upstream format slug. */
26
+ export declare function sttFormatFromExtension(ext: string): string;
27
+ /** Resolve speech_to_text audio from data URL, HTTP URL, or sandboxed local file. */
28
+ export declare function resolveSpeechToTextAudio(audioPath: string): Promise<{
29
+ data: string;
30
+ format: string;
31
+ }>;