@stabgan/openrouter-mcp-multimodal 4.0.1 → 4.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/README.md +95 -8
  2. package/dist/errors.d.ts +26 -8
  3. package/dist/errors.js +17 -12
  4. package/dist/index.js +21 -5
  5. package/dist/logger.d.ts +11 -0
  6. package/dist/logger.js +26 -0
  7. package/dist/model-cache.d.ts +17 -0
  8. package/dist/model-cache.js +27 -1
  9. package/dist/openrouter-api.d.ts +22 -2
  10. package/dist/openrouter-api.js +21 -2
  11. package/dist/tool-handlers/analyze-audio.d.ts +9 -4
  12. package/dist/tool-handlers/analyze-audio.js +29 -15
  13. package/dist/tool-handlers/analyze-image.d.ts +10 -4
  14. package/dist/tool-handlers/analyze-image.js +37 -9
  15. package/dist/tool-handlers/analyze-video.d.ts +9 -4
  16. package/dist/tool-handlers/analyze-video.js +31 -19
  17. package/dist/tool-handlers/cache.d.ts +33 -0
  18. package/dist/tool-handlers/cache.js +54 -0
  19. package/dist/tool-handlers/chat-completion.d.ts +18 -4
  20. package/dist/tool-handlers/chat-completion.js +39 -11
  21. package/dist/tool-handlers/completion-utils.d.ts +29 -0
  22. package/dist/tool-handlers/completion-utils.js +76 -17
  23. package/dist/tool-handlers/generate-audio.d.ts +2 -0
  24. package/dist/tool-handlers/generate-audio.js +18 -1
  25. package/dist/tool-handlers/generate-image.d.ts +2 -0
  26. package/dist/tool-handlers/generate-image.js +15 -0
  27. package/dist/tool-handlers/generate-video.d.ts +43 -0
  28. package/dist/tool-handlers/generate-video.js +46 -3
  29. package/dist/tool-handlers/get-model-info.d.ts +1 -6
  30. package/dist/tool-handlers/get-model-info.js +2 -1
  31. package/dist/tool-handlers/health-check.d.ts +23 -0
  32. package/dist/tool-handlers/health-check.js +35 -0
  33. package/dist/tool-handlers/openai-withresponse.d.ts +16 -0
  34. package/dist/tool-handlers/openai-withresponse.js +16 -0
  35. package/dist/tool-handlers/openrouter-errors.d.ts +5 -1
  36. package/dist/tool-handlers/openrouter-errors.js +72 -11
  37. package/dist/tool-handlers/rerank.d.ts +17 -0
  38. package/dist/tool-handlers/rerank.js +52 -0
  39. package/dist/tool-handlers/search-models.d.ts +18 -7
  40. package/dist/tool-handlers/search-models.js +25 -2
  41. package/dist/tool-handlers/structured-output.d.ts +13 -0
  42. package/dist/tool-handlers/structured-output.js +24 -0
  43. package/dist/tool-handlers/validate-model.d.ts +4 -6
  44. package/dist/tool-handlers/validate-model.js +3 -8
  45. package/dist/tool-handlers.d.ts +1 -0
  46. package/dist/tool-handlers.js +435 -165
  47. package/dist/version.d.ts +16 -0
  48. package/dist/version.js +16 -0
  49. package/package.json +1 -1
@@ -69,6 +69,7 @@ export declare function handleGenerateImage(request: {
69
69
  completion_tokens: number;
70
70
  total_tokens: number;
71
71
  } | undefined;
72
+ server_version: string;
72
73
  save_path: string;
73
74
  mime: string;
74
75
  };
@@ -84,6 +85,7 @@ export declare function handleGenerateImage(request: {
84
85
  completion_tokens: number;
85
86
  total_tokens: number;
86
87
  } | undefined;
88
+ server_version: string;
87
89
  mime: string;
88
90
  };
89
91
  }>;
@@ -3,6 +3,8 @@ import path from 'node:path';
3
3
  import { resolveSafeOutputPath, resolveSafeInputPath, UnsafeOutputPathError } from './path-safety.js';
4
4
  import { parseBase64DataUrl } from './fetch-utils.js';
5
5
  import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
6
+ import { SERVER_VERSION } from '../version.js';
7
+ import { logger } from '../logger.js';
6
8
  import { classifyUpstreamError } from './openrouter-errors.js';
7
9
  const DEFAULT_MODEL = 'google/gemini-2.5-flash-image';
8
10
  // OpenRouter-documented aspect ratios (standard + extended). Extended are
@@ -30,6 +32,17 @@ export async function handleGenerateImage(request, openai) {
30
32
  if (!prompt?.trim()) {
31
33
  return toolError(ErrorCode.INVALID_INPUT, 'prompt is required.');
32
34
  }
35
+ // Audit entry. Bypasses the normal log level so operators always see a
36
+ // record of cost-incurring operations. Prompt preview is hard-capped
37
+ // at 80 chars to avoid PII spillage in log aggregators.
38
+ logger.audit('generate_image.start', {
39
+ model: model || DEFAULT_MODEL,
40
+ prompt_preview: prompt.slice(0, 80),
41
+ aspect_ratio,
42
+ image_size,
43
+ save_path: save_path ? 'provided' : 'none',
44
+ input_images_count: input_images?.length ?? 0,
45
+ });
33
46
  // Validate optional shape fields early so callers get a clear error
34
47
  // instead of a cryptic upstream 400.
35
48
  if (aspect_ratio !== undefined && !VALID_ASPECT_RATIOS.has(aspect_ratio)) {
@@ -126,6 +139,7 @@ export async function handleGenerateImage(request, openai) {
126
139
  { type: 'image', mimeType: base64.mime, data: base64.data },
127
140
  ],
128
141
  _meta: {
142
+ server_version: SERVER_VERSION,
129
143
  save_path: safePathResolved,
130
144
  mime: base64.mime,
131
145
  ...(usage
@@ -144,6 +158,7 @@ export async function handleGenerateImage(request, openai) {
144
158
  return {
145
159
  content: [{ type: 'image', mimeType: base64.mime, data: base64.data }],
146
160
  _meta: {
161
+ server_version: SERVER_VERSION,
147
162
  mime: base64.mime,
148
163
  ...(usage
149
164
  ? {
@@ -39,6 +39,7 @@ export declare function handleGenerateVideo(request: {
39
39
  }[];
40
40
  isError: false;
41
41
  _meta: {
42
+ server_version: string;
42
43
  code: "JOB_STILL_RUNNING";
43
44
  video_id: string;
44
45
  polling_url: string;
@@ -64,12 +65,54 @@ export declare function handleGetVideoStatus(request: {
64
65
  }[];
65
66
  isError: false;
66
67
  _meta: {
68
+ server_version: string;
67
69
  code: "JOB_STILL_RUNNING";
68
70
  video_id: string;
69
71
  last_status: string;
70
72
  progress: number | undefined;
71
73
  };
72
74
  }>;
75
+ /**
76
+ * Image-to-video convenience wrapper. Takes a single `image` argument
77
+ * (first frame) and delegates to `handleGenerateVideo` with the broader
78
+ * parameter surface hidden. Based on arxiv 2511.03497's finding that
79
+ * tool-calling success degrades with parameter count — a narrower tool
80
+ * gives the model a cleaner decision path.
81
+ */
82
+ export interface GenerateVideoFromImageRequest {
83
+ image: string;
84
+ prompt: string;
85
+ model?: string;
86
+ resolution?: string;
87
+ aspect_ratio?: string;
88
+ duration?: number;
89
+ seed?: number;
90
+ save_path?: string;
91
+ max_wait_ms?: number;
92
+ poll_interval_ms?: number;
93
+ }
94
+ export declare function handleGenerateVideoFromImage(request: {
95
+ params: {
96
+ arguments: GenerateVideoFromImageRequest;
97
+ };
98
+ }, apiClient: OpenRouterAPIClient, progress?: ProgressHook): Promise<import("../errors.js").ToolErrorResult | {
99
+ content: {
100
+ type: "text";
101
+ text: string;
102
+ }[];
103
+ isError: false;
104
+ _meta: {
105
+ server_version: string;
106
+ code: "JOB_STILL_RUNNING";
107
+ video_id: string;
108
+ polling_url: string;
109
+ last_status: string | undefined;
110
+ };
111
+ } | {
112
+ content: Record<string, unknown>[];
113
+ _meta: Record<string, unknown>;
114
+ isError?: undefined;
115
+ }>;
73
116
  export declare const _internals: {
74
117
  buildRequestBody: typeof buildRequestBody;
75
118
  stripAndReplaceExt: typeof stripAndReplaceExt;
@@ -1,6 +1,7 @@
1
1
  import { promises as fs } from 'node:fs';
2
2
  import { extname } from 'node:path';
3
3
  import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
4
+ import { SERVER_VERSION } from '../version.js';
4
5
  import { logger } from '../logger.js';
5
6
  import { resolveSafeOutputPath, resolveSafeInputPath, UnsafeOutputPathError, } from './path-safety.js';
6
7
  import { readEnvInt } from './fetch-utils.js';
@@ -177,6 +178,7 @@ async function finalizeCompletedJob(apiClient, status, savePath) {
177
178
  ? '.mpeg'
178
179
  : '.mp4';
179
180
  const baseMeta = {
181
+ server_version: SERVER_VERSION,
180
182
  video_id: status.id,
181
183
  mime,
182
184
  size_bytes: buffer.length,
@@ -232,6 +234,23 @@ export async function handleGenerateVideo(request, apiClient, progress) {
232
234
  if (!args.prompt || !args.prompt.trim()) {
233
235
  return toolError(ErrorCode.INVALID_INPUT, 'prompt is required.');
234
236
  }
237
+ const model = args.model ||
238
+ process.env.OPENROUTER_DEFAULT_VIDEO_GEN_MODEL ||
239
+ FALLBACK_MODEL;
240
+ // Audit entry — video is the most expensive tool we have. Always log
241
+ // model, resolution, duration, and a safe prompt preview so unintended
242
+ // spend can be traced.
243
+ logger.audit('generate_video.start', {
244
+ model,
245
+ prompt_preview: args.prompt.slice(0, 80),
246
+ resolution: args.resolution,
247
+ duration: args.duration,
248
+ aspect_ratio: args.aspect_ratio,
249
+ first_frame: args.first_frame_image ? 'provided' : 'none',
250
+ last_frame: args.last_frame_image ? 'provided' : 'none',
251
+ reference_images: args.reference_images?.length ?? 0,
252
+ save_path: args.save_path ? 'provided' : 'none',
253
+ });
235
254
  // Fail-fast on unsafe save_path BEFORE spending credits on the job.
236
255
  let safeSavePath = null;
237
256
  if (args.save_path) {
@@ -244,9 +263,6 @@ export async function handleGenerateVideo(request, apiClient, progress) {
244
263
  return toolErrorFrom(ErrorCode.INTERNAL, err);
245
264
  }
246
265
  }
247
- const model = args.model ||
248
- process.env.OPENROUTER_DEFAULT_VIDEO_GEN_MODEL ||
249
- FALLBACK_MODEL;
250
266
  const body = buildRequestBody(args, model);
251
267
  try {
252
268
  await attachFrameImages(args, body);
@@ -290,6 +306,7 @@ export async function handleGenerateVideo(request, apiClient, progress) {
290
306
  ],
291
307
  isError: false,
292
308
  _meta: {
309
+ server_version: SERVER_VERSION,
293
310
  code: ErrorCode.JOB_STILL_RUNNING,
294
311
  video_id: envelope.id,
295
312
  polling_url: envelope.polling_url ?? `https://openrouter.ai/api/v1/videos/${envelope.id}`,
@@ -355,6 +372,7 @@ export async function handleGetVideoStatus(request, apiClient) {
355
372
  ],
356
373
  isError: false,
357
374
  _meta: {
375
+ server_version: SERVER_VERSION,
358
376
  code: ErrorCode.JOB_STILL_RUNNING,
359
377
  video_id: id,
360
378
  last_status: status.status,
@@ -362,4 +380,29 @@ export async function handleGetVideoStatus(request, apiClient) {
362
380
  },
363
381
  };
364
382
  }
383
+ export async function handleGenerateVideoFromImage(request, apiClient, progress) {
384
+ const args = request.params.arguments ?? {};
385
+ if (!args.image) {
386
+ return toolError(ErrorCode.INVALID_INPUT, 'image is required.');
387
+ }
388
+ if (!args.prompt || !args.prompt.trim()) {
389
+ return toolError(ErrorCode.INVALID_INPUT, 'prompt is required.');
390
+ }
391
+ return handleGenerateVideo({
392
+ params: {
393
+ arguments: {
394
+ prompt: args.prompt,
395
+ first_frame_image: args.image,
396
+ model: args.model,
397
+ resolution: args.resolution,
398
+ aspect_ratio: args.aspect_ratio,
399
+ duration: args.duration,
400
+ seed: args.seed,
401
+ save_path: args.save_path,
402
+ max_wait_ms: args.max_wait_ms,
403
+ poll_interval_ms: args.poll_interval_ms,
404
+ },
405
+ },
406
+ }, apiClient, progress);
407
+ }
365
408
  export const _internals = { buildRequestBody, stripAndReplaceExt, extractJobError };
@@ -6,9 +6,4 @@ export declare function handleGetModelInfo(request: {
6
6
  model: string;
7
7
  };
8
8
  };
9
- }, modelCache: ModelCache, apiClient?: OpenRouterAPIClient): Promise<import("../errors.js").ToolErrorResult | {
10
- content: {
11
- type: "text";
12
- text: string;
13
- }[];
14
- }>;
9
+ }, modelCache: ModelCache, apiClient?: OpenRouterAPIClient): Promise<import("../errors.js").ToolErrorResult | import("./structured-output.js").StructuredResult<import("../model-cache.js").OpenRouterModelRecord>>;
@@ -1,5 +1,6 @@
1
1
  import { ErrorCode, toolError } from '../errors.js';
2
2
  import { classifyUpstreamError } from './openrouter-errors.js';
3
+ import { buildStructuredResult } from './structured-output.js';
3
4
  export async function handleGetModelInfo(request, modelCache, apiClient) {
4
5
  const { model } = request.params.arguments ?? { model: '' };
5
6
  if (!model || typeof model !== 'string') {
@@ -20,5 +21,5 @@ export async function handleGetModelInfo(request, modelCache, apiClient) {
20
21
  if (!info) {
21
22
  return toolError(ErrorCode.MODEL_NOT_FOUND, `Model '${model}' not found.`);
22
23
  }
23
- return { content: [{ type: 'text', text: JSON.stringify(info, null, 2) }] };
24
+ return buildStructuredResult(info);
24
25
  }
@@ -0,0 +1,23 @@
1
+ import type { OpenRouterAPIClient } from '../openrouter-api.js';
2
+ import { ModelCache } from '../model-cache.js';
3
+ /**
4
+ * Lightweight liveness probe. Runs the following checks:
5
+ * - Hit `/models` via the API client (indirectly validates API key +
6
+ * OpenRouter reachability)
7
+ * - Read cached model count
8
+ * - Report server + protocol versions
9
+ *
10
+ * Returns `{ ok, ... }` so ops can use it as a readiness signal.
11
+ */
12
+ export declare function handleHealthCheck(_request: {
13
+ params: {
14
+ arguments: Record<string, unknown>;
15
+ };
16
+ }, apiClient: OpenRouterAPIClient, modelCache: ModelCache): Promise<import("./structured-output.js").StructuredResult<{
17
+ error?: string | undefined;
18
+ ok: boolean;
19
+ server_version: string;
20
+ protocol_version: string;
21
+ api_key_valid: boolean;
22
+ models_cached: number;
23
+ }>>;
@@ -0,0 +1,35 @@
1
+ import { SERVER_VERSION, MCP_PROTOCOL_VERSION } from '../version.js';
2
+ import { buildStructuredResult } from './structured-output.js';
3
+ /**
4
+ * Lightweight liveness probe. Runs the following checks:
5
+ * - Hit `/models` via the API client (indirectly validates API key +
6
+ * OpenRouter reachability)
7
+ * - Read cached model count
8
+ * - Report server + protocol versions
9
+ *
10
+ * Returns `{ ok, ... }` so ops can use it as a readiness signal.
11
+ */
12
+ export async function handleHealthCheck(_request, apiClient, modelCache) {
13
+ let apiKeyValid = false;
14
+ let errorMessage;
15
+ try {
16
+ await modelCache.ensureFresh(() => apiClient.getModels());
17
+ apiKeyValid = true;
18
+ }
19
+ catch (err) {
20
+ errorMessage = err instanceof Error ? err.message : String(err);
21
+ }
22
+ const modelsCached = modelCache.isValid() ? modelCache.size() : 0;
23
+ // `ok` means the API was reachable and the key was accepted. An empty
24
+ // catalog counts as success (the API just returned no models) — callers
25
+ // branch on `models_cached` if they care about the count.
26
+ const ok = apiKeyValid;
27
+ return buildStructuredResult({
28
+ ok,
29
+ server_version: SERVER_VERSION,
30
+ protocol_version: MCP_PROTOCOL_VERSION,
31
+ api_key_valid: apiKeyValid,
32
+ models_cached: modelsCached,
33
+ ...(errorMessage ? { error: errorMessage } : {}),
34
+ });
35
+ }
@@ -0,0 +1,16 @@
1
+ /**
2
+ * Helper for calling `openai.chat.completions.create()` and getting back
3
+ * BOTH the typed body and the raw fetch `Response` (so we can read the
4
+ * X-OpenRouter-Cache-* headers).
5
+ *
6
+ * The real openai SDK returns an `APIPromise` that exposes `.withResponse()`.
7
+ * Vitest tests typically stub `create()` to return a plain `ChatCompletion`
8
+ * object. This helper handles both cases so tests don't need to mock the
9
+ * chainable.
10
+ */
11
+ import type { ChatCompletion } from 'openai/resources/chat/completions.js';
12
+ export interface ChatCompletionWithHeaders {
13
+ data: ChatCompletion;
14
+ response: Response | undefined;
15
+ }
16
+ export declare function awaitCompletionWithHeaders(call: unknown): Promise<ChatCompletionWithHeaders>;
@@ -0,0 +1,16 @@
1
+ export async function awaitCompletionWithHeaders(call) {
2
+ // Prefer the APIPromise .withResponse() chainable, which returns
3
+ // `{ data, response }` — but only if the object looks like an
4
+ // APIPromise. Mocks that return plain objects get unwrapped via a
5
+ // direct await.
6
+ const maybeChainable = call;
7
+ if (typeof maybeChainable?.withResponse === 'function') {
8
+ const { data, response } = await maybeChainable.withResponse();
9
+ return { data, response };
10
+ }
11
+ // Fallback: await the value directly. Covers both test mocks
12
+ // (which return a plain ChatCompletion via vi.fn().mockResolvedValue())
13
+ // and any exotic SDK shape we don't recognize.
14
+ const data = (await call);
15
+ return { data, response: undefined };
16
+ }
@@ -14,5 +14,9 @@ import { type ToolErrorResult } from '../errors.js';
14
14
  * 2. Message heuristics for common OpenRouter strings (credits, ZDR,
15
15
  * "model does not exist", content policy, etc.).
16
16
  * 3. Default to INTERNAL to avoid leaking raw shapes.
17
+ *
18
+ * When the error carries a `Retry-After` header (on 429 / 503) we populate
19
+ * `_meta.retry_after_seconds` so agents can back off intelligently. We
20
+ * also attach canonical `suggestions[]` for common cases.
17
21
  */
18
- export declare function classifyUpstreamError(err: unknown, _contextMessage?: string): ToolErrorResult;
22
+ export declare function classifyUpstreamError(err: unknown, contextMessage?: string): ToolErrorResult;
@@ -5,6 +5,29 @@
5
5
  * don't drift.
6
6
  */
7
7
  import { ErrorCode, toolError } from '../errors.js';
8
+ function extractRetryAfterSeconds(err) {
9
+ if (typeof err !== 'object' || err === null)
10
+ return undefined;
11
+ const e = err;
12
+ const getHeader = (h) => {
13
+ if (!h)
14
+ return null;
15
+ if (typeof h === 'object' && typeof h.get === 'function') {
16
+ return h.get('retry-after') ?? null;
17
+ }
18
+ const rec = h;
19
+ return rec['retry-after'] ?? rec['Retry-After'] ?? null;
20
+ };
21
+ const raw = getHeader(e.headers) ?? getHeader(e.response?.headers);
22
+ if (!raw)
23
+ return undefined;
24
+ const n = Number(raw);
25
+ if (Number.isFinite(n) && n >= 0)
26
+ return n;
27
+ // Retry-After can also be an HTTP-date; return undefined for those (caller
28
+ // can still retry on its own backoff schedule).
29
+ return undefined;
30
+ }
8
31
  function extractStatus(err) {
9
32
  if (typeof err !== 'object' || err === null)
10
33
  return undefined;
@@ -49,43 +72,78 @@ function extractMessage(err) {
49
72
  * 2. Message heuristics for common OpenRouter strings (credits, ZDR,
50
73
  * "model does not exist", content policy, etc.).
51
74
  * 3. Default to INTERNAL to avoid leaking raw shapes.
75
+ *
76
+ * When the error carries a `Retry-After` header (on 429 / 503) we populate
77
+ * `_meta.retry_after_seconds` so agents can back off intelligently. We
78
+ * also attach canonical `suggestions[]` for common cases.
52
79
  */
53
- export function classifyUpstreamError(err, _contextMessage) {
54
- const msg = extractMessage(err);
80
+ export function classifyUpstreamError(err, contextMessage) {
81
+ const rawMsg = extractMessage(err);
55
82
  const status = extractStatus(err);
56
- const lower = msg.toLowerCase();
57
- const fullMsg = msg;
83
+ const lower = rawMsg.toLowerCase();
84
+ // Prefix every user-visible message with the handler context when the
85
+ // caller supplied one (e.g. `rerank`, `generate_video.submit`). Makes
86
+ // server-side triage possible without digging through logs.
87
+ const fullMsg = contextMessage ? `${contextMessage}: ${rawMsg}` : rawMsg;
88
+ const retryAfterSeconds = extractRetryAfterSeconds(err);
58
89
  // Explicit credit / balance signals.
59
90
  if (lower.includes('insufficient balance') ||
60
91
  lower.includes('insufficient credits') ||
61
92
  lower.includes('requires more credits') ||
62
93
  lower.includes('requires at least') ||
63
94
  status === 402) {
64
- return toolError(ErrorCode.UPSTREAM_REFUSED, fullMsg, { status, reason: 'credits' });
95
+ return toolError(ErrorCode.UPSTREAM_REFUSED, fullMsg, { status, reason: 'credits' }, {
96
+ suggestions: [
97
+ 'Top up credits at https://openrouter.ai/settings/credits',
98
+ 'Switch to a free-tier model (append :free to the slug)',
99
+ ],
100
+ });
65
101
  }
66
102
  // Zero Data Retention.
67
103
  if (lower.includes('zdr') || lower.includes('zero data retention')) {
68
- return toolError(ErrorCode.ZDR_INCOMPATIBLE, fullMsg, { status });
104
+ return toolError(ErrorCode.ZDR_INCOMPATIBLE, fullMsg, { status }, {
105
+ suggestions: [
106
+ 'Pick a provider that supports your ZDR policy',
107
+ 'Set provider.data_collection: "allow" to bypass the restriction',
108
+ ],
109
+ });
69
110
  }
70
111
  // Model lookup failures.
71
112
  if (lower.includes('model') &&
72
113
  (lower.includes('does not exist') || lower.includes('not found') || lower.includes('invalid model'))) {
73
- return toolError(ErrorCode.MODEL_NOT_FOUND, fullMsg, { status });
114
+ return toolError(ErrorCode.MODEL_NOT_FOUND, fullMsg, { status }, {
115
+ suggestions: [
116
+ 'Use search_models to discover valid model ids',
117
+ 'Use validate_model to pre-flight a model id',
118
+ ],
119
+ });
74
120
  }
75
121
  // Content policy / moderation — surface as UPSTREAM_REFUSED so callers can distinguish from 5xx.
76
122
  if (lower.includes('content policy') || lower.includes('moderation') || lower.includes('refused')) {
77
- return toolError(ErrorCode.UPSTREAM_REFUSED, fullMsg, { status, reason: 'policy' });
123
+ return toolError(ErrorCode.UPSTREAM_REFUSED, fullMsg, { status, reason: 'policy' }, {
124
+ suggestions: ['Rephrase the prompt', 'Try a different provider via provider.order'],
125
+ });
78
126
  }
79
127
  // Rate-limit specific.
80
128
  if (status === 429 || lower.includes('rate limit')) {
81
- return toolError(ErrorCode.UPSTREAM_REFUSED, fullMsg, { status, reason: 'rate_limit' });
129
+ return toolError(ErrorCode.UPSTREAM_REFUSED, fullMsg, { status, reason: 'rate_limit' }, {
130
+ suggestions: [
131
+ retryAfterSeconds !== undefined
132
+ ? `Wait ${retryAfterSeconds}s and retry`
133
+ : 'Wait and retry with exponential backoff',
134
+ 'Append :nitro to the model slug to route to a faster provider',
135
+ ],
136
+ retry_after_seconds: retryAfterSeconds,
137
+ });
82
138
  }
83
139
  // Timeouts (AbortError from `AbortSignal.timeout`).
84
140
  if (lower.includes('timed out') ||
85
141
  lower.includes('timeout') ||
86
142
  lower.includes('aborted') ||
87
143
  (err instanceof Error && err.name === 'AbortError')) {
88
- return toolError(ErrorCode.UPSTREAM_TIMEOUT, fullMsg, { status });
144
+ return toolError(ErrorCode.UPSTREAM_TIMEOUT, fullMsg, { status }, {
145
+ suggestions: ['Retry', 'Raise max_wait_ms or max_tokens'],
146
+ });
89
147
  }
90
148
  // Anything in the 4xx band that isn't covered above — user supplied a bad request.
91
149
  if (typeof status === 'number' && status >= 400 && status < 500) {
@@ -93,7 +151,10 @@ export function classifyUpstreamError(err, _contextMessage) {
93
151
  }
94
152
  // 5xx / network errors.
95
153
  if (typeof status === 'number' && status >= 500) {
96
- return toolError(ErrorCode.UPSTREAM_HTTP, fullMsg, { status });
154
+ return toolError(ErrorCode.UPSTREAM_HTTP, fullMsg, { status }, {
155
+ suggestions: ['Retry after a brief delay', 'Check https://status.openrouter.ai'],
156
+ retry_after_seconds: retryAfterSeconds,
157
+ });
97
158
  }
98
159
  return toolError(ErrorCode.UPSTREAM_HTTP, fullMsg);
99
160
  }
@@ -0,0 +1,17 @@
1
+ import type { OpenRouterAPIClient } from '../openrouter-api.js';
2
+ export interface RerankDocumentsRequest {
3
+ query: string;
4
+ documents: string[];
5
+ model?: string;
6
+ top_n?: number;
7
+ /** When true, include the original document text in each result. */
8
+ return_documents?: boolean;
9
+ }
10
+ export declare function handleRerankDocuments(request: {
11
+ params: {
12
+ arguments: RerankDocumentsRequest;
13
+ };
14
+ }, apiClient: OpenRouterAPIClient): Promise<import("../errors.js").ToolErrorResult | import("./structured-output.js").StructuredResult<{
15
+ model: string;
16
+ results: Record<string, unknown>[];
17
+ }>>;
@@ -0,0 +1,52 @@
1
+ import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
2
+ import { classifyUpstreamError } from './openrouter-errors.js';
3
+ import { buildStructuredResult } from './structured-output.js';
4
+ const DEFAULT_MODEL = 'cohere/rerank-english-v3.0';
5
+ export async function handleRerankDocuments(request, apiClient) {
6
+ const args = request.params.arguments ??
7
+ { query: '', documents: [] };
8
+ const { query, documents, model, top_n, return_documents } = args;
9
+ if (!query?.trim()) {
10
+ return toolError(ErrorCode.INVALID_INPUT, 'query is required.');
11
+ }
12
+ if (!Array.isArray(documents) || documents.length === 0) {
13
+ return toolError(ErrorCode.INVALID_INPUT, 'documents must be a non-empty array of strings.');
14
+ }
15
+ if (documents.some((d) => typeof d !== 'string')) {
16
+ return toolError(ErrorCode.INVALID_INPUT, 'every document must be a string.');
17
+ }
18
+ let response;
19
+ try {
20
+ response = await apiClient.rerank({
21
+ model: model || DEFAULT_MODEL,
22
+ query,
23
+ documents,
24
+ top_n,
25
+ });
26
+ }
27
+ catch (err) {
28
+ return classifyUpstreamError(err, 'rerank');
29
+ }
30
+ // Normalize to a stable shape: always expose `score` (OpenRouter
31
+ // providers sometimes return `relevance_score`, sometimes `score`).
32
+ const normalized = (response.results ?? []).map((r) => {
33
+ const score = typeof r.score === 'number' ? r.score : r.relevance_score;
34
+ const out = { index: r.index, score };
35
+ if (return_documents) {
36
+ const doc = typeof r.document === 'string'
37
+ ? r.document
38
+ : r.document?.text ?? documents[r.index];
39
+ out.document = doc;
40
+ }
41
+ return out;
42
+ });
43
+ try {
44
+ return buildStructuredResult({
45
+ model: response.model ?? model ?? DEFAULT_MODEL,
46
+ results: normalized,
47
+ }, response.usage ? { usage: response.usage } : {});
48
+ }
49
+ catch (err) {
50
+ return toolErrorFrom(ErrorCode.INTERNAL, err, 'rerank');
51
+ }
52
+ }
@@ -1,20 +1,31 @@
1
- import { ModelCache } from '../model-cache.js';
1
+ import { ModelCache, type OpenRouterModelRecord } from '../model-cache.js';
2
2
  import { OpenRouterAPIClient } from '../openrouter-api.js';
3
3
  export interface SearchModelsArgs {
4
4
  query?: string;
5
5
  provider?: string;
6
6
  capabilities?: {
7
7
  vision?: boolean;
8
+ audio?: boolean;
9
+ video?: boolean;
8
10
  };
9
11
  limit?: number;
12
+ /**
13
+ * Skip this many matching results before returning `limit`. Paired with
14
+ * `limit` and the returned `next_offset` to let large model lists be
15
+ * paged safely. Follows Phil Schmid's "paginate large results" best
16
+ * practice.
17
+ */
18
+ offset?: number;
10
19
  }
11
20
  export declare function handleSearchModels(request: {
12
21
  params: {
13
22
  arguments: SearchModelsArgs;
14
23
  };
15
- }, apiClient: OpenRouterAPIClient, modelCache: ModelCache): Promise<import("../errors.js").ToolErrorResult | {
16
- content: {
17
- type: "text";
18
- text: string;
19
- }[];
20
- }>;
24
+ }, apiClient: OpenRouterAPIClient, modelCache: ModelCache): Promise<import("../errors.js").ToolErrorResult | import("./structured-output.js").StructuredResult<{
25
+ results: OpenRouterModelRecord[];
26
+ offset: number;
27
+ limit: number;
28
+ total: number;
29
+ has_more: boolean;
30
+ next_offset: number | null;
31
+ }>>;
@@ -1,5 +1,8 @@
1
1
  import { ErrorCode, toolErrorFrom } from '../errors.js';
2
2
  import { classifyUpstreamError } from './openrouter-errors.js';
3
+ import { buildStructuredResult } from './structured-output.js';
4
+ const DEFAULT_LIMIT = 20;
5
+ const MAX_LIMIT = 50;
3
6
  export async function handleSearchModels(request, apiClient, modelCache) {
4
7
  try {
5
8
  await modelCache.ensureFresh(() => apiClient.getModels());
@@ -8,8 +11,28 @@ export async function handleSearchModels(request, apiClient, modelCache) {
8
11
  return classifyUpstreamError(error, 'search_models');
9
12
  }
10
13
  try {
11
- const results = modelCache.search(request.params.arguments ?? {});
12
- return { content: [{ type: 'text', text: JSON.stringify(results, null, 2) }] };
14
+ const args = request.params.arguments ?? {};
15
+ const limit = Math.min(Math.max(1, args.limit ?? DEFAULT_LIMIT), MAX_LIMIT);
16
+ const offset = Math.max(0, args.offset ?? 0);
17
+ // Get the full filtered set, then slice for pagination.
18
+ const all = modelCache.search({
19
+ query: args.query,
20
+ provider: args.provider,
21
+ capabilities: args.capabilities,
22
+ all: true,
23
+ });
24
+ const total = all.length;
25
+ const page = all.slice(offset, offset + limit);
26
+ const nextOffset = offset + limit;
27
+ const hasMore = nextOffset < total;
28
+ return buildStructuredResult({
29
+ results: page,
30
+ offset,
31
+ limit,
32
+ total,
33
+ has_more: hasMore,
34
+ next_offset: hasMore ? nextOffset : null,
35
+ });
13
36
  }
14
37
  catch (error) {
15
38
  return toolErrorFrom(ErrorCode.INTERNAL, error, 'search_models');
@@ -0,0 +1,13 @@
1
+ export interface StructuredResult<T = unknown> {
2
+ content: Array<{
3
+ type: 'text';
4
+ text: string;
5
+ }>;
6
+ structuredContent: T;
7
+ _meta: Record<string, unknown>;
8
+ }
9
+ /**
10
+ * Wrap a JSON-serializable object in the MCP-spec dual-representation
11
+ * format. `meta` is merged on top of the default `server_version` stamp.
12
+ */
13
+ export declare function buildStructuredResult<T>(data: T, meta?: Record<string, unknown>): StructuredResult<T>;