@prismshadow/mmsp 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. package/README.md +121 -0
  2. package/dist/ant_messages/client.d.ts +59 -0
  3. package/dist/ant_messages/client.d.ts.map +1 -0
  4. package/dist/ant_messages/client.js +419 -0
  5. package/dist/ant_messages/index.d.ts +2 -0
  6. package/dist/ant_messages/index.d.ts.map +1 -0
  7. package/dist/ant_messages/index.js +18 -0
  8. package/dist/anthropic_official/client.d.ts +63 -0
  9. package/dist/anthropic_official/client.d.ts.map +1 -0
  10. package/dist/anthropic_official/client.js +540 -0
  11. package/dist/anthropic_official/index.d.ts +2 -0
  12. package/dist/anthropic_official/index.d.ts.map +1 -0
  13. package/dist/anthropic_official/index.js +18 -0
  14. package/dist/autoClient.d.ts +96 -0
  15. package/dist/autoClient.d.ts.map +1 -0
  16. package/dist/autoClient.js +258 -0
  17. package/dist/baseClient.d.ts +111 -0
  18. package/dist/baseClient.d.ts.map +1 -0
  19. package/dist/baseClient.js +284 -0
  20. package/dist/deepseek_official/client.d.ts +60 -0
  21. package/dist/deepseek_official/client.d.ts.map +1 -0
  22. package/dist/deepseek_official/client.js +407 -0
  23. package/dist/deepseek_official/index.d.ts +2 -0
  24. package/dist/deepseek_official/index.d.ts.map +1 -0
  25. package/dist/deepseek_official/index.js +18 -0
  26. package/dist/errors.d.ts +84 -0
  27. package/dist/errors.d.ts.map +1 -0
  28. package/dist/errors.js +140 -0
  29. package/dist/gemini_generate_content/client.d.ts +86 -0
  30. package/dist/gemini_generate_content/client.d.ts.map +1 -0
  31. package/dist/gemini_generate_content/client.js +809 -0
  32. package/dist/gemini_generate_content/index.d.ts +2 -0
  33. package/dist/gemini_generate_content/index.d.ts.map +1 -0
  34. package/dist/gemini_generate_content/index.js +18 -0
  35. package/dist/gemini_official/client.d.ts +87 -0
  36. package/dist/gemini_official/client.d.ts.map +1 -0
  37. package/dist/gemini_official/client.js +780 -0
  38. package/dist/gemini_official/index.d.ts +2 -0
  39. package/dist/gemini_official/index.d.ts.map +1 -0
  40. package/dist/gemini_official/index.js +18 -0
  41. package/dist/index.d.ts +6 -0
  42. package/dist/index.d.ts.map +1 -0
  43. package/dist/index.js +44 -0
  44. package/dist/integration/index.d.ts +1 -0
  45. package/dist/integration/index.d.ts.map +1 -0
  46. package/dist/integration/index.js +14 -0
  47. package/dist/integration/playground.d.ts +17 -0
  48. package/dist/integration/playground.d.ts.map +1 -0
  49. package/dist/integration/playground.js +3246 -0
  50. package/dist/integration/tracer.d.ts +188 -0
  51. package/dist/integration/tracer.d.ts.map +1 -0
  52. package/dist/integration/tracer.js +1928 -0
  53. package/dist/legacy.d.ts +13 -0
  54. package/dist/legacy.d.ts.map +1 -0
  55. package/dist/legacy.js +59 -0
  56. package/dist/minimax_official/client.d.ts +43 -0
  57. package/dist/minimax_official/client.d.ts.map +1 -0
  58. package/dist/minimax_official/client.js +346 -0
  59. package/dist/minimax_official/index.d.ts +2 -0
  60. package/dist/minimax_official/index.d.ts.map +1 -0
  61. package/dist/minimax_official/index.js +18 -0
  62. package/dist/moonshot_official/client.d.ts +73 -0
  63. package/dist/moonshot_official/client.d.ts.map +1 -0
  64. package/dist/moonshot_official/client.js +477 -0
  65. package/dist/moonshot_official/index.d.ts +2 -0
  66. package/dist/moonshot_official/index.d.ts.map +1 -0
  67. package/dist/moonshot_official/index.js +18 -0
  68. package/dist/openai_chat/client.d.ts +67 -0
  69. package/dist/openai_chat/client.d.ts.map +1 -0
  70. package/dist/openai_chat/client.js +419 -0
  71. package/dist/openai_chat/index.d.ts +2 -0
  72. package/dist/openai_chat/index.d.ts.map +1 -0
  73. package/dist/openai_chat/index.js +18 -0
  74. package/dist/openai_chat_vllm_adapter/client.d.ts +15 -0
  75. package/dist/openai_chat_vllm_adapter/client.d.ts.map +1 -0
  76. package/dist/openai_chat_vllm_adapter/client.js +118 -0
  77. package/dist/openai_chat_vllm_adapter/index.d.ts +2 -0
  78. package/dist/openai_chat_vllm_adapter/index.d.ts.map +1 -0
  79. package/dist/openai_chat_vllm_adapter/index.js +5 -0
  80. package/dist/openai_embedding/client.d.ts +46 -0
  81. package/dist/openai_embedding/client.d.ts.map +1 -0
  82. package/dist/openai_embedding/client.js +124 -0
  83. package/dist/openai_embedding/index.d.ts +2 -0
  84. package/dist/openai_embedding/index.d.ts.map +1 -0
  85. package/dist/openai_embedding/index.js +18 -0
  86. package/dist/openai_official/client.d.ts +61 -0
  87. package/dist/openai_official/client.d.ts.map +1 -0
  88. package/dist/openai_official/client.js +464 -0
  89. package/dist/openai_official/index.d.ts +2 -0
  90. package/dist/openai_official/index.d.ts.map +1 -0
  91. package/dist/openai_official/index.js +18 -0
  92. package/dist/openai_responses/client.d.ts +61 -0
  93. package/dist/openai_responses/client.d.ts.map +1 -0
  94. package/dist/openai_responses/client.js +449 -0
  95. package/dist/openai_responses/index.d.ts +2 -0
  96. package/dist/openai_responses/index.d.ts.map +1 -0
  97. package/dist/openai_responses/index.js +18 -0
  98. package/dist/registry.d.ts +49 -0
  99. package/dist/registry.d.ts.map +1 -0
  100. package/dist/registry.js +798 -0
  101. package/dist/streamItems.d.ts +36 -0
  102. package/dist/streamItems.d.ts.map +1 -0
  103. package/dist/streamItems.js +183 -0
  104. package/dist/types.d.ts +190 -0
  105. package/dist/types.d.ts.map +1 -0
  106. package/dist/types.js +37 -0
  107. package/dist/utils.d.ts +99 -0
  108. package/dist/utils.d.ts.map +1 -0
  109. package/dist/utils.js +312 -0
  110. package/dist/zai_official/client.d.ts +78 -0
  111. package/dist/zai_official/client.d.ts.map +1 -0
  112. package/dist/zai_official/client.js +420 -0
  113. package/dist/zai_official/index.d.ts +2 -0
  114. package/dist/zai_official/index.d.ts.map +1 -0
  115. package/dist/zai_official/index.js +18 -0
  116. package/package.json +68 -0
@@ -0,0 +1,419 @@
1
+ "use strict";
2
+ // Copyright 2025 Prism Shadow. and/or its affiliates
3
+ //
4
+ // Licensed under the Apache License, Version 2.0 (the "License");
5
+ // you may not use this file except in compliance with the License.
6
+ // You may obtain a copy of the License at
7
+ //
8
+ // http://www.apache.org/licenses/LICENSE-2.0
9
+ //
10
+ // Unless required by applicable law or agreed to in writing, software
11
+ // distributed under the License is distributed on an "AS IS" BASIS,
12
+ // WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13
+ // See the License for the specific language governing permissions and
14
+ // limitations under the License.
15
+ var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
16
+ if (k2 === undefined) k2 = k;
17
+ var desc = Object.getOwnPropertyDescriptor(m, k);
18
+ if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
19
+ desc = { enumerable: true, get: function() { return m[k]; } };
20
+ }
21
+ Object.defineProperty(o, k2, desc);
22
+ }) : (function(o, m, k, k2) {
23
+ if (k2 === undefined) k2 = k;
24
+ o[k2] = m[k];
25
+ }));
26
+ var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
27
+ Object.defineProperty(o, "default", { enumerable: true, value: v });
28
+ }) : function(o, v) {
29
+ o["default"] = v;
30
+ });
31
+ var __importStar = (this && this.__importStar) || (function () {
32
+ var ownKeys = function(o) {
33
+ ownKeys = Object.getOwnPropertyNames || function (o) {
34
+ var ar = [];
35
+ for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
36
+ return ar;
37
+ };
38
+ return ownKeys(o);
39
+ };
40
+ return function (mod) {
41
+ if (mod && mod.__esModule) return mod;
42
+ var result = {};
43
+ if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
44
+ __setModuleDefault(result, mod);
45
+ return result;
46
+ };
47
+ })();
48
+ var __importDefault = (this && this.__importDefault) || function (mod) {
49
+ return (mod && mod.__esModule) ? mod : { "default": mod };
50
+ };
51
+ Object.defineProperty(exports, "__esModule", { value: true });
52
+ exports.OpenaiChatClient = void 0;
53
+ const openai_1 = __importDefault(require("openai"));
54
+ const path = __importStar(require("path"));
55
+ const baseClient_1 = require("../baseClient");
56
+ const errors_1 = require("../errors");
57
+ const types_1 = require("../types");
58
+ const utils_1 = require("../utils");
59
+ /**
60
+ * OpenAI Chat Completions-compatible client implementation.
61
+ */
62
+ class OpenaiChatClient extends baseClient_1.LLMClient {
63
+ /**
64
+ * Initialize OpenAI-compatible chat client with model, API key, and base URL.
65
+ */
66
+ constructor(options) {
67
+ super();
68
+ this._model = options.model;
69
+ const { apiKey: key, baseUrl: url } = (0, utils_1.resolveCredentials)(this.constructor.name, options, { key: "OPENAI_API_KEY", baseUrl: "OPENAI_BASE_URL" });
70
+ this._client = new openai_1.default({
71
+ apiKey: key,
72
+ baseURL: url,
73
+ defaultHeaders: options.defaultHeaders,
74
+ });
75
+ }
76
+ /**
77
+ * Detect MIME type from URL extension for image.
78
+ */
79
+ _detectImageMimeType(url) {
80
+ const ext = path.extname(url).toLowerCase();
81
+ const mimeTypes = {
82
+ ".bmp": "image/bmp",
83
+ ".gif": "image/gif",
84
+ ".jpg": "image/jpeg",
85
+ ".jpeg": "image/jpeg",
86
+ ".png": "image/png",
87
+ ".svg": "image/svg+xml",
88
+ ".tiff": "image/tiff",
89
+ ".webp": "image/webp",
90
+ };
91
+ return mimeTypes[ext] || "image/jpeg";
92
+ }
93
+ /**
94
+ * Convert image URL to base64-encoded data URL.
95
+ */
96
+ async _convertImageUrlToBase64(url, signal) {
97
+ if (url.startsWith("data:")) {
98
+ return url;
99
+ }
100
+ const response = await fetch(url, { signal });
101
+ if (!response.ok) {
102
+ throw new Error(`Failed to fetch image: ${response.status} ${response.statusText}`);
103
+ }
104
+ const arrayBuffer = await response.arrayBuffer();
105
+ const buffer = Buffer.from(arrayBuffer);
106
+ const mimeType = this._detectImageMimeType(url);
107
+ const base64String = buffer.toString("base64");
108
+ return `data:${mimeType};base64,${base64String}`;
109
+ }
110
+ /**
111
+ * Convert ToolChoice to OpenAI Chat Completions tool_choice format.
112
+ */
113
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
114
+ _convertToolChoice(toolChoice) {
115
+ if (Array.isArray(toolChoice)) {
116
+ return {
117
+ type: "allowed_tools",
118
+ allowed_tools: {
119
+ mode: "auto",
120
+ tools: toolChoice.map((name) => ({
121
+ type: "function",
122
+ function: { name },
123
+ })),
124
+ },
125
+ };
126
+ }
127
+ return toolChoice;
128
+ }
129
+ /**
130
+ * Convert a fetched image to an image_url part, at the detail the API needs
131
+ * to read it.
132
+ */
133
+ _convertImageUrl(dataUrl) {
134
+ const detail = (0, utils_1.openaiImageDetail)(this._model, dataUrl);
135
+ return detail
136
+ ? { type: "image_url", image_url: { url: dataUrl, detail } }
137
+ : { type: "image_url", image_url: { url: dataUrl } };
138
+ }
139
+ /**
140
+ * Transform universal configuration to OpenAI Chat Completions configuration.
141
+ */
142
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
143
+ transformUniConfigToModelConfig(config) {
144
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
145
+ const openaiConfig = {
146
+ model: this._model,
147
+ stream: true,
148
+ stream_options: { include_usage: true },
149
+ };
150
+ if (config.max_tokens !== undefined) {
151
+ openaiConfig.max_completion_tokens = config.max_tokens;
152
+ }
153
+ if (config.temperature !== undefined) {
154
+ openaiConfig.temperature = config.temperature;
155
+ }
156
+ if (config.tools !== undefined) {
157
+ openaiConfig.tools = config.tools.map((tool) => ({
158
+ type: "function",
159
+ function: tool,
160
+ }));
161
+ }
162
+ if (config.tool_choice !== undefined) {
163
+ openaiConfig.tool_choice = this._convertToolChoice(config.tool_choice);
164
+ }
165
+ if (config.fast_mode) {
166
+ openaiConfig.service_tier = "priority";
167
+ }
168
+ if (config.prompt_caching !== undefined &&
169
+ config.prompt_caching !== types_1.PromptCaching.ENABLE) {
170
+ throw new errors_1.UnsupportedParameterError({
171
+ client: this.constructor.name,
172
+ parameter: "prompt_caching",
173
+ message: "prompt_caching must be ENABLE for OpenAI.",
174
+ });
175
+ }
176
+ return openaiConfig;
177
+ }
178
+ /**
179
+ * Transform universal message format to OpenAI Chat Completions message format.
180
+ */
181
+ async transformUniMessageToModelInput(messages, signal) {
182
+ const openaiMessages = [];
183
+ // The reasoning field this upstream has produced so far, if any. A turn the model answered
184
+ // without thinking still has to carry that field, empty, when it made tool calls: DeepSeek
185
+ // stops thinking part-way through a long tool chain, and then rejects the replay of that
186
+ // turn with "the reasoning_content in the thinking mode must be passed back to the API".
187
+ // It waives that only for tool_call ids it issued itself, which a relay that reissues ids
188
+ // takes away. A conversation that never produced a reasoning field never receives one.
189
+ const replayFields = new Set();
190
+ for (const msg of messages) {
191
+ const contentParts = [];
192
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
193
+ const toolCalls = [];
194
+ let thinking = "";
195
+ const thinkingFields = new Set();
196
+ for (const item of msg.content_items) {
197
+ if (item.type === "text.done") {
198
+ contentParts.push({ type: "text", text: item.text });
199
+ }
200
+ else if (item.type === "image_url.done") {
201
+ const base64Image = await this._convertImageUrlToBase64(item.image_url, signal);
202
+ contentParts.push(this._convertImageUrl(base64Image));
203
+ }
204
+ else if (item.type === "thinking.done") {
205
+ thinking += item.thinking;
206
+ thinkingFields.add(item.fidelity?.reasoning_field);
207
+ if (item.thinking)
208
+ replayFields.add(item.fidelity?.reasoning_field);
209
+ }
210
+ else if (item.type === "tool_call.done") {
211
+ toolCalls.push({
212
+ id: item.tool_call_id,
213
+ type: "function",
214
+ function: {
215
+ name: item.name,
216
+ arguments: JSON.stringify(item.arguments, null, 0),
217
+ },
218
+ });
219
+ }
220
+ else if (item.type === "tool_result.done") {
221
+ if (!item.tool_call_id) {
222
+ throw new Error("tool_call_id is required for tool result.");
223
+ }
224
+ // Chat Completions lets a tool message carry text only, and a server that
225
+ // validates the schema rejects the whole request over an image part in one.
226
+ // The images ride in the user message that follows the turn's tool messages,
227
+ // the one place every OpenAI-compatible server reads them.
228
+ if (item.images && item.images.length > 0) {
229
+ for (const imageUrl of item.images) {
230
+ const base64Image = await this._convertImageUrlToBase64(imageUrl, signal);
231
+ contentParts.push(this._convertImageUrl(base64Image));
232
+ }
233
+ }
234
+ // the plain string is the form every OpenAI-compatible server accepts
235
+ openaiMessages.push({
236
+ role: "tool",
237
+ tool_call_id: item.tool_call_id,
238
+ content: item.text,
239
+ });
240
+ }
241
+ else {
242
+ throw new Error(`Unknown item type: ${item.type}`);
243
+ }
244
+ }
245
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
246
+ const message = { role: msg.role };
247
+ if (contentParts.length > 0) {
248
+ message.content = contentParts;
249
+ }
250
+ if (toolCalls.length > 0) {
251
+ message.tool_calls = toolCalls;
252
+ }
253
+ // This turn's own fidelity when it thought; otherwise the field the upstream has
254
+ // produced, with `thinking` still "" — the empty value is what keeps the turn replayable.
255
+ const fields = thinking ? thinkingFields : replayFields;
256
+ if (thinking || (toolCalls.length > 0 && replayFields.size > 0)) {
257
+ // send thinking back through the exact field the upstream produced (recorded
258
+ // in the item fidelity); servers may reject the spelling they did not emit
259
+ if (fields.size === 1 && fields.has("reasoning_content")) {
260
+ message.reasoning_content = thinking;
261
+ }
262
+ else if (fields.size === 1 && fields.has("reasoning")) {
263
+ message.reasoning = thinking;
264
+ }
265
+ else {
266
+ message.reasoning_content = thinking; // vLLM & siliconflow compatibility
267
+ message.reasoning = thinking; // openrouter compatibility
268
+ }
269
+ }
270
+ if (Object.keys(message).length > 1) {
271
+ openaiMessages.push(message);
272
+ }
273
+ }
274
+ return openaiMessages;
275
+ }
276
+ /**
277
+ * Transform one OpenAI Chat Completions streaming chunk into a universal event.
278
+ *
279
+ * Chat Completions gives an item no identity, so each delta's item_id is the wire field
280
+ * that carried it: an item runs until a delta arrives from another field, or names the next
281
+ * tool call.
282
+ */
283
+ transformModelOutputToUniEvent(modelOutput) {
284
+ let eventType = "delta";
285
+ const contentItems = [];
286
+ let usageMetadata = null;
287
+ let finishReason = null;
288
+ // gateways inject content-free heartbeat chunks on long generations, whose
289
+ // choices arrive as undefined rather than an empty list
290
+ if (modelOutput.choices?.length) {
291
+ const choice = modelOutput.choices[0];
292
+ const delta = choice?.delta;
293
+ // the thinking field name differs by server: vLLM & siliconflow use
294
+ // reasoning_content while openrouter uses reasoning; record the wire
295
+ // field that carried each delta so a replay can reproduce exactly the
296
+ // field the upstream produced. The reasoning goes before the content
297
+ // because a chunk may end the reasoning and begin the answer.
298
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
299
+ const reasoningContent = delta?.reasoning_content;
300
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
301
+ const reasoning = delta?.reasoning;
302
+ if (reasoningContent && reasoning) {
303
+ // ambiguous origin: record no reasoning_field so a replay sends both fields back
304
+ contentItems.push({
305
+ type: "thinking.delta",
306
+ thinking: reasoningContent,
307
+ fidelity: { item_id: "reasoning_content" },
308
+ });
309
+ }
310
+ else if (reasoningContent) {
311
+ contentItems.push({
312
+ type: "thinking.delta",
313
+ thinking: reasoningContent,
314
+ fidelity: {
315
+ item_id: "reasoning_content",
316
+ reasoning_field: "reasoning_content",
317
+ },
318
+ });
319
+ }
320
+ else if (reasoning) {
321
+ contentItems.push({
322
+ type: "thinking.delta",
323
+ thinking: reasoning,
324
+ fidelity: { item_id: "reasoning", reasoning_field: "reasoning" },
325
+ });
326
+ }
327
+ if (delta?.content) {
328
+ contentItems.push({
329
+ type: "text.delta",
330
+ text: delta.content,
331
+ fidelity: { item_id: "content" },
332
+ });
333
+ }
334
+ if (delta?.tool_calls) {
335
+ for (const toolCall of delta.tool_calls) {
336
+ contentItems.push({
337
+ type: "tool_call.delta",
338
+ name: toolCall.function?.name || "",
339
+ arguments: toolCall.function?.arguments || "",
340
+ tool_call_id: toolCall.id || toolCall.function?.name || "",
341
+ fidelity: { item_id: "tool_calls" },
342
+ });
343
+ }
344
+ }
345
+ if (choice?.finish_reason) {
346
+ eventType = "stop";
347
+ const finishReasonMapping = {
348
+ stop: "stop",
349
+ length: "length",
350
+ tool_calls: "tool_call",
351
+ content_filter: "stop",
352
+ };
353
+ finishReason = finishReasonMapping[choice.finish_reason] || "unknown";
354
+ }
355
+ }
356
+ if (modelOutput.usage) {
357
+ eventType = "stop";
358
+ const cachedTokens = modelOutput.usage.prompt_tokens_details?.cached_tokens || null;
359
+ const reasoningTokens = modelOutput.usage.completion_tokens_details?.reasoning_tokens || null;
360
+ const promptTokens = cachedTokens !== null
361
+ ? modelOutput.usage.prompt_tokens - cachedTokens
362
+ : modelOutput.usage.prompt_tokens;
363
+ const responseTokens = reasoningTokens !== null
364
+ ? modelOutput.usage.completion_tokens - reasoningTokens
365
+ : modelOutput.usage.completion_tokens;
366
+ usageMetadata = {
367
+ cached_tokens: cachedTokens,
368
+ prompt_tokens: promptTokens,
369
+ thoughts_tokens: reasoningTokens,
370
+ response_tokens: responseTokens,
371
+ };
372
+ usageMetadata = (0, utils_1.fixOpenrouterUsageMetadata)(usageMetadata, this._client.baseURL);
373
+ }
374
+ return {
375
+ role: "assistant",
376
+ event_type: eventType,
377
+ content_items: contentItems,
378
+ usage_metadata: usageMetadata,
379
+ finish_reason: finishReason,
380
+ };
381
+ }
382
+ /**
383
+ * Stream generate using OpenAI Chat Completions-compatible API.
384
+ */
385
+ async *_streamingResponseInternal(options) {
386
+ const openaiConfig = this.transformUniConfigToModelConfig(options.config);
387
+ const openaiMessages = await this.transformUniMessageToModelInput(options.messages, options.signal);
388
+ if (options.config.system_prompt) {
389
+ openaiMessages.unshift({
390
+ role: "system",
391
+ content: options.config.system_prompt,
392
+ });
393
+ }
394
+ const params = {
395
+ ...openaiConfig,
396
+ messages: openaiMessages,
397
+ stream: true,
398
+ };
399
+ const stream = await this._client.chat.completions.create(params, {
400
+ signal: options.signal,
401
+ });
402
+ for await (const chunk of stream) {
403
+ yield this.transformModelOutputToUniEvent(chunk);
404
+ }
405
+ }
406
+ /**
407
+ * List the model ids the configured endpoint serves.
408
+ *
409
+ * @returns The model ids, in the order the endpoint returned them.
410
+ */
411
+ async listModels() {
412
+ const models = [];
413
+ for await (const model of this._client.models.list()) {
414
+ models.push(model.id);
415
+ }
416
+ return models;
417
+ }
418
+ }
419
+ exports.OpenaiChatClient = OpenaiChatClient;
@@ -0,0 +1,2 @@
1
+ export { OpenaiChatClient } from "./client";
2
+ //# sourceMappingURL=index.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../../src/openai_chat/index.ts"],"names":[],"mappings":"AAcA,OAAO,EAAE,gBAAgB,EAAE,MAAM,UAAU,CAAC"}
@@ -0,0 +1,18 @@
1
+ "use strict";
2
+ // Copyright 2025 Prism Shadow. and/or its affiliates
3
+ //
4
+ // Licensed under the Apache License, Version 2.0 (the "License");
5
+ // you may not use this file except in compliance with the License.
6
+ // You may obtain a copy of the License at
7
+ //
8
+ // http://www.apache.org/licenses/LICENSE-2.0
9
+ //
10
+ // Unless required by applicable law or agreed to in writing, software
11
+ // distributed under the License is distributed on an "AS IS" BASIS,
12
+ // WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13
+ // See the License for the specific language governing permissions and
14
+ // limitations under the License.
15
+ Object.defineProperty(exports, "__esModule", { value: true });
16
+ exports.OpenaiChatClient = void 0;
17
+ var client_1 = require("./client");
18
+ Object.defineProperty(exports, "OpenaiChatClient", { enumerable: true, get: function () { return client_1.OpenaiChatClient; } });
@@ -0,0 +1,15 @@
1
+ import { OpenaiChatClient } from "../openai_chat";
2
+ import { UniConfig } from "../types";
3
+ /** Models served through vLLM's OpenAI-compatible Chat Completions API. */
4
+ export declare class OpenaiChatVllmAdapterClient extends OpenaiChatClient {
5
+ /**
6
+ * Return the chat_template_kwargs this model's template reads for the level.
7
+ *
8
+ * A model outside the table falls back to Qwen3's enable_thinking, the most widespread
9
+ * of the conventions and inert on a template that ignores the key.
10
+ */
11
+ private _thinkingChatTemplateKwargs;
12
+ /** Map MMSP's level onto the thinking switch this model's chat template reads. */
13
+ transformUniConfigToModelConfig(config: UniConfig): any;
14
+ }
15
+ //# sourceMappingURL=client.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"client.d.ts","sourceRoot":"","sources":["../../src/openai_chat_vllm_adapter/client.ts"],"names":[],"mappings":"AAcA,OAAO,EAAE,gBAAgB,EAAE,MAAM,gBAAgB,CAAC;AAClD,OAAO,EAAiB,SAAS,EAAE,MAAM,UAAU,CAAC;AAiFpD,2EAA2E;AAC3E,qBAAa,2BAA4B,SAAQ,gBAAgB;IAC/D;;;;;OAKG;IACH,OAAO,CAAC,2BAA2B;IAanC,kFAAkF;IAEzE,+BAA+B,CAAC,MAAM,EAAE,SAAS,GAAG,GAAG;CAcjE"}
@@ -0,0 +1,118 @@
1
+ "use strict";
2
+ // Copyright 2025 Prism Shadow. and/or its affiliates
3
+ //
4
+ // Licensed under the Apache License, Version 2.0 (the "License");
5
+ // you may not use this file except in compliance with the License.
6
+ // You may obtain a copy of the License at
7
+ //
8
+ // http://www.apache.org/licenses/LICENSE-2.0
9
+ //
10
+ // Unless required by applicable law or agreed to in writing, software
11
+ // distributed under the License is distributed on an "AS IS" BASIS,
12
+ // WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13
+ // See the License for the specific language governing permissions and
14
+ // limitations under the License.
15
+ Object.defineProperty(exports, "__esModule", { value: true });
16
+ exports.OpenaiChatVllmAdapterClient = void 0;
17
+ const openai_chat_1 = require("../openai_chat");
18
+ const types_1 = require("../types");
19
+ // vLLM passes chat_template_kwargs straight to the served model's chat template, so the
20
+ // switch that turns thinking on is whatever that template happens to read. Each profile
21
+ // below maps an MMSP level onto one family's kwargs; an empty mapping means the
22
+ // request carries no chat_template_kwargs at all.
23
+ //
24
+ // The upstream artifacts these profiles are read off, and the clamping those artifacts
25
+ // force, are snapshotted in llmsdk_docs/openai_chat_vllm_adapter/. Update that snapshot
26
+ // whenever a model is added here.
27
+ // Qwen3 templates read a single enable_thinking boolean and no effort key at all.
28
+ const QWEN3_THINKING = {
29
+ [types_1.ThinkingLevel.NONE]: { enable_thinking: false },
30
+ [types_1.ThinkingLevel.LOW]: { enable_thinking: true },
31
+ [types_1.ThinkingLevel.MEDIUM]: { enable_thinking: true },
32
+ [types_1.ThinkingLevel.HIGH]: { enable_thinking: true },
33
+ [types_1.ThinkingLevel.XHIGH]: { enable_thinking: true },
34
+ [types_1.ThinkingLevel.MAX]: { enable_thinking: true },
35
+ };
36
+ // Qwen3.8-27B and Qwen3.8-Flash-Next ship the same chat template, byte for byte, so they
37
+ // share a profile. It keeps enable_thinking as the off switch and takes its adaptive modes
38
+ // as reasoning_effort, validated against low/medium/xhigh, so high and max clamp to xhigh.
39
+ // The template defaults the key to xhigh, so a model on this template that is not sent the
40
+ // key runs every level at full effort.
41
+ const QWEN3_8_THINKING = {
42
+ [types_1.ThinkingLevel.NONE]: { enable_thinking: false },
43
+ [types_1.ThinkingLevel.LOW]: { reasoning_effort: "low" },
44
+ [types_1.ThinkingLevel.MEDIUM]: { reasoning_effort: "medium" },
45
+ [types_1.ThinkingLevel.HIGH]: { reasoning_effort: "xhigh" },
46
+ [types_1.ThinkingLevel.XHIGH]: { reasoning_effort: "xhigh" },
47
+ [types_1.ThinkingLevel.MAX]: { reasoning_effort: "xhigh" },
48
+ };
49
+ // DeepSeek V4 publishes no chat template; vLLM reads a thinking flag paired with
50
+ // reasoning_effort, and thinking is off whenever the flag is absent, which is what NONE
51
+ // sends. DeepSeek-V4-Pro and DeepSeek-V4-Flash share an encoding module that asserts
52
+ // reasoning_effort in ['max', None, 'high'], so low is a failed request rather than a
53
+ // weaker answer and high is the lowest value they take. That module then branches on 'max'
54
+ // alone, which means LOW through XHIGH all render the same prompt on these two models.
55
+ const DEEPSEEK_V4_PRO_FLASH_THINKING = {
56
+ [types_1.ThinkingLevel.NONE]: {},
57
+ [types_1.ThinkingLevel.LOW]: { thinking: true, reasoning_effort: "high" },
58
+ [types_1.ThinkingLevel.MEDIUM]: { thinking: true, reasoning_effort: "high" },
59
+ [types_1.ThinkingLevel.HIGH]: { thinking: true, reasoning_effort: "high" },
60
+ [types_1.ThinkingLevel.XHIGH]: { thinking: true, reasoning_effort: "high" },
61
+ [types_1.ThinkingLevel.MAX]: { thinking: true, reasoning_effort: "max" },
62
+ };
63
+ // DeepSeek-V4-Flash-Vision-Exp ships a different copy of that encoding module, one that
64
+ // validates reasoning_effort against a low/high/max table, so it keeps the finer scale;
65
+ // medium and xhigh clamp to high.
66
+ const DEEPSEEK_V4_VISION_EXP_THINKING = {
67
+ [types_1.ThinkingLevel.NONE]: {},
68
+ [types_1.ThinkingLevel.LOW]: { thinking: true, reasoning_effort: "low" },
69
+ [types_1.ThinkingLevel.MEDIUM]: { thinking: true, reasoning_effort: "high" },
70
+ [types_1.ThinkingLevel.HIGH]: { thinking: true, reasoning_effort: "high" },
71
+ [types_1.ThinkingLevel.XHIGH]: { thinking: true, reasoning_effort: "high" },
72
+ [types_1.ThinkingLevel.MAX]: { thinking: true, reasoning_effort: "max" },
73
+ };
74
+ // Keys are matched as substrings of the lowercased model id, so a served id keeps whatever
75
+ // prefix the deployment gave it (Qwen/Qwen3.6-35B-A3B, deepseek-ai/DeepSeek-V4-Pro). The
76
+ // first match wins, so a key that contains another must come first: deepseek-v4-flash is a
77
+ // prefix of deepseek-v4-flash-vision-exp.
78
+ const MODEL_THINKING_PROFILES = [
79
+ ["qwen3.8-flash-next", QWEN3_8_THINKING],
80
+ ["qwen3.8-27b", QWEN3_8_THINKING],
81
+ ["qwen3.6-35b-a3b", QWEN3_THINKING],
82
+ ["qwen3.5-0.8b", QWEN3_THINKING],
83
+ ["qwen3.5-9b", QWEN3_THINKING],
84
+ ["deepseek-v4-flash-vision-exp", DEEPSEEK_V4_VISION_EXP_THINKING],
85
+ ["deepseek-v4-pro", DEEPSEEK_V4_PRO_FLASH_THINKING],
86
+ ["deepseek-v4-flash", DEEPSEEK_V4_PRO_FLASH_THINKING],
87
+ ];
88
+ /** Models served through vLLM's OpenAI-compatible Chat Completions API. */
89
+ class OpenaiChatVllmAdapterClient extends openai_chat_1.OpenaiChatClient {
90
+ /**
91
+ * Return the chat_template_kwargs this model's template reads for the level.
92
+ *
93
+ * A model outside the table falls back to Qwen3's enable_thinking, the most widespread
94
+ * of the conventions and inert on a template that ignores the key.
95
+ */
96
+ _thinkingChatTemplateKwargs(thinkingLevel) {
97
+ const model = this._model.toLowerCase();
98
+ for (const [name, profile] of MODEL_THINKING_PROFILES) {
99
+ if (model.includes(name)) {
100
+ return { ...profile[thinkingLevel] };
101
+ }
102
+ }
103
+ return { ...QWEN3_THINKING[thinkingLevel] };
104
+ }
105
+ /** Map MMSP's level onto the thinking switch this model's chat template reads. */
106
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
107
+ transformUniConfigToModelConfig(config) {
108
+ const vllmConfig = super.transformUniConfigToModelConfig(config);
109
+ if (config.thinking_level !== undefined) {
110
+ const chatTemplateKwargs = this._thinkingChatTemplateKwargs(config.thinking_level);
111
+ if (Object.keys(chatTemplateKwargs).length > 0) {
112
+ vllmConfig.chat_template_kwargs = chatTemplateKwargs;
113
+ }
114
+ }
115
+ return vllmConfig;
116
+ }
117
+ }
118
+ exports.OpenaiChatVllmAdapterClient = OpenaiChatVllmAdapterClient;
@@ -0,0 +1,2 @@
1
+ export { OpenaiChatVllmAdapterClient } from "./client";
2
+ //# sourceMappingURL=index.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../../src/openai_chat_vllm_adapter/index.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,2BAA2B,EAAE,MAAM,UAAU,CAAC"}
@@ -0,0 +1,5 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.OpenaiChatVllmAdapterClient = void 0;
4
+ var client_1 = require("./client");
5
+ Object.defineProperty(exports, "OpenaiChatVllmAdapterClient", { enumerable: true, get: function () { return client_1.OpenaiChatVllmAdapterClient; } });
@@ -0,0 +1,46 @@
1
+ import type { CreateEmbeddingResponse, EmbeddingCreateParams } from "openai/resources/embeddings";
2
+ import { LLMClient } from "../baseClient";
3
+ import { UniConfig, UniEvent, UniMessage } from "../types";
4
+ /**
5
+ * OpenAI Embeddings-compatible client implementation.
6
+ */
7
+ export declare class OpenaiEmbeddingClient extends LLMClient {
8
+ protected _model: string;
9
+ private _client;
10
+ /**
11
+ * Initialize OpenAI-compatible embedding client with model, API key, and base URL.
12
+ */
13
+ constructor(options: {
14
+ model: string;
15
+ apiKey?: string;
16
+ baseUrl?: string | null;
17
+ defaultHeaders?: Record<string, string>;
18
+ });
19
+ /**
20
+ * Transform universal configuration to OpenAI Embeddings configuration.
21
+ */
22
+ transformUniConfigToModelConfig(config: UniConfig): Omit<EmbeddingCreateParams, "input">;
23
+ /**
24
+ * Transform universal messages to OpenAI Embeddings input strings.
25
+ */
26
+ transformUniMessageToModelInput(messages: UniMessage[]): string[];
27
+ /**
28
+ * Transform an OpenAI Embeddings response into a universal event, one complete item per vector.
29
+ */
30
+ transformModelOutputToUniEvent(modelOutput: CreateEmbeddingResponse): UniEvent;
31
+ /**
32
+ * Generate embeddings using OpenAI Embeddings-compatible API.
33
+ */
34
+ _streamingResponseInternal(options: {
35
+ messages: UniMessage[];
36
+ config: UniConfig;
37
+ signal?: AbortSignal;
38
+ }): AsyncGenerator<UniEvent>;
39
+ /**
40
+ * List the model ids the configured endpoint serves.
41
+ *
42
+ * @returns The model ids, in the order the endpoint returned them.
43
+ */
44
+ listModels(): Promise<string[]>;
45
+ }
46
+ //# sourceMappingURL=client.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"client.d.ts","sourceRoot":"","sources":["../../src/openai_embedding/client.ts"],"names":[],"mappings":"AAeA,OAAO,KAAK,EACV,uBAAuB,EACvB,qBAAqB,EACtB,MAAM,6BAA6B,CAAC;AACrC,OAAO,EAAE,SAAS,EAAE,MAAM,eAAe,CAAC;AAE1C,OAAO,EAAE,SAAS,EAAE,QAAQ,EAAE,UAAU,EAAE,MAAM,UAAU,CAAC;AAG3D;;GAEG;AACH,qBAAa,qBAAsB,SAAQ,SAAS;IAClD,SAAS,CAAC,MAAM,EAAE,MAAM,CAAC;IACzB,OAAO,CAAC,OAAO,CAAS;IAExB;;OAEG;gBACS,OAAO,EAAE;QACnB,KAAK,EAAE,MAAM,CAAC;QACd,MAAM,CAAC,EAAE,MAAM,CAAC;QAChB,OAAO,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;QACxB,cAAc,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;KACzC;IAeD;;OAEG;IACH,+BAA+B,CAC7B,MAAM,EAAE,SAAS,GAChB,IAAI,CAAC,qBAAqB,EAAE,OAAO,CAAC;IAmBvC;;OAEG;IACH,+BAA+B,CAAC,QAAQ,EAAE,UAAU,EAAE,GAAG,MAAM,EAAE;IAejE;;OAEG;IACH,8BAA8B,CAC5B,WAAW,EAAE,uBAAuB,GACnC,QAAQ;IAkBX;;OAEG;IACI,0BAA0B,CAAC,OAAO,EAAE;QACzC,QAAQ,EAAE,UAAU,EAAE,CAAC;QACvB,MAAM,EAAE,SAAS,CAAC;QAClB,MAAM,CAAC,EAAE,WAAW,CAAC;KACtB,GAAG,cAAc,CAAC,QAAQ,CAAC;IAW5B;;;;OAIG;IACG,UAAU,IAAI,OAAO,CAAC,MAAM,EAAE,CAAC;CAQtC"}