mslxdff 0.1.143 → 0.1.144

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "mslxdff",
3
- "version": "0.1.143",
3
+ "version": "0.1.144",
4
4
  "description": "测试项目,请勿使用。",
5
5
  "type": "module",
6
6
  "bin": {
@@ -17,6 +17,7 @@ export function chunkToString(c) {
17
17
  return String(c);
18
18
  }
19
19
 
20
+ // content 数组 → 文本(tool_calls 摘要之外的纯文本部分,历史行为)
20
21
  function inputTextOf(content) {
21
22
  if (typeof content === "string") return content;
22
23
  if (Array.isArray(content)) {
@@ -25,6 +26,21 @@ function inputTextOf(content) {
25
26
  return String(content ?? "");
26
27
  }
27
28
 
29
+ // content 数组里的图片(responses 规范 input_image part)→ chat 规范 image_url part。
30
+ // 此前被 inputTextOf 静默丢掉 → 下游根本拿不到图,模型"看不到"(2026-09-22 实测)。
31
+ function inputImagesOf(content) {
32
+ if (!Array.isArray(content)) return [];
33
+ const out = [];
34
+ for (const p of content) {
35
+ if (!p || typeof p !== "object") continue;
36
+ if (p.type === "input_image" || p.type === "image_url") {
37
+ const url = typeof p.image_url === "string" ? p.image_url : p.image_url?.url;
38
+ if (url) out.push({ type: "image_url", image_url: { url } });
39
+ }
40
+ }
41
+ return out;
42
+ }
43
+
28
44
  // POST /v1/responses body → chat completions body(直接喂现有 pipeline)
29
45
  export function responsesToChatBody(req = {}) {
30
46
  const model = String(req.model || "").trim();
@@ -51,7 +67,10 @@ export function responsesToChatBody(req = {}) {
51
67
  }
52
68
  // responses 规范:message item 的 type 可省(AI SDK/opencode 就不发)→ 有 role 即按 message 处理
53
69
  if (it.type === "message" || (!it.type && it.role)) {
70
+ const imgs = inputImagesOf(it.content);
54
71
  const msg = { role: it.role || "user", content: inputTextOf(it.content) };
72
+ // 有图时 content 用 chat 多模态数组形状(text + image_url),下游 responses 通道才能带图上游
73
+ if (imgs.length) msg.content = [...(msg.content ? [{ type: "text", text: msg.content }] : []), ...imgs];
55
74
  if (pendingReasoning.length && msg.role === "assistant") { msg.reasoning_items = pendingReasoning; pendingReasoning = []; }
56
75
  messages.push(msg);
57
76
  } else if (it.type === "function_call") {
@@ -16,6 +16,22 @@ function textOf(content) {
16
16
  return "";
17
17
  }
18
18
 
19
+ // 图片 → AI SDK 的 file part(mediaType 为 image/*)。
20
+ // 为什么不是 {type:"image"}:AI SDK v3 的 prompt 校验没有 image part 类型,会被序列化成
21
+ // null 发给上游 → 400 "input[N].content did not match any supported type"(2026-09-22 实测)。
22
+ // file part + image/* 才是 @ai-sdk/openai responses 适配器产出 input_image 的正道
23
+ // (dist/index.mjs:mediaType.startsWith("image/") → {type:"input_image", image_url})。
24
+ // mediaType 从 data URL 提取真实类型:通配 "image/*" 会被 SDK 强转成 "image/jpeg"。
25
+ function filePartFromImageUrl(url) {
26
+ const m = /^data:([^;,]*)[^,]*,/.exec(url);
27
+ const mediaType = m && m[1] ? m[1] : "image/*";
28
+ try {
29
+ return { type: "file", mediaType, data: new URL(url) };
30
+ } catch {
31
+ return null;
32
+ }
33
+ }
34
+
19
35
  function userContentParts(content) {
20
36
  if (typeof content === "string") return [{ type: "text", text: content }];
21
37
  if (!Array.isArray(content)) return [{ type: "text", text: String(content ?? "") }];
@@ -24,10 +40,11 @@ function userContentParts(content) {
24
40
  if (!p || typeof p !== "object") continue;
25
41
  if (p.type === "text" || p.type === "input_text") {
26
42
  parts.push({ type: "text", text: String(p.text ?? "") });
27
- } else if (p.type === "image_url") {
43
+ } else if (p.type === "image_url" || p.type === "input_image" || p.image_url != null) {
28
44
  const url = typeof p.image_url === "string" ? p.image_url : p.image_url?.url;
29
45
  if (!url) continue;
30
- try { parts.push({ type: "file", mediaType: "image/*", data: new URL(url) }); } catch {}
46
+ const fp = filePartFromImageUrl(url);
47
+ if (fp) parts.push(fp);
31
48
  }
32
49
  }
33
50
  if (!parts.length) parts.push({ type: "text", text: "" });
@@ -28,21 +28,46 @@ export function chatToResponsesBody(chatBody) {
28
28
  const msgs = Array.isArray(chatBody?.messages) ? chatBody.messages : [];
29
29
  const system = msgs.filter((m) => m.role === "system").map((m) => String(m.content || "")).join("\n");
30
30
  const nonSystem = msgs.filter((m) => m.role !== "system");
31
- const inputParts = nonSystem.map((m) => {
31
+ // 图片保留:content 数组里的 image_url / input_image 转 responses 规范的 input_image item
32
+ // (data: base64 原样透传;此前整段拍平成纯文本 → 模型"看不到图",2026-09-22 实测)。
33
+ const imagesOf = (m) => Array.isArray(m.content)
34
+ ? m.content.flatMap((x) => {
35
+ if (!x || typeof x !== "object") return [];
36
+ const url = x.type === "image_url" || x.type === "input_image" || x.image_url != null
37
+ ? (typeof x.image_url === "string" ? x.image_url : x.image_url?.url)
38
+ : null;
39
+ return url ? [{ type: "input_image", image_url: url }] : [];
40
+ })
41
+ : [];
42
+ const inputItems = nonSystem.map((m) => {
32
43
  const c = m.content;
33
- let base;
34
- if (typeof c === "string") base = `${m.role}: ${c}`;
35
- else if (Array.isArray(c)) base = `${m.role}: ${c.map((x) => x.text || x.content || "").join("")}`;
36
- else base = `${m.role}: ${String(c || "")}`;
44
+ let text;
45
+ if (typeof c === "string") text = c;
46
+ else if (Array.isArray(c)) text = c.filter((x) => x && (x.type === "text" || x.type === "input_text")).map((x) => x.text || "").join("");
47
+ else text = String(c ?? "");
48
+ let base = `${m.role}: ${text}`;
37
49
  // 保留 tool_calls / tool 结果,避免多轮丢失
38
50
  if (Array.isArray(m.tool_calls) && m.tool_calls.length) {
39
51
  const tcStr = m.tool_calls.map((tc) => `${tc.function?.name || "tool"}(${tc.function?.arguments || ""})`).join("; ");
40
52
  base += ` [tool_calls: ${tcStr}]`;
41
53
  }
42
54
  if (m.role === "tool" && m.tool_call_id) base += ` (call_id=${m.tool_call_id})`;
43
- return base;
55
+ const imgs = imagesOf(m);
56
+ return { text: base, images: imgs };
44
57
  });
45
- const input = inputParts.join("\n\n") || "hi";
58
+ const inputParts = inputItems.map((x) => x.text);
59
+ const hasText = inputParts.some((t) => t.trim());
60
+ const allImages = inputItems.flatMap((x) => x.images);
61
+ // 有图时 input 必须用 item 数组(text + input_image 混排);纯文本保持串形状(既有行为不变)
62
+ const input = allImages.length
63
+ ? inputItems.flatMap((x) => {
64
+ const items = [];
65
+ if (x.text.trim()) items.push({ type: "input_text", text: x.text });
66
+ items.push(...x.images);
67
+ return items;
68
+ })
69
+ : (inputParts.join("\n\n") || "hi");
70
+ void hasText;
46
71
  // 流式意图透传:客户端要 SSE 就向上游要 SSE(reshapeResponsesSse 负责转回 chat SSE)。
47
72
  // 写死 stream:false 是历史折衷(当时聚合 JSON 直回),已由完整 SSE 转换取代。
48
73
  const out = { model: chatBody.model, input, stream: chatBody?.stream === true };