@vincemakes/kiso-provider-openai 0.1.20 → 0.1.23

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/README.md +11 -0
  2. package/dist/index.js +64 -28
  3. package/package.json +3 -3
package/README.md CHANGED
@@ -7,3 +7,14 @@ connection/timeout/5xx classification.
7
7
 
8
8
  Requires Node >= 22. See the repository README for the framework
9
9
  overview.
10
+
11
+ ## Debug tooling
12
+
13
+ `KISO_DUMP_REQUESTS=<dir>` writes every outgoing request body to
14
+ `<dir>/req-<pid>-<n>.json` before it is sent — the diagnosis instrument
15
+ for the request-prefix (D 区) contract, kept as a permanent debug sink.
16
+ ⚠ The bodies are REAL conversation data (the model may have seen repo
17
+ contents) — never share a dump dir; a dump failure never breaks the
18
+ request. `bench/dumpdiff.py` byte-diffs consecutive dumps and localizes
19
+ the first divergence (healthy = at the older request's last-message end;
20
+ violation = inside an old message).
package/dist/index.js CHANGED
@@ -13,6 +13,8 @@
13
13
  * provider reports usage (not all compat providers do); `stop` reason maps
14
14
  * from finish_reason.
15
15
  */
16
+ import { mkdirSync, writeFileSync } from "node:fs";
17
+ import { join } from "node:path";
16
18
  import OpenAI from "openai";
17
19
  import { mapApiError } from "@vincemakes/kiso-core";
18
20
  /**
@@ -32,23 +34,28 @@ export function createOpenAICompatProvider(config = {}) {
32
34
  export function createOpenAICompatAdapter(client) {
33
35
  return {
34
36
  async *stream(options) {
37
+ // The explicit streaming params type keeps the `stream: true`
38
+ // literal overload — a widened `stream: boolean` would type the
39
+ // create() call as the NON-streaming variant (TS2504).
40
+ const body = {
41
+ model: options.model,
42
+ messages: toOpenAIMessages(options.messages, options.systemPrompt),
43
+ stream: true,
44
+ // D5: request real streaming usage — without this the
45
+ // provider never sends a usage chunk and we would
46
+ // report known:false forever.
47
+ stream_options: { include_usage: true },
48
+ ...(options.tools?.length ? { tools: options.tools.map(toOpenAITool) } : {}),
49
+ ...(options.maxTokens !== undefined ? { max_tokens: options.maxTokens } : {}),
50
+ ...(options.temperature !== undefined ? { temperature: options.temperature } : {}),
51
+ };
52
+ maybeDumpRequest(body);
35
53
  // Area 6: the stream CREATION is inside the error normalization —
36
54
  // a 429/5xx/connection failure before the first byte is a mapped,
37
55
  // retryable StructuredError, so the loop's pre-stream retry works.
38
56
  let stream;
39
57
  try {
40
- stream = await client.chat.completions.create({
41
- model: options.model,
42
- messages: toOpenAIMessages(options.messages, options.systemPrompt),
43
- stream: true,
44
- // D5: request real streaming usage — without this the
45
- // provider never sends a usage chunk and we would
46
- // report known:false forever.
47
- stream_options: { include_usage: true },
48
- ...(options.tools?.length ? { tools: options.tools.map(toOpenAITool) } : {}),
49
- ...(options.maxTokens !== undefined ? { max_tokens: options.maxTokens } : {}),
50
- ...(options.temperature !== undefined ? { temperature: options.temperature } : {}),
51
- },
58
+ stream = await client.chat.completions.create(body,
52
59
  // Phase B: cancellation reaches the SDK; the system prompt is a
53
60
  // first-class message, not a dropped option.
54
61
  options.signal !== undefined ? { signal: options.signal } : undefined);
@@ -226,6 +233,30 @@ export function createOpenAICompatAdapter(client) {
226
233
  },
227
234
  };
228
235
  }
236
+ // ── Request dump (debug tool) ───────────────────────────────────────────
237
+ let dumpSeq = 0;
238
+ /**
239
+ * KISO_DUMP_REQUESTS=<dir> — DEBUG TOOL: writes the FULL request body (the
240
+ * exact JSON the SDK will send) to `<dir>/req-<pid>-<n>.json` before it is
241
+ * sent, numbered per request in process order. Built for the fresh-mystery
242
+ * diagnosis (byte-diff consecutive requests' common prefix); kept as a
243
+ * permanent debug sink. ⚠ The bodies are REAL conversation data (the model
244
+ * may have seen repo contents) — never share a dump dir. A dump failure
245
+ * never breaks the request.
246
+ */
247
+ function maybeDumpRequest(body) {
248
+ const dir = process.env.KISO_DUMP_REQUESTS;
249
+ if (dir === undefined || dir === "")
250
+ return;
251
+ dumpSeq += 1;
252
+ try {
253
+ mkdirSync(dir, { recursive: true });
254
+ writeFileSync(join(dir, `req-${process.pid}-${dumpSeq}.json`), JSON.stringify(body));
255
+ }
256
+ catch {
257
+ // debug tool: silent on failure
258
+ }
259
+ }
229
260
  // ── Mapping helpers ────────────────────────────────────────────────────
230
261
  /**
231
262
  * Exhaustive over the SDK's CLOSED finish_reason union (Area 6): a new SDK
@@ -294,21 +325,22 @@ function toOpenAIMessages(messages, systemPrompt) {
294
325
  if (systemPrompt !== undefined) {
295
326
  out.push({ role: "system", content: systemPrompt });
296
327
  }
297
- // 自举 P1/P2: thinking mode is detected by the presence of ANY
298
- // reasoning in the projection (real OpenAI never emits thinking events,
299
- // so its requests never see the field). In thinking mode, ONLY the
300
- // CURRENT turn's assistant messages (after the last user message) carry
301
- // reasoning_content — their own, or "" when the step produced no
302
- // thinking (the field must still be present, or DeepSeek 400s); OLD
303
- // turns' CoT is never echoed (DeepSeek does not need it, and echoing
304
- // it is token waste).
305
- const thinkingMode = messages.some((m) => m.role === "assistant" && m.reasoning !== undefined);
306
- let lastUser = -1;
307
- for (const [i, m] of messages.entries()) {
308
- if (m.role === "user")
309
- lastUser = i;
310
- }
311
- for (const [i, msg] of messages.entries()) {
328
+ // 合并轮 (0.1.23) C7 修订: `reasoning_content` 的存在性由整个投影的
329
+ // 单调状态决定 — 投影里存在任一 reasoning → 每条 assistant 消息都
330
+ // 携带(自身 reasoning,或 "");否则一条都不带。理由: D 区请求级
331
+ // 字节稳定。旧实现按"当前轮"判定(手感批 C7),轮边界一过,旧轮
332
+ // assistant 消息的字段被剥掉,序列化被改写 — 请求 N 与 N+1 的公共
333
+ // 前缀断在旧消息处,provider 前缀缓存每轮边界断一次(fresh 之谜
334
+ // 实证: 14 请求的会话里两次缓存断点都在轮边界;修复后同会话 0 断
335
+ // 点、逐请求 cached 82-98%)。单调性保证存在性在会话内只翻转一次
336
+ // (首个 thinking 出现时,通常在首轮 — 此前几乎没有旧 assistant
337
+ // 消息),之后从生到死不翻转;真 OpenAI 永不产生 reasoning → 字段
338
+ // 永不出现,其请求路径与旧行为逐字节相同。带 reasoning 的旧轮消息
339
+ // 回传其推理是 DeepSeek 官方推荐的缓存稳定形态;旧内容命中前缀
340
+ // 缓存只按 0.1× 计费,"回传是 token 浪费"的前提在缓存经济下不成立。
341
+ // "" 字段在旧轮同样被接受(真 API 验证: 200 + 2560 cached tokens)。
342
+ const hasReasoning = messages.some((m) => m.role === "assistant" && m.reasoning !== undefined);
343
+ for (const msg of messages) {
312
344
  if (msg.role === "user") {
313
345
  out.push({ role: "user", content: toOpenAIContent(msg.content) });
314
346
  }
@@ -330,7 +362,11 @@ function toOpenAIMessages(messages, systemPrompt) {
330
362
  role: "assistant",
331
363
  content: msg.blocks.filter((b) => b.type === "text").map((b) => b.text).join(""),
332
364
  ...(toolCalls.length ? { tool_calls: toolCalls } : {}),
333
- ...(thinkingMode && i > lastUser ? { reasoning_content: msg.reasoning ?? "" } : {}),
365
+ // DeepSeek extension field — absent from the SDK's param type,
366
+ // added via spread so the literal stays assignable (the union
367
+ // type carries the extra key). Presence follows hasReasoning
368
+ // (the monotone rule above — the field never flips mid-history).
369
+ ...(hasReasoning ? { reasoning_content: msg.reasoning ?? "" } : {}),
334
370
  });
335
371
  }
336
372
  else {
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@vincemakes/kiso-provider-openai",
3
- "version": "0.1.20",
4
- "description": "kiso OpenAI-compatible adapter — OpenAI + the compat family (GLM, Kimi, DeepSeek, OpenRouter) via base_url swap, reasoning dialects digested.",
3
+ "version": "0.1.23",
4
+ "description": "kiso OpenAI-compatible adapter \u2014 OpenAI + the compat family (GLM, Kimi, DeepSeek, OpenRouter) via base_url swap, reasoning dialects digested.",
5
5
  "type": "module",
6
6
  "license": "MIT",
7
7
  "exports": {
@@ -21,7 +21,7 @@
21
21
  "test": "vitest run"
22
22
  },
23
23
  "dependencies": {
24
- "@vincemakes/kiso-core": "0.1.20",
24
+ "@vincemakes/kiso-core": "0.1.23",
25
25
  "openai": "^7.3.0"
26
26
  },
27
27
  "devDependencies": {