@vincemakes/kiso-provider-openai 0.36.0 → 0.38.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +31 -3
- package/package.json +2 -2
package/dist/index.js
CHANGED
|
@@ -99,12 +99,27 @@ export function createOpenAICompatAdapter(client, adapterOpts = {}) {
|
|
|
99
99
|
// usage chunk is still accepted after the finish: some compat
|
|
100
100
|
// providers send it late.
|
|
101
101
|
let finishSeen = false;
|
|
102
|
+
// TRACE-F1: the model the SERVER says it served. Every chunk of an
|
|
103
|
+
// OpenAI-shaped stream carries `model`; we never read it, so an id
|
|
104
|
+
// the vendor silently aliases (a retired name, a migration id)
|
|
105
|
+
// looked identical to the one we asked for. ONE variable, read by
|
|
106
|
+
// every usage exit below — this adapter has three of them, and two
|
|
107
|
+
// places computing the same thing is what W22-R1 was.
|
|
108
|
+
let served = null;
|
|
102
109
|
try {
|
|
103
110
|
for await (const chunk of stream) {
|
|
111
|
+
// before the finishSeen branch below, which continues early.
|
|
112
|
+
// FIRST statement wins: a provider that changes the id
|
|
113
|
+
// mid-stream is telling us something is wrong, and the first
|
|
114
|
+
// answer is the one the request was routed by.
|
|
115
|
+
if (served === null && typeof chunk.model === "string" && chunk.model !== "")
|
|
116
|
+
served = chunk.model;
|
|
104
117
|
if (finishSeen) {
|
|
105
118
|
if (chunk.usage) {
|
|
106
119
|
usageSent = true;
|
|
107
|
-
const
|
|
120
|
+
const u = chunk.usage;
|
|
121
|
+
const details = u.prompt_tokens_details;
|
|
122
|
+
const reasoning = u.completion_tokens_details?.reasoning_tokens;
|
|
108
123
|
yield {
|
|
109
124
|
seq: 0,
|
|
110
125
|
type: "usage",
|
|
@@ -113,6 +128,8 @@ export function createOpenAICompatAdapter(client, adapterOpts = {}) {
|
|
|
113
128
|
cacheRead: details?.cached_tokens ?? null,
|
|
114
129
|
cacheWrite: null,
|
|
115
130
|
known: true,
|
|
131
|
+
...(typeof reasoning === "number" ? { reasoningTokens: reasoning } : {}),
|
|
132
|
+
...(served !== null ? { servedModel: served } : {}),
|
|
116
133
|
};
|
|
117
134
|
}
|
|
118
135
|
continue; // content and finish reasons after the first finish: ignored
|
|
@@ -209,7 +226,14 @@ export function createOpenAICompatAdapter(client, adapterOpts = {}) {
|
|
|
209
226
|
// prompt_tokens_details — an absent value is null, NEVER
|
|
210
227
|
// faked as a zero-cache turn. OpenAI does not report a
|
|
211
228
|
// cache write; null is the honest answer.
|
|
212
|
-
const
|
|
229
|
+
const u = chunk.usage;
|
|
230
|
+
const details = u.prompt_tokens_details;
|
|
231
|
+
// RSN-1: the completion side of the SAME object. We have
|
|
232
|
+
// always read the prompt side for cache and never looked
|
|
233
|
+
// here, so `canonical.reasoning` sat hardcoded null under a
|
|
234
|
+
// comment saying no provider reports a split — true when
|
|
235
|
+
// written, false since this vendor shipped one.
|
|
236
|
+
const reasoning = u.completion_tokens_details?.reasoning_tokens;
|
|
213
237
|
yield {
|
|
214
238
|
seq: 0,
|
|
215
239
|
type: "usage",
|
|
@@ -218,6 +242,8 @@ export function createOpenAICompatAdapter(client, adapterOpts = {}) {
|
|
|
218
242
|
cacheRead: details?.cached_tokens ?? null,
|
|
219
243
|
cacheWrite: null,
|
|
220
244
|
known: true,
|
|
245
|
+
...(typeof reasoning === "number" ? { reasoningTokens: reasoning } : {}),
|
|
246
|
+
...(served !== null ? { servedModel: served } : {}),
|
|
221
247
|
};
|
|
222
248
|
}
|
|
223
249
|
const fr = chunk.choices?.[0]?.finish_reason;
|
|
@@ -255,7 +281,9 @@ export function createOpenAICompatAdapter(client, adapterOpts = {}) {
|
|
|
255
281
|
if (!usageSent) {
|
|
256
282
|
// Area 6: no usage reported is expressed as UNKNOWN — nulls
|
|
257
283
|
// and known:false — never faked as a zero-cost turn.
|
|
258
|
-
|
|
284
|
+
// known:false and a served model are not in tension: the server
|
|
285
|
+
// stated what it ran, it just never reported what it cost.
|
|
286
|
+
yield { seq: 0, type: "usage", inputTokens: null, outputTokens: null, cacheRead: null, cacheWrite: null, known: false, ...(served !== null ? { servedModel: served } : {}) };
|
|
259
287
|
}
|
|
260
288
|
// Area 6 hardening (review finding 4): a stream that ended with
|
|
261
289
|
// NO finish_reason is a TRUNCATED turn — the stop is an explicit
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@vincemakes/kiso-provider-openai",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.38.0",
|
|
4
4
|
"description": "kiso OpenAI-compatible adapter \u2014 OpenAI + the compat family (GLM, Kimi, DeepSeek, OpenRouter) via base_url swap, reasoning dialects digested.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "MIT",
|
|
@@ -21,7 +21,7 @@
|
|
|
21
21
|
"test": "vitest run"
|
|
22
22
|
},
|
|
23
23
|
"dependencies": {
|
|
24
|
-
"@vincemakes/kiso-core": "0.
|
|
24
|
+
"@vincemakes/kiso-core": "0.38.0",
|
|
25
25
|
"openai": "^7.3.0"
|
|
26
26
|
},
|
|
27
27
|
"devDependencies": {
|