@plurnk/plurnk-providers 1.18.0 → 1.19.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +8 -4
- package/README.md +4 -16
- package/SPEC.md +41 -11
- package/dist/AiSdkProvider.d.ts +2 -4
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +7 -17
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/AiSdkRequestBody.d.ts +1 -4
- package/dist/AiSdkRequestBody.d.ts.map +1 -1
- package/dist/AiSdkRequestBody.js +10 -48
- package/dist/AiSdkRequestBody.js.map +1 -1
- package/dist/Mock.d.ts +2 -2
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +3 -3
- package/dist/Mock.js.map +1 -1
- package/dist/ProviderRegistry.js +2 -2
- package/dist/ProviderRegistry.js.map +1 -1
- package/dist/accounting.d.ts +0 -1
- package/dist/accounting.d.ts.map +1 -1
- package/dist/accounting.js +1 -7
- package/dist/accounting.js.map +1 -1
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +79 -7
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +13 -3
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +13 -25
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +1 -0
- package/dist/env.js.map +1 -1
- package/dist/index.d.ts +1 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -1
- package/dist/index.js.map +1 -1
- package/dist/types.d.ts +0 -6
- package/dist/types.d.ts.map +1 -1
- package/package.json +7 -7
- package/src/AiSdkProvider.test.ts +97 -126
- package/src/AiSdkProvider.ts +8 -22
- package/src/AiSdkRequestBody.ts +10 -42
- package/src/Mock.test.ts +11 -3
- package/src/Mock.ts +3 -3
- package/src/ProviderRegistry.test.ts +0 -1
- package/src/ProviderRegistry.ts +1 -1
- package/src/accounting.test.ts +0 -11
- package/src/accounting.ts +0 -8
- package/src/aiSdkTransport.test.ts +63 -1
- package/src/aiSdkTransport.ts +82 -7
- package/src/boundaries.test.ts +1 -1
- package/src/capacity.test.ts +2 -2
- package/src/catalogProvider.test.ts +46 -1
- package/src/catalogProvider.ts +12 -3
- package/src/compatibleProvider.test.ts +6 -6
- package/src/compatibleProvider.ts +13 -28
- package/src/cost.test.ts +3 -3
- package/src/discover.test.ts +11 -2
- package/src/env.test.ts +3 -0
- package/src/env.ts +2 -1
- package/src/errors.test.ts +2 -2
- package/src/index.ts +0 -1
- package/src/inputModalities.test.ts +2 -1
- package/src/sdkModels.test.ts +1 -1
- package/src/types.ts +8 -35
package/src/accounting.ts
CHANGED
|
@@ -1,5 +1,4 @@
|
|
|
1
1
|
import type {
|
|
2
|
-
ChargedCost,
|
|
3
2
|
ProviderAccounting,
|
|
4
3
|
ProviderCostNormalizer,
|
|
5
4
|
ProviderRequestAccounting,
|
|
@@ -7,7 +6,6 @@ import type {
|
|
|
7
6
|
} from "./types.ts";
|
|
8
7
|
import {
|
|
9
8
|
sumProviderCostsUsd,
|
|
10
|
-
validateChargedCost,
|
|
11
9
|
validateProviderCost,
|
|
12
10
|
} from "./cost.ts";
|
|
13
11
|
import { validateProviderUsage } from "./usage.ts";
|
|
@@ -82,12 +80,6 @@ const deepInfraCost: ProviderCostNormalizer = ({ usage }) => {
|
|
|
82
80
|
};
|
|
83
81
|
};
|
|
84
82
|
|
|
85
|
-
// The first-party endpoint owns the direct charged-cost wire field.
|
|
86
|
-
export const plurnkCostNormalizer: ProviderCostNormalizer = ({ charge }) => {
|
|
87
|
-
if (charge === undefined) return undefined;
|
|
88
|
-
return validateChargedCost(charge) as ChargedCost;
|
|
89
|
-
};
|
|
90
|
-
|
|
91
83
|
export const providerCostNormalizer = (
|
|
92
84
|
sdkPackage: string,
|
|
93
85
|
): ProviderCostNormalizer | undefined => {
|
|
@@ -19,7 +19,7 @@ const request = {
|
|
|
19
19
|
captureRawBody: false,
|
|
20
20
|
};
|
|
21
21
|
|
|
22
|
-
test("the transport performs exactly one physical request", async () => {
|
|
22
|
+
test("{§provider-sdk-boundary} the transport performs exactly one physical request", async () => {
|
|
23
23
|
let calls = 0;
|
|
24
24
|
await assert.rejects(
|
|
25
25
|
executeOpenAICompatible({
|
|
@@ -416,3 +416,65 @@ test("inconsistent usage counters refuse normalization without failing the respo
|
|
|
416
416
|
assert.equal(result.usageRefusal, undefined);
|
|
417
417
|
});
|
|
418
418
|
});
|
|
419
|
+
|
|
420
|
+
test("{§google-thought-response}: Gemini's flagged thought content is reasoning, never emission", async (t) => {
|
|
421
|
+
const chunk = (delta: Record<string, unknown>, extra: Record<string, unknown> = {}) => ({
|
|
422
|
+
id: "gemini-1", object: "chat.completion.chunk", created: 1, model: "gemini-3.8-flash",
|
|
423
|
+
choices: [{ index: 0, delta, finish_reason: null }], ...extra,
|
|
424
|
+
});
|
|
425
|
+
const thought = (content: string) => ({ role: "assistant", content, extra_content: { google: { thought: true } } });
|
|
426
|
+
const sse = (chunks: object[]) => `${chunks.map((c) => `data: ${JSON.stringify(c)}\n\n`).join("")}data: [DONE]\n\n`;
|
|
427
|
+
|
|
428
|
+
await t.test("streamed: flagged deltas are observed and settled as reasoning; the answer stays clean", async () => {
|
|
429
|
+
const reasoning: string[] = [];
|
|
430
|
+
const text: string[] = [];
|
|
431
|
+
const result = await executeOpenAICompatible({
|
|
432
|
+
...request,
|
|
433
|
+
streaming: true,
|
|
434
|
+
observeReasoning: (delta) => reasoning.push(delta),
|
|
435
|
+
observeText: (delta) => text.push(delta),
|
|
436
|
+
fetch: async () => new Response(sse([
|
|
437
|
+
chunk(thought("<thought>**Checking 391**\n\n")),
|
|
438
|
+
chunk(thought("17 × 23 = 391.")),
|
|
439
|
+
chunk({ content: "</thought>No. " }),
|
|
440
|
+
chunk({ content: "391 = 17 × 23.", extra_content: { google: { thought_signature: "c2ln" } } }),
|
|
441
|
+
{ ...chunk({}), choices: [{ index: 0, delta: {}, finish_reason: "stop" }], usage: { prompt_tokens: 5, completion_tokens: 8, total_tokens: 90 } },
|
|
442
|
+
]), { headers: { "content-type": "text/event-stream" } }),
|
|
443
|
+
});
|
|
444
|
+
assert.equal(result.content, "No. 391 = 17 × 23.");
|
|
445
|
+
assert.equal(result.reasoning, "**Checking 391**\n\n17 × 23 = 391.");
|
|
446
|
+
assert.equal(result.reasoningProjected, true);
|
|
447
|
+
assert.deepEqual(reasoning, ["**Checking 391**\n\n", "17 × 23 = 391."]);
|
|
448
|
+
assert.deepEqual(text, ["No. ", "391 = 17 × 23."]);
|
|
449
|
+
});
|
|
450
|
+
|
|
451
|
+
await t.test("whole: the flagged message's leading <thought> wrapper is reasoning and the remainder is content", async () => {
|
|
452
|
+
const result = await executeOpenAICompatible({
|
|
453
|
+
...request,
|
|
454
|
+
fetch: async () => new Response(JSON.stringify({
|
|
455
|
+
id: "gemini-2", object: "chat.completion", created: 1, model: "gemini-3.8-flash",
|
|
456
|
+
choices: [{ index: 0, finish_reason: "stop", message: {
|
|
457
|
+
role: "assistant",
|
|
458
|
+
content: "<thought>**Checking 391**\n17 × 23 = 391.</thought>No. It is divisible by 17 and 23.",
|
|
459
|
+
extra_content: { google: { thought: true, thought_signature: "c2ln" } },
|
|
460
|
+
} }],
|
|
461
|
+
usage: { prompt_tokens: 5, completion_tokens: 9, total_tokens: 90 },
|
|
462
|
+
}), { headers: { "content-type": "application/json" } }),
|
|
463
|
+
});
|
|
464
|
+
assert.equal(result.content, "No. It is divisible by 17 and 23.");
|
|
465
|
+
assert.equal(result.reasoning, "**Checking 391**\n17 × 23 = 391.");
|
|
466
|
+
assert.equal(result.reasoningProjected, true);
|
|
467
|
+
});
|
|
468
|
+
|
|
469
|
+
await t.test("unflagged content is never inspected for the wrapper", async () => {
|
|
470
|
+
const result = await executeOpenAICompatible({
|
|
471
|
+
...request,
|
|
472
|
+
fetch: async () => new Response(JSON.stringify({
|
|
473
|
+
id: "plain", object: "chat.completion", created: 1, model: "m",
|
|
474
|
+
choices: [{ index: 0, finish_reason: "stop", message: { role: "assistant", content: "<thought>literal</thought>tail" } }],
|
|
475
|
+
}), { headers: { "content-type": "application/json" } }),
|
|
476
|
+
});
|
|
477
|
+
assert.equal(result.content, "<thought>literal</thought>tail");
|
|
478
|
+
assert.equal(result.reasoning, "");
|
|
479
|
+
});
|
|
480
|
+
});
|
package/src/aiSdkTransport.ts
CHANGED
|
@@ -162,6 +162,38 @@ const recordOf = (value: unknown): Record<string, unknown> | null =>
|
|
|
162
162
|
? value as Record<string, unknown>
|
|
163
163
|
: null;
|
|
164
164
|
|
|
165
|
+
// {§google-thought-response} — Gemini behind an OpenAI-compatible endpoint returns its readable
|
|
166
|
+
// thought summary as ordinary `content` wrapped `<thought>…</thought>` and flagged
|
|
167
|
+
// `extra_content.google.thought`. Whole, the flagged message holds wrapper and answer; streamed,
|
|
168
|
+
// the flagged deltas hold `<thought>` and the thought, and the first unflagged delta opens with
|
|
169
|
+
// `</thought>` before the answer.
|
|
170
|
+
const googleThoughtFlagged = (message: Record<string, unknown> | null): boolean =>
|
|
171
|
+
recordOf(recordOf(message?.extra_content)?.google)?.thought === true;
|
|
172
|
+
|
|
173
|
+
const THOUGHT_OPEN = "<thought>";
|
|
174
|
+
const THOUGHT_CLOSE = "</thought>";
|
|
175
|
+
const splitGoogleThought = (content: string): { thought: string; answer: string } | null => {
|
|
176
|
+
if (!content.startsWith(THOUGHT_OPEN)) return null;
|
|
177
|
+
const close = content.indexOf(THOUGHT_CLOSE);
|
|
178
|
+
if (close === -1) return null;
|
|
179
|
+
return { thought: content.slice(THOUGHT_OPEN.length, close), answer: content.slice(close + THOUGHT_CLOSE.length) };
|
|
180
|
+
};
|
|
181
|
+
|
|
182
|
+
const unwrapThoughtDelta = (text: string): string => {
|
|
183
|
+
const opened = text.startsWith(THOUGHT_OPEN) ? text.slice(THOUGHT_OPEN.length) : text;
|
|
184
|
+
return opened.endsWith(THOUGHT_CLOSE) ? opened.slice(0, -THOUGHT_CLOSE.length) : opened;
|
|
185
|
+
};
|
|
186
|
+
|
|
187
|
+
const rawChunkThought = (value: unknown): boolean => {
|
|
188
|
+
const choices = recordOf(value)?.choices;
|
|
189
|
+
return Array.isArray(choices) && googleThoughtFlagged(recordOf(recordOf(choices[0])?.delta));
|
|
190
|
+
};
|
|
191
|
+
|
|
192
|
+
const wholeResponseThought = (values: readonly unknown[]): boolean => values.some((value) => {
|
|
193
|
+
const choices = recordOf(value)?.choices;
|
|
194
|
+
return Array.isArray(choices) && googleThoughtFlagged(recordOf(recordOf(choices[0])?.message));
|
|
195
|
+
});
|
|
196
|
+
|
|
165
197
|
const metadataOf = (values: readonly unknown[]): Record<string, unknown> => {
|
|
166
198
|
const metadata: Record<string, unknown> = {};
|
|
167
199
|
for (const value of values) {
|
|
@@ -413,6 +445,18 @@ const executeModelOnce = async (
|
|
|
413
445
|
? { role: "assistant", content: chatMessageText(message) }
|
|
414
446
|
: { role: "system", content: chatMessageText(message) };
|
|
415
447
|
});
|
|
448
|
+
// {§provider-connectivity} — a streamed attempt's deadline holds only until semantic content
|
|
449
|
+
// flows; after that, stream-idle catches a stall and the operation deadline bounds the whole.
|
|
450
|
+
// A healthy stream still producing (long reasoning) is never cut off and its tokens wasted.
|
|
451
|
+
const attemptDeadline = request.streaming && request.fetchTimeoutMs > 0 ? new AbortController() : null;
|
|
452
|
+
const attemptTimer = attemptDeadline === null ? null : setTimeout(
|
|
453
|
+
() => attemptDeadline.abort(new ProviderTimeoutError("attempt", request.fetchTimeoutMs)),
|
|
454
|
+
request.fetchTimeoutMs,
|
|
455
|
+
);
|
|
456
|
+
const liftAttemptDeadline = (): void => { if (attemptTimer !== null) clearTimeout(attemptTimer); };
|
|
457
|
+
const abortSignal = attemptDeadline === null
|
|
458
|
+
? request.signal
|
|
459
|
+
: request.signal === undefined ? attemptDeadline.signal : AbortSignal.any([request.signal, attemptDeadline.signal]);
|
|
416
460
|
const common = {
|
|
417
461
|
model,
|
|
418
462
|
...(instructions.length === 0 ? {} : { instructions }),
|
|
@@ -422,10 +466,10 @@ const executeModelOnce = async (
|
|
|
422
466
|
// AiSdkProvider owns retries so every physical request is independently
|
|
423
467
|
// observed and accounted. The SDK transport executes exactly once.
|
|
424
468
|
maxRetries: 0,
|
|
425
|
-
abortSignal
|
|
469
|
+
abortSignal,
|
|
426
470
|
headers: request.headers,
|
|
427
471
|
timeout: {
|
|
428
|
-
...(request.fetchTimeoutMs > 0 ? { totalMs: request.fetchTimeoutMs } : {}),
|
|
472
|
+
...(request.fetchTimeoutMs > 0 && !request.streaming ? { totalMs: request.fetchTimeoutMs } : {}),
|
|
429
473
|
...(request.streaming
|
|
430
474
|
&& request.firstContentTimeoutMs !== undefined
|
|
431
475
|
&& request.firstContentTimeoutMs > 0
|
|
@@ -451,9 +495,10 @@ const executeModelOnce = async (
|
|
|
451
495
|
const accountingUsage = wireUsageEvidenceOf(values);
|
|
452
496
|
const reasoningText = evidence.reasoning || result.reasoningText || "";
|
|
453
497
|
const rawFinishReason = result.rawFinishReason;
|
|
498
|
+
const thought = wholeResponseThought(values) ? splitGoogleThought(result.text) : null;
|
|
454
499
|
return {
|
|
455
500
|
model: result.response.modelId,
|
|
456
|
-
content: result.text,
|
|
501
|
+
content: thought === null ? result.text : thought.answer,
|
|
457
502
|
reasoning: reasoningText,
|
|
458
503
|
reasoningProjected: evidence.reasoningProjected,
|
|
459
504
|
finishReason: finishReasonOf(rawFinishReason),
|
|
@@ -490,15 +535,35 @@ const executeModelOnce = async (
|
|
|
490
535
|
const rawChunks: unknown[] = [];
|
|
491
536
|
let streamError: unknown;
|
|
492
537
|
let outputObserved = false;
|
|
538
|
+
// The SDK enqueues each raw chunk before the deltas it yields, so the latest raw chunk's
|
|
539
|
+
// thought flag classifies the text deltas that follow ({§google-thought-response}).
|
|
540
|
+
let thoughtChunk = false;
|
|
541
|
+
let thoughtSeen = false;
|
|
542
|
+
let answer = "";
|
|
493
543
|
try {
|
|
494
544
|
for await (const part of result.fullStream) {
|
|
495
|
-
if (part.type === "raw")
|
|
545
|
+
if (part.type === "raw") {
|
|
546
|
+
rawChunks.push(part.rawValue);
|
|
547
|
+
thoughtChunk = rawChunkThought(part.rawValue);
|
|
548
|
+
}
|
|
496
549
|
if (part.type === "text-delta" && part.text.length > 0) {
|
|
497
550
|
outputObserved = true;
|
|
498
|
-
|
|
551
|
+
liftAttemptDeadline();
|
|
552
|
+
if (thoughtChunk) {
|
|
553
|
+
thoughtSeen = true;
|
|
554
|
+
const thought = unwrapThoughtDelta(part.text);
|
|
555
|
+
if (thought.length > 0) request.observeReasoning?.(thought);
|
|
556
|
+
} else {
|
|
557
|
+
const text = thoughtSeen && answer.length === 0 && part.text.startsWith(THOUGHT_CLOSE)
|
|
558
|
+
? part.text.slice(THOUGHT_CLOSE.length)
|
|
559
|
+
: part.text;
|
|
560
|
+
answer += text;
|
|
561
|
+
if (text.length > 0) request.observeText?.(text);
|
|
562
|
+
}
|
|
499
563
|
}
|
|
500
564
|
if (part.type === "reasoning-delta" && part.text.length > 0) {
|
|
501
565
|
outputObserved = true;
|
|
566
|
+
liftAttemptDeadline();
|
|
502
567
|
request.observeReasoning?.(part.text);
|
|
503
568
|
}
|
|
504
569
|
if (part.type === "error") streamError ??= part.error;
|
|
@@ -506,6 +571,8 @@ const executeModelOnce = async (
|
|
|
506
571
|
} catch (error) {
|
|
507
572
|
preserveStreamFailure(error, rawChunks, outputObserved);
|
|
508
573
|
throw error;
|
|
574
|
+
} finally {
|
|
575
|
+
liftAttemptDeadline();
|
|
509
576
|
}
|
|
510
577
|
if (streamError !== undefined) {
|
|
511
578
|
preserveStreamFailure(streamError, rawChunks, outputObserved);
|
|
@@ -513,7 +580,7 @@ const executeModelOnce = async (
|
|
|
513
580
|
}
|
|
514
581
|
const evidence = extractEvidence(rawChunks);
|
|
515
582
|
const accountingUsage = wireUsageEvidenceOf(rawChunks);
|
|
516
|
-
const content = await result.text;
|
|
583
|
+
const content = thoughtSeen ? answer : await result.text;
|
|
517
584
|
const reasoningText = evidence.reasoning || (await result.reasoningText) || "";
|
|
518
585
|
const rawFinishReason = await result.rawFinishReason;
|
|
519
586
|
const [response, providerMetadata, warnings] = await Promise.all([
|
|
@@ -671,13 +738,21 @@ const extractEvidence = (values: unknown[]): {
|
|
|
671
738
|
: { token: entry.token, logprob: entry.logprob, top });
|
|
672
739
|
}
|
|
673
740
|
}
|
|
674
|
-
const
|
|
741
|
+
const delta = recordOf(choice.delta);
|
|
742
|
+
const message = delta ?? recordOf(choice.message) ?? {};
|
|
675
743
|
for (const key of ["reasoning_content", "reasoning", "thinking"]) { // lexicon-allow: backend wire fields
|
|
676
744
|
if (typeof message[key] === "string") {
|
|
677
745
|
reasoningProjected = true;
|
|
678
746
|
reasoning += message[key];
|
|
679
747
|
}
|
|
680
748
|
}
|
|
749
|
+
if (googleThoughtFlagged(message) && typeof message.content === "string") {
|
|
750
|
+
const thought = delta !== null ? unwrapThoughtDelta(message.content) : splitGoogleThought(message.content)?.thought;
|
|
751
|
+
if (thought !== undefined) {
|
|
752
|
+
reasoningProjected = true;
|
|
753
|
+
reasoning += thought;
|
|
754
|
+
}
|
|
755
|
+
}
|
|
681
756
|
if (!Array.isArray(message.reasoning_details)) continue;
|
|
682
757
|
for (const value of message.reasoning_details) {
|
|
683
758
|
const detail = recordOf(value);
|
package/src/boundaries.test.ts
CHANGED
|
@@ -64,7 +64,7 @@ test("the OpenAI-compatible entrypoint excludes Node-owned provider machinery",
|
|
|
64
64
|
]));
|
|
65
65
|
});
|
|
66
66
|
|
|
67
|
-
test("the normalized-error entrypoint excludes transport and Node machinery", () => {
|
|
67
|
+
test("{§provider-runtime-neutral-errors} the normalized-error entrypoint excludes transport and Node machinery", () => {
|
|
68
68
|
assertRuntimeNeutralGraph("providerError.ts", new Set([
|
|
69
69
|
"notices.ts",
|
|
70
70
|
"providerError.ts",
|
package/src/capacity.test.ts
CHANGED
|
@@ -27,7 +27,7 @@ test("call-specific output tightening also tightens its reasoning subset", () =>
|
|
|
27
27
|
);
|
|
28
28
|
});
|
|
29
29
|
|
|
30
|
-
test("capacity applies independent input and combined-context limits", () => {
|
|
30
|
+
test("{§provider-capacity-admission} capacity applies independent input and combined-context limits", () => {
|
|
31
31
|
assert.equal(effectiveInputCapacity({
|
|
32
32
|
contextWindow: 100_000,
|
|
33
33
|
maxInputTokens: 70_000,
|
|
@@ -61,7 +61,7 @@ test("a known combined context must leave positive input capacity", () => {
|
|
|
61
61
|
);
|
|
62
62
|
});
|
|
63
63
|
|
|
64
|
-
test("only exact overflow rejects before provider I/O", () => {
|
|
64
|
+
test("{§provider-capacity-admission} only exact overflow rejects before provider I/O", () => {
|
|
65
65
|
const base = {
|
|
66
66
|
contextWindow: 100,
|
|
67
67
|
maxInputTokens: null,
|
|
@@ -865,7 +865,7 @@ test("cataloged unknown model fails unless its context is explicit", () => {
|
|
|
865
865
|
assert.deepEqual(provider?.supportedReasoningPolicies, ["off", "adaptive"]);
|
|
866
866
|
});
|
|
867
867
|
|
|
868
|
-
test("Models.dev is the only fallback rate table", async () => {
|
|
868
|
+
test("{§provider-monetary-evidence} Models.dev is the only fallback rate table", async () => {
|
|
869
869
|
mock.method(globalThis, "fetch", async () => new Response([
|
|
870
870
|
`data: ${JSON.stringify({
|
|
871
871
|
id: "response",
|
|
@@ -963,3 +963,48 @@ test("(#458) declared efforts union into the supported set under the models.dev-
|
|
|
963
963
|
// (#474) "max" joined the portable vocabulary; "off" still requires a declared "none".
|
|
964
964
|
assert.deepEqual(provider?.supportedReasoningPolicies, ["adaptive", "low", "high", "max"]);
|
|
965
965
|
});
|
|
966
|
+
|
|
967
|
+
test("{§provider-reasoning-style} {§google-reasoning-request}: a route-level thinking_config style asks Gemini behind Cloudflare's gateway for readable thoughts", async () => {
|
|
968
|
+
const bodies: Record<string, unknown>[] = [];
|
|
969
|
+
mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
|
|
970
|
+
bodies.push(JSON.parse(String(init?.body)) as Record<string, unknown>);
|
|
971
|
+
return new Response([
|
|
972
|
+
`data: ${JSON.stringify({
|
|
973
|
+
id: "gateway-gemini",
|
|
974
|
+
object: "chat.completion.chunk",
|
|
975
|
+
created: 1,
|
|
976
|
+
model: "gemini-3.8-flash",
|
|
977
|
+
choices: [{ index: 0, delta: { content: "done" }, finish_reason: "stop" }],
|
|
978
|
+
})}`,
|
|
979
|
+
"data: [DONE]",
|
|
980
|
+
].join("\n\n"), { headers: { "content-type": "text/event-stream" } });
|
|
981
|
+
});
|
|
982
|
+
const gatewayEnv = {
|
|
983
|
+
...env,
|
|
984
|
+
CLOUDFLARE_ACCOUNT_ID: "account",
|
|
985
|
+
CLOUDFLARE_API_KEY: "token",
|
|
986
|
+
PLURNK_PROVIDERS_CONTEXT_WINDOW: "1048576",
|
|
987
|
+
PLURNK_PROVIDERS_PROVIDER_CLOUDFLARE_WORKERS_AI_REASONING_STYLE: "effort_required",
|
|
988
|
+
PLURNK_PROVIDERS_REASONING_STYLE: "thinking_config",
|
|
989
|
+
};
|
|
990
|
+
const model = "google-ai-studio/gemini-3.8-flash";
|
|
991
|
+
const adaptive = catalogProviderFromEnv("cloudflare-workers-ai", { ...gatewayEnv, PLURNK_PROVIDERS_REASONING: "adaptive" }, model);
|
|
992
|
+
assert.deepEqual(adaptive?.supportedReasoningPolicies, ["adaptive", "low", "medium", "high"]);
|
|
993
|
+
await adaptive?.generate({ workerId: "gemini-adaptive", messages: [{ role: "user", content: "hello" }] });
|
|
994
|
+
const high = catalogProviderFromEnv("cloudflare-workers-ai", { ...gatewayEnv, PLURNK_PROVIDERS_REASONING: "high" }, model);
|
|
995
|
+
await high?.generate({ workerId: "gemini-high", messages: [{ role: "user", content: "hello" }] });
|
|
996
|
+
|
|
997
|
+
assert.deepEqual(bodies.map((body) => body.extra_body), [
|
|
998
|
+
{ google: { thinking_config: { include_thoughts: true } } },
|
|
999
|
+
{ google: { thinking_config: { include_thoughts: true, thinking_level: "high" } } },
|
|
1000
|
+
]);
|
|
1001
|
+
assert.ok(bodies.every((body) => !("reasoning_effort" in body)), "Gemini refuses reasoning_effort beside a thinking_config");
|
|
1002
|
+
assert.throws(
|
|
1003
|
+
() => catalogProviderFromEnv("cloudflare-workers-ai", { ...gatewayEnv, PLURNK_PROVIDERS_REASONING: "off" }, model),
|
|
1004
|
+
/reasoning policy 'off' is unsupported; supported policies: adaptive, low, medium, high/,
|
|
1005
|
+
);
|
|
1006
|
+
assert.throws(
|
|
1007
|
+
() => catalogProviderFromEnv("cloudflare-workers-ai", { ...gatewayEnv, PLURNK_PROVIDERS_REASONING_STYLE: "gemini" }, model),
|
|
1008
|
+
/cloudflare-workers-ai provider: PLURNK_PROVIDERS_REASONING_STYLE has invalid value "gemini"/,
|
|
1009
|
+
);
|
|
1010
|
+
});
|
package/src/catalogProvider.ts
CHANGED
|
@@ -38,19 +38,25 @@ import type { LanguageModel } from "ai";
|
|
|
38
38
|
import type { AiSdkProviderOptions, CacheAffinity } from "./AiSdkProvider.ts";
|
|
39
39
|
import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
|
|
40
40
|
|
|
41
|
+
// {§provider-reasoning-style} — the provider-wide declaration, unless the bare knob (alias-scopable:
|
|
42
|
+
// PLURNK_PROVIDERS_REASONING_STYLE_<alias>) names the wire for one route; one provider can serve
|
|
43
|
+
// models whose reasoning controls differ (Cloudflare's gateway hosts `@cf/…` and Gemini alike).
|
|
41
44
|
const reasoningStyleFromEnv = (
|
|
42
45
|
env: NodeJS.ProcessEnv,
|
|
43
46
|
name: string,
|
|
44
47
|
): ReasoningStyle | undefined => {
|
|
45
48
|
const prefix = name.replaceAll(/[^a-zA-Z0-9]/g, "_").toUpperCase();
|
|
46
|
-
const
|
|
49
|
+
const routeKey = "PLURNK_PROVIDERS_REASONING_STYLE";
|
|
50
|
+
const providerKey = `PLURNK_PROVIDERS_PROVIDER_${prefix}_REASONING_STYLE`;
|
|
51
|
+
const key = env[routeKey] !== undefined && env[routeKey].length > 0 ? routeKey : providerKey;
|
|
52
|
+
const value = env[key];
|
|
47
53
|
if (value === undefined || value.length === 0) return undefined;
|
|
48
54
|
const styles: readonly ReasoningStyle[] = [
|
|
49
55
|
"none", "think", "include_reasoning", "effort",
|
|
50
|
-
"effort_explicit", "effort_required", "thinking_effort", "template", "anthropic",
|
|
56
|
+
"effort_explicit", "effort_required", "thinking_effort", "thinking_config", "template", "anthropic",
|
|
51
57
|
];
|
|
52
58
|
if (!styles.includes(value as ReasoningStyle)) {
|
|
53
|
-
throw new Error(`${name} provider:
|
|
59
|
+
throw new Error(`${name} provider: ${key} has invalid value "${value}"`);
|
|
54
60
|
}
|
|
55
61
|
return value as ReasoningStyle;
|
|
56
62
|
};
|
|
@@ -167,6 +173,9 @@ const supportedReasoningPolicies = ({
|
|
|
167
173
|
// not, and a word the template does not know fails loudly on the first request.
|
|
168
174
|
if (style === "template") return REASONING_POLICIES;
|
|
169
175
|
if (info !== undefined && info.reasoning !== true) return activationPolicies;
|
|
176
|
+
// {§google-reasoning-request} — the declared wire's whole vocabulary: Gemini reasons
|
|
177
|
+
// unconditionally and takes exactly these levels.
|
|
178
|
+
if (style === "thinking_config") return reasoningWithoutOff;
|
|
170
179
|
if (info?.reasoningOptions !== undefined) {
|
|
171
180
|
return catalogSupportedReasoningPolicies({ info, native, style, declared });
|
|
172
181
|
}
|
|
@@ -44,7 +44,7 @@ test("an undifferentiated compatible endpoint receives no guessed prompt-cache f
|
|
|
44
44
|
return streamedChatResponse("ok");
|
|
45
45
|
});
|
|
46
46
|
|
|
47
|
-
const provider = await compatibleProviderFromEnv(
|
|
47
|
+
const provider = await compatibleProviderFromEnv(env, "local");
|
|
48
48
|
await provider.generate({
|
|
49
49
|
workerId: "worker-affinity",
|
|
50
50
|
messages: [{ role: "user", content: "hello" }],
|
|
@@ -65,7 +65,7 @@ test("the server-wide DRY-off floor emits no DRY request fields", async () => {
|
|
|
65
65
|
return streamedChatResponse("ok");
|
|
66
66
|
});
|
|
67
67
|
|
|
68
|
-
const provider = await compatibleProviderFromEnv(
|
|
68
|
+
const provider = await compatibleProviderFromEnv({
|
|
69
69
|
...env,
|
|
70
70
|
PLURNK_PROVIDERS_DRY_MULTIPLIER: "0",
|
|
71
71
|
// Stale or independently supplied shape values cannot activate DRY.
|
|
@@ -90,7 +90,7 @@ test("(#483) a detected llama-server rail admits the operator's stated effort",
|
|
|
90
90
|
if (url.endsWith("/props")) return new Response(JSON.stringify({ total_slots: 1 }));
|
|
91
91
|
throw new Error(`unexpected request ${url}`);
|
|
92
92
|
});
|
|
93
|
-
const provider = await compatibleProviderFromEnv(
|
|
93
|
+
const provider = await compatibleProviderFromEnv({ ...env, PLURNK_PROVIDERS_REASONING: "medium" }, "local");
|
|
94
94
|
assert.ok(provider.supportedReasoningPolicies.includes("medium"), "the template governs: medium is admitted on a llama-server rail");
|
|
95
95
|
assert.ok(provider.supportedReasoningPolicies.includes("low") && provider.supportedReasoningPolicies.includes("high"), "the whole policy vocabulary rides; the template refuses unknown words itself");
|
|
96
96
|
});
|
|
@@ -118,7 +118,7 @@ test("detected llama-server measures the complete chat request through input_tok
|
|
|
118
118
|
throw new Error(`unexpected request ${url}`);
|
|
119
119
|
});
|
|
120
120
|
|
|
121
|
-
const provider = await compatibleProviderFromEnv(
|
|
121
|
+
const provider = await compatibleProviderFromEnv(env, "local");
|
|
122
122
|
const messages = [
|
|
123
123
|
{ role: "system" as const, content: "system slot" },
|
|
124
124
|
{ role: "user" as const, content: "漢漢漢" },
|
|
@@ -134,7 +134,7 @@ test("detected llama-server measures the complete chat request through input_tok
|
|
|
134
134
|
assert.deepEqual(countBody?.chat_template_kwargs, { enable_thinking: false });
|
|
135
135
|
});
|
|
136
136
|
|
|
137
|
-
test("a missing llama-server input-token endpoint degrades explicitly, never to a claimed bound", async () => {
|
|
137
|
+
test("{§provider-prompt-measurement} a missing llama-server input-token endpoint degrades explicitly, never to a claimed bound", async () => {
|
|
138
138
|
mock.method(globalThis, "fetch", async (input: string | URL | Request) => {
|
|
139
139
|
const url = String(input);
|
|
140
140
|
if (url.endsWith("/models")) {
|
|
@@ -147,7 +147,7 @@ test("a missing llama-server input-token endpoint degrades explicitly, never to
|
|
|
147
147
|
throw new Error(`unexpected request ${url}`);
|
|
148
148
|
});
|
|
149
149
|
|
|
150
|
-
const provider = await compatibleProviderFromEnv(
|
|
150
|
+
const provider = await compatibleProviderFromEnv(env, "local");
|
|
151
151
|
assert.deepEqual(await provider.countPromptTokens([{ role: "user", content: "漢漢漢" }]), {
|
|
152
152
|
kind: "estimate",
|
|
153
153
|
tokens: 2,
|
|
@@ -16,7 +16,6 @@ import {
|
|
|
16
16
|
reasoningResponseStyleFromEnv,
|
|
17
17
|
} from "./env.ts";
|
|
18
18
|
import { providerSource } from "./notices.ts";
|
|
19
|
-
import { plurnkCostNormalizer } from "./accounting.ts";
|
|
20
19
|
import type { Provider } from "./types.ts";
|
|
21
20
|
import { emitWarningOnce } from "./warnings.ts";
|
|
22
21
|
|
|
@@ -27,22 +26,16 @@ type EndpointProbe = {
|
|
|
27
26
|
failed: boolean;
|
|
28
27
|
};
|
|
29
28
|
|
|
30
|
-
const
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
override
|
|
34
|
-
): string => {
|
|
35
|
-
const configured = override
|
|
36
|
-
?? (provider === "openai"
|
|
37
|
-
? env.OPENAI_BASE_URL ?? env.OPENAI_API_BASE
|
|
38
|
-
: env.PLURNK_BASE_URL);
|
|
29
|
+
const provider = "openai";
|
|
30
|
+
|
|
31
|
+
const chatUrl = (env: NodeJS.ProcessEnv, override?: string): string => {
|
|
32
|
+
const configured = override ?? env.OPENAI_BASE_URL ?? env.OPENAI_API_BASE;
|
|
39
33
|
if (configured === undefined || configured.length === 0) {
|
|
40
|
-
throw new Error(`${provider} provider:
|
|
34
|
+
throw new Error(`${provider} provider: OPENAI_BASE_URL or OPENAI_API_BASE must be set`);
|
|
41
35
|
}
|
|
42
36
|
const base = configured.replace(/\/+$/, "");
|
|
43
37
|
if (base.endsWith("/chat/completions")) return base;
|
|
44
|
-
|
|
45
|
-
return `${base}/chat/completions`;
|
|
38
|
+
return `${base.replace(/\/v1$/, "")}/v1/chat/completions`;
|
|
46
39
|
};
|
|
47
40
|
|
|
48
41
|
const probeModels = async (
|
|
@@ -114,18 +107,16 @@ const probeProps = async (
|
|
|
114
107
|
};
|
|
115
108
|
|
|
116
109
|
export const compatibleProviderFromEnv = async (
|
|
117
|
-
provider: "openai" | "plurnk",
|
|
118
110
|
env: NodeJS.ProcessEnv,
|
|
119
111
|
model: string,
|
|
120
112
|
baseUrlOverride?: string,
|
|
121
113
|
): Promise<Provider> => {
|
|
122
|
-
// The knobs remain universal and fail hard when malformed, but this local
|
|
123
|
-
//
|
|
124
|
-
// already owns slot affinity and the first-party endpoint receives worker metadata.
|
|
114
|
+
// The knobs remain universal and fail hard when malformed, but this local compatible
|
|
115
|
+
// route declares no vendor cache projection: llama-server already owns slot affinity.
|
|
125
116
|
cacheAffinityFromEnv(env, provider);
|
|
126
117
|
cacheWritePolicyFromEnv(env, provider);
|
|
127
|
-
const url = chatUrl(
|
|
128
|
-
const apiKey =
|
|
118
|
+
const url = chatUrl(env, baseUrlOverride);
|
|
119
|
+
const apiKey = env.OPENAI_API_KEY;
|
|
129
120
|
const headers: Record<string, string> = apiKey === undefined || apiKey.length === 0
|
|
130
121
|
? {}
|
|
131
122
|
: { Authorization: `Bearer ${apiKey}` };
|
|
@@ -145,8 +136,8 @@ export const compatibleProviderFromEnv = async (
|
|
|
145
136
|
throw new Error(`${provider} provider: PLURNK_PROVIDERS_LLAMA_SERVER must be "1", "0", or unset`);
|
|
146
137
|
}
|
|
147
138
|
const pinned = pinRaw === undefined || pinRaw === "" ? null : pinRaw === "1";
|
|
148
|
-
const llamaServer =
|
|
149
|
-
if (probe.failed && pinned === null
|
|
139
|
+
const llamaServer = pinned ?? probe.llamaServer;
|
|
140
|
+
if (probe.failed && pinned === null) {
|
|
150
141
|
emitWarningOnce(
|
|
151
142
|
`${provider} provider: llama-server detection failed after ${attempts} attempts; pin PLURNK_PROVIDERS_LLAMA_SERVER=1 when this is a llama-server`,
|
|
152
143
|
"PLURNK_PROBE_FAILED",
|
|
@@ -160,7 +151,7 @@ export const compatibleProviderFromEnv = async (
|
|
|
160
151
|
}
|
|
161
152
|
|
|
162
153
|
let grammarStyle: GrammarStyle = "none";
|
|
163
|
-
let reasoningStyle: ReasoningStyle =
|
|
154
|
+
let reasoningStyle: ReasoningStyle = "think";
|
|
164
155
|
let slotCount: number | null = null;
|
|
165
156
|
let eosText: string | undefined;
|
|
166
157
|
let tokenizeUrl: string | undefined;
|
|
@@ -210,17 +201,11 @@ export const compatibleProviderFromEnv = async (
|
|
|
210
201
|
dryBase: parseOptionalFloat(env.PLURNK_PROVIDERS_DRY_BASE, "PLURNK_PROVIDERS_DRY_BASE", provider, 0) ?? undefined,
|
|
211
202
|
dryAllowedLength: parseOptionalInt(env.PLURNK_PROVIDERS_DRY_ALLOWED_LENGTH, "PLURNK_PROVIDERS_DRY_ALLOWED_LENGTH", provider) ?? undefined,
|
|
212
203
|
repeatLastN: parseOptionalInt(env.PLURNK_PROVIDERS_REPEAT_LAST_N, "PLURNK_PROVIDERS_REPEAT_LAST_N", provider) ?? undefined,
|
|
213
|
-
tuningFloors: provider !== "plurnk",
|
|
214
204
|
retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", provider),
|
|
215
205
|
errorDetailLimit: parseRequiredInt(env.PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT, "PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT", provider),
|
|
216
206
|
source: providerSource(provider),
|
|
217
207
|
grammarStyle,
|
|
218
208
|
...dataCaptureFromEnv(env, provider),
|
|
219
|
-
firstPartyMetadata: provider === "plurnk",
|
|
220
|
-
normalizeCost: provider === "plurnk" ? plurnkCostNormalizer : undefined,
|
|
221
|
-
apiKeyRejectedMessage: provider === "plurnk"
|
|
222
|
-
? "PLURNK_API_KEY was rejected by plurnk.ai (invalid or expired)."
|
|
223
|
-
: undefined,
|
|
224
209
|
supportsSlotPinning: llamaServer,
|
|
225
210
|
slotCount,
|
|
226
211
|
eosText,
|
package/src/cost.test.ts
CHANGED
|
@@ -16,7 +16,7 @@ const usage: ProviderUsage = {
|
|
|
16
16
|
totalTokens: 2,
|
|
17
17
|
};
|
|
18
18
|
|
|
19
|
-
test("direct charged evidence wins over a Models.dev estimate", () => {
|
|
19
|
+
test("{§provider-monetary-evidence} direct charged evidence wins over a Models.dev estimate", () => {
|
|
20
20
|
const charged = {
|
|
21
21
|
kind: "charged",
|
|
22
22
|
amount: { amount: "0.0000042", currency: "XMR" },
|
|
@@ -28,7 +28,7 @@ test("direct charged evidence wins over a Models.dev estimate", () => {
|
|
|
28
28
|
assert.equal(providerCostUsd(charged), "0.73");
|
|
29
29
|
});
|
|
30
30
|
|
|
31
|
-
test("an exact zero estimate remains distinguishable from unknown cost", () => {
|
|
31
|
+
test("{§provider-cost} an exact zero estimate remains distinguishable from unknown cost", () => {
|
|
32
32
|
const zero = estimateProviderCost(usage, { input: 0, output: 0 }, "Models.dev");
|
|
33
33
|
const unknown = estimateProviderCost(usage, null, "Models.dev");
|
|
34
34
|
assert.deepEqual(zero, {
|
|
@@ -104,7 +104,7 @@ test("decimal aggregation is exact and becomes unknown if any request is unknown
|
|
|
104
104
|
]), null);
|
|
105
105
|
});
|
|
106
106
|
|
|
107
|
-
test("malformed charged money is rejected instead of coerced", () => {
|
|
107
|
+
test("{§provider-cost} malformed charged money is rejected instead of coerced", () => {
|
|
108
108
|
assert.throws(() => validateChargedCost({
|
|
109
109
|
kind: "charged",
|
|
110
110
|
amount: { amount: "1e3", currency: "usd" },
|
package/src/discover.test.ts
CHANGED
|
@@ -5,6 +5,11 @@ import os from "node:os";
|
|
|
5
5
|
import path from "node:path";
|
|
6
6
|
import { discover } from "./discover.ts";
|
|
7
7
|
|
|
8
|
+
// This file's fixtures are third-party packages, so it exercises the operator who admitted them
|
|
9
|
+
// ({§executor-trust}); the shipped panel admits only `@plurnk/*`. Tests of the gate itself state
|
|
10
|
+
// their own value below and override this one.
|
|
11
|
+
process.env.PLURNK_PLUGINS_TRUSTED_ONLY = "0";
|
|
12
|
+
|
|
8
13
|
// Create a temp dir and register its removal on the test context, so it is
|
|
9
14
|
// cleaned on a GREEN or RED run. A trailing rm after the assertions leaks the dir
|
|
10
15
|
// whenever one throws — thousands accumulate on a shared box at drill frequency.
|
|
@@ -166,13 +171,17 @@ const trustFixture = (t: TestContext) => buildModules(t, {
|
|
|
166
171
|
"@acme/acme-provider-foo": { name: "@acme/acme-provider-foo", plurnk: { kind: "provider", name: "foo" } },
|
|
167
172
|
});
|
|
168
173
|
|
|
169
|
-
test("trust gate OFF (
|
|
174
|
+
test("trust gate OFF ('' or '0'): every provider is trusted; unset defers to the panel", async (t) => {
|
|
170
175
|
const root = await trustFixture(t);
|
|
171
|
-
for (const gate of [
|
|
176
|
+
for (const gate of ["", "0"]) {
|
|
172
177
|
const { registry, skipped } = await discover({ cwd: root, env: { PLURNK_PLUGINS_TRUSTED_ONLY: gate } as NodeJS.ProcessEnv });
|
|
173
178
|
assert.deepEqual([...registry.keys()].sort(), ["foo", "native"]);
|
|
174
179
|
assert.equal(skipped.size, 0);
|
|
175
180
|
}
|
|
181
|
+
// An unset key is answered by @plurnk/plurnk-meta's own panel, which ships `1`.
|
|
182
|
+
const { registry, skipped } = await discover({ cwd: root, env: {} as NodeJS.ProcessEnv });
|
|
183
|
+
assert.deepEqual([...registry.keys()].sort(), ["native"], "only the first-party provider survives the shipped gate");
|
|
184
|
+
assert.equal(skipped.size, 1);
|
|
176
185
|
});
|
|
177
186
|
|
|
178
187
|
test("trust gate ON: @plurnk/* always trusted; third party declined → skipped, not registered", async (t) => {
|
package/src/env.test.ts
CHANGED
|
@@ -113,6 +113,7 @@ test("scopeEnvToAlias: suffixed knob wins, bare is the fallback, other aliases i
|
|
|
113
113
|
PLURNK_PROVIDERS_REASONING_turboderp: "high",
|
|
114
114
|
PLURNK_PROVIDERS_REASONING_BUDGET_TURBODERP: "4096", // case-folds like PLURNK_MODEL_ keys
|
|
115
115
|
PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE_TURBODERP: "think-tags",
|
|
116
|
+
PLURNK_PROVIDERS_REASONING_STYLE_turboderp: "thinking_config",
|
|
116
117
|
PLURNK_PROVIDERS_CONTEXT_WINDOW_turboderp: "8000",
|
|
117
118
|
PLURNK_PROVIDERS_OUTPUT_BUDGET_turboderp: "4096",
|
|
118
119
|
PLURNK_PROVIDERS_CONTEXT_WINDOW_other: "1",
|
|
@@ -121,9 +122,11 @@ test("scopeEnvToAlias: suffixed knob wins, bare is the fallback, other aliases i
|
|
|
121
122
|
assert.equal(scoped.PLURNK_PROVIDERS_REASONING, "high");
|
|
122
123
|
assert.equal(scoped.PLURNK_PROVIDERS_REASONING_BUDGET, "4096");
|
|
123
124
|
assert.equal(scoped.PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE, "think-tags");
|
|
125
|
+
assert.equal(scoped.PLURNK_PROVIDERS_REASONING_STYLE, "thinking_config");
|
|
124
126
|
assert.equal(scoped.PLURNK_PROVIDERS_CONTEXT_WINDOW, "8000");
|
|
125
127
|
assert.equal(scoped.PLURNK_PROVIDERS_OUTPUT_BUDGET, "4096");
|
|
126
128
|
assert.equal(scopeEnvToAlias(env, "plain").PLURNK_PROVIDERS_REASONING, "off"); // fallback intact
|
|
129
|
+
assert.equal(scopeEnvToAlias(env, "plain").PLURNK_PROVIDERS_REASONING_STYLE, undefined, "another alias's style never leaks");
|
|
127
130
|
});
|
|
128
131
|
|
|
129
132
|
test("scopeEnvToAlias: aliases with underscores resolve; a bare knob is never mistaken for a suffix", async () => {
|
package/src/env.ts
CHANGED
|
@@ -243,7 +243,7 @@ export const resolveGenerationEnvelopeFromEnv = (
|
|
|
243
243
|
// PLURNK_PROVIDERS_REASONING_BUDGET optional reasoning subset of the total
|
|
244
244
|
// output budget, used for tier/budget mapping where the backend supports it.
|
|
245
245
|
// The provider maps intent to the backend's mechanism; the consumer states
|
|
246
|
-
// intent, never mechanism.
|
|
246
|
+
// intent, never mechanism. working memory is separate from provider reasoning.
|
|
247
247
|
export type Reasoning = { mode: ReasoningPolicy; budget: number | null };
|
|
248
248
|
|
|
249
249
|
export const parseReasoningPolicy = (value: unknown, label: string): ReasoningPolicy => {
|
|
@@ -294,6 +294,7 @@ export const PROVIDERS_KNOBS = Object.freeze([
|
|
|
294
294
|
"PLURNK_PROVIDERS_COST",
|
|
295
295
|
"PLURNK_PROVIDERS_OUTPUT_BUDGET",
|
|
296
296
|
"PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE",
|
|
297
|
+
"PLURNK_PROVIDERS_REASONING_STYLE",
|
|
297
298
|
"PLURNK_PROVIDERS_REASONING_BUDGET",
|
|
298
299
|
"PLURNK_PROVIDERS_REASONING",
|
|
299
300
|
"PLURNK_PROVIDERS_CONTEXT_WINDOW",
|