@plurnk/plurnk-providers 1.18.0 → 1.19.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. package/.env.defaults +8 -4
  2. package/README.md +4 -16
  3. package/SPEC.md +41 -11
  4. package/dist/AiSdkProvider.d.ts +2 -4
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +7 -17
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/AiSdkRequestBody.d.ts +1 -4
  9. package/dist/AiSdkRequestBody.d.ts.map +1 -1
  10. package/dist/AiSdkRequestBody.js +10 -48
  11. package/dist/AiSdkRequestBody.js.map +1 -1
  12. package/dist/Mock.d.ts +2 -2
  13. package/dist/Mock.d.ts.map +1 -1
  14. package/dist/Mock.js +3 -3
  15. package/dist/Mock.js.map +1 -1
  16. package/dist/ProviderRegistry.js +2 -2
  17. package/dist/ProviderRegistry.js.map +1 -1
  18. package/dist/accounting.d.ts +0 -1
  19. package/dist/accounting.d.ts.map +1 -1
  20. package/dist/accounting.js +1 -7
  21. package/dist/accounting.js.map +1 -1
  22. package/dist/aiSdkTransport.d.ts.map +1 -1
  23. package/dist/aiSdkTransport.js +79 -7
  24. package/dist/aiSdkTransport.js.map +1 -1
  25. package/dist/catalogProvider.d.ts.map +1 -1
  26. package/dist/catalogProvider.js +13 -3
  27. package/dist/catalogProvider.js.map +1 -1
  28. package/dist/compatibleProvider.d.ts +1 -1
  29. package/dist/compatibleProvider.d.ts.map +1 -1
  30. package/dist/compatibleProvider.js +13 -25
  31. package/dist/compatibleProvider.js.map +1 -1
  32. package/dist/env.d.ts.map +1 -1
  33. package/dist/env.js +1 -0
  34. package/dist/env.js.map +1 -1
  35. package/dist/index.d.ts +1 -1
  36. package/dist/index.d.ts.map +1 -1
  37. package/dist/index.js +1 -1
  38. package/dist/index.js.map +1 -1
  39. package/dist/types.d.ts +0 -6
  40. package/dist/types.d.ts.map +1 -1
  41. package/package.json +7 -7
  42. package/src/AiSdkProvider.test.ts +97 -126
  43. package/src/AiSdkProvider.ts +8 -22
  44. package/src/AiSdkRequestBody.ts +10 -42
  45. package/src/Mock.test.ts +11 -3
  46. package/src/Mock.ts +3 -3
  47. package/src/ProviderRegistry.test.ts +0 -1
  48. package/src/ProviderRegistry.ts +1 -1
  49. package/src/accounting.test.ts +0 -11
  50. package/src/accounting.ts +0 -8
  51. package/src/aiSdkTransport.test.ts +63 -1
  52. package/src/aiSdkTransport.ts +82 -7
  53. package/src/boundaries.test.ts +1 -1
  54. package/src/capacity.test.ts +2 -2
  55. package/src/catalogProvider.test.ts +46 -1
  56. package/src/catalogProvider.ts +12 -3
  57. package/src/compatibleProvider.test.ts +6 -6
  58. package/src/compatibleProvider.ts +13 -28
  59. package/src/cost.test.ts +3 -3
  60. package/src/discover.test.ts +11 -2
  61. package/src/env.test.ts +3 -0
  62. package/src/env.ts +2 -1
  63. package/src/errors.test.ts +2 -2
  64. package/src/index.ts +0 -1
  65. package/src/inputModalities.test.ts +2 -1
  66. package/src/sdkModels.test.ts +1 -1
  67. package/src/types.ts +8 -35
package/src/accounting.ts CHANGED
@@ -1,5 +1,4 @@
1
1
  import type {
2
- ChargedCost,
3
2
  ProviderAccounting,
4
3
  ProviderCostNormalizer,
5
4
  ProviderRequestAccounting,
@@ -7,7 +6,6 @@ import type {
7
6
  } from "./types.ts";
8
7
  import {
9
8
  sumProviderCostsUsd,
10
- validateChargedCost,
11
9
  validateProviderCost,
12
10
  } from "./cost.ts";
13
11
  import { validateProviderUsage } from "./usage.ts";
@@ -82,12 +80,6 @@ const deepInfraCost: ProviderCostNormalizer = ({ usage }) => {
82
80
  };
83
81
  };
84
82
 
85
- // The first-party endpoint owns the direct charged-cost wire field.
86
- export const plurnkCostNormalizer: ProviderCostNormalizer = ({ charge }) => {
87
- if (charge === undefined) return undefined;
88
- return validateChargedCost(charge) as ChargedCost;
89
- };
90
-
91
83
  export const providerCostNormalizer = (
92
84
  sdkPackage: string,
93
85
  ): ProviderCostNormalizer | undefined => {
@@ -19,7 +19,7 @@ const request = {
19
19
  captureRawBody: false,
20
20
  };
21
21
 
22
- test("the transport performs exactly one physical request", async () => {
22
+ test("{§provider-sdk-boundary} the transport performs exactly one physical request", async () => {
23
23
  let calls = 0;
24
24
  await assert.rejects(
25
25
  executeOpenAICompatible({
@@ -416,3 +416,65 @@ test("inconsistent usage counters refuse normalization without failing the respo
416
416
  assert.equal(result.usageRefusal, undefined);
417
417
  });
418
418
  });
419
+
420
+ test("{§google-thought-response}: Gemini's flagged thought content is reasoning, never emission", async (t) => {
421
+ const chunk = (delta: Record<string, unknown>, extra: Record<string, unknown> = {}) => ({
422
+ id: "gemini-1", object: "chat.completion.chunk", created: 1, model: "gemini-3.8-flash",
423
+ choices: [{ index: 0, delta, finish_reason: null }], ...extra,
424
+ });
425
+ const thought = (content: string) => ({ role: "assistant", content, extra_content: { google: { thought: true } } });
426
+ const sse = (chunks: object[]) => `${chunks.map((c) => `data: ${JSON.stringify(c)}\n\n`).join("")}data: [DONE]\n\n`;
427
+
428
+ await t.test("streamed: flagged deltas are observed and settled as reasoning; the answer stays clean", async () => {
429
+ const reasoning: string[] = [];
430
+ const text: string[] = [];
431
+ const result = await executeOpenAICompatible({
432
+ ...request,
433
+ streaming: true,
434
+ observeReasoning: (delta) => reasoning.push(delta),
435
+ observeText: (delta) => text.push(delta),
436
+ fetch: async () => new Response(sse([
437
+ chunk(thought("<thought>**Checking 391**\n\n")),
438
+ chunk(thought("17 × 23 = 391.")),
439
+ chunk({ content: "</thought>No. " }),
440
+ chunk({ content: "391 = 17 × 23.", extra_content: { google: { thought_signature: "c2ln" } } }),
441
+ { ...chunk({}), choices: [{ index: 0, delta: {}, finish_reason: "stop" }], usage: { prompt_tokens: 5, completion_tokens: 8, total_tokens: 90 } },
442
+ ]), { headers: { "content-type": "text/event-stream" } }),
443
+ });
444
+ assert.equal(result.content, "No. 391 = 17 × 23.");
445
+ assert.equal(result.reasoning, "**Checking 391**\n\n17 × 23 = 391.");
446
+ assert.equal(result.reasoningProjected, true);
447
+ assert.deepEqual(reasoning, ["**Checking 391**\n\n", "17 × 23 = 391."]);
448
+ assert.deepEqual(text, ["No. ", "391 = 17 × 23."]);
449
+ });
450
+
451
+ await t.test("whole: the flagged message's leading <thought> wrapper is reasoning and the remainder is content", async () => {
452
+ const result = await executeOpenAICompatible({
453
+ ...request,
454
+ fetch: async () => new Response(JSON.stringify({
455
+ id: "gemini-2", object: "chat.completion", created: 1, model: "gemini-3.8-flash",
456
+ choices: [{ index: 0, finish_reason: "stop", message: {
457
+ role: "assistant",
458
+ content: "<thought>**Checking 391**\n17 × 23 = 391.</thought>No. It is divisible by 17 and 23.",
459
+ extra_content: { google: { thought: true, thought_signature: "c2ln" } },
460
+ } }],
461
+ usage: { prompt_tokens: 5, completion_tokens: 9, total_tokens: 90 },
462
+ }), { headers: { "content-type": "application/json" } }),
463
+ });
464
+ assert.equal(result.content, "No. It is divisible by 17 and 23.");
465
+ assert.equal(result.reasoning, "**Checking 391**\n17 × 23 = 391.");
466
+ assert.equal(result.reasoningProjected, true);
467
+ });
468
+
469
+ await t.test("unflagged content is never inspected for the wrapper", async () => {
470
+ const result = await executeOpenAICompatible({
471
+ ...request,
472
+ fetch: async () => new Response(JSON.stringify({
473
+ id: "plain", object: "chat.completion", created: 1, model: "m",
474
+ choices: [{ index: 0, finish_reason: "stop", message: { role: "assistant", content: "<thought>literal</thought>tail" } }],
475
+ }), { headers: { "content-type": "application/json" } }),
476
+ });
477
+ assert.equal(result.content, "<thought>literal</thought>tail");
478
+ assert.equal(result.reasoning, "");
479
+ });
480
+ });
@@ -162,6 +162,38 @@ const recordOf = (value: unknown): Record<string, unknown> | null =>
162
162
  ? value as Record<string, unknown>
163
163
  : null;
164
164
 
165
+ // {§google-thought-response} — Gemini behind an OpenAI-compatible endpoint returns its readable
166
+ // thought summary as ordinary `content` wrapped `<thought>…</thought>` and flagged
167
+ // `extra_content.google.thought`. Whole, the flagged message holds wrapper and answer; streamed,
168
+ // the flagged deltas hold `<thought>` and the thought, and the first unflagged delta opens with
169
+ // `</thought>` before the answer.
170
+ const googleThoughtFlagged = (message: Record<string, unknown> | null): boolean =>
171
+ recordOf(recordOf(message?.extra_content)?.google)?.thought === true;
172
+
173
+ const THOUGHT_OPEN = "<thought>";
174
+ const THOUGHT_CLOSE = "</thought>";
175
+ const splitGoogleThought = (content: string): { thought: string; answer: string } | null => {
176
+ if (!content.startsWith(THOUGHT_OPEN)) return null;
177
+ const close = content.indexOf(THOUGHT_CLOSE);
178
+ if (close === -1) return null;
179
+ return { thought: content.slice(THOUGHT_OPEN.length, close), answer: content.slice(close + THOUGHT_CLOSE.length) };
180
+ };
181
+
182
+ const unwrapThoughtDelta = (text: string): string => {
183
+ const opened = text.startsWith(THOUGHT_OPEN) ? text.slice(THOUGHT_OPEN.length) : text;
184
+ return opened.endsWith(THOUGHT_CLOSE) ? opened.slice(0, -THOUGHT_CLOSE.length) : opened;
185
+ };
186
+
187
+ const rawChunkThought = (value: unknown): boolean => {
188
+ const choices = recordOf(value)?.choices;
189
+ return Array.isArray(choices) && googleThoughtFlagged(recordOf(recordOf(choices[0])?.delta));
190
+ };
191
+
192
+ const wholeResponseThought = (values: readonly unknown[]): boolean => values.some((value) => {
193
+ const choices = recordOf(value)?.choices;
194
+ return Array.isArray(choices) && googleThoughtFlagged(recordOf(recordOf(choices[0])?.message));
195
+ });
196
+
165
197
  const metadataOf = (values: readonly unknown[]): Record<string, unknown> => {
166
198
  const metadata: Record<string, unknown> = {};
167
199
  for (const value of values) {
@@ -413,6 +445,18 @@ const executeModelOnce = async (
413
445
  ? { role: "assistant", content: chatMessageText(message) }
414
446
  : { role: "system", content: chatMessageText(message) };
415
447
  });
448
+ // {§provider-connectivity} — a streamed attempt's deadline holds only until semantic content
449
+ // flows; after that, stream-idle catches a stall and the operation deadline bounds the whole.
450
+ // A healthy stream still producing (long reasoning) is never cut off and its tokens wasted.
451
+ const attemptDeadline = request.streaming && request.fetchTimeoutMs > 0 ? new AbortController() : null;
452
+ const attemptTimer = attemptDeadline === null ? null : setTimeout(
453
+ () => attemptDeadline.abort(new ProviderTimeoutError("attempt", request.fetchTimeoutMs)),
454
+ request.fetchTimeoutMs,
455
+ );
456
+ const liftAttemptDeadline = (): void => { if (attemptTimer !== null) clearTimeout(attemptTimer); };
457
+ const abortSignal = attemptDeadline === null
458
+ ? request.signal
459
+ : request.signal === undefined ? attemptDeadline.signal : AbortSignal.any([request.signal, attemptDeadline.signal]);
416
460
  const common = {
417
461
  model,
418
462
  ...(instructions.length === 0 ? {} : { instructions }),
@@ -422,10 +466,10 @@ const executeModelOnce = async (
422
466
  // AiSdkProvider owns retries so every physical request is independently
423
467
  // observed and accounted. The SDK transport executes exactly once.
424
468
  maxRetries: 0,
425
- abortSignal: request.signal,
469
+ abortSignal,
426
470
  headers: request.headers,
427
471
  timeout: {
428
- ...(request.fetchTimeoutMs > 0 ? { totalMs: request.fetchTimeoutMs } : {}),
472
+ ...(request.fetchTimeoutMs > 0 && !request.streaming ? { totalMs: request.fetchTimeoutMs } : {}),
429
473
  ...(request.streaming
430
474
  && request.firstContentTimeoutMs !== undefined
431
475
  && request.firstContentTimeoutMs > 0
@@ -451,9 +495,10 @@ const executeModelOnce = async (
451
495
  const accountingUsage = wireUsageEvidenceOf(values);
452
496
  const reasoningText = evidence.reasoning || result.reasoningText || "";
453
497
  const rawFinishReason = result.rawFinishReason;
498
+ const thought = wholeResponseThought(values) ? splitGoogleThought(result.text) : null;
454
499
  return {
455
500
  model: result.response.modelId,
456
- content: result.text,
501
+ content: thought === null ? result.text : thought.answer,
457
502
  reasoning: reasoningText,
458
503
  reasoningProjected: evidence.reasoningProjected,
459
504
  finishReason: finishReasonOf(rawFinishReason),
@@ -490,15 +535,35 @@ const executeModelOnce = async (
490
535
  const rawChunks: unknown[] = [];
491
536
  let streamError: unknown;
492
537
  let outputObserved = false;
538
+ // The SDK enqueues each raw chunk before the deltas it yields, so the latest raw chunk's
539
+ // thought flag classifies the text deltas that follow ({§google-thought-response}).
540
+ let thoughtChunk = false;
541
+ let thoughtSeen = false;
542
+ let answer = "";
493
543
  try {
494
544
  for await (const part of result.fullStream) {
495
- if (part.type === "raw") rawChunks.push(part.rawValue);
545
+ if (part.type === "raw") {
546
+ rawChunks.push(part.rawValue);
547
+ thoughtChunk = rawChunkThought(part.rawValue);
548
+ }
496
549
  if (part.type === "text-delta" && part.text.length > 0) {
497
550
  outputObserved = true;
498
- request.observeText?.(part.text);
551
+ liftAttemptDeadline();
552
+ if (thoughtChunk) {
553
+ thoughtSeen = true;
554
+ const thought = unwrapThoughtDelta(part.text);
555
+ if (thought.length > 0) request.observeReasoning?.(thought);
556
+ } else {
557
+ const text = thoughtSeen && answer.length === 0 && part.text.startsWith(THOUGHT_CLOSE)
558
+ ? part.text.slice(THOUGHT_CLOSE.length)
559
+ : part.text;
560
+ answer += text;
561
+ if (text.length > 0) request.observeText?.(text);
562
+ }
499
563
  }
500
564
  if (part.type === "reasoning-delta" && part.text.length > 0) {
501
565
  outputObserved = true;
566
+ liftAttemptDeadline();
502
567
  request.observeReasoning?.(part.text);
503
568
  }
504
569
  if (part.type === "error") streamError ??= part.error;
@@ -506,6 +571,8 @@ const executeModelOnce = async (
506
571
  } catch (error) {
507
572
  preserveStreamFailure(error, rawChunks, outputObserved);
508
573
  throw error;
574
+ } finally {
575
+ liftAttemptDeadline();
509
576
  }
510
577
  if (streamError !== undefined) {
511
578
  preserveStreamFailure(streamError, rawChunks, outputObserved);
@@ -513,7 +580,7 @@ const executeModelOnce = async (
513
580
  }
514
581
  const evidence = extractEvidence(rawChunks);
515
582
  const accountingUsage = wireUsageEvidenceOf(rawChunks);
516
- const content = await result.text;
583
+ const content = thoughtSeen ? answer : await result.text;
517
584
  const reasoningText = evidence.reasoning || (await result.reasoningText) || "";
518
585
  const rawFinishReason = await result.rawFinishReason;
519
586
  const [response, providerMetadata, warnings] = await Promise.all([
@@ -671,13 +738,21 @@ const extractEvidence = (values: unknown[]): {
671
738
  : { token: entry.token, logprob: entry.logprob, top });
672
739
  }
673
740
  }
674
- const message = recordOf(choice.delta) ?? recordOf(choice.message) ?? {};
741
+ const delta = recordOf(choice.delta);
742
+ const message = delta ?? recordOf(choice.message) ?? {};
675
743
  for (const key of ["reasoning_content", "reasoning", "thinking"]) { // lexicon-allow: backend wire fields
676
744
  if (typeof message[key] === "string") {
677
745
  reasoningProjected = true;
678
746
  reasoning += message[key];
679
747
  }
680
748
  }
749
+ if (googleThoughtFlagged(message) && typeof message.content === "string") {
750
+ const thought = delta !== null ? unwrapThoughtDelta(message.content) : splitGoogleThought(message.content)?.thought;
751
+ if (thought !== undefined) {
752
+ reasoningProjected = true;
753
+ reasoning += thought;
754
+ }
755
+ }
681
756
  if (!Array.isArray(message.reasoning_details)) continue;
682
757
  for (const value of message.reasoning_details) {
683
758
  const detail = recordOf(value);
@@ -64,7 +64,7 @@ test("the OpenAI-compatible entrypoint excludes Node-owned provider machinery",
64
64
  ]));
65
65
  });
66
66
 
67
- test("the normalized-error entrypoint excludes transport and Node machinery", () => {
67
+ test("{§provider-runtime-neutral-errors} the normalized-error entrypoint excludes transport and Node machinery", () => {
68
68
  assertRuntimeNeutralGraph("providerError.ts", new Set([
69
69
  "notices.ts",
70
70
  "providerError.ts",
@@ -27,7 +27,7 @@ test("call-specific output tightening also tightens its reasoning subset", () =>
27
27
  );
28
28
  });
29
29
 
30
- test("capacity applies independent input and combined-context limits", () => {
30
+ test("{§provider-capacity-admission} capacity applies independent input and combined-context limits", () => {
31
31
  assert.equal(effectiveInputCapacity({
32
32
  contextWindow: 100_000,
33
33
  maxInputTokens: 70_000,
@@ -61,7 +61,7 @@ test("a known combined context must leave positive input capacity", () => {
61
61
  );
62
62
  });
63
63
 
64
- test("only exact overflow rejects before provider I/O", () => {
64
+ test("{§provider-capacity-admission} only exact overflow rejects before provider I/O", () => {
65
65
  const base = {
66
66
  contextWindow: 100,
67
67
  maxInputTokens: null,
@@ -865,7 +865,7 @@ test("cataloged unknown model fails unless its context is explicit", () => {
865
865
  assert.deepEqual(provider?.supportedReasoningPolicies, ["off", "adaptive"]);
866
866
  });
867
867
 
868
- test("Models.dev is the only fallback rate table", async () => {
868
+ test("{§provider-monetary-evidence} Models.dev is the only fallback rate table", async () => {
869
869
  mock.method(globalThis, "fetch", async () => new Response([
870
870
  `data: ${JSON.stringify({
871
871
  id: "response",
@@ -963,3 +963,48 @@ test("(#458) declared efforts union into the supported set under the models.dev-
963
963
  // (#474) "max" joined the portable vocabulary; "off" still requires a declared "none".
964
964
  assert.deepEqual(provider?.supportedReasoningPolicies, ["adaptive", "low", "high", "max"]);
965
965
  });
966
+
967
+ test("{§provider-reasoning-style} {§google-reasoning-request}: a route-level thinking_config style asks Gemini behind Cloudflare's gateway for readable thoughts", async () => {
968
+ const bodies: Record<string, unknown>[] = [];
969
+ mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
970
+ bodies.push(JSON.parse(String(init?.body)) as Record<string, unknown>);
971
+ return new Response([
972
+ `data: ${JSON.stringify({
973
+ id: "gateway-gemini",
974
+ object: "chat.completion.chunk",
975
+ created: 1,
976
+ model: "gemini-3.8-flash",
977
+ choices: [{ index: 0, delta: { content: "done" }, finish_reason: "stop" }],
978
+ })}`,
979
+ "data: [DONE]",
980
+ ].join("\n\n"), { headers: { "content-type": "text/event-stream" } });
981
+ });
982
+ const gatewayEnv = {
983
+ ...env,
984
+ CLOUDFLARE_ACCOUNT_ID: "account",
985
+ CLOUDFLARE_API_KEY: "token",
986
+ PLURNK_PROVIDERS_CONTEXT_WINDOW: "1048576",
987
+ PLURNK_PROVIDERS_PROVIDER_CLOUDFLARE_WORKERS_AI_REASONING_STYLE: "effort_required",
988
+ PLURNK_PROVIDERS_REASONING_STYLE: "thinking_config",
989
+ };
990
+ const model = "google-ai-studio/gemini-3.8-flash";
991
+ const adaptive = catalogProviderFromEnv("cloudflare-workers-ai", { ...gatewayEnv, PLURNK_PROVIDERS_REASONING: "adaptive" }, model);
992
+ assert.deepEqual(adaptive?.supportedReasoningPolicies, ["adaptive", "low", "medium", "high"]);
993
+ await adaptive?.generate({ workerId: "gemini-adaptive", messages: [{ role: "user", content: "hello" }] });
994
+ const high = catalogProviderFromEnv("cloudflare-workers-ai", { ...gatewayEnv, PLURNK_PROVIDERS_REASONING: "high" }, model);
995
+ await high?.generate({ workerId: "gemini-high", messages: [{ role: "user", content: "hello" }] });
996
+
997
+ assert.deepEqual(bodies.map((body) => body.extra_body), [
998
+ { google: { thinking_config: { include_thoughts: true } } },
999
+ { google: { thinking_config: { include_thoughts: true, thinking_level: "high" } } },
1000
+ ]);
1001
+ assert.ok(bodies.every((body) => !("reasoning_effort" in body)), "Gemini refuses reasoning_effort beside a thinking_config");
1002
+ assert.throws(
1003
+ () => catalogProviderFromEnv("cloudflare-workers-ai", { ...gatewayEnv, PLURNK_PROVIDERS_REASONING: "off" }, model),
1004
+ /reasoning policy 'off' is unsupported; supported policies: adaptive, low, medium, high/,
1005
+ );
1006
+ assert.throws(
1007
+ () => catalogProviderFromEnv("cloudflare-workers-ai", { ...gatewayEnv, PLURNK_PROVIDERS_REASONING_STYLE: "gemini" }, model),
1008
+ /cloudflare-workers-ai provider: PLURNK_PROVIDERS_REASONING_STYLE has invalid value "gemini"/,
1009
+ );
1010
+ });
@@ -38,19 +38,25 @@ import type { LanguageModel } from "ai";
38
38
  import type { AiSdkProviderOptions, CacheAffinity } from "./AiSdkProvider.ts";
39
39
  import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
40
40
 
41
+ // {§provider-reasoning-style} — the provider-wide declaration, unless the bare knob (alias-scopable:
42
+ // PLURNK_PROVIDERS_REASONING_STYLE_<alias>) names the wire for one route; one provider can serve
43
+ // models whose reasoning controls differ (Cloudflare's gateway hosts `@cf/…` and Gemini alike).
41
44
  const reasoningStyleFromEnv = (
42
45
  env: NodeJS.ProcessEnv,
43
46
  name: string,
44
47
  ): ReasoningStyle | undefined => {
45
48
  const prefix = name.replaceAll(/[^a-zA-Z0-9]/g, "_").toUpperCase();
46
- const value = env[`PLURNK_PROVIDERS_PROVIDER_${prefix}_REASONING_STYLE`];
49
+ const routeKey = "PLURNK_PROVIDERS_REASONING_STYLE";
50
+ const providerKey = `PLURNK_PROVIDERS_PROVIDER_${prefix}_REASONING_STYLE`;
51
+ const key = env[routeKey] !== undefined && env[routeKey].length > 0 ? routeKey : providerKey;
52
+ const value = env[key];
47
53
  if (value === undefined || value.length === 0) return undefined;
48
54
  const styles: readonly ReasoningStyle[] = [
49
55
  "none", "think", "include_reasoning", "effort",
50
- "effort_explicit", "effort_required", "thinking_effort", "template", "anthropic",
56
+ "effort_explicit", "effort_required", "thinking_effort", "thinking_config", "template", "anthropic",
51
57
  ];
52
58
  if (!styles.includes(value as ReasoningStyle)) {
53
- throw new Error(`${name} provider: PLURNK_PROVIDERS_PROVIDER_${prefix}_REASONING_STYLE has invalid value "${value}"`);
59
+ throw new Error(`${name} provider: ${key} has invalid value "${value}"`);
54
60
  }
55
61
  return value as ReasoningStyle;
56
62
  };
@@ -167,6 +173,9 @@ const supportedReasoningPolicies = ({
167
173
  // not, and a word the template does not know fails loudly on the first request.
168
174
  if (style === "template") return REASONING_POLICIES;
169
175
  if (info !== undefined && info.reasoning !== true) return activationPolicies;
176
+ // {§google-reasoning-request} — the declared wire's whole vocabulary: Gemini reasons
177
+ // unconditionally and takes exactly these levels.
178
+ if (style === "thinking_config") return reasoningWithoutOff;
170
179
  if (info?.reasoningOptions !== undefined) {
171
180
  return catalogSupportedReasoningPolicies({ info, native, style, declared });
172
181
  }
@@ -44,7 +44,7 @@ test("an undifferentiated compatible endpoint receives no guessed prompt-cache f
44
44
  return streamedChatResponse("ok");
45
45
  });
46
46
 
47
- const provider = await compatibleProviderFromEnv("openai", env, "local");
47
+ const provider = await compatibleProviderFromEnv(env, "local");
48
48
  await provider.generate({
49
49
  workerId: "worker-affinity",
50
50
  messages: [{ role: "user", content: "hello" }],
@@ -65,7 +65,7 @@ test("the server-wide DRY-off floor emits no DRY request fields", async () => {
65
65
  return streamedChatResponse("ok");
66
66
  });
67
67
 
68
- const provider = await compatibleProviderFromEnv("openai", {
68
+ const provider = await compatibleProviderFromEnv({
69
69
  ...env,
70
70
  PLURNK_PROVIDERS_DRY_MULTIPLIER: "0",
71
71
  // Stale or independently supplied shape values cannot activate DRY.
@@ -90,7 +90,7 @@ test("(#483) a detected llama-server rail admits the operator's stated effort",
90
90
  if (url.endsWith("/props")) return new Response(JSON.stringify({ total_slots: 1 }));
91
91
  throw new Error(`unexpected request ${url}`);
92
92
  });
93
- const provider = await compatibleProviderFromEnv("openai", { ...env, PLURNK_PROVIDERS_REASONING: "medium" }, "local");
93
+ const provider = await compatibleProviderFromEnv({ ...env, PLURNK_PROVIDERS_REASONING: "medium" }, "local");
94
94
  assert.ok(provider.supportedReasoningPolicies.includes("medium"), "the template governs: medium is admitted on a llama-server rail");
95
95
  assert.ok(provider.supportedReasoningPolicies.includes("low") && provider.supportedReasoningPolicies.includes("high"), "the whole policy vocabulary rides; the template refuses unknown words itself");
96
96
  });
@@ -118,7 +118,7 @@ test("detected llama-server measures the complete chat request through input_tok
118
118
  throw new Error(`unexpected request ${url}`);
119
119
  });
120
120
 
121
- const provider = await compatibleProviderFromEnv("openai", env, "local");
121
+ const provider = await compatibleProviderFromEnv(env, "local");
122
122
  const messages = [
123
123
  { role: "system" as const, content: "system slot" },
124
124
  { role: "user" as const, content: "漢漢漢" },
@@ -134,7 +134,7 @@ test("detected llama-server measures the complete chat request through input_tok
134
134
  assert.deepEqual(countBody?.chat_template_kwargs, { enable_thinking: false });
135
135
  });
136
136
 
137
- test("a missing llama-server input-token endpoint degrades explicitly, never to a claimed bound", async () => {
137
+ test("{§provider-prompt-measurement} a missing llama-server input-token endpoint degrades explicitly, never to a claimed bound", async () => {
138
138
  mock.method(globalThis, "fetch", async (input: string | URL | Request) => {
139
139
  const url = String(input);
140
140
  if (url.endsWith("/models")) {
@@ -147,7 +147,7 @@ test("a missing llama-server input-token endpoint degrades explicitly, never to
147
147
  throw new Error(`unexpected request ${url}`);
148
148
  });
149
149
 
150
- const provider = await compatibleProviderFromEnv("openai", env, "local");
150
+ const provider = await compatibleProviderFromEnv(env, "local");
151
151
  assert.deepEqual(await provider.countPromptTokens([{ role: "user", content: "漢漢漢" }]), {
152
152
  kind: "estimate",
153
153
  tokens: 2,
@@ -16,7 +16,6 @@ import {
16
16
  reasoningResponseStyleFromEnv,
17
17
  } from "./env.ts";
18
18
  import { providerSource } from "./notices.ts";
19
- import { plurnkCostNormalizer } from "./accounting.ts";
20
19
  import type { Provider } from "./types.ts";
21
20
  import { emitWarningOnce } from "./warnings.ts";
22
21
 
@@ -27,22 +26,16 @@ type EndpointProbe = {
27
26
  failed: boolean;
28
27
  };
29
28
 
30
- const chatUrl = (
31
- provider: "openai" | "plurnk",
32
- env: NodeJS.ProcessEnv,
33
- override?: string,
34
- ): string => {
35
- const configured = override
36
- ?? (provider === "openai"
37
- ? env.OPENAI_BASE_URL ?? env.OPENAI_API_BASE
38
- : env.PLURNK_BASE_URL);
29
+ const provider = "openai";
30
+
31
+ const chatUrl = (env: NodeJS.ProcessEnv, override?: string): string => {
32
+ const configured = override ?? env.OPENAI_BASE_URL ?? env.OPENAI_API_BASE;
39
33
  if (configured === undefined || configured.length === 0) {
40
- throw new Error(`${provider} provider: ${provider === "openai" ? "OPENAI_BASE_URL or OPENAI_API_BASE" : "PLURNK_BASE_URL"} must be set`);
34
+ throw new Error(`${provider} provider: OPENAI_BASE_URL or OPENAI_API_BASE must be set`);
41
35
  }
42
36
  const base = configured.replace(/\/+$/, "");
43
37
  if (base.endsWith("/chat/completions")) return base;
44
- if (provider === "openai") return `${base.replace(/\/v1$/, "")}/v1/chat/completions`;
45
- return `${base}/chat/completions`;
38
+ return `${base.replace(/\/v1$/, "")}/v1/chat/completions`;
46
39
  };
47
40
 
48
41
  const probeModels = async (
@@ -114,18 +107,16 @@ const probeProps = async (
114
107
  };
115
108
 
116
109
  export const compatibleProviderFromEnv = async (
117
- provider: "openai" | "plurnk",
118
110
  env: NodeJS.ProcessEnv,
119
111
  model: string,
120
112
  baseUrlOverride?: string,
121
113
  ): Promise<Provider> => {
122
- // The knobs remain universal and fail hard when malformed, but this local /
123
- // first-party compatible route declares no vendor cache projection. llama-server
124
- // already owns slot affinity and the first-party endpoint receives worker metadata.
114
+ // The knobs remain universal and fail hard when malformed, but this local compatible
115
+ // route declares no vendor cache projection: llama-server already owns slot affinity.
125
116
  cacheAffinityFromEnv(env, provider);
126
117
  cacheWritePolicyFromEnv(env, provider);
127
- const url = chatUrl(provider, env, baseUrlOverride);
128
- const apiKey = provider === "openai" ? env.OPENAI_API_KEY : env.PLURNK_API_KEY;
118
+ const url = chatUrl(env, baseUrlOverride);
119
+ const apiKey = env.OPENAI_API_KEY;
129
120
  const headers: Record<string, string> = apiKey === undefined || apiKey.length === 0
130
121
  ? {}
131
122
  : { Authorization: `Bearer ${apiKey}` };
@@ -145,8 +136,8 @@ export const compatibleProviderFromEnv = async (
145
136
  throw new Error(`${provider} provider: PLURNK_PROVIDERS_LLAMA_SERVER must be "1", "0", or unset`);
146
137
  }
147
138
  const pinned = pinRaw === undefined || pinRaw === "" ? null : pinRaw === "1";
148
- const llamaServer = provider === "openai" && (pinned ?? probe.llamaServer);
149
- if (probe.failed && pinned === null && provider === "openai") {
139
+ const llamaServer = pinned ?? probe.llamaServer;
140
+ if (probe.failed && pinned === null) {
150
141
  emitWarningOnce(
151
142
  `${provider} provider: llama-server detection failed after ${attempts} attempts; pin PLURNK_PROVIDERS_LLAMA_SERVER=1 when this is a llama-server`,
152
143
  "PLURNK_PROBE_FAILED",
@@ -160,7 +151,7 @@ export const compatibleProviderFromEnv = async (
160
151
  }
161
152
 
162
153
  let grammarStyle: GrammarStyle = "none";
163
- let reasoningStyle: ReasoningStyle = provider === "openai" ? "think" : "none";
154
+ let reasoningStyle: ReasoningStyle = "think";
164
155
  let slotCount: number | null = null;
165
156
  let eosText: string | undefined;
166
157
  let tokenizeUrl: string | undefined;
@@ -210,17 +201,11 @@ export const compatibleProviderFromEnv = async (
210
201
  dryBase: parseOptionalFloat(env.PLURNK_PROVIDERS_DRY_BASE, "PLURNK_PROVIDERS_DRY_BASE", provider, 0) ?? undefined,
211
202
  dryAllowedLength: parseOptionalInt(env.PLURNK_PROVIDERS_DRY_ALLOWED_LENGTH, "PLURNK_PROVIDERS_DRY_ALLOWED_LENGTH", provider) ?? undefined,
212
203
  repeatLastN: parseOptionalInt(env.PLURNK_PROVIDERS_REPEAT_LAST_N, "PLURNK_PROVIDERS_REPEAT_LAST_N", provider) ?? undefined,
213
- tuningFloors: provider !== "plurnk",
214
204
  retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", provider),
215
205
  errorDetailLimit: parseRequiredInt(env.PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT, "PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT", provider),
216
206
  source: providerSource(provider),
217
207
  grammarStyle,
218
208
  ...dataCaptureFromEnv(env, provider),
219
- firstPartyMetadata: provider === "plurnk",
220
- normalizeCost: provider === "plurnk" ? plurnkCostNormalizer : undefined,
221
- apiKeyRejectedMessage: provider === "plurnk"
222
- ? "PLURNK_API_KEY was rejected by plurnk.ai (invalid or expired)."
223
- : undefined,
224
209
  supportsSlotPinning: llamaServer,
225
210
  slotCount,
226
211
  eosText,
package/src/cost.test.ts CHANGED
@@ -16,7 +16,7 @@ const usage: ProviderUsage = {
16
16
  totalTokens: 2,
17
17
  };
18
18
 
19
- test("direct charged evidence wins over a Models.dev estimate", () => {
19
+ test("{§provider-monetary-evidence} direct charged evidence wins over a Models.dev estimate", () => {
20
20
  const charged = {
21
21
  kind: "charged",
22
22
  amount: { amount: "0.0000042", currency: "XMR" },
@@ -28,7 +28,7 @@ test("direct charged evidence wins over a Models.dev estimate", () => {
28
28
  assert.equal(providerCostUsd(charged), "0.73");
29
29
  });
30
30
 
31
- test("an exact zero estimate remains distinguishable from unknown cost", () => {
31
+ test("{§provider-cost} an exact zero estimate remains distinguishable from unknown cost", () => {
32
32
  const zero = estimateProviderCost(usage, { input: 0, output: 0 }, "Models.dev");
33
33
  const unknown = estimateProviderCost(usage, null, "Models.dev");
34
34
  assert.deepEqual(zero, {
@@ -104,7 +104,7 @@ test("decimal aggregation is exact and becomes unknown if any request is unknown
104
104
  ]), null);
105
105
  });
106
106
 
107
- test("malformed charged money is rejected instead of coerced", () => {
107
+ test("{§provider-cost} malformed charged money is rejected instead of coerced", () => {
108
108
  assert.throws(() => validateChargedCost({
109
109
  kind: "charged",
110
110
  amount: { amount: "1e3", currency: "usd" },
@@ -5,6 +5,11 @@ import os from "node:os";
5
5
  import path from "node:path";
6
6
  import { discover } from "./discover.ts";
7
7
 
8
+ // This file's fixtures are third-party packages, so it exercises the operator who admitted them
9
+ // ({§executor-trust}); the shipped panel admits only `@plurnk/*`. Tests of the gate itself state
10
+ // their own value below and override this one.
11
+ process.env.PLURNK_PLUGINS_TRUSTED_ONLY = "0";
12
+
8
13
  // Create a temp dir and register its removal on the test context, so it is
9
14
  // cleaned on a GREEN or RED run. A trailing rm after the assertions leaks the dir
10
15
  // whenever one throws — thousands accumulate on a shared box at drill frequency.
@@ -166,13 +171,17 @@ const trustFixture = (t: TestContext) => buildModules(t, {
166
171
  "@acme/acme-provider-foo": { name: "@acme/acme-provider-foo", plurnk: { kind: "provider", name: "foo" } },
167
172
  });
168
173
 
169
- test("trust gate OFF (unset/empty/0): every provider is trusted", async (t) => {
174
+ test("trust gate OFF ('' or '0'): every provider is trusted; unset defers to the panel", async (t) => {
170
175
  const root = await trustFixture(t);
171
- for (const gate of [undefined, "", "0"]) {
176
+ for (const gate of ["", "0"]) {
172
177
  const { registry, skipped } = await discover({ cwd: root, env: { PLURNK_PLUGINS_TRUSTED_ONLY: gate } as NodeJS.ProcessEnv });
173
178
  assert.deepEqual([...registry.keys()].sort(), ["foo", "native"]);
174
179
  assert.equal(skipped.size, 0);
175
180
  }
181
+ // An unset key is answered by @plurnk/plurnk-meta's own panel, which ships `1`.
182
+ const { registry, skipped } = await discover({ cwd: root, env: {} as NodeJS.ProcessEnv });
183
+ assert.deepEqual([...registry.keys()].sort(), ["native"], "only the first-party provider survives the shipped gate");
184
+ assert.equal(skipped.size, 1);
176
185
  });
177
186
 
178
187
  test("trust gate ON: @plurnk/* always trusted; third party declined → skipped, not registered", async (t) => {
package/src/env.test.ts CHANGED
@@ -113,6 +113,7 @@ test("scopeEnvToAlias: suffixed knob wins, bare is the fallback, other aliases i
113
113
  PLURNK_PROVIDERS_REASONING_turboderp: "high",
114
114
  PLURNK_PROVIDERS_REASONING_BUDGET_TURBODERP: "4096", // case-folds like PLURNK_MODEL_ keys
115
115
  PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE_TURBODERP: "think-tags",
116
+ PLURNK_PROVIDERS_REASONING_STYLE_turboderp: "thinking_config",
116
117
  PLURNK_PROVIDERS_CONTEXT_WINDOW_turboderp: "8000",
117
118
  PLURNK_PROVIDERS_OUTPUT_BUDGET_turboderp: "4096",
118
119
  PLURNK_PROVIDERS_CONTEXT_WINDOW_other: "1",
@@ -121,9 +122,11 @@ test("scopeEnvToAlias: suffixed knob wins, bare is the fallback, other aliases i
121
122
  assert.equal(scoped.PLURNK_PROVIDERS_REASONING, "high");
122
123
  assert.equal(scoped.PLURNK_PROVIDERS_REASONING_BUDGET, "4096");
123
124
  assert.equal(scoped.PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE, "think-tags");
125
+ assert.equal(scoped.PLURNK_PROVIDERS_REASONING_STYLE, "thinking_config");
124
126
  assert.equal(scoped.PLURNK_PROVIDERS_CONTEXT_WINDOW, "8000");
125
127
  assert.equal(scoped.PLURNK_PROVIDERS_OUTPUT_BUDGET, "4096");
126
128
  assert.equal(scopeEnvToAlias(env, "plain").PLURNK_PROVIDERS_REASONING, "off"); // fallback intact
129
+ assert.equal(scopeEnvToAlias(env, "plain").PLURNK_PROVIDERS_REASONING_STYLE, undefined, "another alias's style never leaks");
127
130
  });
128
131
 
129
132
  test("scopeEnvToAlias: aliases with underscores resolve; a bare knob is never mistaken for a suffix", async () => {
package/src/env.ts CHANGED
@@ -243,7 +243,7 @@ export const resolveGenerationEnvelopeFromEnv = (
243
243
  // PLURNK_PROVIDERS_REASONING_BUDGET optional reasoning subset of the total
244
244
  // output budget, used for tier/budget mapping where the backend supports it.
245
245
  // The provider maps intent to the backend's mechanism; the consumer states
246
- // intent, never mechanism. TASK inventory is separate from provider reasoning.
246
+ // intent, never mechanism. working memory is separate from provider reasoning.
247
247
  export type Reasoning = { mode: ReasoningPolicy; budget: number | null };
248
248
 
249
249
  export const parseReasoningPolicy = (value: unknown, label: string): ReasoningPolicy => {
@@ -294,6 +294,7 @@ export const PROVIDERS_KNOBS = Object.freeze([
294
294
  "PLURNK_PROVIDERS_COST",
295
295
  "PLURNK_PROVIDERS_OUTPUT_BUDGET",
296
296
  "PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE",
297
+ "PLURNK_PROVIDERS_REASONING_STYLE",
297
298
  "PLURNK_PROVIDERS_REASONING_BUDGET",
298
299
  "PLURNK_PROVIDERS_REASONING",
299
300
  "PLURNK_PROVIDERS_CONTEXT_WINDOW",