@plurnk/plurnk-providers 1.14.1 → 1.14.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +21 -9
- package/SPEC.md +54 -12
- package/dist/AiSdkProvider.d.ts +2 -2
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +24 -13
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +2 -0
- package/dist/Pool.js.map +1 -1
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +29 -32
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/capacity.d.ts +8 -0
- package/dist/capacity.d.ts.map +1 -1
- package/dist/capacity.js +21 -0
- package/dist/capacity.js.map +1 -1
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +8 -3
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +6 -3
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/errors.d.ts.map +1 -1
- package/dist/errors.js +23 -2
- package/dist/errors.js.map +1 -1
- package/dist/types.d.ts +1 -0
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js +9 -1
- package/dist/types.js.map +1 -1
- package/package.json +6 -6
- package/src/AiSdkProvider.test.ts +138 -112
- package/src/AiSdkProvider.ts +28 -17
- package/src/Pool.test.ts +2 -0
- package/src/Pool.ts +2 -0
- package/src/aiSdkTransport.test.ts +31 -16
- package/src/aiSdkTransport.ts +32 -34
- package/src/capacity.test.ts +33 -1
- package/src/capacity.ts +34 -0
- package/src/catalogProvider.test.ts +22 -0
- package/src/catalogProvider.ts +7 -2
- package/src/compatibleProvider.test.ts +12 -0
- package/src/compatibleProvider.ts +6 -3
- package/src/errors.test.ts +1 -0
- package/src/errors.ts +22 -1
- package/src/types.ts +12 -1
package/dist/types.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"types.d.ts","sourceRoot":"","sources":["../src/types.ts"],"names":[],"mappings":"AAIA,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,cAAc,CAAC;AACnD,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,IAAI,CAAC;AACxC,OAAO,KAAK,EACR,iBAAiB,EACjB,wBAAwB,EACxB,uBAAuB,EAC1B,MAAM,qBAAqB,CAAC;AAC7B,OAAO,KAAK,EACR,YAAY,EACZ,yBAAyB,EACzB,uBAAuB,EACvB,eAAe,EAClB,MAAM,0BAA0B,CAAC;AAElC,YAAY,EACR,kBAAkB,EAClB,YAAY,EACZ,yBAAyB,EACzB,uBAAuB,EACvB,uBAAuB,EACvB,yBAAyB,EACzB,aAAa,EACb,eAAe,GAClB,MAAM,0BAA0B,CAAC;AAElC,qBAAa,+BAAgC,SAAQ,KAAK;IACtD,QAAQ,CAAC,MAAM,EAAE,eAAe,CAAC;IACjC,QAAQ,CAAC,SAAS,EAAE,SAAS,eAAe,EAAE,CAAC;IAE/C,YAAY,MAAM,EAAE,MAAM,EAAE,MAAM,EAAE,eAAe,EAAE,SAAS,EAAE,SAAS,eAAe,EAAE,
|
|
1
|
+
{"version":3,"file":"types.d.ts","sourceRoot":"","sources":["../src/types.ts"],"names":[],"mappings":"AAIA,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,cAAc,CAAC;AACnD,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,IAAI,CAAC;AACxC,OAAO,KAAK,EACR,iBAAiB,EACjB,wBAAwB,EACxB,uBAAuB,EAC1B,MAAM,qBAAqB,CAAC;AAC7B,OAAO,KAAK,EACR,YAAY,EACZ,yBAAyB,EACzB,uBAAuB,EACvB,eAAe,EAClB,MAAM,0BAA0B,CAAC;AAElC,YAAY,EACR,kBAAkB,EAClB,YAAY,EACZ,yBAAyB,EACzB,uBAAuB,EACvB,uBAAuB,EACvB,yBAAyB,EACzB,aAAa,EACb,eAAe,GAClB,MAAM,0BAA0B,CAAC;AAElC,qBAAa,+BAAgC,SAAQ,KAAK;IACtD,QAAQ,CAAC,MAAM,EAAE,eAAe,CAAC;IACjC,QAAQ,CAAC,SAAS,EAAE,SAAS,eAAe,EAAE,CAAC;IAE/C,YAAY,MAAM,EAAE,MAAM,EAAE,MAAM,EAAE,eAAe,EAAE,SAAS,EAAE,SAAS,eAAe,EAAE,EAazF;CACJ;AAED,MAAM,WAAW,WAAW;IACxB,IAAI,EAAE,QAAQ,GAAG,MAAM,GAAG,WAAW,CAAC;IACtC,OAAO,EAAE,MAAM,CAAC;CACnB;AAKD,MAAM,MAAM,gBAAgB,GAAG,UAAU,GAAG,MAAM,CAAC;AAInD,MAAM,MAAM,sBAAsB,GAC5B;IACE,QAAQ,CAAC,IAAI,EAAE,OAAO,GAAG,aAAa,CAAC;IACvC,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;IACxB,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;CAC3B,GACC;IACE,QAAQ,CAAC,IAAI,EAAE,UAAU,CAAC;IAC1B,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;IACxB,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;IACxB,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;CAC3B,GACC;IACE,QAAQ,CAAC,IAAI,EAAE,aAAa,CAAC;IAC7B,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;IACxB,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;CAC3B,CAAC;AAEN,MAAM,MAAM,+BAA+B,GAAG,OAAO,GAAG,OAAO,GAAG,QAAQ,CAAC;AAK3E,MAAM,WAAW,uBAAuB;IACpC,QAAQ,CAAC,QAAQ,EAAE,+BAA+B,CAAC;IACnD,QAAQ,CAAC,aAAa,EAAE,MAAM,GAAG,IAAI,CAAC;IACtC,QAAQ,CAAC,cAAc,EAAE,MAAM,GAAG,IAAI,CAAC;IACvC,QAAQ,CAAC,eAAe,EAAE,MAAM,GAAG,IAAI,CAAC;IACxC,QAAQ,CAAC,YAAY,EAAE,MAAM,GAAG,IAAI,CAAC;IACrC,QAAQ,CAAC,eAAe,EAAE,MAAM,GAAG,IAAI,CAAC;IACxC,QAAQ,CAAC,aAAa,EAAE,MAAM,GAAG,IAAI,CAAC;IAGtC,QAAQ,CAAC,WAAW,EAAE,MAAM,GAAG,IAAI,CAAC;IACpC,QAAQ,CAAC,MAAM,EAAE,sBAAsB,CAAC;CAC3C;AAED,MAAM,MAAM,WAAW,GAAG,OAAO,CAAC,YAAY,EAAE;IAAE,IAAI,EAAE,SAAS,CAAA;CAAE,CAAC,CAAC;AAKrE,MAAM,WAAW,sBAAsB;IACnC,QAAQ,CAAC,gBAAgB,CAAC,EAAE,OAAO,CAAC;IAGpC,QAAQ,CAAC,MAAM,CAAC,EAAE,OAAO,CAAC;IAI1B,QAAQ,CAAC,KAAK,CAAC,EAAE,OAAO,CAAC;IACzB,QAAQ,CAAC,QAAQ,EAAE;QACf,QAAQ,CAAC,EAAE,CAAC,EAAE,MAAM,CAAC;QACrB,QAAQ,CAAC,OAAO,CAAC,EAAE,QAAQ,CAAC,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC,CAAC;KACvD,CAAC;CACL;AAED,MAAM,MAAM,sBAAsB,GAAG,CACjC,QAAQ,EAAE,sBAAsB,KAC/B,YAAY,GAAG,SAAS,CAAC;AAK9B,MAAM,MAAM,yBAAyB,GAAG,CAAC,KAAK,EAAE,MAAM,KAAK,IAAI,CAAC;AAIhE,MAAM,MAAM,YAAY,GAAG,MAAM,GAAG,QAAQ,GAAG,YAAY,GAAG,gBAAgB,GAAG,IAAI,CAAC;AACtF,MAAM,MAAM,2BAA2B,GAAG,YAAY,GAAG,sBAAsB,CAAC;AAUhF,MAAM,WAAW,gBAAgB;IAC7B,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,QAAQ,CAAC,OAAO,EAAE,MAAM,CAAC;CAC5B;AACD,MAAM,WAAW,YAAY;IACzB,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,QAAQ,CAAC,OAAO,EAAE,MAAM,CAAC;IACzB,QAAQ,CAAC,GAAG,CAAC,EAAE,SAAS,gBAAgB,EAAE,CAAC;CAC9C;AAKD,MAAM,WAAW,8BAA8B;IAC3C,QAAQ,CAAC,EAAE,EAAE,MAAM,GAAG,IAAI,CAAC;IAC3B,QAAQ,CAAC,OAAO,EAAE,MAAM,CAAC;IACzB,QAAQ,CAAC,SAAS,EAAE,aAAa,CAAC;QAAE,IAAI,EAAE,MAAM,CAAC;QAAC,MAAM,EAAE,MAAM,GAAG,IAAI,CAAA;KAAE,CAAC,CAAC;CAC9E;AAED,MAAM,WAAW,iBAAiB,CAAC,OAAO,SAAS,2BAA2B,GAAG,YAAY;IACzF,QAAQ,CAAC,OAAO,EAAE,MAAM,CAAC;IACzB,QAAQ,CAAC,SAAS,EAAE,MAAM,GAAG,IAAI,CAAC;IAElC,QAAQ,CAAC,kBAAkB,CAAC,EAAE,aAAa,CAAC,8BAA8B,CAAC,CAAC;IAC5E,QAAQ,CAAC,YAAY,EAAE,OAAO,CAAC;IAC/B,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IAIvB,QAAQ,CAAC,QAAQ,CAAC,EAAE,SAAS,YAAY,EAAE,CAAC;IAE5C,QAAQ,CAAC,WAAW,CAAC,EAAE,MAAM,CAAC;CACjC;AAED,MAAM,WAAW,eAAe;IAI5B,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,QAAQ,CAAC,YAAY,EAAE,MAAM,CAAC;IAC9B,QAAQ,CAAC,WAAW,EAAE,OAAO,CAAC;CACjC;AAED,MAAM,WAAW,gBAAgB,CAAC,OAAO,SAAS,2BAA2B,GAAG,YAAY;IACxF,QAAQ,CAAC,SAAS,EAAE,iBAAiB,CAAC,OAAO,CAAC,CAAC;IAC/C,QAAQ,CAAC,YAAY,EAAE,OAAO,CAAC;IAG/B,QAAQ,CAAC,UAAU,EAAE,SAAS,yBAAyB,EAAE,CAAC;IAC1D,QAAQ,CAAC,QAAQ,EAAE,uBAAuB,CAAC;IAE3C,QAAQ,CAAC,eAAe,CAAC,EAAE,eAAe,CAAC;IAO3C,QAAQ,CAAC,IAAI,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;IAOxC,QAAQ,CAAC,OAAO,CAAC,EAAE,OAAO,CAAC;IAI3B,QAAQ,CAAC,OAAO,CAAC,EAAE,SAAS,cAAc,EAAE,CAAC;CAChD;AAED,MAAM,MAAM,eAAe,GAAG,gBAAgB,CAAC,2BAA2B,CAAC,CAAC;AAE5E,MAAM,WAAW,oBAAoB;IACjC,QAAQ,CAAC,QAAQ,EAAE,WAAW,EAAE,CAAC;IACjC,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IAC1B,QAAQ,CAAC,eAAe,CAAC,EAAE,MAAM,CAAC;IAClC,QAAQ,CAAC,MAAM,CAAC,EAAE,WAAW,CAAC;IAC9B,QAAQ,CAAC,OAAO,CAAC,EAAE,MAAM,CAAC;IAC1B,QAAQ,CAAC,eAAe,CAAC,EAAE,MAAM,CAAC;IAClC,QAAQ,CAAC,YAAY,CAAC,EAAE,MAAM,EAAE,CAAC;IACjC,QAAQ,CAAC,MAAM,CAAC,EAAE,MAAM,CAAC;IACzB,QAAQ,CAAC,OAAO,CAAC,EAAE,MAAM,CAAC;IAC1B,QAAQ,CAAC,WAAW,CAAC,EAAE,MAAM,CAAC;IAC9B,QAAQ,CAAC,IAAI,CAAC,EAAE,MAAM,CAAC;IACvB,QAAQ,CAAC,IAAI,CAAC,EAAE,MAAM,CAAC;IACvB,QAAQ,CAAC,QAAQ,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;IAC5C,QAAQ,CAAC,cAAc,CAAC,EAAE,uBAAuB,CAAC;IAClD,QAAQ,CAAC,gBAAgB,CAAC,EAAE,yBAAyB,CAAC;IACtD,QAAQ,CAAC,QAAQ,CAAC,EAAE,gBAAgB,CAAC;CACxC;AAED,MAAM,WAAW,QAAQ;IAGrB,YAAY,CAAC,CAAC,OAAO,EAAE,wBAAwB,GAAG,iBAAiB,CAAC;IAsDpE,QAAQ,CAAC,IAAI,EAAE,oBAAoB,GAAG,OAAO,CAAC,gBAAgB,CAAC,CAAC;IAIhE,QAAQ,CAAC,aAAa,EAAE,MAAM,GAAG,IAAI,CAAC;IACtC,QAAQ,CAAC,cAAc,EAAE,MAAM,GAAG,IAAI,CAAC;IACvC,QAAQ,CAAC,eAAe,EAAE,MAAM,GAAG,IAAI,CAAC;IACxC,QAAQ,CAAC,YAAY,EAAE,MAAM,GAAG,IAAI,CAAC;IACrC,QAAQ,CAAC,eAAe,EAAE,MAAM,GAAG,IAAI,CAAC;IAExC,QAAQ,CAAC,0BAA0B,EAAE,SAAS,eAAe,EAAE,CAAC;IAChE,QAAQ,CAAC,aAAa,EAAE,MAAM,GAAG,IAAI,CAAC;IACtC,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IAMvB,QAAQ,CAAC,WAAW,CAAC,EAAE,MAAM,CAAC;IAK9B,QAAQ,CAAC,gBAAgB,CAAC,EAAE,OAAO,CAAC;IAQpC,QAAQ,CAAC,oBAAoB,CAAC,EAAE,OAAO,CAAC;IAKxC,qBAAqB,CACjB,QAAQ,EAAE,SAAS,WAAW,EAAE,EAChC,eAAe,CAAC,EAAE,MAAM,EACxB,MAAM,CAAC,EAAE,WAAW,GACrB,OAAO,CAAC,uBAAuB,CAAC,CAAC;IAIpC,iBAAiB,CAAC,QAAQ,EAAE,SAAS,WAAW,EAAE,EAAE,MAAM,CAAC,EAAE,WAAW,GAAG,OAAO,CAAC,sBAAsB,CAAC,CAAC;IAM3G,QAAQ,CAAC,CAAC,IAAI,EAAE,MAAM,GAAG,OAAO,CAAC,MAAM,EAAE,CAAC,CAAC;CAC9C;AAUD,MAAM,WAAW,eAAe;IAC5B,QAAQ,CAAC,OAAO,CAAC,EAAE,MAAM,CAAC;CAC7B;AAID,MAAM,WAAW,mBAAoB,SAAQ,uBAAuB;IAChE,aAAa,CAAC,KAAK,EAAE,MAAM,GAAG,aAAa,CAAC;CAC/C"}
|
package/dist/types.js
CHANGED
|
@@ -5,7 +5,15 @@ export class UnsupportedReasoningPolicyError extends Error {
|
|
|
5
5
|
policy;
|
|
6
6
|
supported;
|
|
7
7
|
constructor(source, policy, supported) {
|
|
8
|
-
|
|
8
|
+
// A graded-effort refusal names the operator's declaration lever (#472):
|
|
9
|
+
// the catalog is authoritative per model, and where a provider's API
|
|
10
|
+
// accepts more than its catalog entry lists, the remedy is the
|
|
11
|
+
// operator's, never a vendored fact in shipped defaults.
|
|
12
|
+
const providerName = /^provider:(.+)$/.exec(source)?.[1];
|
|
13
|
+
const lever = policy === "off" || policy === "adaptive"
|
|
14
|
+
? ""
|
|
15
|
+
: ` If the provider's API documents this effort beyond its catalog entry, declare it: PLURNK_PROVIDERS_PROVIDER_${(providerName ?? "<NAME>").replaceAll(/[^a-zA-Z0-9<>]/g, "_").toUpperCase()}_REASONING_EFFORTS=${policy} (comma-separated).`;
|
|
16
|
+
super(`${source}: reasoning policy '${policy}' is unsupported; supported policies: ${supported.join(", ")}.${lever}`);
|
|
9
17
|
this.name = "UnsupportedReasoningPolicyError";
|
|
10
18
|
this.policy = policy;
|
|
11
19
|
this.supported = supported;
|
package/dist/types.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"types.js","sourceRoot":"","sources":["../src/types.ts"],"names":[],"mappings":"AAAA,wEAAwE;AACxE,6EAA6E;AAC7E,iCAAiC;AA2BjC,MAAM,OAAO,+BAAgC,SAAQ,KAAK;IAC7C,MAAM,CAAkB;IACxB,SAAS,CAA6B;IAE/C,YAAY,MAAc,EAAE,MAAuB,EAAE,SAAqC;QACtF,KAAK,CAAC,GAAG,MAAM,uBAAuB,MAAM,yCAAyC,SAAS,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;
|
|
1
|
+
{"version":3,"file":"types.js","sourceRoot":"","sources":["../src/types.ts"],"names":[],"mappings":"AAAA,wEAAwE;AACxE,6EAA6E;AAC7E,iCAAiC;AA2BjC,MAAM,OAAO,+BAAgC,SAAQ,KAAK;IAC7C,MAAM,CAAkB;IACxB,SAAS,CAA6B;IAE/C,YAAY,MAAc,EAAE,MAAuB,EAAE,SAAqC;QACtF,yEAAyE;QACzE,qEAAqE;QACrE,+DAA+D;QAC/D,yDAAyD;QACzD,MAAM,YAAY,GAAG,iBAAiB,CAAC,IAAI,CAAC,MAAM,CAAC,EAAE,CAAC,CAAC,CAAC,CAAC;QACzD,MAAM,KAAK,GAAG,MAAM,KAAK,KAAK,IAAI,MAAM,KAAK,UAAU;YACnD,CAAC,CAAC,EAAE;YACJ,CAAC,CAAC,gHAAgH,CAAC,YAAY,IAAI,QAAQ,CAAC,CAAC,UAAU,CAAC,iBAAiB,EAAE,GAAG,CAAC,CAAC,WAAW,EAAE,sBAAsB,MAAM,qBAAqB,CAAC;QACnP,KAAK,CAAC,GAAG,MAAM,uBAAuB,MAAM,yCAAyC,SAAS,CAAC,IAAI,CAAC,IAAI,CAAC,IAAI,KAAK,EAAE,CAAC,CAAC;QACtH,IAAI,CAAC,IAAI,GAAG,iCAAiC,CAAC;QAC9C,IAAI,CAAC,MAAM,GAAG,MAAM,CAAC;QACrB,IAAI,CAAC,SAAS,GAAG,SAAS,CAAC;IAC/B,CAAC;CACJ"}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@plurnk/plurnk-providers",
|
|
3
|
-
"version": "1.14.
|
|
3
|
+
"version": "1.14.2",
|
|
4
4
|
"description": "PLURNK's stable model-provider contract and AI SDK adapter.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"plurnk",
|
|
@@ -84,11 +84,11 @@
|
|
|
84
84
|
"@ai-sdk/togetherai": "^3.0.0",
|
|
85
85
|
"@ai-sdk/xai": "^4.0.0",
|
|
86
86
|
"@openrouter/ai-sdk-provider": "^3.0.0",
|
|
87
|
-
"@plurnk/gbnf": "1.14.
|
|
88
|
-
"@plurnk/plurnk-aliases": "1.14.
|
|
89
|
-
"@plurnk/plurnk-contracts": "1.14.
|
|
90
|
-
"@plurnk/plurnk-meta": "1.14.
|
|
91
|
-
"@plurnk/plurnk-models": "1.14.
|
|
87
|
+
"@plurnk/gbnf": "1.14.2",
|
|
88
|
+
"@plurnk/plurnk-aliases": "1.14.2",
|
|
89
|
+
"@plurnk/plurnk-contracts": "1.14.2",
|
|
90
|
+
"@plurnk/plurnk-meta": "1.14.2",
|
|
91
|
+
"@plurnk/plurnk-models": "1.14.2",
|
|
92
92
|
"ai": "^7.0.37",
|
|
93
93
|
"zod": "^4.4.3"
|
|
94
94
|
},
|
|
@@ -192,7 +192,7 @@ test("per-instance fetch owns tokenization and retry attempts", async () => {
|
|
|
192
192
|
return new Response(JSON.stringify({ tokens: [10, 20] }), { status: 200 });
|
|
193
193
|
}
|
|
194
194
|
generationAttempts++;
|
|
195
|
-
if (generationAttempts === 1) return new Response("busy", { status: 503 });
|
|
195
|
+
if (generationAttempts === 1) return new Response("busy", { status: 503, headers: { "retry-after": "0" } });
|
|
196
196
|
return new Response(sseStream([
|
|
197
197
|
{ model: "m", choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] },
|
|
198
198
|
{ choices: [], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } },
|
|
@@ -214,6 +214,63 @@ test("per-instance fetch owns tokenization and retry attempts", async () => {
|
|
|
214
214
|
]);
|
|
215
215
|
});
|
|
216
216
|
|
|
217
|
+
test("(#482) an exact-counting transport flexes the wire grant; the floor holds at capacity", async () => {
|
|
218
|
+
const seen: Array<Record<string, unknown>> = [];
|
|
219
|
+
let promptTokens = 100;
|
|
220
|
+
const providerFetch: typeof globalThis.fetch = async (input, init) => {
|
|
221
|
+
if (String(input).endsWith("/input-tokens")) {
|
|
222
|
+
return new Response(JSON.stringify({ input_tokens: promptTokens }), { status: 200 });
|
|
223
|
+
}
|
|
224
|
+
seen.push(JSON.parse(String(init?.body)));
|
|
225
|
+
return new Response(sseStream([
|
|
226
|
+
{ model: "m", choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] },
|
|
227
|
+
]), { status: 200 });
|
|
228
|
+
};
|
|
229
|
+
const provider = testProvider({
|
|
230
|
+
...injectedBase,
|
|
231
|
+
fetch: providerFetch,
|
|
232
|
+
promptTokensUrl: "https://example.test/input-tokens",
|
|
233
|
+
contextWindow: 48_000,
|
|
234
|
+
outputBudget: 8_000,
|
|
235
|
+
});
|
|
236
|
+
await provider.generate({ workerId: "flex", messages: [{ role: "user", content: "small" }] });
|
|
237
|
+
assert.equal(seen[0]?.max_tokens, 48_000 - 100 - 256, "a small exact prompt harvests the window slack — silent wire tolerance");
|
|
238
|
+
|
|
239
|
+
promptTokens = 40_000;
|
|
240
|
+
await provider.generate({ workerId: "flex", messages: [{ role: "user", content: "full" }] });
|
|
241
|
+
assert.equal(seen[1]?.max_tokens, 8_000, "a packet at capacity keeps the guaranteed floor");
|
|
242
|
+
});
|
|
243
|
+
|
|
244
|
+
test("(#482) conformance judges the grant: tolerated overflow is accepted, past-the-grant is refused", async () => {
|
|
245
|
+
let reportOutput = 9_648;
|
|
246
|
+
const providerFetch: typeof globalThis.fetch = async (input) => {
|
|
247
|
+
if (String(input).endsWith("/input-tokens")) {
|
|
248
|
+
return new Response(JSON.stringify({ input_tokens: 100 }), { status: 200 });
|
|
249
|
+
}
|
|
250
|
+
return new Response(sseStream([
|
|
251
|
+
{ model: "m", choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] },
|
|
252
|
+
{ choices: [], usage: { prompt_tokens: 100, completion_tokens: reportOutput, total_tokens: 100 + reportOutput } },
|
|
253
|
+
]), { status: 200 });
|
|
254
|
+
};
|
|
255
|
+
const provider = testProvider({
|
|
256
|
+
...injectedBase,
|
|
257
|
+
fetch: providerFetch,
|
|
258
|
+
promptTokensUrl: "https://example.test/input-tokens",
|
|
259
|
+
contextWindow: 48_000,
|
|
260
|
+
outputBudget: 8_000,
|
|
261
|
+
});
|
|
262
|
+
// 9,648 > the 8,000 floor but inside the 47,644 grant: run7's loop-death shape, now tolerated.
|
|
263
|
+
const ok = await provider.generate({ workerId: "flex", messages: [{ role: "user", content: "small" }] });
|
|
264
|
+
assert.equal(ok.assistant.content, "ok", "output between floor and grant is the tolerance working");
|
|
265
|
+
|
|
266
|
+
reportOutput = 47_700;
|
|
267
|
+
await assert.rejects(
|
|
268
|
+
provider.generate({ workerId: "flex", messages: [{ role: "user", content: "small" }] }),
|
|
269
|
+
/granted output allowance of 47644/,
|
|
270
|
+
"output past the grant remains a provider fault, named by the grant",
|
|
271
|
+
);
|
|
272
|
+
});
|
|
273
|
+
|
|
217
274
|
test("request-observer open failures preserve the durability cause and issue no provider I/O", async () => {
|
|
218
275
|
const root = new Error("durable request open failed");
|
|
219
276
|
let calls = 0;
|
|
@@ -722,7 +779,7 @@ test("native SDK accounting metadata becomes a normalized charge in buffered and
|
|
|
722
779
|
});
|
|
723
780
|
});
|
|
724
781
|
|
|
725
|
-
test("native SDK providers share the first-content
|
|
782
|
+
test("native SDK providers share the first-content surface-at-once contract (#479)", async () => {
|
|
726
783
|
let calls = 0;
|
|
727
784
|
const usage = {
|
|
728
785
|
inputTokens: { total: 1, noCache: 1, cacheRead: 0, cacheWrite: 0 },
|
|
@@ -784,10 +841,12 @@ test("native SDK providers share the first-content retry contract", async () =>
|
|
|
784
841
|
source: "provider:test-native",
|
|
785
842
|
});
|
|
786
843
|
|
|
787
|
-
|
|
788
|
-
|
|
789
|
-
|
|
790
|
-
|
|
844
|
+
await assert.rejects(
|
|
845
|
+
provider.generate({ workerId: "native-retry", messages: [] }),
|
|
846
|
+
(error: ProviderError) => error.kind === "network_failure"
|
|
847
|
+
&& error.problem.timeoutPhase === "first_content",
|
|
848
|
+
);
|
|
849
|
+
assert.equal(calls, 1, "a deadline surfaces at once; no transport retry (#479)");
|
|
791
850
|
});
|
|
792
851
|
|
|
793
852
|
test("compatible xAI wire usage becomes an exact tick charge without raw-body capture", async () => {
|
|
@@ -1337,6 +1396,27 @@ test("DRY + repeat_last_n ride the llamacpp path when set; unset leaves the box
|
|
|
1337
1396
|
mock.restoreAll();
|
|
1338
1397
|
});
|
|
1339
1398
|
|
|
1399
|
+
test("(#480) unset sampling passes through: no temperature or repetition field on the wire, grammar included", async () => {
|
|
1400
|
+
// Cloud shape: nothing configured, nothing sent — the provider default governs.
|
|
1401
|
+
const cloud = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: null, repeatPenalty: null, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
1402
|
+
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1403
|
+
await cloud.generate({ workerId: "r", messages: [] });
|
|
1404
|
+
let body = JSON.parse(calls[0].init.body as string);
|
|
1405
|
+
assert.equal("temperature" in body, false);
|
|
1406
|
+
assert.equal("repeat_penalty" in body, false);
|
|
1407
|
+
assert.equal("frequency_penalty" in body, false);
|
|
1408
|
+
mock.restoreAll();
|
|
1409
|
+
// llamacpp grammar path: the grammar rides alone; the box's own defaults decode.
|
|
1410
|
+
const llama = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: null, repeatPenalty: null, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
|
|
1411
|
+
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1412
|
+
await llama.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
1413
|
+
body = JSON.parse(calls[0].init.body as string);
|
|
1414
|
+
assert.ok(body.grammar, "the grammar still rides");
|
|
1415
|
+
assert.equal("temperature" in body, false);
|
|
1416
|
+
assert.equal("repeat_penalty" in body, false);
|
|
1417
|
+
mock.restoreAll();
|
|
1418
|
+
});
|
|
1419
|
+
|
|
1340
1420
|
test("llamacpp grammar path: temperature default + the managed repeat-penalty floor", async () => {
|
|
1341
1421
|
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
|
|
1342
1422
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
@@ -1563,7 +1643,7 @@ test("reasoningStyle 'template' carries the explicit reasoning subset and reject
|
|
|
1563
1643
|
);
|
|
1564
1644
|
});
|
|
1565
1645
|
|
|
1566
|
-
test("reasoningStyle 'template'
|
|
1646
|
+
test("reasoningStyle 'template' forwards a fixed effort as the template's own variable beside the budget", async () => {
|
|
1567
1647
|
const p = testProvider({
|
|
1568
1648
|
model: "m",
|
|
1569
1649
|
url: "http://x/v1/chat/completions",
|
|
@@ -1580,8 +1660,8 @@ test("reasoningStyle 'template' explicit activation uses the configured reasonin
|
|
|
1580
1660
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1581
1661
|
await p.generate({ workerId: "r", messages: [] });
|
|
1582
1662
|
const body = JSON.parse(calls[0].init.body as string);
|
|
1583
|
-
assert.deepEqual(body.chat_template_kwargs, { enable_thinking: true });
|
|
1584
|
-
assert.equal(body.thinking_budget_tokens, 64);
|
|
1663
|
+
assert.deepEqual(body.chat_template_kwargs, { enable_thinking: true, reasoning_effort: "high" }, "the stated intent reaches the template; adaptive alone leaves the template default");
|
|
1664
|
+
assert.equal(body.thinking_budget_tokens, 64, "the budget still rides beside the effort");
|
|
1585
1665
|
});
|
|
1586
1666
|
|
|
1587
1667
|
test("reasoning off suppresses effort and include_reasoning controls", async () => {
|
|
@@ -1993,19 +2073,11 @@ test("retry: a transient failure retries and a later success resolves", async ()
|
|
|
1993
2073
|
assert.equal(res.notices, undefined, "retries before semantic output do not manufacture a degraded-response warning");
|
|
1994
2074
|
});
|
|
1995
2075
|
|
|
1996
|
-
test("streamed-body silence
|
|
2076
|
+
test("streamed-body silence surfaces on the first failure; the budget is not consumed (#479)", async () => {
|
|
1997
2077
|
let calls = 0;
|
|
1998
2078
|
mock.method(globalThis, "fetch", async () => {
|
|
1999
2079
|
calls++;
|
|
2000
|
-
|
|
2001
|
-
return new Response(new ReadableStream({
|
|
2002
|
-
start(controller) {
|
|
2003
|
-
controller.enqueue(new TextEncoder().encode(
|
|
2004
|
-
'data: {"id":"second","object":"chat.completion.chunk","created":2,"model":"m","choices":[{"index":0,"delta":{"content":"recovered"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n',
|
|
2005
|
-
));
|
|
2006
|
-
controller.close();
|
|
2007
|
-
},
|
|
2008
|
-
}), { status: 200 });
|
|
2080
|
+
return stalledStreamResponse();
|
|
2009
2081
|
});
|
|
2010
2082
|
const p = testProvider({
|
|
2011
2083
|
model: "m",
|
|
@@ -2018,30 +2090,29 @@ test("streamed-body silence retries and returns the retry's complete output", as
|
|
|
2018
2090
|
retryAttempts: 1,
|
|
2019
2091
|
source: "provider:test",
|
|
2020
2092
|
});
|
|
2021
|
-
|
|
2022
|
-
|
|
2023
|
-
|
|
2093
|
+
await assert.rejects(
|
|
2094
|
+
p.generate({ workerId: "r", messages: [] }),
|
|
2095
|
+
(error: ProviderError) => error.kind === "network_failure"
|
|
2096
|
+
&& error.problem.timeoutPhase === "stream_idle"
|
|
2097
|
+
&& error.problem.timeoutMs === 10,
|
|
2098
|
+
);
|
|
2099
|
+
assert.equal(calls, 1, "a stalled stream is the engine's to recover, not the transport's to replay");
|
|
2024
2100
|
mock.restoreAll();
|
|
2025
2101
|
});
|
|
2026
2102
|
|
|
2027
|
-
test("an Undici stream termination
|
|
2103
|
+
test("an Undici stream termination surfaces on the first failure (#479)", async () => {
|
|
2028
2104
|
let calls = 0;
|
|
2029
2105
|
mock.method(globalThis, "fetch", async () => {
|
|
2030
2106
|
calls++;
|
|
2031
|
-
|
|
2032
|
-
|
|
2033
|
-
|
|
2034
|
-
|
|
2035
|
-
|
|
2036
|
-
|
|
2037
|
-
|
|
2038
|
-
|
|
2039
|
-
|
|
2040
|
-
}), { status: 200 });
|
|
2041
|
-
}
|
|
2042
|
-
return new Response(sseStream([
|
|
2043
|
-
{ choices: [{ delta: { reasoning_content: "accepted reasoning", content: "recovered" }, finish_reason: "stop" }] },
|
|
2044
|
-
]), { status: 200 });
|
|
2107
|
+
return new Response(new ReadableStream({
|
|
2108
|
+
start(controller) {
|
|
2109
|
+
controller.enqueue(new TextEncoder().encode(
|
|
2110
|
+
'data: {"id":"terminated","object":"chat.completion.chunk","created":1,"model":"m","choices":[{"index":0,"delta":{"reasoning_content":"abandoned reasoning","content":"partial"},"finish_reason":null}]}\n\n',
|
|
2111
|
+
));
|
|
2112
|
+
const socket = Object.assign(new Error("other side closed"), { code: "UND_ERR_SOCKET" });
|
|
2113
|
+
setTimeout(() => controller.error(new TypeError("terminated", { cause: socket })), 10);
|
|
2114
|
+
},
|
|
2115
|
+
}), { status: 200 });
|
|
2045
2116
|
});
|
|
2046
2117
|
const p = testProvider({
|
|
2047
2118
|
model: "m",
|
|
@@ -2055,23 +2126,12 @@ test("an Undici stream termination retries and returns the retry's complete outp
|
|
|
2055
2126
|
source: "provider:test",
|
|
2056
2127
|
});
|
|
2057
2128
|
const reasoning: string[] = [];
|
|
2058
|
-
|
|
2059
|
-
workerId: "r",
|
|
2060
|
-
|
|
2061
|
-
|
|
2062
|
-
|
|
2063
|
-
assert.equal(
|
|
2064
|
-
assert.equal(result.assistant.reasoning, "accepted reasoning", "durable reasoning belongs only to the complete retry");
|
|
2065
|
-
assert.deepEqual(reasoning, ["abandoned reasoning", "accepted reasoning"], "transient observation preserves both physical requests for request-scoped presentation");
|
|
2066
|
-
assert.equal(calls, 2, "the terminated stream retried once and the retry succeeded");
|
|
2067
|
-
assert.deepEqual(result.accounting.map(({ outcome }) => outcome), ["error", "response"]);
|
|
2068
|
-
assert.deepEqual(result.notices, [{
|
|
2069
|
-
source: "provider:test",
|
|
2070
|
-
kind: "provider_retry",
|
|
2071
|
-
level: "warn",
|
|
2072
|
-
message: "A provider stream failed after model output began; this response is a complete retry, not a continuation.",
|
|
2073
|
-
position: null,
|
|
2074
|
-
}], "a recovered partial-output failure is attributed on the accepted response");
|
|
2129
|
+
await assert.rejects(
|
|
2130
|
+
p.generate({ workerId: "r", messages: [], observeReasoning: (delta) => reasoning.push(delta) }),
|
|
2131
|
+
(error: ProviderError) => error.kind === "network_failure",
|
|
2132
|
+
);
|
|
2133
|
+
assert.deepEqual(reasoning, ["abandoned reasoning"], "transient observation preserves the failed request's partial for request-scoped presentation");
|
|
2134
|
+
assert.equal(calls, 1, "a peer-terminated body is the engine's to recover (#479)");
|
|
2075
2135
|
mock.restoreAll();
|
|
2076
2136
|
});
|
|
2077
2137
|
|
|
@@ -2102,49 +2162,15 @@ test("streamed-body silence does not replay when retries are disabled", async ()
|
|
|
2102
2162
|
mock.restoreAll();
|
|
2103
2163
|
});
|
|
2104
2164
|
|
|
2105
|
-
test("
|
|
2106
|
-
let calls = 0;
|
|
2107
|
-
mock.method(globalThis, "fetch", async () => {
|
|
2108
|
-
calls++;
|
|
2109
|
-
return stalledStreamResponse();
|
|
2110
|
-
});
|
|
2111
|
-
const p = testProvider({
|
|
2112
|
-
model: "m",
|
|
2113
|
-
url: "http://x/v1/chat/completions",
|
|
2114
|
-
fetchTimeoutMs: 5000,
|
|
2115
|
-
streamIdleTimeoutMs: 10,
|
|
2116
|
-
temperature: 0.2,
|
|
2117
|
-
repeatPenalty: 1.15,
|
|
2118
|
-
reasoning: { mode: "off", budget: null },
|
|
2119
|
-
retryAttempts: 1,
|
|
2120
|
-
source: "provider:test",
|
|
2121
|
-
});
|
|
2122
|
-
await assert.rejects(
|
|
2123
|
-
p.generate({ workerId: "r", messages: [] }),
|
|
2124
|
-
(error: ProviderError) => error.kind === "network_failure"
|
|
2125
|
-
&& error.problem.attempts === 2
|
|
2126
|
-
&& error.problem.retryExhausted === true
|
|
2127
|
-
&& error.problem.retryable === false,
|
|
2128
|
-
);
|
|
2129
|
-
assert.equal(calls, 2, "one configured retry permits exactly two provider requests");
|
|
2130
|
-
mock.restoreAll();
|
|
2131
|
-
});
|
|
2132
|
-
|
|
2133
|
-
test("an attempt timeout retries within the larger operation deadline and settles every physical request", async () => {
|
|
2165
|
+
test("an attempt timeout surfaces on the first failure and settles its physical request (#479)", async () => {
|
|
2134
2166
|
let calls = 0;
|
|
2135
2167
|
mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
|
|
2136
2168
|
calls++;
|
|
2137
|
-
if (calls > 1) {
|
|
2138
|
-
return new Response(sseStream([
|
|
2139
|
-
{ choices: [{ delta: { content: "recovered" }, finish_reason: "stop" }] },
|
|
2140
|
-
]), { status: 200 });
|
|
2141
|
-
}
|
|
2142
2169
|
return await new Promise<Response>((_resolve, reject) => {
|
|
2143
2170
|
const signal = init?.signal;
|
|
2144
2171
|
signal?.addEventListener("abort", () => reject(signal.reason), { once: true });
|
|
2145
2172
|
});
|
|
2146
2173
|
});
|
|
2147
|
-
const connectivity = { operationTimeoutMs: 5_000 };
|
|
2148
2174
|
const settled: Array<{ outcome: string }> = [];
|
|
2149
2175
|
const p = testProvider({
|
|
2150
2176
|
model: "m",
|
|
@@ -2156,37 +2182,33 @@ test("an attempt timeout retries within the larger operation deadline and settle
|
|
|
2156
2182
|
reasoning: { mode: "off", budget: null },
|
|
2157
2183
|
retryAttempts: 1,
|
|
2158
2184
|
source: "provider:test",
|
|
2159
|
-
|
|
2160
|
-
});
|
|
2161
|
-
const result = await p.generate({
|
|
2162
|
-
workerId: "r",
|
|
2163
|
-
messages: [],
|
|
2164
|
-
observeRequest: async () => async (accounting) => { settled.push(accounting); },
|
|
2185
|
+
operationTimeoutMs: 5_000,
|
|
2165
2186
|
});
|
|
2166
|
-
assert.
|
|
2167
|
-
|
|
2168
|
-
|
|
2169
|
-
|
|
2170
|
-
|
|
2187
|
+
await assert.rejects(
|
|
2188
|
+
p.generate({
|
|
2189
|
+
workerId: "r",
|
|
2190
|
+
messages: [],
|
|
2191
|
+
observeRequest: async () => async (accounting) => { settled.push(accounting); },
|
|
2192
|
+
}),
|
|
2193
|
+
(error: ProviderError) => error.kind === "network_failure"
|
|
2194
|
+
&& error.problem.timeoutPhase === "attempt"
|
|
2195
|
+
&& error.problem.timeoutMs === 10,
|
|
2196
|
+
);
|
|
2197
|
+
assert.equal(calls, 1, "the deadline surfaces at once (#479)");
|
|
2198
|
+
assert.deepEqual(settled.map(({ outcome }) => outcome), ["error"], "the failed physical request is durably settled");
|
|
2171
2199
|
mock.restoreAll();
|
|
2172
2200
|
});
|
|
2173
2201
|
|
|
2174
|
-
test("first-content silence
|
|
2202
|
+
test("first-content silence surfaces independently of the stream-idle deadline (#479)", async () => {
|
|
2175
2203
|
let calls = 0;
|
|
2176
2204
|
mock.method(globalThis, "fetch", async () => {
|
|
2177
2205
|
calls++;
|
|
2178
|
-
if (calls > 1) {
|
|
2179
|
-
return new Response(sseStream([
|
|
2180
|
-
{ choices: [{ delta: { content: "recovered" }, finish_reason: "stop" }] },
|
|
2181
|
-
]), { status: 200 });
|
|
2182
|
-
}
|
|
2183
2206
|
return new Response(new ReadableStream({
|
|
2184
2207
|
start(controller) {
|
|
2185
2208
|
setTimeout(() => controller.close(), 100);
|
|
2186
2209
|
},
|
|
2187
2210
|
}), { status: 200 });
|
|
2188
2211
|
});
|
|
2189
|
-
const connectivity = { operationTimeoutMs: 5_000, firstContentTimeoutMs: 10 };
|
|
2190
2212
|
const p = testProvider({
|
|
2191
2213
|
model: "m",
|
|
2192
2214
|
url: "http://x/v1/chat/completions",
|
|
@@ -2197,12 +2219,16 @@ test("first-content silence retries independently of the stream-idle deadline",
|
|
|
2197
2219
|
reasoning: { mode: "off", budget: null },
|
|
2198
2220
|
retryAttempts: 1,
|
|
2199
2221
|
source: "provider:test",
|
|
2200
|
-
|
|
2222
|
+
operationTimeoutMs: 5_000,
|
|
2223
|
+
firstContentTimeoutMs: 10,
|
|
2201
2224
|
});
|
|
2202
|
-
|
|
2203
|
-
|
|
2204
|
-
|
|
2205
|
-
|
|
2225
|
+
await assert.rejects(
|
|
2226
|
+
p.generate({ workerId: "r", messages: [] }),
|
|
2227
|
+
(error: ProviderError) => error.kind === "network_failure"
|
|
2228
|
+
&& error.problem.timeoutPhase === "first_content"
|
|
2229
|
+
&& error.problem.timeoutMs === 10,
|
|
2230
|
+
);
|
|
2231
|
+
assert.equal(calls, 1, "the first-content deadline surfaces at once (#479)");
|
|
2206
2232
|
mock.restoreAll();
|
|
2207
2233
|
});
|
|
2208
2234
|
|
package/src/AiSdkProvider.ts
CHANGED
|
@@ -173,8 +173,8 @@ export type AiSdkProviderConfig = {
|
|
|
173
173
|
// `repeatPenalty` is the FLOOR the provider manages wherever a grammar rides
|
|
174
174
|
// (greedy-under-mask loops without it) — the VALUE is operator config;
|
|
175
175
|
// WHERE it applies stays mechanism.
|
|
176
|
-
temperature: number;
|
|
177
|
-
repeatPenalty: number;
|
|
176
|
+
temperature: number | null;
|
|
177
|
+
repeatPenalty: number | null;
|
|
178
178
|
// Anti-degeneration guard on the cloud path (grammarStyle "none"), where the
|
|
179
179
|
// repeat_penalty multiplier isn't available - the OpenAI-standard frequency_penalty.
|
|
180
180
|
// Optional (default 0 = off) so an out-of-date plugin that omits it just runs unguarded
|
|
@@ -395,8 +395,8 @@ export default class AiSdkProvider implements Provider {
|
|
|
395
395
|
#compatibleAdaptiveReasoning: CompatibleReasoningEffort | "provider-default";
|
|
396
396
|
#compatibleOffReasoning: "none" | undefined;
|
|
397
397
|
#adaptiveReasoningProviderOptions: AiSdkProviderOptions | undefined;
|
|
398
|
-
#temperature: number;
|
|
399
|
-
#repeatPenalty: number;
|
|
398
|
+
#temperature: number | null;
|
|
399
|
+
#repeatPenalty: number | null;
|
|
400
400
|
#frequencyPenalty: number;
|
|
401
401
|
#dryMultiplier: number | undefined;
|
|
402
402
|
#dryBase: number | undefined;
|
|
@@ -482,8 +482,8 @@ export default class AiSdkProvider implements Provider {
|
|
|
482
482
|
// Loud guard: an out-of-date consumer (stale plugin dist) omitting the
|
|
483
483
|
// required tuning fields must fail at construction, not silently send
|
|
484
484
|
// undefined sampling on every grammar request.
|
|
485
|
-
if (
|
|
486
|
-
throw new Error(`${config.source ?? "provider"}: AiSdkProviderConfig requires temperature + repeatPenalty (PLURNK_PROVIDERS_TEMPERATURE / _REPEAT_PENALTY)`);
|
|
485
|
+
if (config.temperature === undefined || config.repeatPenalty === undefined) {
|
|
486
|
+
throw new Error(`${config.source ?? "provider"}: AiSdkProviderConfig requires temperature + repeatPenalty declared (PLURNK_PROVIDERS_TEMPERATURE / _REPEAT_PENALTY; null = provider default)`);
|
|
487
487
|
}
|
|
488
488
|
this.#temperature = config.temperature;
|
|
489
489
|
this.#repeatPenalty = config.repeatPenalty;
|
|
@@ -718,8 +718,11 @@ export default class AiSdkProvider implements Provider {
|
|
|
718
718
|
const allowance = mode === "off"
|
|
719
719
|
? 0
|
|
720
720
|
: budget;
|
|
721
|
+
// A fixed effort rides into the template as its own variable; adaptive
|
|
722
|
+
// and off send none and leave the template's default in force.
|
|
723
|
+
const templateEffort = mode === "off" || mode === "adaptive" ? {} : { reasoning_effort: fixedEffort(mode) };
|
|
721
724
|
return {
|
|
722
|
-
chat_template_kwargs: { enable_thinking: on },
|
|
725
|
+
chat_template_kwargs: { enable_thinking: on, ...templateEffort },
|
|
723
726
|
reasoning_format: preserveGrammarSentence ? "none" : "auto",
|
|
724
727
|
...(allowance === null ? {} : { thinking_budget_tokens: allowance }),
|
|
725
728
|
};
|
|
@@ -813,9 +816,9 @@ export default class AiSdkProvider implements Provider {
|
|
|
813
816
|
#grammarBody(grammar: string | undefined): Record<string, unknown> {
|
|
814
817
|
if (grammar === undefined) return {};
|
|
815
818
|
switch (this.#grammarStyle) {
|
|
816
|
-
//
|
|
817
|
-
//
|
|
818
|
-
case "llamacpp": return { grammar, repeat_penalty: this.#repeatPenalty };
|
|
819
|
+
// Grammar-constrained decoding can loop under the mask; a configured
|
|
820
|
+
// per-alias repeat_penalty is the measured remedy ({§provider-sampling-passthrough}).
|
|
821
|
+
case "llamacpp": return { grammar, ...(this.#repeatPenalty !== null ? { repeat_penalty: this.#repeatPenalty } : {}) };
|
|
819
822
|
case "none": return {};
|
|
820
823
|
}
|
|
821
824
|
}
|
|
@@ -835,7 +838,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
835
838
|
// repeat_last_n window — the loop-breaking tools a llama.cpp backend serves.
|
|
836
839
|
// Each rides only when its operator knob is set; absent = the box's default.
|
|
837
840
|
case "llamacpp": return {
|
|
838
|
-
repeat_penalty: this.#repeatPenalty,
|
|
841
|
+
...(this.#repeatPenalty !== null ? { repeat_penalty: this.#repeatPenalty } : {}),
|
|
839
842
|
...(this.#repeatLastN !== undefined ? { repeat_last_n: this.#repeatLastN } : {}),
|
|
840
843
|
...(this.#dryMultiplier !== undefined && this.#dryMultiplier > 0 ? {
|
|
841
844
|
dry_multiplier: this.#dryMultiplier,
|
|
@@ -1033,7 +1036,9 @@ export default class AiSdkProvider implements Provider {
|
|
|
1033
1036
|
{ capacity, extensions: { capacityStage: "preflight", capacity } },
|
|
1034
1037
|
);
|
|
1035
1038
|
}
|
|
1036
|
-
|
|
1039
|
+
// {§provider-flexed-allowance} (#482): the wire grants the flexed
|
|
1040
|
+
// allowance — the floor, or the exactly-measured slack above it.
|
|
1041
|
+
const effectiveMaxOutputTokens = capacity.responseMax ?? capacity.outputBudget ?? undefined;
|
|
1037
1042
|
const nativeReasoningBudget = this.#nativeReasoningBudget(
|
|
1038
1043
|
capacity.outputBudget,
|
|
1039
1044
|
capacity.reasoningBudget,
|
|
@@ -1046,7 +1051,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
1046
1051
|
const body: Record<string, unknown> = {
|
|
1047
1052
|
// Floors are suppressed on router-owned-tuning providers (plurnk) —
|
|
1048
1053
|
// the router's per-model tuning must not be overridden by client floors.
|
|
1049
|
-
...(this.#tuningFloors ? { temperature: this.#temperature, ...this.#repetitionPenaltyBody() } : {}),
|
|
1054
|
+
...(this.#tuningFloors ? { ...(this.#temperature !== null ? { temperature: this.#temperature } : {}), ...this.#repetitionPenaltyBody() } : {}),
|
|
1050
1055
|
...this.#samplingBody(sampling),
|
|
1051
1056
|
...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
|
|
1052
1057
|
model: this.#model,
|
|
@@ -1182,7 +1187,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
1182
1187
|
captureRawBody: this.#rawBody,
|
|
1183
1188
|
...(observeRequestReasoning === undefined ? {} : { observeReasoning: observeRequestReasoning }),
|
|
1184
1189
|
temperature: this.#tuningFloors
|
|
1185
|
-
? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
|
|
1190
|
+
? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature ?? undefined)
|
|
1186
1191
|
: typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
|
|
1187
1192
|
topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
|
|
1188
1193
|
topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
|
|
@@ -1397,9 +1402,14 @@ export default class AiSdkProvider implements Provider {
|
|
|
1397
1402
|
...(meta !== undefined ? { meta } : {}),
|
|
1398
1403
|
...(notices !== undefined ? { notices } : {}),
|
|
1399
1404
|
};
|
|
1400
|
-
|
|
1405
|
+
// {§provider-flexed-allowance} (#482): conformance judges the GRANT the
|
|
1406
|
+
// wire actually sent, not the configured floor — output between the two
|
|
1407
|
+
// is overflow tolerance working, never a provider fault. Run7 loop-death
|
|
1408
|
+
// was this guard still holding the floor after the flex landed.
|
|
1409
|
+
const grantedOutput = capacity.responseMax ?? capacity.outputBudget;
|
|
1410
|
+
if (grantedOutput !== null
|
|
1401
1411
|
&& usage?.outputTokens !== undefined
|
|
1402
|
-
&& usage.outputTokens >
|
|
1412
|
+
&& usage.outputTokens > grantedOutput) {
|
|
1403
1413
|
const attempt: ProviderAttempt = {
|
|
1404
1414
|
assistant: { ...assistant, finishReason: raw.finishReason },
|
|
1405
1415
|
...evidence,
|
|
@@ -1407,13 +1417,14 @@ export default class AiSdkProvider implements Provider {
|
|
|
1407
1417
|
throw new ProviderError(
|
|
1408
1418
|
this.#source,
|
|
1409
1419
|
"invalid_response",
|
|
1410
|
-
`The provider reported ${usage.outputTokens} output tokens after receiving a
|
|
1420
|
+
`The provider reported ${usage.outputTokens} output tokens after receiving a granted output allowance of ${grantedOutput}.`,
|
|
1411
1421
|
{
|
|
1412
1422
|
attempt,
|
|
1413
1423
|
accounting,
|
|
1414
1424
|
extensions: {
|
|
1415
1425
|
stage: "provider-response",
|
|
1416
1426
|
outputBudget: capacity.outputBudget,
|
|
1427
|
+
grantedOutput,
|
|
1417
1428
|
reportedOutputTokens: usage.outputTokens,
|
|
1418
1429
|
},
|
|
1419
1430
|
},
|
package/src/Pool.test.ts
CHANGED
|
@@ -31,6 +31,7 @@ const RESP: ProviderResponse = {
|
|
|
31
31
|
outputBudget: 12_000,
|
|
32
32
|
reasoningBudget: null,
|
|
33
33
|
inputCapacity: 36_000,
|
|
34
|
+
responseMax: 12_000,
|
|
34
35
|
prompt: { kind: "exact", tokens: 0, source: "test:exact" },
|
|
35
36
|
},
|
|
36
37
|
};
|
|
@@ -76,6 +77,7 @@ const backend = (opts: FakeOpts = {}) => {
|
|
|
76
77
|
outputBudget: maxOutputTokens ?? opts.outputBudget ?? null,
|
|
77
78
|
reasoningBudget: opts.reasoningBudget ?? null,
|
|
78
79
|
inputCapacity: null,
|
|
80
|
+
responseMax: maxOutputTokens ?? opts.outputBudget ?? null,
|
|
79
81
|
prompt: opts.promptMeasurement ?? {
|
|
80
82
|
kind: "exact",
|
|
81
83
|
tokens: messages.reduce((sum, { content }) => sum + content.length, 0),
|
package/src/Pool.ts
CHANGED
|
@@ -190,6 +190,8 @@ export default class Pool implements Provider {
|
|
|
190
190
|
outputBudget: minimum(envelopes.map((envelope) => envelope.outputBudget)),
|
|
191
191
|
reasoningBudget: minimum(envelopes.map((envelope) => envelope.reasoningBudget)),
|
|
192
192
|
inputCapacity,
|
|
193
|
+
// A pool spans members whose windows differ; it never flexes ({§provider-flexed-allowance}).
|
|
194
|
+
responseMax: minimum(envelopes.map((envelope) => envelope.outputBudget)),
|
|
193
195
|
prompt: measurement,
|
|
194
196
|
};
|
|
195
197
|
}
|