@plurnk/plurnk-providers 1.5.0 → 1.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. package/.env.defaults +41 -34
  2. package/README.md +15 -0
  3. package/SPEC.md +242 -89
  4. package/dist/AiSdkProvider.d.ts +33 -33
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +442 -133
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +10 -11
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +87 -25
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +9 -24
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +86 -25
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/accounting.d.ts +5 -2
  17. package/dist/accounting.d.ts.map +1 -1
  18. package/dist/accounting.js +100 -16
  19. package/dist/accounting.js.map +1 -1
  20. package/dist/accountingPublic.d.ts +5 -0
  21. package/dist/accountingPublic.d.ts.map +1 -0
  22. package/dist/accountingPublic.js +3 -0
  23. package/dist/accountingPublic.js.map +1 -0
  24. package/dist/aiSdkTransport.d.ts +9 -2
  25. package/dist/aiSdkTransport.d.ts.map +1 -1
  26. package/dist/aiSdkTransport.js +160 -62
  27. package/dist/aiSdkTransport.js.map +1 -1
  28. package/dist/capacity.d.ts +26 -0
  29. package/dist/capacity.d.ts.map +1 -0
  30. package/dist/capacity.js +90 -0
  31. package/dist/capacity.js.map +1 -0
  32. package/dist/catalogProvider.d.ts +8 -3
  33. package/dist/catalogProvider.d.ts.map +1 -1
  34. package/dist/catalogProvider.js +45 -41
  35. package/dist/catalogProvider.js.map +1 -1
  36. package/dist/compatibleProvider.d.ts.map +1 -1
  37. package/dist/compatibleProvider.js +26 -12
  38. package/dist/compatibleProvider.js.map +1 -1
  39. package/dist/cost.d.ts +10 -10
  40. package/dist/cost.d.ts.map +1 -1
  41. package/dist/cost.js +90 -42
  42. package/dist/cost.js.map +1 -1
  43. package/dist/env.d.ts +13 -11
  44. package/dist/env.d.ts.map +1 -1
  45. package/dist/env.js +83 -46
  46. package/dist/env.js.map +1 -1
  47. package/dist/errors.d.ts +17 -3
  48. package/dist/errors.d.ts.map +1 -1
  49. package/dist/errors.js +91 -8
  50. package/dist/errors.js.map +1 -1
  51. package/dist/index.d.ts +7 -6
  52. package/dist/index.d.ts.map +1 -1
  53. package/dist/index.js +5 -3
  54. package/dist/index.js.map +1 -1
  55. package/dist/ollama.js +3 -3
  56. package/dist/ollama.js.map +1 -1
  57. package/dist/promptTokens.d.ts.map +1 -1
  58. package/dist/promptTokens.js +7 -4
  59. package/dist/promptTokens.js.map +1 -1
  60. package/dist/sdkModels.d.ts +7 -2
  61. package/dist/sdkModels.d.ts.map +1 -1
  62. package/dist/sdkModels.js +43 -13
  63. package/dist/sdkModels.js.map +1 -1
  64. package/dist/types.d.ts +55 -33
  65. package/dist/types.d.ts.map +1 -1
  66. package/dist/usage.d.ts +22 -5
  67. package/dist/usage.d.ts.map +1 -1
  68. package/dist/usage.js +169 -83
  69. package/dist/usage.js.map +1 -1
  70. package/package.json +18 -7
  71. package/src/AiSdkProvider.test.ts +964 -206
  72. package/src/AiSdkProvider.ts +545 -155
  73. package/src/Mock.test.ts +69 -30
  74. package/src/Mock.ts +99 -29
  75. package/src/Pool.test.ts +90 -19
  76. package/src/Pool.ts +96 -27
  77. package/src/ProviderRegistry.test.ts +16 -11
  78. package/src/accounting.test.ts +58 -22
  79. package/src/accounting.ts +119 -18
  80. package/src/accountingPublic.ts +9 -0
  81. package/src/aiSdkTransport.test.ts +42 -49
  82. package/src/aiSdkTransport.ts +174 -62
  83. package/src/boundaries.test.ts +2 -0
  84. package/src/capacity.test.ts +92 -0
  85. package/src/capacity.ts +140 -0
  86. package/src/catalogProvider.test.ts +339 -30
  87. package/src/catalogProvider.ts +65 -47
  88. package/src/compatibleProvider.test.ts +7 -5
  89. package/src/compatibleProvider.ts +29 -13
  90. package/src/cost.test.ts +86 -36
  91. package/src/cost.ts +111 -50
  92. package/src/defaults.test.ts +13 -3
  93. package/src/env.test.ts +103 -25
  94. package/src/env.ts +153 -65
  95. package/src/errors.test.ts +80 -2
  96. package/src/errors.ts +107 -8
  97. package/src/index.ts +26 -7
  98. package/src/ollama.test.ts +5 -3
  99. package/src/ollama.ts +3 -3
  100. package/src/promptTokens.ts +8 -5
  101. package/src/sdkModels.test.ts +77 -8
  102. package/src/sdkModels.ts +51 -15
  103. package/src/types.ts +112 -51
  104. package/src/usage.test.ts +112 -116
  105. package/src/usage.ts +214 -93
package/.env.defaults CHANGED
@@ -13,18 +13,21 @@
13
13
  # and model facts. {§model-fact-resolution} defines precedence per fact; there is
14
14
  # no live price fetch. Secret VALUES never belong here.
15
15
 
16
+ # --- Generation envelope ({§provider-generation-envelope}, #242) ---
17
+ # OUTPUT_BUDGET is one total response ceiling, including hidden reasoning. It accepts a
18
+ # percentage of the effective context window or an absolute token count and is always
19
+ # capped by known model output limits. REASONING_BUDGET is an optional subset of that
20
+ # total; leave it unset for provider-adaptive depth. Both are alias-scopable.
21
+ PLURNK_PROVIDERS_OUTPUT_BUDGET=35%
22
+ # PLURNK_PROVIDERS_REASONING_BUDGET=8192
23
+
16
24
  # --- Side-channel reasoning (SPEC §4, #32/#33/#399) ---
17
25
  # ACTIVATION and BUDGET are separate so a numeric can never silently flip wire flags.
18
26
  # off | adaptive | on. The provider maps intent to each backend's native mechanism
19
27
  # (reasoning_effort, enable_thinking, think, ...). Default ADAPTIVE (#399):
20
- # reasoning ACTIVE on a fresh install, each backend's own adaptive depth,
21
- # no shipped magnitude - a reasoning model that ships un-reasoning blind-edits and
22
- # declares done.
28
+ # defer activation and depth to the backend's documented default. Use an alias-scoped
29
+ # ON when a reasoning-capable model defaults off and the operator wants it enabled.
23
30
  PLURNK_PROVIDERS_REASONING=adaptive
24
- # Positive int, REQUIRED iff REASONING=on - the magnitude for tier/budget mapping.
25
- # On llama-server it is a request-scoped reasoning allowance and may tighten, but
26
- # cannot exceed, the resolved PLURNK_PROVIDERS_REASONING_RESERVE.
27
- # PLURNK_PROVIDERS_REASONING_BUDGET=4096
28
31
 
29
32
  # Response-content interpretation ({§provider-tagged-reasoning}) is independent
30
33
  # from request-side reasoning activation. The portable floor trusts only
@@ -48,10 +51,14 @@ PLURNK_PROVIDERS_FREQUENCY_PENALTY=0
48
51
  # Fixed provider service tier. Fireworks accepts auto|default|flex|priority;
49
52
  # normally set per alias so a paid routing choice is explicit.
50
53
  # PLURNK_PROVIDERS_SERVICE_TIER_myfireworks=priority
51
- # Compatible endpoints verified to accept the OpenAI prompt_cache_key use the
52
- # stable worker id for replica-local prefix affinity. Official native SDKs own
53
- # their cache mechanisms and do not receive this compatible extension.
54
- PLURNK_PROVIDERS_PROMPT_CACHE_KEY=1
54
+ # Route adapters project the stable worker identity through only the provider's
55
+ # documented session/cache-affinity control. Unknown compatible routes receive
56
+ # no guessed field. Disable globally or per alias only for operational diagnosis.
57
+ PLURNK_PROVIDERS_CACHE_AFFINITY=1
58
+ # Explicit cache writes are distinct from affinity and may affect billing.
59
+ # stable-system marks only the reusable system boundary on supported Claude
60
+ # routes; off requests no explicit cache write. Provider-default lifetime is 5m.
61
+ PLURNK_PROVIDERS_CACHE_WRITE_POLICY=stable-system
55
62
  # #567: DRY is a llama.cpp-only repeated-sequence penalty. It can reduce
56
63
  # degenerate loops, but it can also corrupt exact source, identifiers, quoted
57
64
  # evidence, and other repetition required by PLURNK operations. The portable
@@ -64,16 +71,27 @@ PLURNK_PROVIDERS_DRY_MULTIPLIER=0
64
71
  # Set it per alias only from measured model behavior.
65
72
  # PLURNK_PROVIDERS_REPEAT_LAST_N_<alias>=512
66
73
 
67
- # --- Transport budgets (§4, #18) ---
68
- # Total generation-operation timeout (ms), including retries; caller cancellation also spans the operation.
74
+ # --- Connectivity budgets ({§provider-connectivity}, #240) ---
75
+ # Complete logical generation deadline (ms), spanning every physical attempt
76
+ # and retry delay. Zero disables this outer deadline; caller cancellation still
77
+ # spans the operation. The floor leaves all four ten-minute attempts plus normal
78
+ # backoff available under the three-retry floor.
79
+ PLURNK_PROVIDERS_OPERATION_TIMEOUT=2700000
80
+ # Maximum duration (ms) of one physical generation attempt. Zero disables the
81
+ # per-attempt deadline without changing the operation deadline.
69
82
  PLURNK_PROVIDERS_FETCH_TIMEOUT=600000
70
- # Maximum silence (ms) between streamed response-body chunks after response
71
- # streaming begins. Two minutes: generous for slow local inference, fast enough
72
- # to catch a dead connection before the total timeout. A detected stall consumes
73
- # the same configured retry budget as every other retryable transport failure.
83
+ # Maximum wait (ms) from response-stream start to first semantic model content.
84
+ # Transport metadata and empty deltas do not satisfy it. The portable floor
85
+ # allows endpoints that intentionally buffer a full attempt; tighten per alias
86
+ # from measured time-to-first-content. Zero disables it.
87
+ PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT=600000
88
+ # Maximum silence (ms) between semantic streamed-content chunks after content
89
+ # begins. Zero disables it. Both content deadlines consume the ordinary retry
90
+ # budget when they expire.
74
91
  PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT=120000
75
92
  # Transient-failure retries: 0 = surface the first failure; N = retries on
76
- # 429/5xx/stream-idle timeout with the AI SDK's backoff (Retry-After wins).
93
+ # network failure, 408/409/429/ordinary 5xx, or an inner deadline, with the AI
94
+ # SDK's backoff (Retry-After wins).
77
95
  PLURNK_PROVIDERS_RETRY_ATTEMPTS=3
78
96
  # Maximum characters retained from an upstream provider diagnostic in the
79
97
  # public Problem detail. Structured failure facts are not truncated.
@@ -89,16 +107,16 @@ PLURNK_PROVIDERS_PROBE_DELAY=250
89
107
  # Unset by default. Set per alias only for a local llama-server whose GBNF
90
108
  # transport is detected or pinned. Cloud and endpoint-managed model settings do
91
109
  # not use this knob.
92
- # PLURNK_PROVIDERS_GBNF=plurnk.gbnf
110
+ # PLURNK_PROVIDERS_GBNF=plurnk.qwen.gbnf
93
111
  # Debug toggle: validate but withhold a configured local GBNF, then report the
94
112
  # unconstrained output's divergence. Development aid; leave unset in production.
95
113
  # PLURNK_PROVIDERS_GBNF_DEBUG=0
96
114
 
97
115
  # --- Window ({§model-fact-resolution}) ---
98
- # Physical provider context only. Unset derives from a live endpoint probe (n_ctx) or
99
- # models.dev, else null (surfaced once via PLURNK_CONTEXT_UNKNOWN). A configured value
100
- # caps detected physics or declares it when unknown. Prompt packing and model-facing
101
- # pressure are consumer policy, not provider configuration.
116
+ # Effective total context envelope. Unset derives natural capacity from a live endpoint probe
117
+ # (n_ctx) or models.dev, else null (surfaced once via PLURNK_CONTEXT_UNKNOWN). A configured
118
+ # value is a final hard cap on known capacity or declares the envelope when unknown. Model-facing
119
+ # curation pressure is separate consumer policy.
102
120
  # PLURNK_PROVIDERS_CONTEXT_WINDOW=200000
103
121
 
104
122
  # --- llama-server detection pin (#34) ---
@@ -162,14 +180,3 @@ OPENAI_BASE_URL=https://api.openai.com/v1
162
180
  # PLURNK_API_KEY is an optional bearer; the endpoint is eventually keyless.
163
181
  PLURNK_BASE_URL=https://plurnk.ai/v1
164
182
  # PLURNK_BASE_URL=http://plurnksnr2kihuukt6v22ko72r34dxeatbsfhgow3hvnlw6btanxphad.onion/v1 # Tor
165
-
166
- # --- Generation envelope (#507) - sane defaults from the DETECTED window ---
167
- # When a backend advertises its context window (llama-server n_ctx, the plurnk.ai router,
168
- # a cataloged cloud model), the reserves derive from it automatically - ZERO operator
169
- # tuning. Each accepts a percentage of the window ("10%") or an absolute token count
170
- # ("4096"; absolutes win outright, alias-scopable for measured envelopes). The prompt
171
- # budget is window - reasoning - completion - the consumer's own safety margin.
172
- # On llama-server, the resolved reasoning reserve is also the adaptive per-response
173
- # reasoning ceiling. It is one cumulative allowance across every reasoning block.
174
- PLURNK_PROVIDERS_REASONING_RESERVE=10%
175
- PLURNK_PROVIDERS_COMPLETION_RESERVE=25%
package/README.md CHANGED
@@ -21,6 +21,21 @@ reasoning activation, and estimated prices resolve independently
21
21
  ({§model-fact-resolution}). PLURNK does not fetch live per-token prices, and the
22
22
  local estimate is not an authoritative relay-settled charge.
23
23
 
24
+ ## Runtime-neutral accounting
25
+
26
+ Browser and edge Workers import the accounting contract through its dedicated
27
+ runtime-neutral subpath ({§provider-runtime-neutral-accounting}):
28
+
29
+ ```js
30
+ import {
31
+ aggregateProviderAccounting,
32
+ estimateProviderCost,
33
+ } from "@plurnk/plurnk-providers/accounting";
34
+ ```
35
+
36
+ The package root composes the complete Node provider runtime, including plugin
37
+ discovery and environment-file defaults.
38
+
24
39
  ## Configure a model
25
40
 
26
41
  Declare an alias, then select it: