@pwguler/pi-pengepul-provider 0.5.0 → 0.5.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -0
- package/extensions/credential.ts +28 -0
- package/extensions/models.ts +55 -12
- package/extensions/provider.ts +14 -1
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -58,6 +58,10 @@ prefixed `pengepul/<id>`. To try it without installing, use
|
|
|
58
58
|
relay already serves — takes them from the same family's previous minor, so
|
|
59
59
|
it does not sit at `high` until the next pi catalog update. Nothing to
|
|
60
60
|
configure: `thinkingLevelMap` is resolved per id here.
|
|
61
|
+
- Publishes Anthropic's cache lifetimes (5 minutes, 1 hour) on the Messages
|
|
62
|
+
wire, so pi's cache warmer can keep a Claude conversation's entry alive
|
|
63
|
+
instead of re-billing its whole prefix after five idle minutes. See
|
|
64
|
+
[Prompt cache](#prompt-cache).
|
|
61
65
|
- Caches the catalog in pi's model store, so startup does not wait on the
|
|
62
66
|
network and a briefly absent relay is covered. The cached models are
|
|
63
67
|
re-pointed at the relay base configured now, so moving the relay does not
|
|
@@ -107,6 +111,52 @@ to every model of the provider, which collapses the two wires onto one URL —
|
|
|
107
111
|
Anthropic Messages traffic would be sent to `/v1` and Chat Completions traffic
|
|
108
112
|
to `/`.
|
|
109
113
|
|
|
114
|
+
## Prompt cache
|
|
115
|
+
|
|
116
|
+
The upstream prompt cache is one entry per account and lives five minutes by
|
|
117
|
+
default, so a conversation that waits longer than that pays for its whole prefix
|
|
118
|
+
again — on a 700k-token Opus session, dollars per return, logged by the relay as
|
|
119
|
+
a `cache_write` with a near-zero `cache_read`.
|
|
120
|
+
|
|
121
|
+
This provider declares Anthropic's lifetimes (5 minutes, 1 hour) on every
|
|
122
|
+
Messages-wire model, which is what lets pi keep an entry alive at all: pi
|
|
123
|
+
refreshes an entry by re-sending the prompt with a one-token cap before it
|
|
124
|
+
expires, and it only warms a model whose lifetime it can resolve. Two pi
|
|
125
|
+
settings decide when that happens, and neither covers an idle gap out of the
|
|
126
|
+
box:
|
|
127
|
+
|
|
128
|
+
| Setting | Where | Effect |
|
|
129
|
+
|---|---|---|
|
|
130
|
+
| `cacheWarming: "idle"` | `~/.pi/agent/settings.json` | Warms between agent runs. The default, `"streaming"`, stops as soon as a run settles — exactly when a five-minute entry is about to expire. |
|
|
131
|
+
| `env: { "PI_CACHE_RETENTION": "long" }` | the `pengepul` credential in `auth.json` | Asks for the 1-hour tier instead of the 5-minute one. The relay forwards the extended TTL to Anthropic. |
|
|
132
|
+
|
|
133
|
+
```json
|
|
134
|
+
{
|
|
135
|
+
"pengepul": {
|
|
136
|
+
"type": "api_key",
|
|
137
|
+
"key": "sk-local-...",
|
|
138
|
+
"baseUrl": "http://127.0.0.1:8317",
|
|
139
|
+
"env": { "PI_CACHE_RETENTION": "long" }
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
`env` is read off the credential by this provider, not by pi core: a provider
|
|
145
|
+
that brings its own `resolve()` has to hand the values back, and pengepul does.
|
|
146
|
+
Re-add them after `/login pengepul` — pi replaces the whole credential on login,
|
|
147
|
+
so a key rotation drops the block and, with it, the tier.
|
|
148
|
+
|
|
149
|
+
Idle warming covers gaps up to 30 minutes, pi's own safety limit, and pi only
|
|
150
|
+
triggers it when the expected saving clears a threshold of its own — a small
|
|
151
|
+
prompt is left to expire rather than refreshed. The long tier is what carries a
|
|
152
|
+
longer break. Both are cheap next to a miss: a refresh is one cache read plus a
|
|
153
|
+
one-token completion, while the long tier only raises the write premium, on the
|
|
154
|
+
few hundred tokens a turn actually adds.
|
|
155
|
+
|
|
156
|
+
`/session` reports the mode, the refresh cost, and the miss penalty it is
|
|
157
|
+
weighing, and `showCacheMissNotices` in `settings.json` prints each miss with
|
|
158
|
+
its re-billed token count.
|
|
159
|
+
|
|
110
160
|
## Upgrading from 0.4.0 or earlier
|
|
111
161
|
|
|
112
162
|
This release is breaking: the relay's own `~/.pengepul/config.yaml` is no longer
|
package/extensions/credential.ts
CHANGED
|
@@ -20,6 +20,8 @@ export interface PengepulCredential {
|
|
|
20
20
|
type?: unknown
|
|
21
21
|
key?: unknown
|
|
22
22
|
baseUrl?: unknown
|
|
23
|
+
/** Provider-scoped environment values pi stores alongside the key. */
|
|
24
|
+
env?: unknown
|
|
23
25
|
}
|
|
24
26
|
|
|
25
27
|
function nonEmptyString(value: unknown): string | undefined {
|
|
@@ -38,6 +40,32 @@ export function credentialApiKey(credential: PengepulCredential | undefined): st
|
|
|
38
40
|
return nonEmptyString(credential?.key)
|
|
39
41
|
}
|
|
40
42
|
|
|
43
|
+
/**
|
|
44
|
+
* The provider-scoped environment the credential carries, when it holds any.
|
|
45
|
+
*
|
|
46
|
+
* pi core fills an `AuthResult.env` from here only for providers that bring no
|
|
47
|
+
* `resolve()` of their own; this provider brings one, so the values have to be
|
|
48
|
+
* passed through or they never reach a request. They are how a user sets
|
|
49
|
+
* `PI_CACHE_RETENTION=long` for the relay without exporting it in every shell
|
|
50
|
+
* that starts pi, and losing them is silent: the cache quietly falls back to
|
|
51
|
+
* the five-minute tier and re-bills whole prefixes.
|
|
52
|
+
*
|
|
53
|
+
* Non-string values are dropped rather than stringified — an env var is a
|
|
54
|
+
* string, and a nested object here is a typo, not a value.
|
|
55
|
+
*/
|
|
56
|
+
export function credentialEnv(
|
|
57
|
+
credential: PengepulCredential | undefined,
|
|
58
|
+
): Record<string, string> | undefined {
|
|
59
|
+
const raw = credential?.env
|
|
60
|
+
if (typeof raw !== "object" || raw === null || Array.isArray(raw)) return undefined
|
|
61
|
+
|
|
62
|
+
const env: Record<string, string> = {}
|
|
63
|
+
for (const [name, value] of Object.entries(raw)) {
|
|
64
|
+
if (typeof value === "string") env[name] = value
|
|
65
|
+
}
|
|
66
|
+
return Object.keys(env).length > 0 ? env : undefined
|
|
67
|
+
}
|
|
68
|
+
|
|
41
69
|
export interface RelayBaseSources {
|
|
42
70
|
/** The `PENGEPUL_BASE_URL` environment value. */
|
|
43
71
|
environment?: string | undefined
|
package/extensions/models.ts
CHANGED
|
@@ -522,8 +522,22 @@ export function modelsFromApiResponse(
|
|
|
522
522
|
|
|
523
523
|
import type { Api, Model } from "@earendil-works/pi-ai"
|
|
524
524
|
|
|
525
|
+
/**
|
|
526
|
+
* Prompt cache lifetime in seconds for each retention tier a request can ask
|
|
527
|
+
* for, the shape pi reads off a model to decide when to refresh its cache
|
|
528
|
+
* entry. Declared here because pi-ai 0.85.1's `Model` has no `promptCache`;
|
|
529
|
+
* a host that predates the field ignores it.
|
|
530
|
+
*/
|
|
531
|
+
export interface PromptCacheTiers {
|
|
532
|
+
short: number
|
|
533
|
+
long: number
|
|
534
|
+
}
|
|
535
|
+
|
|
525
536
|
/** The pi-ai model shapes this provider serves: one per dialect the relay speaks. */
|
|
526
|
-
export type PengepulModelEntry =
|
|
537
|
+
export type PengepulModelEntry = (
|
|
538
|
+
| Model<"anthropic-messages">
|
|
539
|
+
| Model<"openai-completions">
|
|
540
|
+
) & { promptCache?: PromptCacheTiers }
|
|
527
541
|
|
|
528
542
|
/** The provider id every pengepul model is stamped with. */
|
|
529
543
|
export const PENGEPUL_PROVIDER_ID = "pengepul"
|
|
@@ -562,18 +576,17 @@ function sharedModelFields(model: PengepulModel, relayBase: string) {
|
|
|
562
576
|
* openai-completions carries `prompt_cache_key: <sessionId>` (and
|
|
563
577
|
* `prompt_cache_retention: "24h"`) under PI_CACHE_RETENTION=long, so the body
|
|
564
578
|
* field alone would name the conversation; anthropic-messages carries no
|
|
565
|
-
* `prompt_cache_key` at all, and pi-ai hardcodes `x-session-affinity` there
|
|
566
|
-
* which this relay does not read. Messages traffic therefore rests
|
|
567
|
-
* the relay's prefix fallback until
|
|
568
|
-
*
|
|
579
|
+
* `prompt_cache_key` at all, and pi-ai hardcodes `x-session-affinity` there up
|
|
580
|
+
* to 0.85.1, which this relay does not read. Messages traffic therefore rests
|
|
581
|
+
* on the relay's prefix fallback until 0.87, where `sessionAffinityFormat`
|
|
582
|
+
* starts being read on that dialect too and the same `openrouter` value lands
|
|
583
|
+
* as `x-session-id`.
|
|
569
584
|
*
|
|
570
|
-
* The
|
|
571
|
-
* `
|
|
572
|
-
*
|
|
573
|
-
*
|
|
574
|
-
*
|
|
575
|
-
* zero. Pin now rather than later — the cost of forgetting is silently
|
|
576
|
-
* re-billed Claude prefixes.
|
|
585
|
+
* The field is re-declared below rather than taken from
|
|
586
|
+
* `AnthropicMessagesCompat` so the pin does not depend on the host's pi-ai
|
|
587
|
+
* version: 0.85.1 ignores a value it never types, and the cost of an ignored
|
|
588
|
+
* header is zero. Pin now rather than later — the cost of forgetting is
|
|
589
|
+
* silently re-billed Claude prefixes.
|
|
577
590
|
*/
|
|
578
591
|
function affinityPin() {
|
|
579
592
|
return {
|
|
@@ -582,6 +595,32 @@ function affinityPin() {
|
|
|
582
595
|
}
|
|
583
596
|
}
|
|
584
597
|
|
|
598
|
+
/**
|
|
599
|
+
* Anthropic's prompt cache lifetimes: five minutes by default, one hour when
|
|
600
|
+
* the request asks for the extended TTL. Copied per model rather than shared,
|
|
601
|
+
* so nothing downstream can mutate one model's lifetimes into another's.
|
|
602
|
+
*
|
|
603
|
+
* pi reads `model.promptCache[tier]` to know when the entry a request wrote
|
|
604
|
+
* expires, and skips warming a model whose lifetime it cannot resolve
|
|
605
|
+
* (`cache-warmer.js` `getPromptCacheTtlMs`). No pi catalog answers for a relay
|
|
606
|
+
* id, so a Claude model served here would otherwise never be warmed: any idle
|
|
607
|
+
* past the TTL — five minutes on the default tier — re-bills the whole prefix
|
|
608
|
+
* at the write rate, which for a 700k-token Opus conversation is dollars per
|
|
609
|
+
* miss.
|
|
610
|
+
*
|
|
611
|
+
* Only the Messages wire gets them. Those ids are the ones the relay forwards
|
|
612
|
+
* to Anthropic's own API, whose lifetimes these are; the Chat Completions wire
|
|
613
|
+
* serves aggregated vendors (commandcode, openrouter) through upstreams whose
|
|
614
|
+
* cache lifetimes this provider has no measurement for, and pi does not warm a
|
|
615
|
+
* lifetime it cannot resolve.
|
|
616
|
+
*
|
|
617
|
+
* Measured, not assumed: a `cache_control` ttl of `1h` on this wire comes back
|
|
618
|
+
* with the whole prefix in `usage.cache_creation.ephemeral_1h_input_tokens`,
|
|
619
|
+
* so the relay passes the extended TTL through and `PI_CACHE_RETENTION=long`
|
|
620
|
+
* buys the hour declared here.
|
|
621
|
+
*/
|
|
622
|
+
const ANTHROPIC_PROMPT_CACHE: PromptCacheTiers = { short: 300, long: 3600 }
|
|
623
|
+
|
|
585
624
|
/**
|
|
586
625
|
* Anthropic compat as pi-ai 0.85.1 types it, plus the affinity format a later
|
|
587
626
|
* pi reads. Declared here so the pin does not depend on the host's pi-ai
|
|
@@ -633,6 +672,10 @@ export function toPengepulModels(
|
|
|
633
672
|
// The 1h cache TTL is a Messages-dialect feature too: `cache_control.ttl`
|
|
634
673
|
// has nowhere to go on the Chat Completions wire. Reasoning is not part
|
|
635
674
|
// of it — a non-reasoning Claude model caches the same way.
|
|
675
|
+
//
|
|
676
|
+
// The lifetimes pi's cache warmer schedules its refreshes against, on
|
|
677
|
+
// this wire only, for the reason just given.
|
|
678
|
+
promptCache: { ...ANTHROPIC_PROMPT_CACHE },
|
|
636
679
|
compat: {
|
|
637
680
|
...affinityPin(),
|
|
638
681
|
...(adaptive ? { forceAdaptiveThinking: true as const } : {}),
|
package/extensions/provider.ts
CHANGED
|
@@ -30,6 +30,7 @@ import { anthropicMessagesApi, lazyStream, openAICompletionsApi } from "@earendi
|
|
|
30
30
|
import { API_KEY_ENV, RELAY_BASE_ENV } from "./config.ts"
|
|
31
31
|
import {
|
|
32
32
|
credentialApiKey,
|
|
33
|
+
credentialEnv,
|
|
33
34
|
credentialRelayBase,
|
|
34
35
|
resolveApiKey,
|
|
35
36
|
resolveRelayBase,
|
|
@@ -211,9 +212,21 @@ export function createPengepulProvider(options: PengepulProviderOptions): Provid
|
|
|
211
212
|
resolve: async (input): Promise<AuthResult | undefined> => {
|
|
212
213
|
const resolved = resolveKey(input.credential)
|
|
213
214
|
if (!resolved) return undefined
|
|
215
|
+
const env = credentialEnv(input.credential as PengepulCredential | undefined)
|
|
214
216
|
// No `baseUrl` here on purpose: pi would apply it to every model and
|
|
215
217
|
// collapse the two dialect base URLs into one.
|
|
216
|
-
|
|
218
|
+
//
|
|
219
|
+
// `env` is here because pi only reads it off the credential itself
|
|
220
|
+
// when the provider brings no `resolve()` of its own — this one does,
|
|
221
|
+
// so a value like PI_CACHE_RETENTION=long is dropped unless it is
|
|
222
|
+
// handed back. Dropping it is silent and costs money: the cache falls
|
|
223
|
+
// back to the five-minute tier and re-bills the prefix on every
|
|
224
|
+
// return.
|
|
225
|
+
return {
|
|
226
|
+
auth: { apiKey: resolved.key },
|
|
227
|
+
...(env ? { env } : {}),
|
|
228
|
+
source: resolved.source,
|
|
229
|
+
}
|
|
217
230
|
},
|
|
218
231
|
},
|
|
219
232
|
},
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@pwguler/pi-pengepul-provider",
|
|
3
|
-
"version": "0.5.
|
|
3
|
+
"version": "0.5.2",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "pi custom provider for pengepul, a relay that pools your Claude/Codex subscriptions. Key and relay base live in auth.json; both native wires (Anthropic Messages, OpenAI Chat Completions) come from one catalog.",
|
|
6
6
|
"license": "MIT",
|