@pwguler/pi-pengepul-provider 0.5.0 → 0.5.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -58,6 +58,10 @@ prefixed `pengepul/<id>`. To try it without installing, use
58
58
  relay already serves — takes them from the same family's previous minor, so
59
59
  it does not sit at `high` until the next pi catalog update. Nothing to
60
60
  configure: `thinkingLevelMap` is resolved per id here.
61
+ - Publishes Anthropic's cache lifetimes (5 minutes, 1 hour) on the Messages
62
+ wire, so pi's cache warmer can keep a Claude conversation's entry alive
63
+ instead of re-billing its whole prefix after five idle minutes. See
64
+ [Prompt cache](#prompt-cache).
61
65
  - Caches the catalog in pi's model store, so startup does not wait on the
62
66
  network and a briefly absent relay is covered. The cached models are
63
67
  re-pointed at the relay base configured now, so moving the relay does not
@@ -107,6 +111,52 @@ to every model of the provider, which collapses the two wires onto one URL —
107
111
  Anthropic Messages traffic would be sent to `/v1` and Chat Completions traffic
108
112
  to `/`.
109
113
 
114
+ ## Prompt cache
115
+
116
+ The upstream prompt cache is one entry per account and lives five minutes by
117
+ default, so a conversation that waits longer than that pays for its whole prefix
118
+ again — on a 700k-token Opus session, dollars per return, logged by the relay as
119
+ a `cache_write` with a near-zero `cache_read`.
120
+
121
+ This provider declares Anthropic's lifetimes (5 minutes, 1 hour) on every
122
+ Messages-wire model, which is what lets pi keep an entry alive at all: pi
123
+ refreshes an entry by re-sending the prompt with a one-token cap before it
124
+ expires, and it only warms a model whose lifetime it can resolve. Two pi
125
+ settings decide when that happens, and neither covers an idle gap out of the
126
+ box:
127
+
128
+ | Setting | Where | Effect |
129
+ |---|---|---|
130
+ | `cacheWarming: "idle"` | `~/.pi/agent/settings.json` | Warms between agent runs. The default, `"streaming"`, stops as soon as a run settles — exactly when a five-minute entry is about to expire. |
131
+ | `env: { "PI_CACHE_RETENTION": "long" }` | the `pengepul` credential in `auth.json` | Asks for the 1-hour tier instead of the 5-minute one. The relay forwards the extended TTL to Anthropic. |
132
+
133
+ ```json
134
+ {
135
+ "pengepul": {
136
+ "type": "api_key",
137
+ "key": "sk-local-...",
138
+ "baseUrl": "http://127.0.0.1:8317",
139
+ "env": { "PI_CACHE_RETENTION": "long" }
140
+ }
141
+ }
142
+ ```
143
+
144
+ `env` is read off the credential by this provider, not by pi core: a provider
145
+ that brings its own `resolve()` has to hand the values back, and pengepul does.
146
+ Re-add them after `/login pengepul` — pi replaces the whole credential on login,
147
+ so a key rotation drops the block and, with it, the tier.
148
+
149
+ Idle warming covers gaps up to 30 minutes, pi's own safety limit, and pi only
150
+ triggers it when the expected saving clears a threshold of its own — a small
151
+ prompt is left to expire rather than refreshed. The long tier is what carries a
152
+ longer break. Both are cheap next to a miss: a refresh is one cache read plus a
153
+ one-token completion, while the long tier only raises the write premium, on the
154
+ few hundred tokens a turn actually adds.
155
+
156
+ `/session` reports the mode, the refresh cost, and the miss penalty it is
157
+ weighing, and `showCacheMissNotices` in `settings.json` prints each miss with
158
+ its re-billed token count.
159
+
110
160
  ## Upgrading from 0.4.0 or earlier
111
161
 
112
162
  This release is breaking: the relay's own `~/.pengepul/config.yaml` is no longer
@@ -20,6 +20,8 @@ export interface PengepulCredential {
20
20
  type?: unknown
21
21
  key?: unknown
22
22
  baseUrl?: unknown
23
+ /** Provider-scoped environment values pi stores alongside the key. */
24
+ env?: unknown
23
25
  }
24
26
 
25
27
  function nonEmptyString(value: unknown): string | undefined {
@@ -38,6 +40,32 @@ export function credentialApiKey(credential: PengepulCredential | undefined): st
38
40
  return nonEmptyString(credential?.key)
39
41
  }
40
42
 
43
+ /**
44
+ * The provider-scoped environment the credential carries, when it holds any.
45
+ *
46
+ * pi core fills an `AuthResult.env` from here only for providers that bring no
47
+ * `resolve()` of their own; this provider brings one, so the values have to be
48
+ * passed through or they never reach a request. They are how a user sets
49
+ * `PI_CACHE_RETENTION=long` for the relay without exporting it in every shell
50
+ * that starts pi, and losing them is silent: the cache quietly falls back to
51
+ * the five-minute tier and re-bills whole prefixes.
52
+ *
53
+ * Non-string values are dropped rather than stringified — an env var is a
54
+ * string, and a nested object here is a typo, not a value.
55
+ */
56
+ export function credentialEnv(
57
+ credential: PengepulCredential | undefined,
58
+ ): Record<string, string> | undefined {
59
+ const raw = credential?.env
60
+ if (typeof raw !== "object" || raw === null || Array.isArray(raw)) return undefined
61
+
62
+ const env: Record<string, string> = {}
63
+ for (const [name, value] of Object.entries(raw)) {
64
+ if (typeof value === "string") env[name] = value
65
+ }
66
+ return Object.keys(env).length > 0 ? env : undefined
67
+ }
68
+
41
69
  export interface RelayBaseSources {
42
70
  /** The `PENGEPUL_BASE_URL` environment value. */
43
71
  environment?: string | undefined
@@ -522,8 +522,22 @@ export function modelsFromApiResponse(
522
522
 
523
523
  import type { Api, Model } from "@earendil-works/pi-ai"
524
524
 
525
+ /**
526
+ * Prompt cache lifetime in seconds for each retention tier a request can ask
527
+ * for, the shape pi reads off a model to decide when to refresh its cache
528
+ * entry. Declared here because pi-ai 0.85.1's `Model` has no `promptCache`;
529
+ * a host that predates the field ignores it.
530
+ */
531
+ export interface PromptCacheTiers {
532
+ short: number
533
+ long: number
534
+ }
535
+
525
536
  /** The pi-ai model shapes this provider serves: one per dialect the relay speaks. */
526
- export type PengepulModelEntry = Model<"anthropic-messages"> | Model<"openai-completions">
537
+ export type PengepulModelEntry = (
538
+ | Model<"anthropic-messages">
539
+ | Model<"openai-completions">
540
+ ) & { promptCache?: PromptCacheTiers }
527
541
 
528
542
  /** The provider id every pengepul model is stamped with. */
529
543
  export const PENGEPUL_PROVIDER_ID = "pengepul"
@@ -562,18 +576,17 @@ function sharedModelFields(model: PengepulModel, relayBase: string) {
562
576
  * openai-completions carries `prompt_cache_key: <sessionId>` (and
563
577
  * `prompt_cache_retention: "24h"`) under PI_CACHE_RETENTION=long, so the body
564
578
  * field alone would name the conversation; anthropic-messages carries no
565
- * `prompt_cache_key` at all, and pi-ai hardcodes `x-session-affinity` there,
566
- * which this relay does not read. Messages traffic therefore rests entirely on
567
- * the relay's prefix fallback until a pi release honours
568
- * `sessionAffinityFormat` on that dialect.
579
+ * `prompt_cache_key` at all, and pi-ai hardcodes `x-session-affinity` there up
580
+ * to 0.85.1, which this relay does not read. Messages traffic therefore rests
581
+ * on the relay's prefix fallback until 0.87, where `sessionAffinityFormat`
582
+ * starts being read on that dialect too and the same `openrouter` value lands
583
+ * as `x-session-id`.
569
584
  *
570
- * The two dialects do not land at the same time. openai-completions honors
571
- * `sessionAffinityFormat` in every released pi. anthropic-messages only reads it
572
- * from the unreleased change on pi main (commit bbb61e34a), which is why the
573
- * field is re-declared below rather than taken from `AnthropicMessagesCompat`:
574
- * against pi-ai 0.85.1 the pin is inert, and the cost of an ignored header is
575
- * zero. Pin now rather than later — the cost of forgetting is silently
576
- * re-billed Claude prefixes.
585
+ * The field is re-declared below rather than taken from
586
+ * `AnthropicMessagesCompat` so the pin does not depend on the host's pi-ai
587
+ * version: 0.85.1 ignores a value it never types, and the cost of an ignored
588
+ * header is zero. Pin now rather than later — the cost of forgetting is
589
+ * silently re-billed Claude prefixes.
577
590
  */
578
591
  function affinityPin() {
579
592
  return {
@@ -582,6 +595,32 @@ function affinityPin() {
582
595
  }
583
596
  }
584
597
 
598
+ /**
599
+ * Anthropic's prompt cache lifetimes: five minutes by default, one hour when
600
+ * the request asks for the extended TTL. Copied per model rather than shared,
601
+ * so nothing downstream can mutate one model's lifetimes into another's.
602
+ *
603
+ * pi reads `model.promptCache[tier]` to know when the entry a request wrote
604
+ * expires, and skips warming a model whose lifetime it cannot resolve
605
+ * (`cache-warmer.js` `getPromptCacheTtlMs`). No pi catalog answers for a relay
606
+ * id, so a Claude model served here would otherwise never be warmed: any idle
607
+ * past the TTL — five minutes on the default tier — re-bills the whole prefix
608
+ * at the write rate, which for a 700k-token Opus conversation is dollars per
609
+ * miss.
610
+ *
611
+ * Only the Messages wire gets them. Those ids are the ones the relay forwards
612
+ * to Anthropic's own API, whose lifetimes these are; the Chat Completions wire
613
+ * serves aggregated vendors (commandcode, openrouter) through upstreams whose
614
+ * cache lifetimes this provider has no measurement for, and pi does not warm a
615
+ * lifetime it cannot resolve.
616
+ *
617
+ * Measured, not assumed: a `cache_control` ttl of `1h` on this wire comes back
618
+ * with the whole prefix in `usage.cache_creation.ephemeral_1h_input_tokens`,
619
+ * so the relay passes the extended TTL through and `PI_CACHE_RETENTION=long`
620
+ * buys the hour declared here.
621
+ */
622
+ const ANTHROPIC_PROMPT_CACHE: PromptCacheTiers = { short: 300, long: 3600 }
623
+
585
624
  /**
586
625
  * Anthropic compat as pi-ai 0.85.1 types it, plus the affinity format a later
587
626
  * pi reads. Declared here so the pin does not depend on the host's pi-ai
@@ -633,6 +672,10 @@ export function toPengepulModels(
633
672
  // The 1h cache TTL is a Messages-dialect feature too: `cache_control.ttl`
634
673
  // has nowhere to go on the Chat Completions wire. Reasoning is not part
635
674
  // of it — a non-reasoning Claude model caches the same way.
675
+ //
676
+ // The lifetimes pi's cache warmer schedules its refreshes against, on
677
+ // this wire only, for the reason just given.
678
+ promptCache: { ...ANTHROPIC_PROMPT_CACHE },
636
679
  compat: {
637
680
  ...affinityPin(),
638
681
  ...(adaptive ? { forceAdaptiveThinking: true as const } : {}),
@@ -30,6 +30,7 @@ import { anthropicMessagesApi, lazyStream, openAICompletionsApi } from "@earendi
30
30
  import { API_KEY_ENV, RELAY_BASE_ENV } from "./config.ts"
31
31
  import {
32
32
  credentialApiKey,
33
+ credentialEnv,
33
34
  credentialRelayBase,
34
35
  resolveApiKey,
35
36
  resolveRelayBase,
@@ -211,9 +212,21 @@ export function createPengepulProvider(options: PengepulProviderOptions): Provid
211
212
  resolve: async (input): Promise<AuthResult | undefined> => {
212
213
  const resolved = resolveKey(input.credential)
213
214
  if (!resolved) return undefined
215
+ const env = credentialEnv(input.credential as PengepulCredential | undefined)
214
216
  // No `baseUrl` here on purpose: pi would apply it to every model and
215
217
  // collapse the two dialect base URLs into one.
216
- return { auth: { apiKey: resolved.key }, source: resolved.source }
218
+ //
219
+ // `env` is here because pi only reads it off the credential itself
220
+ // when the provider brings no `resolve()` of its own — this one does,
221
+ // so a value like PI_CACHE_RETENTION=long is dropped unless it is
222
+ // handed back. Dropping it is silent and costs money: the cache falls
223
+ // back to the five-minute tier and re-bills the prefix on every
224
+ // return.
225
+ return {
226
+ auth: { apiKey: resolved.key },
227
+ ...(env ? { env } : {}),
228
+ source: resolved.source,
229
+ }
217
230
  },
218
231
  },
219
232
  },
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@pwguler/pi-pengepul-provider",
3
- "version": "0.5.0",
3
+ "version": "0.5.2",
4
4
  "type": "module",
5
5
  "description": "pi custom provider for pengepul, a relay that pools your Claude/Codex subscriptions. Key and relay base live in auth.json; both native wires (Anthropic Messages, OpenAI Chat Completions) come from one catalog.",
6
6
  "license": "MIT",