@pwguler/pi-pengepul-provider 0.2.3 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -20,10 +20,9 @@
20
20
  * The network/cache are injected so the catalog logic stays testable.
21
21
  */
22
22
 
23
- import { mkdir, readFile, rename, rm, writeFile } from "node:fs/promises"
24
- import { randomUUID } from "node:crypto"
25
- import { dirname } from "node:path"
23
+ import { readFile } from "node:fs/promises"
26
24
 
25
+ import { MODELS_TIMEOUT_MS_ENV } from "./config.ts"
27
26
  import { baseUrlForDialect, dialectForModelId } from "./dialect.ts"
28
27
  import type { PengepulDialect } from "./dialect.ts"
29
28
 
@@ -65,7 +64,7 @@ export interface BuiltinModelMeta {
65
64
  */
66
65
  export type BuiltinModelLookup = (id: string, dialect: PengepulDialect) => BuiltinModelMeta | undefined
67
66
 
68
- /** A pengepul model ready to become a pi `ProviderModelConfig`. */
67
+ /** A pengepul model ready to become a pi model entry. */
69
68
  export interface PengepulModel {
70
69
  id: string
71
70
  name: string
@@ -79,12 +78,6 @@ export interface PengepulModel {
79
78
  thinkingLevelMap?: Record<string, string | null>
80
79
  }
81
80
 
82
- export interface PengepulModelSource {
83
- models: readonly PengepulModel[]
84
- /** "live" = fetched from the relay; "cache" = read from disk; "empty" = none. */
85
- source: "live" | "cache" | "empty"
86
- warning?: string
87
- }
88
81
 
89
82
  /** `anthropic/claude-opus-5` -> `claude-opus-5` (the id upstream actually serves). */
90
83
  export function bareId(id: string): string {
@@ -140,6 +133,26 @@ function heuristicMeta(id: string): BuiltinModelMeta {
140
133
  }
141
134
  }
142
135
 
136
+ /**
137
+ * The catalog-id shapes to try for a relay id, in the order they are tried.
138
+ *
139
+ * The order is load-bearing, and it is not "most specific first": the last
140
+ * segment is tried before the namespace-stripped id, which is what the
141
+ * catalogs were measured to give. For 141 of the relay's 527 ids the two
142
+ * shapes resolve to different entries and the last segment wins -
143
+ * `commandcode/deepseek/deepseek-v4-pro` takes the deepseek catalog's bare
144
+ * `deepseek-v4-pro` (input 0.435) rather than openrouter's
145
+ * `deepseek/deepseek-v4-pro` (input 0.890, and a different level map).
146
+ * Reordering these shapes repoints those ids, so the test pins the order.
147
+ *
148
+ * The namespace-stripped shape is the third fallback: the only one that
149
+ * reaches an id whose catalog entry both the full id and the last segment
150
+ * miss, which is three of the relay's 527 at the time of writing.
151
+ */
152
+ export function catalogIdForms(id: string): readonly string[] {
153
+ return [...new Set([id, modelName(id), bareId(id)])]
154
+ }
155
+
143
156
  /**
144
157
  * Metadata pengepul itself advertises for a model (pengepul >= 0.6.0).
145
158
  * Every field is optional: the rollout is partial and older relays send none.
@@ -209,6 +222,43 @@ function optionalRate(value: unknown): number | undefined {
209
222
  return typeof value === "number" && Number.isFinite(value) && value >= 0 ? value : undefined
210
223
  }
211
224
 
225
+ /**
226
+ * Extended effort levels for an id no pi catalog carries, on the one relay
227
+ * namespace where that vocabulary has been measured.
228
+ *
229
+ * pi hides `xhigh` and `max` unless a model's map names them, so an id no
230
+ * catalog answers for dropped to low/medium/high and could never send the top
231
+ * of the scale. Two of the 296 reasoning openai-completions ids the provider
232
+ * registers rest on this today; the lookup's namespace-stripped shapes answer
233
+ * for the rest.
234
+ *
235
+ * Only `max` is named. DeepSeek documents low/high/max for the
236
+ * OpenAI-compatible wire and folds `minimal` into low, `medium` and `xhigh`
237
+ * into high - which is what pi's own deepseek catalog encodes, hiding `xhigh`
238
+ * outright and nulling `medium` - but the relay validates one enum for every
239
+ * family it routes, so the namespace rule cannot tell a vendor alias from a
240
+ * level in its own right. Naming the top of the scale is the part that is
241
+ * true whatever the upstream does with it; the rest stays with pi's default.
242
+ *
243
+ * Measured against the running relay: on `commandcode/` ids every requested
244
+ * effort except `minimal` is accepted, and one error text — the relay's own
245
+ * enum — answers all nine families probed (deepseek, Qwen, MiniMax, google,
246
+ * moonshotai, xiaomi, stepfun, nvidia, inclusionai), nine of the eighteen the
247
+ * namespace carries, so the vocabulary belongs to the relay rather than to any
248
+ * one model. `minimal` 400s, and the overlay below nulls it.
249
+ *
250
+ * `openrouter/` ids are left alone, but not because that namespace is
251
+ * unmeasurable: it validates no effort enum at all, so a probe there answers
252
+ * 200 whether the level exists upstream or not. What it serves is also
253
+ * heterogeneous - image generators, R1-class models that take no effort
254
+ * parameter at all - and pi's openrouter entries spell DeepSeek's top level
255
+ * `xhigh` rather than `max`, so a namespace-wide rule there has nothing solid
256
+ * to stand on.
257
+ */
258
+ function fallbackLevelMap(id: string): Record<string, string | null> | undefined {
259
+ return id.toLowerCase().startsWith("commandcode/") ? { max: "max" } : undefined
260
+ }
261
+
212
262
  /**
213
263
  * Resolve a model's metadata, most trustworthy source first:
214
264
  * 1. what pengepul advertises (first-party for this relay),
@@ -223,9 +273,28 @@ function metaFor(
223
273
  dialect: PengepulDialect,
224
274
  lookup: BuiltinModelLookup | undefined,
225
275
  ): BuiltinModelMeta {
226
- const base = lookup?.(id, dialect) ?? heuristicMeta(id)
276
+ const known = lookup?.(id, dialect)
277
+ const base = known ?? heuristicMeta(id)
227
278
  const relay = metaFromRelayEntry(entry)
228
- return relay ? { ...base, ...relay } : base
279
+ const meta = relay ? { ...base, ...relay } : base
280
+
281
+ // A model the relay tags `anthropic` is Claude-family, hence reasoning-
282
+ // capable, even when pi's catalog does not know its exact id yet.
283
+ const reasoning = meta.reasoning || entry["owned_by"] === "anthropic"
284
+
285
+ // The lookup searches every provider's catalog, so a hit need not come from
286
+ // the id's own vendor: github-copilot answers for
287
+ // `commandcode/google/gemini-3.8-flash` and carries no map, which is silence
288
+ // about another provider, not an answer about this relay. The fallback still
289
+ // stops there, deliberately. Measuring this relay shows which efforts it
290
+ // accepts and never which ones the upstream honours, so an entry's silence
291
+ // is left as silence and those ids keep pi's default until per-vendor
292
+ // evidence exists - the way DeepSeek's documented scale settled the ids the
293
+ // fallback does serve.
294
+ const fallback =
295
+ known || !reasoning || dialect !== "openai-completions" ? undefined : fallbackLevelMap(id)
296
+
297
+ return { ...meta, reasoning, ...(fallback ? { thinkingLevelMap: fallback } : {}) }
229
298
  }
230
299
 
231
300
  function toPengepulModel(
@@ -235,25 +304,25 @@ function toPengepulModel(
235
304
  const id = stringField(entry, "id")
236
305
  const dialect = dialectForModelId(id)
237
306
  const meta = metaFor(entry, id, dialect, lookup)
238
- const ownedBy = entry["owned_by"]
239
-
240
- // A model the relay tags `anthropic` is Claude-family, hence reasoning-
241
- // capable, even when pi's catalog does not know its exact id yet.
242
- const reasoning = meta.reasoning || ownedBy === "anthropic"
243
-
244
- // The relay enforces reasoning_effort low|medium|high|xhigh|max at its
245
- // request layer, uniformly across families: `minimal` 400s and its thinking
246
- // toggle never actually disables thinking. That holds no matter where the
247
- // reasoning knowledge came from, so every openai-completions reasoning model
248
- // gets the relay's shape: inherited strings case-folded to the enum (pi's
249
- // catalogs spell Google efforts `HIGH` and Qwen's `default`), values that
250
- // match under no casing hidden, and off/minimal always null. Overlay, never
251
- // replace: inherited nulls keep their levels hidden. The Messages dialect
252
- // needs none of this: it folds `minimal` into `low` and already nulls `off`
253
- // via forceAdaptiveThinking.
307
+ const reasoning = meta.reasoning
308
+
309
+ // The relay's request layer validates reasoning_effort on `commandcode/`
310
+ // ids: `minimal` 400s there, and its thinking toggle never actually disables
311
+ // thinking. On `openrouter/` ids it validates nothing, so `minimal` is a real
312
+ // level there and is left in place. `off` is different and stays nulled
313
+ // everywhere: `none` is refused by some upstreams (gemini-3.8-flash 400s on
314
+ // it) and accepted by others (deepseek), which is no basis for sending it.
315
+ // Either way every openai-completions reasoning model gets the relay's shape:
316
+ // inherited strings case-folded to the enum (pi's catalogs spell Google
317
+ // efforts `HIGH` and Qwen's `default`), values that match under no casing
318
+ // hidden. Overlay, never replace: inherited nulls keep their levels hidden.
319
+ // The Messages dialect needs none of this: it folds `minimal` into `low` and
320
+ // already nulls `off` via forceAdaptiveThinking.
321
+ const enforceRelayEnum = reasoning && dialect === "openai-completions"
254
322
  const thinkingLevelMap = relaySafeLevelMap(
255
323
  meta.thinkingLevelMap,
256
- reasoning && dialect === "openai-completions",
324
+ enforceRelayEnum,
325
+ relayAcceptsMinimal(id),
257
326
  )
258
327
 
259
328
  return {
@@ -272,32 +341,51 @@ function toPengepulModel(
272
341
  /** The efforts the relay accepts; anything else is rejected at its request layer. */
273
342
  const RELAY_EFFORTS = new Set(["low", "medium", "high", "xhigh", "max"])
274
343
 
344
+ /**
345
+ * Whether the relay's request layer takes `minimal` for this id.
346
+ *
347
+ * Measured: `commandcode/` validates its enum and answers 400 Invalid option:
348
+ * expected one of "low"|"medium"|"high"|"xhigh"|"max", while `openrouter/`
349
+ * validates nothing and answers 200 - so hiding `minimal` there costs 166 of
350
+ * the 227 openrouter models that reason, a level the relay accepts.
351
+ *
352
+ * `none` is not the same story: gemini-3.8-flash answers 400 on it while
353
+ * deepseek accepts it, so `off` stays hidden in every namespace. An unmeasured
354
+ * namespace keeps the conservative default too.
355
+ */
356
+ function relayAcceptsMinimal(id: string): boolean {
357
+ return id.toLowerCase().startsWith("openrouter/")
358
+ }
359
+
275
360
  /**
276
361
  * Shape an inherited level map for the relay's wire. With `enforce` (an
277
362
  * openai-completions reasoning model) the relay's enum is the only vocabulary
278
363
  * that reaches it: inherited strings are case-folded to the enum or hidden,
279
- * and off/minimal are always hidden. Without it the map passes through.
364
+ * and `off` is always hidden. `minimal` is hidden unless the namespace is
365
+ * known to take it. Without `enforce` the map passes through.
280
366
  */
281
367
  function relaySafeLevelMap(
282
368
  inherited: Record<string, string | null> | undefined,
283
369
  enforce: boolean,
370
+ acceptsMinimal = false,
284
371
  ): Record<string, string | null> | undefined {
285
372
  if (!enforce) return inherited ? { ...inherited } : undefined
286
373
 
287
374
  const map: Record<string, string | null> = {}
288
375
  for (const [level, mapped] of Object.entries(inherited ?? {})) {
289
- map[level] = typeof mapped === "string" ? relayEffort(mapped) : mapped
376
+ map[level] = typeof mapped === "string" ? relayEffort(mapped, acceptsMinimal) : mapped
290
377
  }
291
378
  map["off"] = null
292
- map["minimal"] = null
379
+ if (!acceptsMinimal) map["minimal"] = null
293
380
  return map
294
381
  }
295
382
 
296
383
  /** The relay's effort enum, case-folded; null keeps the level out of the picker. */
297
- function relayEffort(value: string): string | null {
384
+ function relayEffort(value: string, acceptsMinimal: boolean): string | null {
298
385
  if (RELAY_EFFORTS.has(value)) return value
299
386
  const lower = value.toLowerCase()
300
- return RELAY_EFFORTS.has(lower) ? lower : null
387
+ if (RELAY_EFFORTS.has(lower)) return lower
388
+ return acceptsMinimal && lower === "minimal" ? lower : null
301
389
  }
302
390
 
303
391
  function isRecord(value: unknown): value is Record<string, unknown> {
@@ -312,6 +400,31 @@ function stringField(record: Record<string, unknown>, key: string): string {
312
400
  return value
313
401
  }
314
402
 
403
+ /**
404
+ * The relay lists OpenRouter's batch routes as models, and OpenRouter refuses
405
+ * them on the chat wire with 404 "This model is only available through the
406
+ * Batch API". Confirmed on 10 of the relay's 77 batch ids across anthropic,
407
+ * openai, qwen, deepseek and z-ai; the rest are inferred from the same route
408
+ * rule, because probing them in bulk does not work - the refusals put the
409
+ * relay's pooled openrouter account on cooldown, which turns every following
410
+ * probe into 503 "no available openrouter account" and measures the cooldown
411
+ * rather than the model. Nothing pi sends can reach them, so they are left
412
+ * out of the catalog rather than offered as an entry that always fails.
413
+ */
414
+ function isBatchRoute(id: string): boolean {
415
+ return id.endsWith(":batch")
416
+ }
417
+
418
+ /**
419
+ * Drop the ids this relay cannot serve on either wire. Applied on the way in
420
+ * from the relay and on the way in from the cache: the cache is what covers a
421
+ * briefly absent relay, so it must not be the path that resurrects a route
422
+ * the live catalog would have dropped.
423
+ */
424
+ function servableModels(models: readonly PengepulModel[]): PengepulModel[] {
425
+ return models.filter((model) => !isBatchRoute(model.id))
426
+ }
427
+
315
428
  /** Parse the raw `/v1/models` body into models. Throws on a malformed body. */
316
429
  export function modelsFromApiResponse(
317
430
  value: unknown,
@@ -322,118 +435,177 @@ export function modelsFromApiResponse(
322
435
 
323
436
  const data = value["data"]
324
437
  if (!Array.isArray(data)) throw new Error("Expected models response data to be an array")
325
- if (data.length === 0) throw new Error("pengepul returned an empty model catalog")
326
438
 
327
- return data.map((entry) => {
328
- if (!isRecord(entry)) throw new Error("Expected model entry to be an object")
329
- return toPengepulModel(entry, lookupBuiltin)
330
- })
439
+ const models = servableModels(
440
+ data.map((entry) => {
441
+ if (!isRecord(entry)) throw new Error("Expected model entry to be an object")
442
+ return toPengepulModel(entry, lookupBuiltin)
443
+ }),
444
+ )
445
+
446
+ if (models.length === 0) throw new Error("pengepul returned an empty model catalog")
447
+ return models
331
448
  }
332
449
 
333
- /** Map models to pi `ProviderModelConfig` entries. Pure. */
334
- export function toProviderModelConfigs(
335
- models: readonly PengepulModel[],
336
- relayBase: string,
337
- ): Array<{
338
- id: string
339
- name: string
340
- api: PengepulDialect
341
- baseUrl: string
342
- reasoning: boolean
343
- input: ("text" | "image")[]
344
- cost: {
345
- input: number
346
- output: number
347
- cacheRead: number
348
- cacheWrite: number
450
+ import type { Api, Model } from "@earendil-works/pi-ai"
451
+
452
+ /** The pi-ai model shapes this provider serves: one per dialect the relay speaks. */
453
+ export type PengepulModelEntry = Model<"anthropic-messages"> | Model<"openai-completions">
454
+
455
+ /** The provider id every pengepul model is stamped with. */
456
+ export const PENGEPUL_PROVIDER_ID = "pengepul"
457
+
458
+ /** What every model carries regardless of wire: identity, pricing, and the limits. */
459
+ function sharedModelFields(model: PengepulModel, relayBase: string) {
460
+ return {
461
+ id: model.id,
462
+ name: model.name,
463
+ provider: PENGEPUL_PROVIDER_ID,
464
+ baseUrl: baseUrlForDialect(relayBase, model.dialect),
465
+ reasoning: model.reasoning,
466
+ input: model.input,
467
+ cost: model.cost,
468
+ contextWindow: model.contextWindow,
469
+ maxTokens: model.maxTokens,
349
470
  }
350
- contextWindow: number
351
- maxTokens: number
352
- thinkingLevelMap?: Record<string, string | null>
353
- compat?: {
354
- forceAdaptiveThinking?: boolean
355
- supportsLongCacheRetention?: boolean
356
- sendSessionAffinityHeaders?: boolean
357
- sessionAffinityFormat?: "openai" | "openai-nosession" | "openrouter"
471
+ }
472
+
473
+ /**
474
+ * The relay's prompt-cache affinity pin, on both wires.
475
+ *
476
+ * The relay's conversation_key resolves `x-claude-code-session-id`, then
477
+ * `x-session-id`, then the body's `prompt_cache_key`, then a hash of the
478
+ * cacheable prefix. pi emits one of those headers only for the `openrouter`
479
+ * affinity format, and only when the send flag is set. Both auto-detected
480
+ * defaults are wrong here: openai-completions picks `openai` (session_id +
481
+ * x-client-request-id + x-session-affinity) and anthropic-messages picks
482
+ * nothing at all. The header is the cheaper and more explicit of the two
483
+ * signals and it outranks the body field, so the pin keeps a session's account
484
+ * stable by the relay's first rule rather than its third. Losing it costs a
485
+ * session that migrates between pooled accounts its whole prefix: the upstream
486
+ * cache is per account.
487
+ *
488
+ * Measured, not assumed — `test/affinity-wire.test.ts` dumps both bodies:
489
+ * openai-completions carries `prompt_cache_key: <sessionId>` (and
490
+ * `prompt_cache_retention: "24h"`) under PI_CACHE_RETENTION=long, so the body
491
+ * field alone would name the conversation; anthropic-messages carries no
492
+ * `prompt_cache_key` at all, and pi-ai hardcodes `x-session-affinity` there,
493
+ * which this relay does not read. Messages traffic therefore rests entirely on
494
+ * the relay's prefix fallback until a pi release honours
495
+ * `sessionAffinityFormat` on that dialect.
496
+ *
497
+ * The two dialects do not land at the same time. openai-completions honors
498
+ * `sessionAffinityFormat` in every released pi. anthropic-messages only reads it
499
+ * from the unreleased change on pi main (commit bbb61e34a), which is why the
500
+ * field is re-declared below rather than taken from `AnthropicMessagesCompat`:
501
+ * against pi-ai 0.85.1 the pin is inert, and the cost of an ignored header is
502
+ * zero. Pin now rather than later — the cost of forgetting is silently
503
+ * re-billed Claude prefixes.
504
+ */
505
+ function affinityPin() {
506
+ return {
507
+ sendSessionAffinityHeaders: true as const,
508
+ sessionAffinityFormat: "openrouter" as const,
358
509
  }
359
- }> {
510
+ }
511
+
512
+ /**
513
+ * Anthropic compat as pi-ai 0.85.1 types it, plus the affinity format a later
514
+ * pi reads. Declared here so the pin does not depend on the host's pi-ai
515
+ * version; the field is optional, so a host that predates it ignores the value.
516
+ */
517
+ type AnthropicCompat = NonNullable<Model<"anthropic-messages">["compat"]> & {
518
+ sessionAffinityFormat?: "openrouter"
519
+ }
520
+
521
+ /**
522
+ * Map the relay catalog to pi models, one base URL per dialect. Pure.
523
+ *
524
+ * The per-model `baseUrl` is the only place the dialect split can live: pi
525
+ * applies a base URL returned from `auth.resolve()` to every model at once
526
+ * (`models.js` `applyAuth`), so a provider-wide value would send Anthropic
527
+ * Messages traffic to `/v1` and Chat Completions traffic to the root.
528
+ */
529
+ export function toPengepulModels(
530
+ models: readonly PengepulModel[],
531
+ relayBase: string,
532
+ ): PengepulModelEntry[] {
360
533
  return models.map((model) => {
361
- const adaptive = model.dialect === "anthropic-messages" && model.reasoning
362
- // The 1h cache TTL is a Messages-dialect feature: `cache_control.ttl`
363
- // has nowhere to go on the Chat Completions wire. Reasoning is not part
364
- // of it — a non-reasoning Claude model caches the same way.
365
- const longCacheRetention = model.dialect === "anthropic-messages"
534
+ const shared = sharedModelFields(model, relayBase)
535
+ if (model.dialect === "anthropic-messages") {
536
+ const adaptive = model.reasoning
537
+ return {
538
+ ...shared,
539
+ api: "anthropic-messages" as const,
540
+ // Inherited level mapping (e.g. deepseek {high:"high"}) flows through;
541
+ // adaptive Claude models additionally mark "off" unsupported so the
542
+ // stream omits thinking:{type:"disabled"} (upstream rejects it).
543
+ ...(model.thinkingLevelMap || adaptive
544
+ ? {
545
+ thinkingLevelMap: {
546
+ ...(model.thinkingLevelMap ?? {}),
547
+ ...(adaptive ? { off: null } : {}),
548
+ },
549
+ }
550
+ : {}),
551
+ // Reasoning-capable Claude models run on the adaptive-thinking wire:
552
+ // pi's streamSimple always passes thinkingEnabled:false when no level is
553
+ // selected, and the stream would send thinking:{type:"disabled"}, which
554
+ // the upstream rejects (400: "thinking.type.disabled is not supported
555
+ // for this model"). thinkingLevelMap.off = null marks "off" as
556
+ // unsupported so pi omits the thinking param entirely (server default
557
+ // = adaptive), and forceAdaptiveThinking routes an explicit level to
558
+ // {type:"adaptive"} + effort instead of budget_tokens.
559
+ //
560
+ // The 1h cache TTL is a Messages-dialect feature too: `cache_control.ttl`
561
+ // has nowhere to go on the Chat Completions wire. Reasoning is not part
562
+ // of it — a non-reasoning Claude model caches the same way.
563
+ compat: {
564
+ ...affinityPin(),
565
+ ...(adaptive ? { forceAdaptiveThinking: true as const } : {}),
566
+ supportsLongCacheRetention: true as const,
567
+ } satisfies AnthropicCompat,
568
+ }
569
+ }
570
+
366
571
  return {
367
- id: model.id,
368
- name: model.name,
369
- api: model.dialect,
370
- baseUrl: baseUrlForDialect(relayBase, model.dialect),
371
- reasoning: model.reasoning,
372
- input: model.input,
373
- cost: model.cost,
374
- contextWindow: model.contextWindow,
375
- maxTokens: model.maxTokens,
376
- // Inherited level mapping (e.g. deepseek {high:"high"}) flows through;
377
- // adaptive Claude models additionally mark "off" unsupported so the
378
- // stream omits thinking:{type:"disabled"} (upstream rejects it).
379
- ...(model.thinkingLevelMap || adaptive
380
- ? { thinkingLevelMap: { ...(model.thinkingLevelMap ?? {}), ...(adaptive ? { off: null } : {}) } }
381
- : {}),
382
- // Reasoning-capable Claude models run on the adaptive-thinking wire:
383
- // pi's streamSimple always passes thinkingEnabled:false when no level is
384
- // selected, and the stream would send thinking:{type:"disabled"}, which
385
- // the upstream rejects (400: "thinking.type.disabled is not supported
386
- // for this model"). thinkingLevelMap.off = null marks "off" as
387
- // unsupported so pi omits the thinking param entirely (server default
388
- // = adaptive), and forceAdaptiveThinking routes an explicit level to
389
- // {type:"adaptive"} + effort instead of budget_tokens.
390
- // Both dialects pin both fields, and that is why `compat` is
391
- // unconditional. The relay's prompt-cache affinity key resolves in this
392
- // order: `x-claude-code-session-id`, `x-session-id`, the body's
393
- // `prompt_cache_key`, then a hash of the cacheable request prefix
394
- // (app.rs `conversation_key`). pi emits `x-session-id` only for the
395
- // `openrouter` affinity format, and only when the send flag is set;
396
- // the auto-detected defaults are wrong here in both cases
397
- // (openai-completions picks `openai`: session_id + x-client-request-id +
398
- // x-session-affinity; anthropic-messages picks nothing). The header is
399
- // the cheaper and more explicit of the two signals and it outranks the
400
- // body field, so pinning it keeps a session's account stable by the
401
- // relay's first rule rather than its third. Losing the pin costs a
402
- // session that migrates between pooled accounts its whole prefix: the
403
- // upstream cache is per account.
404
- //
405
- // Measured, not assumed — `test/affinity-wire.test.ts` dumps both bodies:
406
- // openai-completions carries `prompt_cache_key: <sessionId>` (and
407
- // `prompt_cache_retention: "24h"`) under PI_CACHE_RETENTION=long, so the
408
- // body field alone would name the conversation; anthropic-messages
409
- // carries no `prompt_cache_key` at all, and pi-ai hardcodes
410
- // `x-session-affinity` there, which this relay does not read. Messages
411
- // traffic therefore rests entirely on the relay's prefix fallback until
412
- // a pi release honours `sessionAffinityFormat` on that dialect.
413
- //
414
- // The two dialects do not land at the same time. openai-completions
415
- // honors `sessionAffinityFormat` in every released pi. anthropic-messages
416
- // only reads it from the unreleased change on pi main (commit
417
- // bbb61e34a); through pi-ai 0.85.1 that client hardcodes the header name
418
- // `x-session-affinity`, which this relay does not read, so the Claude
419
- // pin below is inert until pi ships it. Pin now rather than later: the
420
- // cost of an ignored header is zero, and the cost of forgetting is
421
- // silently re-billed Claude prefixes.
422
- compat: {
423
- ...(adaptive ? { forceAdaptiveThinking: true as const } : {}),
424
- ...(longCacheRetention ? { supportsLongCacheRetention: true as const } : {}),
425
- sendSessionAffinityHeaders: true as const,
426
- sessionAffinityFormat: "openrouter" as const,
427
- },
572
+ ...shared,
573
+ api: "openai-completions" as const,
574
+ ...(model.thinkingLevelMap ? { thinkingLevelMap: model.thinkingLevelMap } : {}),
575
+ compat: { ...affinityPin() },
428
576
  }
429
577
  })
430
578
  }
431
579
 
580
+ /** Whether a stored pi model is one this provider published. */
581
+ export function isPengepulModelEntry(model: Model<Api>): model is PengepulModelEntry {
582
+ return (
583
+ model.provider === PENGEPUL_PROVIDER_ID &&
584
+ (model.api === "anthropic-messages" || model.api === "openai-completions")
585
+ )
586
+ }
587
+
588
+ /**
589
+ * Re-derive every model's base URL from the base currently configured.
590
+ *
591
+ * pi's model store keeps whole models, baseUrl included, and replays them
592
+ * before the network phase. A relay that moved would otherwise be reached at
593
+ * its old address until a fetch succeeds — which never happens when the old
594
+ * address is gone.
595
+ */
596
+ export function restampRelayBase(
597
+ models: readonly PengepulModelEntry[],
598
+ relayBase: string,
599
+ ): PengepulModelEntry[] {
600
+ return models.map((model) => ({
601
+ ...model,
602
+ baseUrl: baseUrlForDialect(relayBase, model.api),
603
+ }))
604
+ }
605
+
432
606
  /** Picker label: the bare model part of a relay id, suffixed. `anthropic/claude-opus-5` -> `claude-opus-5 (pengepul)`. */
433
607
  function displayName(id: string): string {
434
- const slash = id.indexOf("/")
435
- const bare = slash === -1 ? id : id.slice(slash + 1)
436
- return `${bare} (pengepul)`
608
+ return `${bareId(id)} (pengepul)`
437
609
  }
438
610
 
439
611
  interface FetchModelsOptions {
@@ -466,7 +638,7 @@ function configuredTimeoutMs(timeoutMs: number | undefined): number {
466
638
  }
467
639
 
468
640
  export function getModelsTimeoutMs(env: NodeJS.ProcessEnv = process.env): number {
469
- const raw = env["PENGEPUL_MODELS_TIMEOUT_MS"]
641
+ const raw = env[MODELS_TIMEOUT_MS_ENV]
470
642
  if (!raw) return DEFAULT_MODELS_TIMEOUT_MS
471
643
  const parsed = Number(raw)
472
644
  return configuredTimeoutMs(parsed)
@@ -552,7 +724,7 @@ export async function fetchPengepulModels(
552
724
  throw new Error(
553
725
  `pengepul rejected the API key (${
554
726
  response.status
555
- }). Set PENGEPUL_API_KEY or check ~/.pengepul/config.yaml.`,
727
+ }). Run /login pengepul, or set the key in ~/.pi/agent/auth.json or PENGEPUL_API_KEY.`,
556
728
  )
557
729
  }
558
730
  if (!response.ok) {
@@ -639,8 +811,9 @@ export function modelsFromCache(value: unknown): readonly PengepulModel[] {
639
811
  : {}),
640
812
  }
641
813
  })
642
- if (parsed.length === 0) throw new Error("pengepul cache holds no valid models")
643
- return parsed
814
+ const servable = servableModels(parsed)
815
+ if (servable.length === 0) throw new Error("pengepul cache holds no valid models")
816
+ return servable
644
817
  }
645
818
 
646
819
  async function readCache(cachePath: string): Promise<readonly PengepulModel[]> {
@@ -659,65 +832,4 @@ export async function loadCachedPengepulModels(
659
832
  }
660
833
  }
661
834
 
662
- async function writeCache(cachePath: string, models: readonly PengepulModel[]): Promise<void> {
663
- await mkdir(dirname(cachePath), { recursive: true })
664
- // Unique per write: the runtime can issue two overlapping writes in one
665
- // process (cache-first + background refresh), and a shared pid-keyed name
666
- // would let the first rename remove the second's source mid-flight.
667
- const temporaryPath = `${cachePath}.${process.pid}.${randomUUID()}.tmp`
668
-
669
- try {
670
- await writeFile(
671
- temporaryPath,
672
- `${JSON.stringify({ version: MODEL_CACHE_VERSION, models }, null, 2)}\n`,
673
- { encoding: "utf-8", mode: 0o600 },
674
- )
675
- await rename(temporaryPath, cachePath)
676
- } finally {
677
- try {
678
- await rm(temporaryPath, { force: true })
679
- } catch {
680
- // Best-effort cleanup must not hide the original cache write error.
681
- }
682
- }
683
- }
684
-
685
- export async function loadPengepulModels(
686
- options: LoadModelsOptions,
687
- ): Promise<PengepulModelSource> {
688
- const cachePath = options.cachePath
689
-
690
- try {
691
- const models = await fetchPengepulModels(options)
692
-
693
- try {
694
- await writeCache(cachePath, models)
695
- return { models, source: "live" }
696
- } catch (error) {
697
- return {
698
- models,
699
- source: "live",
700
- warning: `Loaded the live pengepul model catalog but could not update ${cachePath}: ${errorMessage(error)}`,
701
- }
702
- }
703
- } catch (liveError) {
704
- if (options.signal?.aborted) throw abortError(options.signal.reason ?? liveError)
705
-
706
- try {
707
- const models = await readCache(cachePath)
708
- return {
709
- models,
710
- source: "cache",
711
- warning: `Could not refresh the pengepul model catalog (${errorMessage(liveError)}). Using the cached catalog from ${cachePath}.`,
712
- }
713
- } catch (cacheError) {
714
- return {
715
- models: [],
716
- source: "empty",
717
- warning: `Could not refresh the pengepul model catalog (${errorMessage(liveError)}), and no valid cached catalog is available at ${cachePath} (${errorMessage(cacheError)}). pengepul models will remain unavailable until the next startup refresh succeeds.`,
718
- }
719
- }
720
- }
721
- }
722
-
723
835