@pwguler/pi-pengepul-provider 0.2.4 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -20,10 +20,9 @@
20
20
  * The network/cache are injected so the catalog logic stays testable.
21
21
  */
22
22
 
23
- import { mkdir, readFile, rename, rm, writeFile } from "node:fs/promises"
24
- import { randomUUID } from "node:crypto"
25
- import { dirname } from "node:path"
23
+ import { readFile } from "node:fs/promises"
26
24
 
25
+ import { MODELS_TIMEOUT_MS_ENV } from "./config.ts"
27
26
  import { baseUrlForDialect, dialectForModelId } from "./dialect.ts"
28
27
  import type { PengepulDialect } from "./dialect.ts"
29
28
 
@@ -65,7 +64,7 @@ export interface BuiltinModelMeta {
65
64
  */
66
65
  export type BuiltinModelLookup = (id: string, dialect: PengepulDialect) => BuiltinModelMeta | undefined
67
66
 
68
- /** A pengepul model ready to become a pi `ProviderModelConfig`. */
67
+ /** A pengepul model ready to become a pi model entry. */
69
68
  export interface PengepulModel {
70
69
  id: string
71
70
  name: string
@@ -79,12 +78,6 @@ export interface PengepulModel {
79
78
  thinkingLevelMap?: Record<string, string | null>
80
79
  }
81
80
 
82
- export interface PengepulModelSource {
83
- models: readonly PengepulModel[]
84
- /** "live" = fetched from the relay; "cache" = read from disk; "empty" = none. */
85
- source: "live" | "cache" | "empty"
86
- warning?: string
87
- }
88
81
 
89
82
  /** `anthropic/claude-opus-5` -> `claude-opus-5` (the id upstream actually serves). */
90
83
  export function bareId(id: string): string {
@@ -266,6 +259,63 @@ function fallbackLevelMap(id: string): Record<string, string | null> | undefined
266
259
  return id.toLowerCase().startsWith("commandcode/") ? { max: "max" } : undefined
267
260
  }
268
261
 
262
+ /** The levels pi hides unless a map names them. */
263
+ const INHERITED_LEVELS = ["xhigh", "max"] as const
264
+
265
+ /**
266
+ * Extended levels a family's previous minor already carries.
267
+ *
268
+ * pi hides `xhigh` and `max` unless a model's map names them, so an id the
269
+ * catalogs have not caught up with - a Claude point release the relay serves
270
+ * today, before pi's catalog carries it - dropped to low/medium/high and could
271
+ * never send the top of a ladder its own family publishes: in the catalog Opus
272
+ * 4.5 names neither extension, 4.6 names max, and 4.7, 4.8 and 5 name xhigh and
273
+ * max. The previous minor's own entry is that answer, and the only evidence
274
+ * there is: the relay advertises no effort metadata, so `GET /v1/models` has
275
+ * nothing to say about levels and pi's map is the whole vocabulary.
276
+ *
277
+ * Add-only by construction. Only levels the sibling maps to a string come
278
+ * across, so an id that reaches this path can gain the top of the scale but
279
+ * never lose a level that works today, and never takes on a sibling's spelling
280
+ * for off/low/medium/high - the dialect rules downstream own those.
281
+ *
282
+ * Measured against the live catalog at the time of writing: exactly two ids
283
+ * enter this path, `anthropic/claude-opus-5-5` and
284
+ * `openrouter/anthropic/claude-opus-5.5`, each inheriting xhigh and max from
285
+ * `claude-opus-5`. No other family in that catalog was touched.
286
+ */
287
+ function inheritedLevelMap(
288
+ id: string,
289
+ dialect: PengepulDialect,
290
+ lookup: BuiltinModelLookup | undefined,
291
+ ): Record<string, string | null> | undefined {
292
+ const siblingId = previousMinorId(id)
293
+ const siblingMap = siblingId ? lookup?.(siblingId, dialect)?.thinkingLevelMap : undefined
294
+ if (!siblingMap) return undefined
295
+
296
+ const inherited: Record<string, string | null> = {}
297
+ for (const level of INHERITED_LEVELS) {
298
+ const mapped = siblingMap[level]
299
+ if (typeof mapped === "string") inherited[level] = mapped
300
+ }
301
+ return Object.keys(inherited).length > 0 ? inherited : undefined
302
+ }
303
+
304
+ /**
305
+ * The same id one minor older, keeping the routing namespace:
306
+ * `anthropic/claude-opus-5-5` -> `anthropic/claude-opus-5`, and the dotted
307
+ * `claude-opus-5.5` -> `claude-opus-5` for the Chat Completions spelling.
308
+ * Undefined when the name carries no trailing version segment to drop, which
309
+ * keeps qualifiers (`-preview`, `-flash`, `-fast`) out of the derivation.
310
+ */
311
+ function previousMinorId(id: string): string | undefined {
312
+ const slash = id.lastIndexOf("/")
313
+ const name = slash === -1 ? id : id.slice(slash + 1)
314
+ const stripped = name.replace(/[.-]\d+$/, "")
315
+ if (stripped === name || stripped === "") return undefined
316
+ return slash === -1 ? stripped : `${id.slice(0, slash + 1)}${stripped}`
317
+ }
318
+
269
319
  /**
270
320
  * Resolve a model's metadata, most trustworthy source first:
271
321
  * 1. what pengepul advertises (first-party for this relay),
@@ -273,6 +323,11 @@ function fallbackLevelMap(id: string): Record<string, string | null> | undefined
273
323
  * 3. family heuristics.
274
324
  * Sources merge per field, so a relay that sends only `context_window` still
275
325
  * picks up pricing and modalities from the catalog.
326
+ *
327
+ * A catalog miss on a reasoning model still gets its extended levels from the
328
+ * family's previous minor where the catalog carries one (`inheritedLevelMap`);
329
+ * the namespace fallback for the Chat Completions wire then takes precedence
330
+ * over both.
276
331
  */
277
332
  function metaFor(
278
333
  entry: Record<string, unknown>,
@@ -301,7 +356,20 @@ function metaFor(
301
356
  const fallback =
302
357
  known || !reasoning || dialect !== "openai-completions" ? undefined : fallbackLevelMap(id)
303
358
 
304
- return { ...meta, reasoning, ...(fallback ? { thinkingLevelMap: fallback } : {}) }
359
+ // A catalog miss is not always silence: the family's previous minor answers
360
+ // for the levels the catalog has not named yet. Whatever the base resolved
361
+ // wins over it, so a relay that starts advertising a map still outranks this.
362
+ const inherited = known || !reasoning ? undefined : inheritedLevelMap(id, dialect, lookup)
363
+ const levelMap = inherited
364
+ ? { ...inherited, ...(meta.thinkingLevelMap ?? {}) }
365
+ : meta.thinkingLevelMap
366
+
367
+ return {
368
+ ...meta,
369
+ reasoning,
370
+ ...(levelMap ? { thinkingLevelMap: levelMap } : {}),
371
+ ...(fallback ? { thinkingLevelMap: fallback } : {}),
372
+ }
305
373
  }
306
374
 
307
375
  function toPengepulModel(
@@ -454,105 +522,162 @@ export function modelsFromApiResponse(
454
522
  return models
455
523
  }
456
524
 
457
- /** Map models to pi `ProviderModelConfig` entries. Pure. */
458
- export function toProviderModelConfigs(
459
- models: readonly PengepulModel[],
460
- relayBase: string,
461
- ): Array<{
462
- id: string
463
- name: string
464
- api: PengepulDialect
465
- baseUrl: string
466
- reasoning: boolean
467
- input: ("text" | "image")[]
468
- cost: {
469
- input: number
470
- output: number
471
- cacheRead: number
472
- cacheWrite: number
525
+ import type { Api, Model } from "@earendil-works/pi-ai"
526
+
527
+ /** The pi-ai model shapes this provider serves: one per dialect the relay speaks. */
528
+ export type PengepulModelEntry = Model<"anthropic-messages"> | Model<"openai-completions">
529
+
530
+ /** The provider id every pengepul model is stamped with. */
531
+ export const PENGEPUL_PROVIDER_ID = "pengepul"
532
+
533
+ /** What every model carries regardless of wire: identity, pricing, and the limits. */
534
+ function sharedModelFields(model: PengepulModel, relayBase: string) {
535
+ return {
536
+ id: model.id,
537
+ name: model.name,
538
+ provider: PENGEPUL_PROVIDER_ID,
539
+ baseUrl: baseUrlForDialect(relayBase, model.dialect),
540
+ reasoning: model.reasoning,
541
+ input: model.input,
542
+ cost: model.cost,
543
+ contextWindow: model.contextWindow,
544
+ maxTokens: model.maxTokens,
473
545
  }
474
- contextWindow: number
475
- maxTokens: number
476
- thinkingLevelMap?: Record<string, string | null>
477
- compat?: {
478
- forceAdaptiveThinking?: boolean
479
- supportsLongCacheRetention?: boolean
480
- sendSessionAffinityHeaders?: boolean
481
- sessionAffinityFormat?: "openai" | "openai-nosession" | "openrouter"
546
+ }
547
+
548
+ /**
549
+ * The relay's prompt-cache affinity pin, on both wires.
550
+ *
551
+ * The relay's conversation_key resolves `x-claude-code-session-id`, then
552
+ * `x-session-id`, then the body's `prompt_cache_key`, then a hash of the
553
+ * cacheable prefix. pi emits one of those headers only for the `openrouter`
554
+ * affinity format, and only when the send flag is set. Both auto-detected
555
+ * defaults are wrong here: openai-completions picks `openai` (session_id +
556
+ * x-client-request-id + x-session-affinity) and anthropic-messages picks
557
+ * nothing at all. The header is the cheaper and more explicit of the two
558
+ * signals and it outranks the body field, so the pin keeps a session's account
559
+ * stable by the relay's first rule rather than its third. Losing it costs a
560
+ * session that migrates between pooled accounts its whole prefix: the upstream
561
+ * cache is per account.
562
+ *
563
+ * Measured, not assumed — `test/affinity-wire.test.ts` dumps both bodies:
564
+ * openai-completions carries `prompt_cache_key: <sessionId>` (and
565
+ * `prompt_cache_retention: "24h"`) under PI_CACHE_RETENTION=long, so the body
566
+ * field alone would name the conversation; anthropic-messages carries no
567
+ * `prompt_cache_key` at all, and pi-ai hardcodes `x-session-affinity` there,
568
+ * which this relay does not read. Messages traffic therefore rests entirely on
569
+ * the relay's prefix fallback until a pi release honours
570
+ * `sessionAffinityFormat` on that dialect.
571
+ *
572
+ * The two dialects do not land at the same time. openai-completions honors
573
+ * `sessionAffinityFormat` in every released pi. anthropic-messages only reads it
574
+ * from the unreleased change on pi main (commit bbb61e34a), which is why the
575
+ * field is re-declared below rather than taken from `AnthropicMessagesCompat`:
576
+ * against pi-ai 0.85.1 the pin is inert, and the cost of an ignored header is
577
+ * zero. Pin now rather than later — the cost of forgetting is silently
578
+ * re-billed Claude prefixes.
579
+ */
580
+ function affinityPin() {
581
+ return {
582
+ sendSessionAffinityHeaders: true as const,
583
+ sessionAffinityFormat: "openrouter" as const,
482
584
  }
483
- }> {
585
+ }
586
+
587
+ /**
588
+ * Anthropic compat as pi-ai 0.85.1 types it, plus the affinity format a later
589
+ * pi reads. Declared here so the pin does not depend on the host's pi-ai
590
+ * version; the field is optional, so a host that predates it ignores the value.
591
+ */
592
+ type AnthropicCompat = NonNullable<Model<"anthropic-messages">["compat"]> & {
593
+ sessionAffinityFormat?: "openrouter"
594
+ }
595
+
596
+ /**
597
+ * Map the relay catalog to pi models, one base URL per dialect. Pure.
598
+ *
599
+ * The per-model `baseUrl` is the only place the dialect split can live: pi
600
+ * applies a base URL returned from `auth.resolve()` to every model at once
601
+ * (`models.js` `applyAuth`), so a provider-wide value would send Anthropic
602
+ * Messages traffic to `/v1` and Chat Completions traffic to the root.
603
+ */
604
+ export function toPengepulModels(
605
+ models: readonly PengepulModel[],
606
+ relayBase: string,
607
+ ): PengepulModelEntry[] {
484
608
  return models.map((model) => {
485
- const adaptive = model.dialect === "anthropic-messages" && model.reasoning
486
- // The 1h cache TTL is a Messages-dialect feature: `cache_control.ttl`
487
- // has nowhere to go on the Chat Completions wire. Reasoning is not part
488
- // of it — a non-reasoning Claude model caches the same way.
489
- const longCacheRetention = model.dialect === "anthropic-messages"
609
+ const shared = sharedModelFields(model, relayBase)
610
+ if (model.dialect === "anthropic-messages") {
611
+ const adaptive = model.reasoning
612
+ return {
613
+ ...shared,
614
+ api: "anthropic-messages" as const,
615
+ // Inherited level mapping (e.g. deepseek {high:"high"}) flows through;
616
+ // adaptive Claude models additionally mark "off" unsupported so the
617
+ // stream omits thinking:{type:"disabled"} (upstream rejects it).
618
+ ...(model.thinkingLevelMap || adaptive
619
+ ? {
620
+ thinkingLevelMap: {
621
+ ...(model.thinkingLevelMap ?? {}),
622
+ ...(adaptive ? { off: null } : {}),
623
+ },
624
+ }
625
+ : {}),
626
+ // Reasoning-capable Claude models run on the adaptive-thinking wire:
627
+ // pi's streamSimple always passes thinkingEnabled:false when no level is
628
+ // selected, and the stream would send thinking:{type:"disabled"}, which
629
+ // the upstream rejects (400: "thinking.type.disabled is not supported
630
+ // for this model"). thinkingLevelMap.off = null marks "off" as
631
+ // unsupported so pi omits the thinking param entirely (server default
632
+ // = adaptive), and forceAdaptiveThinking routes an explicit level to
633
+ // {type:"adaptive"} + effort instead of budget_tokens.
634
+ //
635
+ // The 1h cache TTL is a Messages-dialect feature too: `cache_control.ttl`
636
+ // has nowhere to go on the Chat Completions wire. Reasoning is not part
637
+ // of it — a non-reasoning Claude model caches the same way.
638
+ compat: {
639
+ ...affinityPin(),
640
+ ...(adaptive ? { forceAdaptiveThinking: true as const } : {}),
641
+ supportsLongCacheRetention: true as const,
642
+ } satisfies AnthropicCompat,
643
+ }
644
+ }
645
+
490
646
  return {
491
- id: model.id,
492
- name: model.name,
493
- api: model.dialect,
494
- baseUrl: baseUrlForDialect(relayBase, model.dialect),
495
- reasoning: model.reasoning,
496
- input: model.input,
497
- cost: model.cost,
498
- contextWindow: model.contextWindow,
499
- maxTokens: model.maxTokens,
500
- // Inherited level mapping (e.g. deepseek {high:"high"}) flows through;
501
- // adaptive Claude models additionally mark "off" unsupported so the
502
- // stream omits thinking:{type:"disabled"} (upstream rejects it).
503
- ...(model.thinkingLevelMap || adaptive
504
- ? { thinkingLevelMap: { ...(model.thinkingLevelMap ?? {}), ...(adaptive ? { off: null } : {}) } }
505
- : {}),
506
- // Reasoning-capable Claude models run on the adaptive-thinking wire:
507
- // pi's streamSimple always passes thinkingEnabled:false when no level is
508
- // selected, and the stream would send thinking:{type:"disabled"}, which
509
- // the upstream rejects (400: "thinking.type.disabled is not supported
510
- // for this model"). thinkingLevelMap.off = null marks "off" as
511
- // unsupported so pi omits the thinking param entirely (server default
512
- // = adaptive), and forceAdaptiveThinking routes an explicit level to
513
- // {type:"adaptive"} + effort instead of budget_tokens.
514
- // Both dialects pin both fields, and that is why `compat` is
515
- // unconditional. The relay's prompt-cache affinity key resolves in this
516
- // order: `x-claude-code-session-id`, `x-session-id`, the body's
517
- // `prompt_cache_key`, then a hash of the cacheable request prefix
518
- // (app.rs `conversation_key`). pi emits `x-session-id` only for the
519
- // `openrouter` affinity format, and only when the send flag is set;
520
- // the auto-detected defaults are wrong here in both cases
521
- // (openai-completions picks `openai`: session_id + x-client-request-id +
522
- // x-session-affinity; anthropic-messages picks nothing). The header is
523
- // the cheaper and more explicit of the two signals and it outranks the
524
- // body field, so pinning it keeps a session's account stable by the
525
- // relay's first rule rather than its third. Losing the pin costs a
526
- // session that migrates between pooled accounts its whole prefix: the
527
- // upstream cache is per account.
528
- //
529
- // Measured, not assumed — `test/affinity-wire.test.ts` dumps both bodies:
530
- // openai-completions carries `prompt_cache_key: <sessionId>` (and
531
- // `prompt_cache_retention: "24h"`) under PI_CACHE_RETENTION=long, so the
532
- // body field alone would name the conversation; anthropic-messages
533
- // carries no `prompt_cache_key` at all, and pi-ai hardcodes
534
- // `x-session-affinity` there, which this relay does not read. Messages
535
- // traffic therefore rests entirely on the relay's prefix fallback until
536
- // a pi release honours `sessionAffinityFormat` on that dialect.
537
- //
538
- // The two dialects do not land at the same time. openai-completions
539
- // honors `sessionAffinityFormat` in every released pi. anthropic-messages
540
- // only reads it from the unreleased change on pi main (commit
541
- // bbb61e34a); through pi-ai 0.85.1 that client hardcodes the header name
542
- // `x-session-affinity`, which this relay does not read, so the Claude
543
- // pin below is inert until pi ships it. Pin now rather than later: the
544
- // cost of an ignored header is zero, and the cost of forgetting is
545
- // silently re-billed Claude prefixes.
546
- compat: {
547
- ...(adaptive ? { forceAdaptiveThinking: true as const } : {}),
548
- ...(longCacheRetention ? { supportsLongCacheRetention: true as const } : {}),
549
- sendSessionAffinityHeaders: true as const,
550
- sessionAffinityFormat: "openrouter" as const,
551
- },
647
+ ...shared,
648
+ api: "openai-completions" as const,
649
+ ...(model.thinkingLevelMap ? { thinkingLevelMap: model.thinkingLevelMap } : {}),
650
+ compat: { ...affinityPin() },
552
651
  }
553
652
  })
554
653
  }
555
654
 
655
+ /** Whether a stored pi model is one this provider published. */
656
+ export function isPengepulModelEntry(model: Model<Api>): model is PengepulModelEntry {
657
+ return (
658
+ model.provider === PENGEPUL_PROVIDER_ID &&
659
+ (model.api === "anthropic-messages" || model.api === "openai-completions")
660
+ )
661
+ }
662
+
663
+ /**
664
+ * Re-derive every model's base URL from the base currently configured.
665
+ *
666
+ * pi's model store keeps whole models, baseUrl included, and replays them
667
+ * before the network phase. A relay that moved would otherwise be reached at
668
+ * its old address until a fetch succeeds — which never happens when the old
669
+ * address is gone.
670
+ */
671
+ export function restampRelayBase(
672
+ models: readonly PengepulModelEntry[],
673
+ relayBase: string,
674
+ ): PengepulModelEntry[] {
675
+ return models.map((model) => ({
676
+ ...model,
677
+ baseUrl: baseUrlForDialect(relayBase, model.api),
678
+ }))
679
+ }
680
+
556
681
  /** Picker label: the bare model part of a relay id, suffixed. `anthropic/claude-opus-5` -> `claude-opus-5 (pengepul)`. */
557
682
  function displayName(id: string): string {
558
683
  return `${bareId(id)} (pengepul)`
@@ -588,7 +713,7 @@ function configuredTimeoutMs(timeoutMs: number | undefined): number {
588
713
  }
589
714
 
590
715
  export function getModelsTimeoutMs(env: NodeJS.ProcessEnv = process.env): number {
591
- const raw = env["PENGEPUL_MODELS_TIMEOUT_MS"]
716
+ const raw = env[MODELS_TIMEOUT_MS_ENV]
592
717
  if (!raw) return DEFAULT_MODELS_TIMEOUT_MS
593
718
  const parsed = Number(raw)
594
719
  return configuredTimeoutMs(parsed)
@@ -674,7 +799,7 @@ export async function fetchPengepulModels(
674
799
  throw new Error(
675
800
  `pengepul rejected the API key (${
676
801
  response.status
677
- }). Set PENGEPUL_API_KEY or check ~/.pengepul/config.yaml.`,
802
+ }). Run /login pengepul, or set the key in ~/.pi/agent/auth.json or PENGEPUL_API_KEY.`,
678
803
  )
679
804
  }
680
805
  if (!response.ok) {
@@ -782,65 +907,4 @@ export async function loadCachedPengepulModels(
782
907
  }
783
908
  }
784
909
 
785
- async function writeCache(cachePath: string, models: readonly PengepulModel[]): Promise<void> {
786
- await mkdir(dirname(cachePath), { recursive: true })
787
- // Unique per write: the runtime can issue two overlapping writes in one
788
- // process (cache-first + background refresh), and a shared pid-keyed name
789
- // would let the first rename remove the second's source mid-flight.
790
- const temporaryPath = `${cachePath}.${process.pid}.${randomUUID()}.tmp`
791
-
792
- try {
793
- await writeFile(
794
- temporaryPath,
795
- `${JSON.stringify({ version: MODEL_CACHE_VERSION, models }, null, 2)}\n`,
796
- { encoding: "utf-8", mode: 0o600 },
797
- )
798
- await rename(temporaryPath, cachePath)
799
- } finally {
800
- try {
801
- await rm(temporaryPath, { force: true })
802
- } catch {
803
- // Best-effort cleanup must not hide the original cache write error.
804
- }
805
- }
806
- }
807
-
808
- export async function loadPengepulModels(
809
- options: LoadModelsOptions,
810
- ): Promise<PengepulModelSource> {
811
- const cachePath = options.cachePath
812
-
813
- try {
814
- const models = await fetchPengepulModels(options)
815
-
816
- try {
817
- await writeCache(cachePath, models)
818
- return { models, source: "live" }
819
- } catch (error) {
820
- return {
821
- models,
822
- source: "live",
823
- warning: `Loaded the live pengepul model catalog but could not update ${cachePath}: ${errorMessage(error)}`,
824
- }
825
- }
826
- } catch (liveError) {
827
- if (options.signal?.aborted) throw abortError(options.signal.reason ?? liveError)
828
-
829
- try {
830
- const models = await readCache(cachePath)
831
- return {
832
- models,
833
- source: "cache",
834
- warning: `Could not refresh the pengepul model catalog (${errorMessage(liveError)}). Using the cached catalog from ${cachePath}.`,
835
- }
836
- } catch (cacheError) {
837
- return {
838
- models: [],
839
- source: "empty",
840
- warning: `Could not refresh the pengepul model catalog (${errorMessage(liveError)}), and no valid cached catalog is available at ${cachePath} (${errorMessage(cacheError)}). pengepul models will remain unavailable until the next startup refresh succeeds.`,
841
- }
842
- }
843
- }
844
- }
845
-
846
910