@pwguler/pi-pengepul-provider 0.2.3 → 0.2.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -86,11 +86,33 @@ is set.
86
86
  ## Development
87
87
 
88
88
  ```sh
89
+ npm install
89
90
  bun test
90
91
  npx tsc --noEmit
91
92
  bun scripts/e2e-live.ts # live e2e against a running relay; sends one tiny completion
92
93
  ```
93
94
 
95
+ The `@earendil-works/*` copies `npm install` writes into `node_modules` are for
96
+ `tsc` and the tests only. At runtime pi serves the extension those modules from
97
+ its own bundle (`loader.js` maps `@earendil-works/pi-ai/providers/all` to a
98
+ virtual module), so the local copies exist to be *type-checked against*, not to
99
+ be the ones in play.
100
+
101
+ Nothing here can enforce that the two match: the peer ranges are `*` and the
102
+ host's version is not knowable at install time. The lockfile records whichever
103
+ version was current when it was last refreshed, so the comparison is a manual
104
+ one, worth making whenever a measurement has to be trusted:
105
+
106
+ ```sh
107
+ pi --version # host pi release
108
+ cat node_modules/@earendil-works/pi-ai/package.json # local copy
109
+ ```
110
+
111
+ When they drift, anything measured from this directory describes a different
112
+ model catalog than the running extension sees — the 0.84.4 copy carried 40
113
+ `:batch` entries where 0.85.1 carries 68 — and local verification quietly
114
+ disagrees with production.
115
+
94
116
  ## Releasing
95
117
 
96
118
  Releases publish from CI on a version tag. Bump, tag, push:
@@ -19,6 +19,7 @@ import { resolveApiKey } from "./api-key.ts"
19
19
  import { resolveSettings } from "./config.ts"
20
20
  import { modelsUrl } from "./dialect.ts"
21
21
  import {
22
+ catalogIdForms,
22
23
  loadCachedPengepulModels,
23
24
  loadPengepulModels,
24
25
  toProviderModelConfigs,
@@ -53,11 +54,10 @@ function metaFromModel(model: NonNullable<ReturnType<typeof getBuiltinModel>>) {
53
54
  }
54
55
 
55
56
  /**
56
- * Multi-catalog lookup over pi's builtin models. A commandcode id can live
57
- * in several catalogs: verbatim under an aggregator (`openrouter`, `baseten`,
58
- * `together`, ...), bare under a vendor catalog (`deepseek`, `google`,
59
- * `xai`, ...), or bare lowercased. Try those shapes in that order and take
60
- * the first hit; reasoning metadata and the thinkingLevelMap flow from it.
57
+ * Multi-catalog lookup over pi's builtin models. A relay id can match several
58
+ * catalogs, so each of the id's shapes is tried in turn - verbatim, bare under
59
+ * a vendor catalog, last segment, and with the routing namespace removed - and
60
+ * the first hit wins. Reasoning metadata and the thinkingLevelMap flow from it.
61
61
  */
62
62
  function createBuiltinLookup(): (id: string, dialect: string) => ReturnType<typeof metaFromModel> | undefined {
63
63
  type Entry = { provider: string; id: string };
@@ -77,27 +77,22 @@ function createBuiltinLookup(): (id: string, dialect: string) => ReturnType<type
77
77
  }
78
78
 
79
79
  return (id, dialect) => {
80
- const candidates: Array<Entry | undefined> = [
81
- exact.get(id),
82
- exact.get(bareOf(id)),
83
- segments.get(bareOf(id)),
84
- lower.get(id.toLowerCase()),
85
- lower.get(bareOf(id).toLowerCase()),
86
- ]
87
- for (const candidate of candidates) {
88
- if (candidate === undefined) continue
89
- const model = getBuiltinModel(candidate.provider as never, candidate.id as never)
90
- if (model) return metaFromModel(model)
80
+ for (const form of catalogIdForms(id)) {
81
+ const candidates: Array<Entry | undefined> = [
82
+ exact.get(form),
83
+ segments.get(form),
84
+ lower.get(form.toLowerCase()),
85
+ ]
86
+ for (const candidate of candidates) {
87
+ if (candidate === undefined) continue
88
+ const model = getBuiltinModel(candidate.provider as never, candidate.id as never)
89
+ if (model) return metaFromModel(model)
90
+ }
91
91
  }
92
92
  return undefined
93
93
  }
94
94
  }
95
95
 
96
- function bareOf(id: string): string {
97
- const slash = id.lastIndexOf("/")
98
- return slash === -1 ? id : id.slice(slash + 1)
99
- }
100
-
101
96
  function createProviderConfigFactory(relayBase: string, apiKey: string | undefined) {
102
97
  return (models: readonly PengepulModel[]): ProviderConfig => ({
103
98
  name: "Pengepul",
@@ -140,6 +140,26 @@ function heuristicMeta(id: string): BuiltinModelMeta {
140
140
  }
141
141
  }
142
142
 
143
+ /**
144
+ * The catalog-id shapes to try for a relay id, in the order they are tried.
145
+ *
146
+ * The order is load-bearing, and it is not "most specific first": the last
147
+ * segment is tried before the namespace-stripped id, which is what the
148
+ * catalogs were measured to give. For 141 of the relay's 527 ids the two
149
+ * shapes resolve to different entries and the last segment wins -
150
+ * `commandcode/deepseek/deepseek-v4-pro` takes the deepseek catalog's bare
151
+ * `deepseek-v4-pro` (input 0.435) rather than openrouter's
152
+ * `deepseek/deepseek-v4-pro` (input 0.890, and a different level map).
153
+ * Reordering these shapes repoints those ids, so the test pins the order.
154
+ *
155
+ * The namespace-stripped shape is the third fallback: the only one that
156
+ * reaches an id whose catalog entry both the full id and the last segment
157
+ * miss, which is three of the relay's 527 at the time of writing.
158
+ */
159
+ export function catalogIdForms(id: string): readonly string[] {
160
+ return [...new Set([id, modelName(id), bareId(id)])]
161
+ }
162
+
143
163
  /**
144
164
  * Metadata pengepul itself advertises for a model (pengepul >= 0.6.0).
145
165
  * Every field is optional: the rollout is partial and older relays send none.
@@ -209,6 +229,43 @@ function optionalRate(value: unknown): number | undefined {
209
229
  return typeof value === "number" && Number.isFinite(value) && value >= 0 ? value : undefined
210
230
  }
211
231
 
232
+ /**
233
+ * Extended effort levels for an id no pi catalog carries, on the one relay
234
+ * namespace where that vocabulary has been measured.
235
+ *
236
+ * pi hides `xhigh` and `max` unless a model's map names them, so an id no
237
+ * catalog answers for dropped to low/medium/high and could never send the top
238
+ * of the scale. Two of the 296 reasoning openai-completions ids the provider
239
+ * registers rest on this today; the lookup's namespace-stripped shapes answer
240
+ * for the rest.
241
+ *
242
+ * Only `max` is named. DeepSeek documents low/high/max for the
243
+ * OpenAI-compatible wire and folds `minimal` into low, `medium` and `xhigh`
244
+ * into high - which is what pi's own deepseek catalog encodes, hiding `xhigh`
245
+ * outright and nulling `medium` - but the relay validates one enum for every
246
+ * family it routes, so the namespace rule cannot tell a vendor alias from a
247
+ * level in its own right. Naming the top of the scale is the part that is
248
+ * true whatever the upstream does with it; the rest stays with pi's default.
249
+ *
250
+ * Measured against the running relay: on `commandcode/` ids every requested
251
+ * effort except `minimal` is accepted, and one error text — the relay's own
252
+ * enum — answers all nine families probed (deepseek, Qwen, MiniMax, google,
253
+ * moonshotai, xiaomi, stepfun, nvidia, inclusionai), nine of the eighteen the
254
+ * namespace carries, so the vocabulary belongs to the relay rather than to any
255
+ * one model. `minimal` 400s, and the overlay below nulls it.
256
+ *
257
+ * `openrouter/` ids are left alone, but not because that namespace is
258
+ * unmeasurable: it validates no effort enum at all, so a probe there answers
259
+ * 200 whether the level exists upstream or not. What it serves is also
260
+ * heterogeneous - image generators, R1-class models that take no effort
261
+ * parameter at all - and pi's openrouter entries spell DeepSeek's top level
262
+ * `xhigh` rather than `max`, so a namespace-wide rule there has nothing solid
263
+ * to stand on.
264
+ */
265
+ function fallbackLevelMap(id: string): Record<string, string | null> | undefined {
266
+ return id.toLowerCase().startsWith("commandcode/") ? { max: "max" } : undefined
267
+ }
268
+
212
269
  /**
213
270
  * Resolve a model's metadata, most trustworthy source first:
214
271
  * 1. what pengepul advertises (first-party for this relay),
@@ -223,9 +280,28 @@ function metaFor(
223
280
  dialect: PengepulDialect,
224
281
  lookup: BuiltinModelLookup | undefined,
225
282
  ): BuiltinModelMeta {
226
- const base = lookup?.(id, dialect) ?? heuristicMeta(id)
283
+ const known = lookup?.(id, dialect)
284
+ const base = known ?? heuristicMeta(id)
227
285
  const relay = metaFromRelayEntry(entry)
228
- return relay ? { ...base, ...relay } : base
286
+ const meta = relay ? { ...base, ...relay } : base
287
+
288
+ // A model the relay tags `anthropic` is Claude-family, hence reasoning-
289
+ // capable, even when pi's catalog does not know its exact id yet.
290
+ const reasoning = meta.reasoning || entry["owned_by"] === "anthropic"
291
+
292
+ // The lookup searches every provider's catalog, so a hit need not come from
293
+ // the id's own vendor: github-copilot answers for
294
+ // `commandcode/google/gemini-3.8-flash` and carries no map, which is silence
295
+ // about another provider, not an answer about this relay. The fallback still
296
+ // stops there, deliberately. Measuring this relay shows which efforts it
297
+ // accepts and never which ones the upstream honours, so an entry's silence
298
+ // is left as silence and those ids keep pi's default until per-vendor
299
+ // evidence exists - the way DeepSeek's documented scale settled the ids the
300
+ // fallback does serve.
301
+ const fallback =
302
+ known || !reasoning || dialect !== "openai-completions" ? undefined : fallbackLevelMap(id)
303
+
304
+ return { ...meta, reasoning, ...(fallback ? { thinkingLevelMap: fallback } : {}) }
229
305
  }
230
306
 
231
307
  function toPengepulModel(
@@ -235,25 +311,25 @@ function toPengepulModel(
235
311
  const id = stringField(entry, "id")
236
312
  const dialect = dialectForModelId(id)
237
313
  const meta = metaFor(entry, id, dialect, lookup)
238
- const ownedBy = entry["owned_by"]
239
-
240
- // A model the relay tags `anthropic` is Claude-family, hence reasoning-
241
- // capable, even when pi's catalog does not know its exact id yet.
242
- const reasoning = meta.reasoning || ownedBy === "anthropic"
243
-
244
- // The relay enforces reasoning_effort low|medium|high|xhigh|max at its
245
- // request layer, uniformly across families: `minimal` 400s and its thinking
246
- // toggle never actually disables thinking. That holds no matter where the
247
- // reasoning knowledge came from, so every openai-completions reasoning model
248
- // gets the relay's shape: inherited strings case-folded to the enum (pi's
249
- // catalogs spell Google efforts `HIGH` and Qwen's `default`), values that
250
- // match under no casing hidden, and off/minimal always null. Overlay, never
251
- // replace: inherited nulls keep their levels hidden. The Messages dialect
252
- // needs none of this: it folds `minimal` into `low` and already nulls `off`
253
- // via forceAdaptiveThinking.
314
+ const reasoning = meta.reasoning
315
+
316
+ // The relay's request layer validates reasoning_effort on `commandcode/`
317
+ // ids: `minimal` 400s there, and its thinking toggle never actually disables
318
+ // thinking. On `openrouter/` ids it validates nothing, so `minimal` is a real
319
+ // level there and is left in place. `off` is different and stays nulled
320
+ // everywhere: `none` is refused by some upstreams (gemini-3.8-flash 400s on
321
+ // it) and accepted by others (deepseek), which is no basis for sending it.
322
+ // Either way every openai-completions reasoning model gets the relay's shape:
323
+ // inherited strings case-folded to the enum (pi's catalogs spell Google
324
+ // efforts `HIGH` and Qwen's `default`), values that match under no casing
325
+ // hidden. Overlay, never replace: inherited nulls keep their levels hidden.
326
+ // The Messages dialect needs none of this: it folds `minimal` into `low` and
327
+ // already nulls `off` via forceAdaptiveThinking.
328
+ const enforceRelayEnum = reasoning && dialect === "openai-completions"
254
329
  const thinkingLevelMap = relaySafeLevelMap(
255
330
  meta.thinkingLevelMap,
256
- reasoning && dialect === "openai-completions",
331
+ enforceRelayEnum,
332
+ relayAcceptsMinimal(id),
257
333
  )
258
334
 
259
335
  return {
@@ -272,32 +348,51 @@ function toPengepulModel(
272
348
  /** The efforts the relay accepts; anything else is rejected at its request layer. */
273
349
  const RELAY_EFFORTS = new Set(["low", "medium", "high", "xhigh", "max"])
274
350
 
351
+ /**
352
+ * Whether the relay's request layer takes `minimal` for this id.
353
+ *
354
+ * Measured: `commandcode/` validates its enum and answers 400 Invalid option:
355
+ * expected one of "low"|"medium"|"high"|"xhigh"|"max", while `openrouter/`
356
+ * validates nothing and answers 200 - so hiding `minimal` there costs 166 of
357
+ * the 227 openrouter models that reason, a level the relay accepts.
358
+ *
359
+ * `none` is not the same story: gemini-3.8-flash answers 400 on it while
360
+ * deepseek accepts it, so `off` stays hidden in every namespace. An unmeasured
361
+ * namespace keeps the conservative default too.
362
+ */
363
+ function relayAcceptsMinimal(id: string): boolean {
364
+ return id.toLowerCase().startsWith("openrouter/")
365
+ }
366
+
275
367
  /**
276
368
  * Shape an inherited level map for the relay's wire. With `enforce` (an
277
369
  * openai-completions reasoning model) the relay's enum is the only vocabulary
278
370
  * that reaches it: inherited strings are case-folded to the enum or hidden,
279
- * and off/minimal are always hidden. Without it the map passes through.
371
+ * and `off` is always hidden. `minimal` is hidden unless the namespace is
372
+ * known to take it. Without `enforce` the map passes through.
280
373
  */
281
374
  function relaySafeLevelMap(
282
375
  inherited: Record<string, string | null> | undefined,
283
376
  enforce: boolean,
377
+ acceptsMinimal = false,
284
378
  ): Record<string, string | null> | undefined {
285
379
  if (!enforce) return inherited ? { ...inherited } : undefined
286
380
 
287
381
  const map: Record<string, string | null> = {}
288
382
  for (const [level, mapped] of Object.entries(inherited ?? {})) {
289
- map[level] = typeof mapped === "string" ? relayEffort(mapped) : mapped
383
+ map[level] = typeof mapped === "string" ? relayEffort(mapped, acceptsMinimal) : mapped
290
384
  }
291
385
  map["off"] = null
292
- map["minimal"] = null
386
+ if (!acceptsMinimal) map["minimal"] = null
293
387
  return map
294
388
  }
295
389
 
296
390
  /** The relay's effort enum, case-folded; null keeps the level out of the picker. */
297
- function relayEffort(value: string): string | null {
391
+ function relayEffort(value: string, acceptsMinimal: boolean): string | null {
298
392
  if (RELAY_EFFORTS.has(value)) return value
299
393
  const lower = value.toLowerCase()
300
- return RELAY_EFFORTS.has(lower) ? lower : null
394
+ if (RELAY_EFFORTS.has(lower)) return lower
395
+ return acceptsMinimal && lower === "minimal" ? lower : null
301
396
  }
302
397
 
303
398
  function isRecord(value: unknown): value is Record<string, unknown> {
@@ -312,6 +407,31 @@ function stringField(record: Record<string, unknown>, key: string): string {
312
407
  return value
313
408
  }
314
409
 
410
+ /**
411
+ * The relay lists OpenRouter's batch routes as models, and OpenRouter refuses
412
+ * them on the chat wire with 404 "This model is only available through the
413
+ * Batch API". Confirmed on 10 of the relay's 77 batch ids across anthropic,
414
+ * openai, qwen, deepseek and z-ai; the rest are inferred from the same route
415
+ * rule, because probing them in bulk does not work - the refusals put the
416
+ * relay's pooled openrouter account on cooldown, which turns every following
417
+ * probe into 503 "no available openrouter account" and measures the cooldown
418
+ * rather than the model. Nothing pi sends can reach them, so they are left
419
+ * out of the catalog rather than offered as an entry that always fails.
420
+ */
421
+ function isBatchRoute(id: string): boolean {
422
+ return id.endsWith(":batch")
423
+ }
424
+
425
+ /**
426
+ * Drop the ids this relay cannot serve on either wire. Applied on the way in
427
+ * from the relay and on the way in from the cache: the cache is what covers a
428
+ * briefly absent relay, so it must not be the path that resurrects a route
429
+ * the live catalog would have dropped.
430
+ */
431
+ function servableModels(models: readonly PengepulModel[]): PengepulModel[] {
432
+ return models.filter((model) => !isBatchRoute(model.id))
433
+ }
434
+
315
435
  /** Parse the raw `/v1/models` body into models. Throws on a malformed body. */
316
436
  export function modelsFromApiResponse(
317
437
  value: unknown,
@@ -322,12 +442,16 @@ export function modelsFromApiResponse(
322
442
 
323
443
  const data = value["data"]
324
444
  if (!Array.isArray(data)) throw new Error("Expected models response data to be an array")
325
- if (data.length === 0) throw new Error("pengepul returned an empty model catalog")
326
445
 
327
- return data.map((entry) => {
328
- if (!isRecord(entry)) throw new Error("Expected model entry to be an object")
329
- return toPengepulModel(entry, lookupBuiltin)
330
- })
446
+ const models = servableModels(
447
+ data.map((entry) => {
448
+ if (!isRecord(entry)) throw new Error("Expected model entry to be an object")
449
+ return toPengepulModel(entry, lookupBuiltin)
450
+ }),
451
+ )
452
+
453
+ if (models.length === 0) throw new Error("pengepul returned an empty model catalog")
454
+ return models
331
455
  }
332
456
 
333
457
  /** Map models to pi `ProviderModelConfig` entries. Pure. */
@@ -431,9 +555,7 @@ export function toProviderModelConfigs(
431
555
 
432
556
  /** Picker label: the bare model part of a relay id, suffixed. `anthropic/claude-opus-5` -> `claude-opus-5 (pengepul)`. */
433
557
  function displayName(id: string): string {
434
- const slash = id.indexOf("/")
435
- const bare = slash === -1 ? id : id.slice(slash + 1)
436
- return `${bare} (pengepul)`
558
+ return `${bareId(id)} (pengepul)`
437
559
  }
438
560
 
439
561
  interface FetchModelsOptions {
@@ -639,8 +761,9 @@ export function modelsFromCache(value: unknown): readonly PengepulModel[] {
639
761
  : {}),
640
762
  }
641
763
  })
642
- if (parsed.length === 0) throw new Error("pengepul cache holds no valid models")
643
- return parsed
764
+ const servable = servableModels(parsed)
765
+ if (servable.length === 0) throw new Error("pengepul cache holds no valid models")
766
+ return servable
644
767
  }
645
768
 
646
769
  async function readCache(cachePath: string): Promise<readonly PengepulModel[]> {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@pwguler/pi-pengepul-provider",
3
- "version": "0.2.3",
3
+ "version": "0.2.4",
4
4
  "type": "module",
5
5
  "description": "pi custom provider for pengepul, a local relay that pools your Claude/Codex subscriptions. Connects pi to http://127.0.0.1:8317 over the native Anthropic Messages and OpenAI Chat Completions wires.",
6
6
  "license": "MIT",