@pwguler/pi-pengepul-provider 0.2.3 → 0.2.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +22 -0
- package/extensions/index.ts +16 -21
- package/extensions/models.ts +157 -34
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -86,11 +86,33 @@ is set.
|
|
|
86
86
|
## Development
|
|
87
87
|
|
|
88
88
|
```sh
|
|
89
|
+
npm install
|
|
89
90
|
bun test
|
|
90
91
|
npx tsc --noEmit
|
|
91
92
|
bun scripts/e2e-live.ts # live e2e against a running relay; sends one tiny completion
|
|
92
93
|
```
|
|
93
94
|
|
|
95
|
+
The `@earendil-works/*` copies `npm install` writes into `node_modules` are for
|
|
96
|
+
`tsc` and the tests only. At runtime pi serves the extension those modules from
|
|
97
|
+
its own bundle (`loader.js` maps `@earendil-works/pi-ai/providers/all` to a
|
|
98
|
+
virtual module), so the local copies exist to be *type-checked against*, not to
|
|
99
|
+
be the ones in play.
|
|
100
|
+
|
|
101
|
+
Nothing here can enforce that the two match: the peer ranges are `*` and the
|
|
102
|
+
host's version is not knowable at install time. The lockfile records whichever
|
|
103
|
+
version was current when it was last refreshed, so the comparison is a manual
|
|
104
|
+
one, worth making whenever a measurement has to be trusted:
|
|
105
|
+
|
|
106
|
+
```sh
|
|
107
|
+
pi --version # host pi release
|
|
108
|
+
cat node_modules/@earendil-works/pi-ai/package.json # local copy
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
When they drift, anything measured from this directory describes a different
|
|
112
|
+
model catalog than the running extension sees — the 0.84.4 copy carried 40
|
|
113
|
+
`:batch` entries where 0.85.1 carries 68 — and local verification quietly
|
|
114
|
+
disagrees with production.
|
|
115
|
+
|
|
94
116
|
## Releasing
|
|
95
117
|
|
|
96
118
|
Releases publish from CI on a version tag. Bump, tag, push:
|
package/extensions/index.ts
CHANGED
|
@@ -19,6 +19,7 @@ import { resolveApiKey } from "./api-key.ts"
|
|
|
19
19
|
import { resolveSettings } from "./config.ts"
|
|
20
20
|
import { modelsUrl } from "./dialect.ts"
|
|
21
21
|
import {
|
|
22
|
+
catalogIdForms,
|
|
22
23
|
loadCachedPengepulModels,
|
|
23
24
|
loadPengepulModels,
|
|
24
25
|
toProviderModelConfigs,
|
|
@@ -53,11 +54,10 @@ function metaFromModel(model: NonNullable<ReturnType<typeof getBuiltinModel>>) {
|
|
|
53
54
|
}
|
|
54
55
|
|
|
55
56
|
/**
|
|
56
|
-
* Multi-catalog lookup over pi's builtin models. A
|
|
57
|
-
*
|
|
58
|
-
*
|
|
59
|
-
*
|
|
60
|
-
* the first hit; reasoning metadata and the thinkingLevelMap flow from it.
|
|
57
|
+
* Multi-catalog lookup over pi's builtin models. A relay id can match several
|
|
58
|
+
* catalogs, so each of the id's shapes is tried in turn - verbatim, bare under
|
|
59
|
+
* a vendor catalog, last segment, and with the routing namespace removed - and
|
|
60
|
+
* the first hit wins. Reasoning metadata and the thinkingLevelMap flow from it.
|
|
61
61
|
*/
|
|
62
62
|
function createBuiltinLookup(): (id: string, dialect: string) => ReturnType<typeof metaFromModel> | undefined {
|
|
63
63
|
type Entry = { provider: string; id: string };
|
|
@@ -77,27 +77,22 @@ function createBuiltinLookup(): (id: string, dialect: string) => ReturnType<type
|
|
|
77
77
|
}
|
|
78
78
|
|
|
79
79
|
return (id, dialect) => {
|
|
80
|
-
const
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
80
|
+
for (const form of catalogIdForms(id)) {
|
|
81
|
+
const candidates: Array<Entry | undefined> = [
|
|
82
|
+
exact.get(form),
|
|
83
|
+
segments.get(form),
|
|
84
|
+
lower.get(form.toLowerCase()),
|
|
85
|
+
]
|
|
86
|
+
for (const candidate of candidates) {
|
|
87
|
+
if (candidate === undefined) continue
|
|
88
|
+
const model = getBuiltinModel(candidate.provider as never, candidate.id as never)
|
|
89
|
+
if (model) return metaFromModel(model)
|
|
90
|
+
}
|
|
91
91
|
}
|
|
92
92
|
return undefined
|
|
93
93
|
}
|
|
94
94
|
}
|
|
95
95
|
|
|
96
|
-
function bareOf(id: string): string {
|
|
97
|
-
const slash = id.lastIndexOf("/")
|
|
98
|
-
return slash === -1 ? id : id.slice(slash + 1)
|
|
99
|
-
}
|
|
100
|
-
|
|
101
96
|
function createProviderConfigFactory(relayBase: string, apiKey: string | undefined) {
|
|
102
97
|
return (models: readonly PengepulModel[]): ProviderConfig => ({
|
|
103
98
|
name: "Pengepul",
|
package/extensions/models.ts
CHANGED
|
@@ -140,6 +140,26 @@ function heuristicMeta(id: string): BuiltinModelMeta {
|
|
|
140
140
|
}
|
|
141
141
|
}
|
|
142
142
|
|
|
143
|
+
/**
|
|
144
|
+
* The catalog-id shapes to try for a relay id, in the order they are tried.
|
|
145
|
+
*
|
|
146
|
+
* The order is load-bearing, and it is not "most specific first": the last
|
|
147
|
+
* segment is tried before the namespace-stripped id, which is what the
|
|
148
|
+
* catalogs were measured to give. For 141 of the relay's 527 ids the two
|
|
149
|
+
* shapes resolve to different entries and the last segment wins -
|
|
150
|
+
* `commandcode/deepseek/deepseek-v4-pro` takes the deepseek catalog's bare
|
|
151
|
+
* `deepseek-v4-pro` (input 0.435) rather than openrouter's
|
|
152
|
+
* `deepseek/deepseek-v4-pro` (input 0.890, and a different level map).
|
|
153
|
+
* Reordering these shapes repoints those ids, so the test pins the order.
|
|
154
|
+
*
|
|
155
|
+
* The namespace-stripped shape is the third fallback: the only one that
|
|
156
|
+
* reaches an id whose catalog entry both the full id and the last segment
|
|
157
|
+
* miss, which is three of the relay's 527 at the time of writing.
|
|
158
|
+
*/
|
|
159
|
+
export function catalogIdForms(id: string): readonly string[] {
|
|
160
|
+
return [...new Set([id, modelName(id), bareId(id)])]
|
|
161
|
+
}
|
|
162
|
+
|
|
143
163
|
/**
|
|
144
164
|
* Metadata pengepul itself advertises for a model (pengepul >= 0.6.0).
|
|
145
165
|
* Every field is optional: the rollout is partial and older relays send none.
|
|
@@ -209,6 +229,43 @@ function optionalRate(value: unknown): number | undefined {
|
|
|
209
229
|
return typeof value === "number" && Number.isFinite(value) && value >= 0 ? value : undefined
|
|
210
230
|
}
|
|
211
231
|
|
|
232
|
+
/**
|
|
233
|
+
* Extended effort levels for an id no pi catalog carries, on the one relay
|
|
234
|
+
* namespace where that vocabulary has been measured.
|
|
235
|
+
*
|
|
236
|
+
* pi hides `xhigh` and `max` unless a model's map names them, so an id no
|
|
237
|
+
* catalog answers for dropped to low/medium/high and could never send the top
|
|
238
|
+
* of the scale. Two of the 296 reasoning openai-completions ids the provider
|
|
239
|
+
* registers rest on this today; the lookup's namespace-stripped shapes answer
|
|
240
|
+
* for the rest.
|
|
241
|
+
*
|
|
242
|
+
* Only `max` is named. DeepSeek documents low/high/max for the
|
|
243
|
+
* OpenAI-compatible wire and folds `minimal` into low, `medium` and `xhigh`
|
|
244
|
+
* into high - which is what pi's own deepseek catalog encodes, hiding `xhigh`
|
|
245
|
+
* outright and nulling `medium` - but the relay validates one enum for every
|
|
246
|
+
* family it routes, so the namespace rule cannot tell a vendor alias from a
|
|
247
|
+
* level in its own right. Naming the top of the scale is the part that is
|
|
248
|
+
* true whatever the upstream does with it; the rest stays with pi's default.
|
|
249
|
+
*
|
|
250
|
+
* Measured against the running relay: on `commandcode/` ids every requested
|
|
251
|
+
* effort except `minimal` is accepted, and one error text — the relay's own
|
|
252
|
+
* enum — answers all nine families probed (deepseek, Qwen, MiniMax, google,
|
|
253
|
+
* moonshotai, xiaomi, stepfun, nvidia, inclusionai), nine of the eighteen the
|
|
254
|
+
* namespace carries, so the vocabulary belongs to the relay rather than to any
|
|
255
|
+
* one model. `minimal` 400s, and the overlay below nulls it.
|
|
256
|
+
*
|
|
257
|
+
* `openrouter/` ids are left alone, but not because that namespace is
|
|
258
|
+
* unmeasurable: it validates no effort enum at all, so a probe there answers
|
|
259
|
+
* 200 whether the level exists upstream or not. What it serves is also
|
|
260
|
+
* heterogeneous - image generators, R1-class models that take no effort
|
|
261
|
+
* parameter at all - and pi's openrouter entries spell DeepSeek's top level
|
|
262
|
+
* `xhigh` rather than `max`, so a namespace-wide rule there has nothing solid
|
|
263
|
+
* to stand on.
|
|
264
|
+
*/
|
|
265
|
+
function fallbackLevelMap(id: string): Record<string, string | null> | undefined {
|
|
266
|
+
return id.toLowerCase().startsWith("commandcode/") ? { max: "max" } : undefined
|
|
267
|
+
}
|
|
268
|
+
|
|
212
269
|
/**
|
|
213
270
|
* Resolve a model's metadata, most trustworthy source first:
|
|
214
271
|
* 1. what pengepul advertises (first-party for this relay),
|
|
@@ -223,9 +280,28 @@ function metaFor(
|
|
|
223
280
|
dialect: PengepulDialect,
|
|
224
281
|
lookup: BuiltinModelLookup | undefined,
|
|
225
282
|
): BuiltinModelMeta {
|
|
226
|
-
const
|
|
283
|
+
const known = lookup?.(id, dialect)
|
|
284
|
+
const base = known ?? heuristicMeta(id)
|
|
227
285
|
const relay = metaFromRelayEntry(entry)
|
|
228
|
-
|
|
286
|
+
const meta = relay ? { ...base, ...relay } : base
|
|
287
|
+
|
|
288
|
+
// A model the relay tags `anthropic` is Claude-family, hence reasoning-
|
|
289
|
+
// capable, even when pi's catalog does not know its exact id yet.
|
|
290
|
+
const reasoning = meta.reasoning || entry["owned_by"] === "anthropic"
|
|
291
|
+
|
|
292
|
+
// The lookup searches every provider's catalog, so a hit need not come from
|
|
293
|
+
// the id's own vendor: github-copilot answers for
|
|
294
|
+
// `commandcode/google/gemini-3.8-flash` and carries no map, which is silence
|
|
295
|
+
// about another provider, not an answer about this relay. The fallback still
|
|
296
|
+
// stops there, deliberately. Measuring this relay shows which efforts it
|
|
297
|
+
// accepts and never which ones the upstream honours, so an entry's silence
|
|
298
|
+
// is left as silence and those ids keep pi's default until per-vendor
|
|
299
|
+
// evidence exists - the way DeepSeek's documented scale settled the ids the
|
|
300
|
+
// fallback does serve.
|
|
301
|
+
const fallback =
|
|
302
|
+
known || !reasoning || dialect !== "openai-completions" ? undefined : fallbackLevelMap(id)
|
|
303
|
+
|
|
304
|
+
return { ...meta, reasoning, ...(fallback ? { thinkingLevelMap: fallback } : {}) }
|
|
229
305
|
}
|
|
230
306
|
|
|
231
307
|
function toPengepulModel(
|
|
@@ -235,25 +311,25 @@ function toPengepulModel(
|
|
|
235
311
|
const id = stringField(entry, "id")
|
|
236
312
|
const dialect = dialectForModelId(id)
|
|
237
313
|
const meta = metaFor(entry, id, dialect, lookup)
|
|
238
|
-
const
|
|
239
|
-
|
|
240
|
-
//
|
|
241
|
-
//
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
//
|
|
245
|
-
//
|
|
246
|
-
//
|
|
247
|
-
//
|
|
248
|
-
//
|
|
249
|
-
//
|
|
250
|
-
//
|
|
251
|
-
//
|
|
252
|
-
|
|
253
|
-
// via forceAdaptiveThinking.
|
|
314
|
+
const reasoning = meta.reasoning
|
|
315
|
+
|
|
316
|
+
// The relay's request layer validates reasoning_effort on `commandcode/`
|
|
317
|
+
// ids: `minimal` 400s there, and its thinking toggle never actually disables
|
|
318
|
+
// thinking. On `openrouter/` ids it validates nothing, so `minimal` is a real
|
|
319
|
+
// level there and is left in place. `off` is different and stays nulled
|
|
320
|
+
// everywhere: `none` is refused by some upstreams (gemini-3.8-flash 400s on
|
|
321
|
+
// it) and accepted by others (deepseek), which is no basis for sending it.
|
|
322
|
+
// Either way every openai-completions reasoning model gets the relay's shape:
|
|
323
|
+
// inherited strings case-folded to the enum (pi's catalogs spell Google
|
|
324
|
+
// efforts `HIGH` and Qwen's `default`), values that match under no casing
|
|
325
|
+
// hidden. Overlay, never replace: inherited nulls keep their levels hidden.
|
|
326
|
+
// The Messages dialect needs none of this: it folds `minimal` into `low` and
|
|
327
|
+
// already nulls `off` via forceAdaptiveThinking.
|
|
328
|
+
const enforceRelayEnum = reasoning && dialect === "openai-completions"
|
|
254
329
|
const thinkingLevelMap = relaySafeLevelMap(
|
|
255
330
|
meta.thinkingLevelMap,
|
|
256
|
-
|
|
331
|
+
enforceRelayEnum,
|
|
332
|
+
relayAcceptsMinimal(id),
|
|
257
333
|
)
|
|
258
334
|
|
|
259
335
|
return {
|
|
@@ -272,32 +348,51 @@ function toPengepulModel(
|
|
|
272
348
|
/** The efforts the relay accepts; anything else is rejected at its request layer. */
|
|
273
349
|
const RELAY_EFFORTS = new Set(["low", "medium", "high", "xhigh", "max"])
|
|
274
350
|
|
|
351
|
+
/**
|
|
352
|
+
* Whether the relay's request layer takes `minimal` for this id.
|
|
353
|
+
*
|
|
354
|
+
* Measured: `commandcode/` validates its enum and answers 400 Invalid option:
|
|
355
|
+
* expected one of "low"|"medium"|"high"|"xhigh"|"max", while `openrouter/`
|
|
356
|
+
* validates nothing and answers 200 - so hiding `minimal` there costs 166 of
|
|
357
|
+
* the 227 openrouter models that reason, a level the relay accepts.
|
|
358
|
+
*
|
|
359
|
+
* `none` is not the same story: gemini-3.8-flash answers 400 on it while
|
|
360
|
+
* deepseek accepts it, so `off` stays hidden in every namespace. An unmeasured
|
|
361
|
+
* namespace keeps the conservative default too.
|
|
362
|
+
*/
|
|
363
|
+
function relayAcceptsMinimal(id: string): boolean {
|
|
364
|
+
return id.toLowerCase().startsWith("openrouter/")
|
|
365
|
+
}
|
|
366
|
+
|
|
275
367
|
/**
|
|
276
368
|
* Shape an inherited level map for the relay's wire. With `enforce` (an
|
|
277
369
|
* openai-completions reasoning model) the relay's enum is the only vocabulary
|
|
278
370
|
* that reaches it: inherited strings are case-folded to the enum or hidden,
|
|
279
|
-
* and off
|
|
371
|
+
* and `off` is always hidden. `minimal` is hidden unless the namespace is
|
|
372
|
+
* known to take it. Without `enforce` the map passes through.
|
|
280
373
|
*/
|
|
281
374
|
function relaySafeLevelMap(
|
|
282
375
|
inherited: Record<string, string | null> | undefined,
|
|
283
376
|
enforce: boolean,
|
|
377
|
+
acceptsMinimal = false,
|
|
284
378
|
): Record<string, string | null> | undefined {
|
|
285
379
|
if (!enforce) return inherited ? { ...inherited } : undefined
|
|
286
380
|
|
|
287
381
|
const map: Record<string, string | null> = {}
|
|
288
382
|
for (const [level, mapped] of Object.entries(inherited ?? {})) {
|
|
289
|
-
map[level] = typeof mapped === "string" ? relayEffort(mapped) : mapped
|
|
383
|
+
map[level] = typeof mapped === "string" ? relayEffort(mapped, acceptsMinimal) : mapped
|
|
290
384
|
}
|
|
291
385
|
map["off"] = null
|
|
292
|
-
map["minimal"] = null
|
|
386
|
+
if (!acceptsMinimal) map["minimal"] = null
|
|
293
387
|
return map
|
|
294
388
|
}
|
|
295
389
|
|
|
296
390
|
/** The relay's effort enum, case-folded; null keeps the level out of the picker. */
|
|
297
|
-
function relayEffort(value: string): string | null {
|
|
391
|
+
function relayEffort(value: string, acceptsMinimal: boolean): string | null {
|
|
298
392
|
if (RELAY_EFFORTS.has(value)) return value
|
|
299
393
|
const lower = value.toLowerCase()
|
|
300
|
-
|
|
394
|
+
if (RELAY_EFFORTS.has(lower)) return lower
|
|
395
|
+
return acceptsMinimal && lower === "minimal" ? lower : null
|
|
301
396
|
}
|
|
302
397
|
|
|
303
398
|
function isRecord(value: unknown): value is Record<string, unknown> {
|
|
@@ -312,6 +407,31 @@ function stringField(record: Record<string, unknown>, key: string): string {
|
|
|
312
407
|
return value
|
|
313
408
|
}
|
|
314
409
|
|
|
410
|
+
/**
|
|
411
|
+
* The relay lists OpenRouter's batch routes as models, and OpenRouter refuses
|
|
412
|
+
* them on the chat wire with 404 "This model is only available through the
|
|
413
|
+
* Batch API". Confirmed on 10 of the relay's 77 batch ids across anthropic,
|
|
414
|
+
* openai, qwen, deepseek and z-ai; the rest are inferred from the same route
|
|
415
|
+
* rule, because probing them in bulk does not work - the refusals put the
|
|
416
|
+
* relay's pooled openrouter account on cooldown, which turns every following
|
|
417
|
+
* probe into 503 "no available openrouter account" and measures the cooldown
|
|
418
|
+
* rather than the model. Nothing pi sends can reach them, so they are left
|
|
419
|
+
* out of the catalog rather than offered as an entry that always fails.
|
|
420
|
+
*/
|
|
421
|
+
function isBatchRoute(id: string): boolean {
|
|
422
|
+
return id.endsWith(":batch")
|
|
423
|
+
}
|
|
424
|
+
|
|
425
|
+
/**
|
|
426
|
+
* Drop the ids this relay cannot serve on either wire. Applied on the way in
|
|
427
|
+
* from the relay and on the way in from the cache: the cache is what covers a
|
|
428
|
+
* briefly absent relay, so it must not be the path that resurrects a route
|
|
429
|
+
* the live catalog would have dropped.
|
|
430
|
+
*/
|
|
431
|
+
function servableModels(models: readonly PengepulModel[]): PengepulModel[] {
|
|
432
|
+
return models.filter((model) => !isBatchRoute(model.id))
|
|
433
|
+
}
|
|
434
|
+
|
|
315
435
|
/** Parse the raw `/v1/models` body into models. Throws on a malformed body. */
|
|
316
436
|
export function modelsFromApiResponse(
|
|
317
437
|
value: unknown,
|
|
@@ -322,12 +442,16 @@ export function modelsFromApiResponse(
|
|
|
322
442
|
|
|
323
443
|
const data = value["data"]
|
|
324
444
|
if (!Array.isArray(data)) throw new Error("Expected models response data to be an array")
|
|
325
|
-
if (data.length === 0) throw new Error("pengepul returned an empty model catalog")
|
|
326
445
|
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
446
|
+
const models = servableModels(
|
|
447
|
+
data.map((entry) => {
|
|
448
|
+
if (!isRecord(entry)) throw new Error("Expected model entry to be an object")
|
|
449
|
+
return toPengepulModel(entry, lookupBuiltin)
|
|
450
|
+
}),
|
|
451
|
+
)
|
|
452
|
+
|
|
453
|
+
if (models.length === 0) throw new Error("pengepul returned an empty model catalog")
|
|
454
|
+
return models
|
|
331
455
|
}
|
|
332
456
|
|
|
333
457
|
/** Map models to pi `ProviderModelConfig` entries. Pure. */
|
|
@@ -431,9 +555,7 @@ export function toProviderModelConfigs(
|
|
|
431
555
|
|
|
432
556
|
/** Picker label: the bare model part of a relay id, suffixed. `anthropic/claude-opus-5` -> `claude-opus-5 (pengepul)`. */
|
|
433
557
|
function displayName(id: string): string {
|
|
434
|
-
|
|
435
|
-
const bare = slash === -1 ? id : id.slice(slash + 1)
|
|
436
|
-
return `${bare} (pengepul)`
|
|
558
|
+
return `${bareId(id)} (pengepul)`
|
|
437
559
|
}
|
|
438
560
|
|
|
439
561
|
interface FetchModelsOptions {
|
|
@@ -639,8 +761,9 @@ export function modelsFromCache(value: unknown): readonly PengepulModel[] {
|
|
|
639
761
|
: {}),
|
|
640
762
|
}
|
|
641
763
|
})
|
|
642
|
-
|
|
643
|
-
|
|
764
|
+
const servable = servableModels(parsed)
|
|
765
|
+
if (servable.length === 0) throw new Error("pengepul cache holds no valid models")
|
|
766
|
+
return servable
|
|
644
767
|
}
|
|
645
768
|
|
|
646
769
|
async function readCache(cachePath: string): Promise<readonly PengepulModel[]> {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@pwguler/pi-pengepul-provider",
|
|
3
|
-
"version": "0.2.
|
|
3
|
+
"version": "0.2.4",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "pi custom provider for pengepul, a local relay that pools your Claude/Codex subscriptions. Connects pi to http://127.0.0.1:8317 over the native Anthropic Messages and OpenAI Chat Completions wires.",
|
|
6
6
|
"license": "MIT",
|