@pwguler/pi-pengepul-provider 0.2.4 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -12
- package/extensions/api-key.ts +6 -37
- package/extensions/config.ts +15 -21
- package/extensions/credential.ts +113 -0
- package/extensions/index.ts +22 -54
- package/extensions/models.ts +229 -165
- package/extensions/provider.ts +269 -0
- package/package.json +2 -2
- package/extensions/runtime.ts +0 -211
package/extensions/models.ts
CHANGED
|
@@ -20,10 +20,9 @@
|
|
|
20
20
|
* The network/cache are injected so the catalog logic stays testable.
|
|
21
21
|
*/
|
|
22
22
|
|
|
23
|
-
import {
|
|
24
|
-
import { randomUUID } from "node:crypto"
|
|
25
|
-
import { dirname } from "node:path"
|
|
23
|
+
import { readFile } from "node:fs/promises"
|
|
26
24
|
|
|
25
|
+
import { MODELS_TIMEOUT_MS_ENV } from "./config.ts"
|
|
27
26
|
import { baseUrlForDialect, dialectForModelId } from "./dialect.ts"
|
|
28
27
|
import type { PengepulDialect } from "./dialect.ts"
|
|
29
28
|
|
|
@@ -65,7 +64,7 @@ export interface BuiltinModelMeta {
|
|
|
65
64
|
*/
|
|
66
65
|
export type BuiltinModelLookup = (id: string, dialect: PengepulDialect) => BuiltinModelMeta | undefined
|
|
67
66
|
|
|
68
|
-
/** A pengepul model ready to become a pi
|
|
67
|
+
/** A pengepul model ready to become a pi model entry. */
|
|
69
68
|
export interface PengepulModel {
|
|
70
69
|
id: string
|
|
71
70
|
name: string
|
|
@@ -79,12 +78,6 @@ export interface PengepulModel {
|
|
|
79
78
|
thinkingLevelMap?: Record<string, string | null>
|
|
80
79
|
}
|
|
81
80
|
|
|
82
|
-
export interface PengepulModelSource {
|
|
83
|
-
models: readonly PengepulModel[]
|
|
84
|
-
/** "live" = fetched from the relay; "cache" = read from disk; "empty" = none. */
|
|
85
|
-
source: "live" | "cache" | "empty"
|
|
86
|
-
warning?: string
|
|
87
|
-
}
|
|
88
81
|
|
|
89
82
|
/** `anthropic/claude-opus-5` -> `claude-opus-5` (the id upstream actually serves). */
|
|
90
83
|
export function bareId(id: string): string {
|
|
@@ -266,6 +259,63 @@ function fallbackLevelMap(id: string): Record<string, string | null> | undefined
|
|
|
266
259
|
return id.toLowerCase().startsWith("commandcode/") ? { max: "max" } : undefined
|
|
267
260
|
}
|
|
268
261
|
|
|
262
|
+
/** The levels pi hides unless a map names them. */
|
|
263
|
+
const INHERITED_LEVELS = ["xhigh", "max"] as const
|
|
264
|
+
|
|
265
|
+
/**
|
|
266
|
+
* Extended levels a family's previous minor already carries.
|
|
267
|
+
*
|
|
268
|
+
* pi hides `xhigh` and `max` unless a model's map names them, so an id the
|
|
269
|
+
* catalogs have not caught up with - a Claude point release the relay serves
|
|
270
|
+
* today, before pi's catalog carries it - dropped to low/medium/high and could
|
|
271
|
+
* never send the top of a ladder its own family publishes: in the catalog Opus
|
|
272
|
+
* 4.5 names neither extension, 4.6 names max, and 4.7, 4.8 and 5 name xhigh and
|
|
273
|
+
* max. The previous minor's own entry is that answer, and the only evidence
|
|
274
|
+
* there is: the relay advertises no effort metadata, so `GET /v1/models` has
|
|
275
|
+
* nothing to say about levels and pi's map is the whole vocabulary.
|
|
276
|
+
*
|
|
277
|
+
* Add-only by construction. Only levels the sibling maps to a string come
|
|
278
|
+
* across, so an id that reaches this path can gain the top of the scale but
|
|
279
|
+
* never lose a level that works today, and never takes on a sibling's spelling
|
|
280
|
+
* for off/low/medium/high - the dialect rules downstream own those.
|
|
281
|
+
*
|
|
282
|
+
* Measured against the live catalog at the time of writing: exactly two ids
|
|
283
|
+
* enter this path, `anthropic/claude-opus-5-5` and
|
|
284
|
+
* `openrouter/anthropic/claude-opus-5.5`, each inheriting xhigh and max from
|
|
285
|
+
* `claude-opus-5`. No other family in that catalog was touched.
|
|
286
|
+
*/
|
|
287
|
+
function inheritedLevelMap(
|
|
288
|
+
id: string,
|
|
289
|
+
dialect: PengepulDialect,
|
|
290
|
+
lookup: BuiltinModelLookup | undefined,
|
|
291
|
+
): Record<string, string | null> | undefined {
|
|
292
|
+
const siblingId = previousMinorId(id)
|
|
293
|
+
const siblingMap = siblingId ? lookup?.(siblingId, dialect)?.thinkingLevelMap : undefined
|
|
294
|
+
if (!siblingMap) return undefined
|
|
295
|
+
|
|
296
|
+
const inherited: Record<string, string | null> = {}
|
|
297
|
+
for (const level of INHERITED_LEVELS) {
|
|
298
|
+
const mapped = siblingMap[level]
|
|
299
|
+
if (typeof mapped === "string") inherited[level] = mapped
|
|
300
|
+
}
|
|
301
|
+
return Object.keys(inherited).length > 0 ? inherited : undefined
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
/**
|
|
305
|
+
* The same id one minor older, keeping the routing namespace:
|
|
306
|
+
* `anthropic/claude-opus-5-5` -> `anthropic/claude-opus-5`, and the dotted
|
|
307
|
+
* `claude-opus-5.5` -> `claude-opus-5` for the Chat Completions spelling.
|
|
308
|
+
* Undefined when the name carries no trailing version segment to drop, which
|
|
309
|
+
* keeps qualifiers (`-preview`, `-flash`, `-fast`) out of the derivation.
|
|
310
|
+
*/
|
|
311
|
+
function previousMinorId(id: string): string | undefined {
|
|
312
|
+
const slash = id.lastIndexOf("/")
|
|
313
|
+
const name = slash === -1 ? id : id.slice(slash + 1)
|
|
314
|
+
const stripped = name.replace(/[.-]\d+$/, "")
|
|
315
|
+
if (stripped === name || stripped === "") return undefined
|
|
316
|
+
return slash === -1 ? stripped : `${id.slice(0, slash + 1)}${stripped}`
|
|
317
|
+
}
|
|
318
|
+
|
|
269
319
|
/**
|
|
270
320
|
* Resolve a model's metadata, most trustworthy source first:
|
|
271
321
|
* 1. what pengepul advertises (first-party for this relay),
|
|
@@ -273,6 +323,11 @@ function fallbackLevelMap(id: string): Record<string, string | null> | undefined
|
|
|
273
323
|
* 3. family heuristics.
|
|
274
324
|
* Sources merge per field, so a relay that sends only `context_window` still
|
|
275
325
|
* picks up pricing and modalities from the catalog.
|
|
326
|
+
*
|
|
327
|
+
* A catalog miss on a reasoning model still gets its extended levels from the
|
|
328
|
+
* family's previous minor where the catalog carries one (`inheritedLevelMap`);
|
|
329
|
+
* the namespace fallback for the Chat Completions wire then takes precedence
|
|
330
|
+
* over both.
|
|
276
331
|
*/
|
|
277
332
|
function metaFor(
|
|
278
333
|
entry: Record<string, unknown>,
|
|
@@ -301,7 +356,20 @@ function metaFor(
|
|
|
301
356
|
const fallback =
|
|
302
357
|
known || !reasoning || dialect !== "openai-completions" ? undefined : fallbackLevelMap(id)
|
|
303
358
|
|
|
304
|
-
|
|
359
|
+
// A catalog miss is not always silence: the family's previous minor answers
|
|
360
|
+
// for the levels the catalog has not named yet. Whatever the base resolved
|
|
361
|
+
// wins over it, so a relay that starts advertising a map still outranks this.
|
|
362
|
+
const inherited = known || !reasoning ? undefined : inheritedLevelMap(id, dialect, lookup)
|
|
363
|
+
const levelMap = inherited
|
|
364
|
+
? { ...inherited, ...(meta.thinkingLevelMap ?? {}) }
|
|
365
|
+
: meta.thinkingLevelMap
|
|
366
|
+
|
|
367
|
+
return {
|
|
368
|
+
...meta,
|
|
369
|
+
reasoning,
|
|
370
|
+
...(levelMap ? { thinkingLevelMap: levelMap } : {}),
|
|
371
|
+
...(fallback ? { thinkingLevelMap: fallback } : {}),
|
|
372
|
+
}
|
|
305
373
|
}
|
|
306
374
|
|
|
307
375
|
function toPengepulModel(
|
|
@@ -454,105 +522,162 @@ export function modelsFromApiResponse(
|
|
|
454
522
|
return models
|
|
455
523
|
}
|
|
456
524
|
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
525
|
+
import type { Api, Model } from "@earendil-works/pi-ai"
|
|
526
|
+
|
|
527
|
+
/** The pi-ai model shapes this provider serves: one per dialect the relay speaks. */
|
|
528
|
+
export type PengepulModelEntry = Model<"anthropic-messages"> | Model<"openai-completions">
|
|
529
|
+
|
|
530
|
+
/** The provider id every pengepul model is stamped with. */
|
|
531
|
+
export const PENGEPUL_PROVIDER_ID = "pengepul"
|
|
532
|
+
|
|
533
|
+
/** What every model carries regardless of wire: identity, pricing, and the limits. */
|
|
534
|
+
function sharedModelFields(model: PengepulModel, relayBase: string) {
|
|
535
|
+
return {
|
|
536
|
+
id: model.id,
|
|
537
|
+
name: model.name,
|
|
538
|
+
provider: PENGEPUL_PROVIDER_ID,
|
|
539
|
+
baseUrl: baseUrlForDialect(relayBase, model.dialect),
|
|
540
|
+
reasoning: model.reasoning,
|
|
541
|
+
input: model.input,
|
|
542
|
+
cost: model.cost,
|
|
543
|
+
contextWindow: model.contextWindow,
|
|
544
|
+
maxTokens: model.maxTokens,
|
|
473
545
|
}
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
546
|
+
}
|
|
547
|
+
|
|
548
|
+
/**
|
|
549
|
+
* The relay's prompt-cache affinity pin, on both wires.
|
|
550
|
+
*
|
|
551
|
+
* The relay's conversation_key resolves `x-claude-code-session-id`, then
|
|
552
|
+
* `x-session-id`, then the body's `prompt_cache_key`, then a hash of the
|
|
553
|
+
* cacheable prefix. pi emits one of those headers only for the `openrouter`
|
|
554
|
+
* affinity format, and only when the send flag is set. Both auto-detected
|
|
555
|
+
* defaults are wrong here: openai-completions picks `openai` (session_id +
|
|
556
|
+
* x-client-request-id + x-session-affinity) and anthropic-messages picks
|
|
557
|
+
* nothing at all. The header is the cheaper and more explicit of the two
|
|
558
|
+
* signals and it outranks the body field, so the pin keeps a session's account
|
|
559
|
+
* stable by the relay's first rule rather than its third. Losing it costs a
|
|
560
|
+
* session that migrates between pooled accounts its whole prefix: the upstream
|
|
561
|
+
* cache is per account.
|
|
562
|
+
*
|
|
563
|
+
* Measured, not assumed — `test/affinity-wire.test.ts` dumps both bodies:
|
|
564
|
+
* openai-completions carries `prompt_cache_key: <sessionId>` (and
|
|
565
|
+
* `prompt_cache_retention: "24h"`) under PI_CACHE_RETENTION=long, so the body
|
|
566
|
+
* field alone would name the conversation; anthropic-messages carries no
|
|
567
|
+
* `prompt_cache_key` at all, and pi-ai hardcodes `x-session-affinity` there,
|
|
568
|
+
* which this relay does not read. Messages traffic therefore rests entirely on
|
|
569
|
+
* the relay's prefix fallback until a pi release honours
|
|
570
|
+
* `sessionAffinityFormat` on that dialect.
|
|
571
|
+
*
|
|
572
|
+
* The two dialects do not land at the same time. openai-completions honors
|
|
573
|
+
* `sessionAffinityFormat` in every released pi. anthropic-messages only reads it
|
|
574
|
+
* from the unreleased change on pi main (commit bbb61e34a), which is why the
|
|
575
|
+
* field is re-declared below rather than taken from `AnthropicMessagesCompat`:
|
|
576
|
+
* against pi-ai 0.85.1 the pin is inert, and the cost of an ignored header is
|
|
577
|
+
* zero. Pin now rather than later — the cost of forgetting is silently
|
|
578
|
+
* re-billed Claude prefixes.
|
|
579
|
+
*/
|
|
580
|
+
function affinityPin() {
|
|
581
|
+
return {
|
|
582
|
+
sendSessionAffinityHeaders: true as const,
|
|
583
|
+
sessionAffinityFormat: "openrouter" as const,
|
|
482
584
|
}
|
|
483
|
-
}
|
|
585
|
+
}
|
|
586
|
+
|
|
587
|
+
/**
|
|
588
|
+
* Anthropic compat as pi-ai 0.85.1 types it, plus the affinity format a later
|
|
589
|
+
* pi reads. Declared here so the pin does not depend on the host's pi-ai
|
|
590
|
+
* version; the field is optional, so a host that predates it ignores the value.
|
|
591
|
+
*/
|
|
592
|
+
type AnthropicCompat = NonNullable<Model<"anthropic-messages">["compat"]> & {
|
|
593
|
+
sessionAffinityFormat?: "openrouter"
|
|
594
|
+
}
|
|
595
|
+
|
|
596
|
+
/**
|
|
597
|
+
* Map the relay catalog to pi models, one base URL per dialect. Pure.
|
|
598
|
+
*
|
|
599
|
+
* The per-model `baseUrl` is the only place the dialect split can live: pi
|
|
600
|
+
* applies a base URL returned from `auth.resolve()` to every model at once
|
|
601
|
+
* (`models.js` `applyAuth`), so a provider-wide value would send Anthropic
|
|
602
|
+
* Messages traffic to `/v1` and Chat Completions traffic to the root.
|
|
603
|
+
*/
|
|
604
|
+
export function toPengepulModels(
|
|
605
|
+
models: readonly PengepulModel[],
|
|
606
|
+
relayBase: string,
|
|
607
|
+
): PengepulModelEntry[] {
|
|
484
608
|
return models.map((model) => {
|
|
485
|
-
const
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
|
|
489
|
-
|
|
609
|
+
const shared = sharedModelFields(model, relayBase)
|
|
610
|
+
if (model.dialect === "anthropic-messages") {
|
|
611
|
+
const adaptive = model.reasoning
|
|
612
|
+
return {
|
|
613
|
+
...shared,
|
|
614
|
+
api: "anthropic-messages" as const,
|
|
615
|
+
// Inherited level mapping (e.g. deepseek {high:"high"}) flows through;
|
|
616
|
+
// adaptive Claude models additionally mark "off" unsupported so the
|
|
617
|
+
// stream omits thinking:{type:"disabled"} (upstream rejects it).
|
|
618
|
+
...(model.thinkingLevelMap || adaptive
|
|
619
|
+
? {
|
|
620
|
+
thinkingLevelMap: {
|
|
621
|
+
...(model.thinkingLevelMap ?? {}),
|
|
622
|
+
...(adaptive ? { off: null } : {}),
|
|
623
|
+
},
|
|
624
|
+
}
|
|
625
|
+
: {}),
|
|
626
|
+
// Reasoning-capable Claude models run on the adaptive-thinking wire:
|
|
627
|
+
// pi's streamSimple always passes thinkingEnabled:false when no level is
|
|
628
|
+
// selected, and the stream would send thinking:{type:"disabled"}, which
|
|
629
|
+
// the upstream rejects (400: "thinking.type.disabled is not supported
|
|
630
|
+
// for this model"). thinkingLevelMap.off = null marks "off" as
|
|
631
|
+
// unsupported so pi omits the thinking param entirely (server default
|
|
632
|
+
// = adaptive), and forceAdaptiveThinking routes an explicit level to
|
|
633
|
+
// {type:"adaptive"} + effort instead of budget_tokens.
|
|
634
|
+
//
|
|
635
|
+
// The 1h cache TTL is a Messages-dialect feature too: `cache_control.ttl`
|
|
636
|
+
// has nowhere to go on the Chat Completions wire. Reasoning is not part
|
|
637
|
+
// of it — a non-reasoning Claude model caches the same way.
|
|
638
|
+
compat: {
|
|
639
|
+
...affinityPin(),
|
|
640
|
+
...(adaptive ? { forceAdaptiveThinking: true as const } : {}),
|
|
641
|
+
supportsLongCacheRetention: true as const,
|
|
642
|
+
} satisfies AnthropicCompat,
|
|
643
|
+
}
|
|
644
|
+
}
|
|
645
|
+
|
|
490
646
|
return {
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
reasoning: model.reasoning,
|
|
496
|
-
input: model.input,
|
|
497
|
-
cost: model.cost,
|
|
498
|
-
contextWindow: model.contextWindow,
|
|
499
|
-
maxTokens: model.maxTokens,
|
|
500
|
-
// Inherited level mapping (e.g. deepseek {high:"high"}) flows through;
|
|
501
|
-
// adaptive Claude models additionally mark "off" unsupported so the
|
|
502
|
-
// stream omits thinking:{type:"disabled"} (upstream rejects it).
|
|
503
|
-
...(model.thinkingLevelMap || adaptive
|
|
504
|
-
? { thinkingLevelMap: { ...(model.thinkingLevelMap ?? {}), ...(adaptive ? { off: null } : {}) } }
|
|
505
|
-
: {}),
|
|
506
|
-
// Reasoning-capable Claude models run on the adaptive-thinking wire:
|
|
507
|
-
// pi's streamSimple always passes thinkingEnabled:false when no level is
|
|
508
|
-
// selected, and the stream would send thinking:{type:"disabled"}, which
|
|
509
|
-
// the upstream rejects (400: "thinking.type.disabled is not supported
|
|
510
|
-
// for this model"). thinkingLevelMap.off = null marks "off" as
|
|
511
|
-
// unsupported so pi omits the thinking param entirely (server default
|
|
512
|
-
// = adaptive), and forceAdaptiveThinking routes an explicit level to
|
|
513
|
-
// {type:"adaptive"} + effort instead of budget_tokens.
|
|
514
|
-
// Both dialects pin both fields, and that is why `compat` is
|
|
515
|
-
// unconditional. The relay's prompt-cache affinity key resolves in this
|
|
516
|
-
// order: `x-claude-code-session-id`, `x-session-id`, the body's
|
|
517
|
-
// `prompt_cache_key`, then a hash of the cacheable request prefix
|
|
518
|
-
// (app.rs `conversation_key`). pi emits `x-session-id` only for the
|
|
519
|
-
// `openrouter` affinity format, and only when the send flag is set;
|
|
520
|
-
// the auto-detected defaults are wrong here in both cases
|
|
521
|
-
// (openai-completions picks `openai`: session_id + x-client-request-id +
|
|
522
|
-
// x-session-affinity; anthropic-messages picks nothing). The header is
|
|
523
|
-
// the cheaper and more explicit of the two signals and it outranks the
|
|
524
|
-
// body field, so pinning it keeps a session's account stable by the
|
|
525
|
-
// relay's first rule rather than its third. Losing the pin costs a
|
|
526
|
-
// session that migrates between pooled accounts its whole prefix: the
|
|
527
|
-
// upstream cache is per account.
|
|
528
|
-
//
|
|
529
|
-
// Measured, not assumed — `test/affinity-wire.test.ts` dumps both bodies:
|
|
530
|
-
// openai-completions carries `prompt_cache_key: <sessionId>` (and
|
|
531
|
-
// `prompt_cache_retention: "24h"`) under PI_CACHE_RETENTION=long, so the
|
|
532
|
-
// body field alone would name the conversation; anthropic-messages
|
|
533
|
-
// carries no `prompt_cache_key` at all, and pi-ai hardcodes
|
|
534
|
-
// `x-session-affinity` there, which this relay does not read. Messages
|
|
535
|
-
// traffic therefore rests entirely on the relay's prefix fallback until
|
|
536
|
-
// a pi release honours `sessionAffinityFormat` on that dialect.
|
|
537
|
-
//
|
|
538
|
-
// The two dialects do not land at the same time. openai-completions
|
|
539
|
-
// honors `sessionAffinityFormat` in every released pi. anthropic-messages
|
|
540
|
-
// only reads it from the unreleased change on pi main (commit
|
|
541
|
-
// bbb61e34a); through pi-ai 0.85.1 that client hardcodes the header name
|
|
542
|
-
// `x-session-affinity`, which this relay does not read, so the Claude
|
|
543
|
-
// pin below is inert until pi ships it. Pin now rather than later: the
|
|
544
|
-
// cost of an ignored header is zero, and the cost of forgetting is
|
|
545
|
-
// silently re-billed Claude prefixes.
|
|
546
|
-
compat: {
|
|
547
|
-
...(adaptive ? { forceAdaptiveThinking: true as const } : {}),
|
|
548
|
-
...(longCacheRetention ? { supportsLongCacheRetention: true as const } : {}),
|
|
549
|
-
sendSessionAffinityHeaders: true as const,
|
|
550
|
-
sessionAffinityFormat: "openrouter" as const,
|
|
551
|
-
},
|
|
647
|
+
...shared,
|
|
648
|
+
api: "openai-completions" as const,
|
|
649
|
+
...(model.thinkingLevelMap ? { thinkingLevelMap: model.thinkingLevelMap } : {}),
|
|
650
|
+
compat: { ...affinityPin() },
|
|
552
651
|
}
|
|
553
652
|
})
|
|
554
653
|
}
|
|
555
654
|
|
|
655
|
+
/** Whether a stored pi model is one this provider published. */
|
|
656
|
+
export function isPengepulModelEntry(model: Model<Api>): model is PengepulModelEntry {
|
|
657
|
+
return (
|
|
658
|
+
model.provider === PENGEPUL_PROVIDER_ID &&
|
|
659
|
+
(model.api === "anthropic-messages" || model.api === "openai-completions")
|
|
660
|
+
)
|
|
661
|
+
}
|
|
662
|
+
|
|
663
|
+
/**
|
|
664
|
+
* Re-derive every model's base URL from the base currently configured.
|
|
665
|
+
*
|
|
666
|
+
* pi's model store keeps whole models, baseUrl included, and replays them
|
|
667
|
+
* before the network phase. A relay that moved would otherwise be reached at
|
|
668
|
+
* its old address until a fetch succeeds — which never happens when the old
|
|
669
|
+
* address is gone.
|
|
670
|
+
*/
|
|
671
|
+
export function restampRelayBase(
|
|
672
|
+
models: readonly PengepulModelEntry[],
|
|
673
|
+
relayBase: string,
|
|
674
|
+
): PengepulModelEntry[] {
|
|
675
|
+
return models.map((model) => ({
|
|
676
|
+
...model,
|
|
677
|
+
baseUrl: baseUrlForDialect(relayBase, model.api),
|
|
678
|
+
}))
|
|
679
|
+
}
|
|
680
|
+
|
|
556
681
|
/** Picker label: the bare model part of a relay id, suffixed. `anthropic/claude-opus-5` -> `claude-opus-5 (pengepul)`. */
|
|
557
682
|
function displayName(id: string): string {
|
|
558
683
|
return `${bareId(id)} (pengepul)`
|
|
@@ -588,7 +713,7 @@ function configuredTimeoutMs(timeoutMs: number | undefined): number {
|
|
|
588
713
|
}
|
|
589
714
|
|
|
590
715
|
export function getModelsTimeoutMs(env: NodeJS.ProcessEnv = process.env): number {
|
|
591
|
-
const raw = env[
|
|
716
|
+
const raw = env[MODELS_TIMEOUT_MS_ENV]
|
|
592
717
|
if (!raw) return DEFAULT_MODELS_TIMEOUT_MS
|
|
593
718
|
const parsed = Number(raw)
|
|
594
719
|
return configuredTimeoutMs(parsed)
|
|
@@ -674,7 +799,7 @@ export async function fetchPengepulModels(
|
|
|
674
799
|
throw new Error(
|
|
675
800
|
`pengepul rejected the API key (${
|
|
676
801
|
response.status
|
|
677
|
-
}).
|
|
802
|
+
}). Run /login pengepul, or set the key in ~/.pi/agent/auth.json or PENGEPUL_API_KEY.`,
|
|
678
803
|
)
|
|
679
804
|
}
|
|
680
805
|
if (!response.ok) {
|
|
@@ -782,65 +907,4 @@ export async function loadCachedPengepulModels(
|
|
|
782
907
|
}
|
|
783
908
|
}
|
|
784
909
|
|
|
785
|
-
async function writeCache(cachePath: string, models: readonly PengepulModel[]): Promise<void> {
|
|
786
|
-
await mkdir(dirname(cachePath), { recursive: true })
|
|
787
|
-
// Unique per write: the runtime can issue two overlapping writes in one
|
|
788
|
-
// process (cache-first + background refresh), and a shared pid-keyed name
|
|
789
|
-
// would let the first rename remove the second's source mid-flight.
|
|
790
|
-
const temporaryPath = `${cachePath}.${process.pid}.${randomUUID()}.tmp`
|
|
791
|
-
|
|
792
|
-
try {
|
|
793
|
-
await writeFile(
|
|
794
|
-
temporaryPath,
|
|
795
|
-
`${JSON.stringify({ version: MODEL_CACHE_VERSION, models }, null, 2)}\n`,
|
|
796
|
-
{ encoding: "utf-8", mode: 0o600 },
|
|
797
|
-
)
|
|
798
|
-
await rename(temporaryPath, cachePath)
|
|
799
|
-
} finally {
|
|
800
|
-
try {
|
|
801
|
-
await rm(temporaryPath, { force: true })
|
|
802
|
-
} catch {
|
|
803
|
-
// Best-effort cleanup must not hide the original cache write error.
|
|
804
|
-
}
|
|
805
|
-
}
|
|
806
|
-
}
|
|
807
|
-
|
|
808
|
-
export async function loadPengepulModels(
|
|
809
|
-
options: LoadModelsOptions,
|
|
810
|
-
): Promise<PengepulModelSource> {
|
|
811
|
-
const cachePath = options.cachePath
|
|
812
|
-
|
|
813
|
-
try {
|
|
814
|
-
const models = await fetchPengepulModels(options)
|
|
815
|
-
|
|
816
|
-
try {
|
|
817
|
-
await writeCache(cachePath, models)
|
|
818
|
-
return { models, source: "live" }
|
|
819
|
-
} catch (error) {
|
|
820
|
-
return {
|
|
821
|
-
models,
|
|
822
|
-
source: "live",
|
|
823
|
-
warning: `Loaded the live pengepul model catalog but could not update ${cachePath}: ${errorMessage(error)}`,
|
|
824
|
-
}
|
|
825
|
-
}
|
|
826
|
-
} catch (liveError) {
|
|
827
|
-
if (options.signal?.aborted) throw abortError(options.signal.reason ?? liveError)
|
|
828
|
-
|
|
829
|
-
try {
|
|
830
|
-
const models = await readCache(cachePath)
|
|
831
|
-
return {
|
|
832
|
-
models,
|
|
833
|
-
source: "cache",
|
|
834
|
-
warning: `Could not refresh the pengepul model catalog (${errorMessage(liveError)}). Using the cached catalog from ${cachePath}.`,
|
|
835
|
-
}
|
|
836
|
-
} catch (cacheError) {
|
|
837
|
-
return {
|
|
838
|
-
models: [],
|
|
839
|
-
source: "empty",
|
|
840
|
-
warning: `Could not refresh the pengepul model catalog (${errorMessage(liveError)}), and no valid cached catalog is available at ${cachePath} (${errorMessage(cacheError)}). pengepul models will remain unavailable until the next startup refresh succeeds.`,
|
|
841
|
-
}
|
|
842
|
-
}
|
|
843
|
-
}
|
|
844
|
-
}
|
|
845
|
-
|
|
846
910
|
|