@billjr99/pi-openai-compat 1.1.11 → 1.1.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/README.md +34 -11
  2. package/index.ts +133 -15
  3. package/package.json +1 -1
package/README.md CHANGED
@@ -49,24 +49,34 @@ If pi is already running when you install, type `/reload` first.
49
49
  | **Venice AI** | `https://api.venice.ai/api/v1` | API key from venice.ai/settings/api |
50
50
  | **Fireworks AI** | `https://api.fireworks.ai/inference/v1` | API key from fireworks.ai/account/api-keys |
51
51
  | **Together AI** | `https://api.together.xyz/v1` | API key from api.together.ai/settings/api-keys |
52
- | **Cloudflare Workers AI** | `https://api.cloudflare.com/client/v4/accounts/{account}/ai/v1` | API token from dash.cloudflare.com |
53
- | **Cloudflare AI Gateway** | `https://gateway.ai.cloudflare.com/v1/{account}/{gateway}/openai` | API token from dash.cloudflare.com |
52
+ | **GitHub Models** | `https://models.github.ai/inference` | GitHub PAT fine-grained, `Models → read` |
53
+ | **Cloudflare Workers AI** | `https://api.cloudflare.com/client/v4/accounts/{account}/ai/v1` | API token from dash.cloudflare.com (`Workers AI: Read`) |
54
+ | **Cloudflare AI Gateway** | `https://gateway.ai.cloudflare.com/v1/{account}/{gateway}/openai` | API token from dash.cloudflare.com (`AI Gateway: Run` + `Workers AI: Read`) |
54
55
  | **Vercel AI Gateway** | `https://ai-gateway.vercel.sh/v1` | API key from vercel.com |
55
56
  | **OpenCode Zen** | `https://opencode.ai/zen/v1` | API key from opencode.ai |
56
57
  | **Ollama (local)** | `http://localhost:11434/v1` | Keyless |
57
58
  | **Ollama Cloud** | `https://ollama.com/v1` | Ollama Cloud API key from ollama.com |
58
59
  | **Custom** | Any URL you supply | Optional bearer token |
59
60
 
60
- > **Providers with built-in fallback model lists (as of May 2026)**
61
- > Some providers do not support `GET /v1/models` or return a non-JSON response.
62
- > For these, the extension substitutes a built-in list and shows a warning during
63
- > `/compat-login`.
61
+ > **Providers whose model catalog lives at a non-standard `/models` path (as of June 2026)**
62
+ > Some providers don't return models at `<base_url>/models`. The extension
63
+ > handles them in one of two ways:
64
64
  >
65
- > | Provider | Reason |
66
- > |---|---|
67
- > | **Cloudflare Workers AI** | Returns HTTP 405 method not supported |
68
- > | **Cloudflare AI Gateway** | Returns HTTP 401 — no anonymous model enumeration |
69
- > | **Hugging Face** | Returns HTML rather than JSON for `/v1/models` |
65
+ > 1. **Live discovery via a per-template URL override** (`modelsUrl`,
66
+ > `modelsIdField`, `modelsKeepTask` on the template). Used for **GitHub
67
+ > Models** and **Cloudflare Workers AI** — `/compat-login` fetches the real
68
+ > catalog and registers actual current models, not a curated guess.
69
+ > 2. **Built-in fallback list** when no working catalog endpoint exists. Used
70
+ > when discovery fails or when the upstream simply has no `/models` at all
71
+ > (e.g. **Hugging Face** returns HTML, **Cloudflare AI Gateway** has no
72
+ > catalog endpoint).
73
+ >
74
+ > | Provider | Default `/models` symptom | Handling |
75
+ > |---|---|---|
76
+ > | **GitHub Models** | HTTP 404 at `<base_url>/models` (catalog is at `/catalog/models`, not `/inference/models`) | Live discovery — `modelsUrl: https://models.github.ai/catalog/models` |
77
+ > | **Cloudflare Workers AI** | HTTP 405 (no `GET /v1/models`; real catalog at `/ai/models/search`, id field is `name`, mixed task types) | Live discovery — `modelsUrl: …/ai/models/search`, `modelsIdField: name`, `modelsKeepTask: "Text Generation"` |
78
+ > | **Cloudflare AI Gateway** | HTTP 401 (token missing `AI Gateway: Run`) or HTTP 400 *"Please configure AI Gateway"* (gateway slug doesn't exist, or the upstream isn't configured on it) | Built-in fallback list |
79
+ > | **Hugging Face** | Returns HTML rather than JSON | Built-in fallback list |
70
80
 
71
81
  ---
72
82
 
@@ -272,6 +282,19 @@ pi remove npm:@billjr99/pi-openai-compat # uninstall
272
282
  Verify the base URL does not have a trailing slash and ends with `/v1`.
273
283
  Confirm the API key is correct and has model-access permissions.
274
284
  For Ollama: ensure `ollama serve` is running.
285
+ For self-hosted Open WebUI: the OpenAI-compatible API is under `/api/v1`, not
286
+ the SPA root — e.g. `https://your-host/api/v1`, not `https://your-host`.
287
+ Hitting the root returns the HTML SPA and `/compat-login` fails with a JSON
288
+ parse error.
289
+
290
+ **Cloudflare AI Gateway returns 401 Unauthorized**
291
+ Check both: (1) the API token has `AI Gateway: Run` *and* `Workers AI: Read`
292
+ under "Permissions", scoped to the correct account; and (2) the gateway slug
293
+ in the URL actually exists under that account — list with
294
+ `curl https://api.cloudflare.com/client/v4/accounts/$ACCOUNT_ID/ai-gateway/gateways
295
+ -H "Authorization: Bearer $TOKEN"`. An empty `result` array means the gateway
296
+ isn't there (you may need to create it in **dash.cloudflare.com → AI → AI
297
+ Gateway**, or you're querying the wrong account).
275
298
 
276
299
  **No models appear after login**
277
300
  For Ollama: pull at least one model first (`ollama pull llama3`).
package/index.ts CHANGED
@@ -35,6 +35,14 @@ interface ProviderConfig {
35
35
  baseUrl: string;
36
36
  apiKey: string | null;
37
37
  cachedModels: CachedModel[];
38
+ // Optional model-discovery overrides for providers whose /models endpoint
39
+ // lives at a non-standard path / shape (e.g. GitHub Models' /catalog/models,
40
+ // Cloudflare Workers AI's /ai/models/search). Persisted on the provider so
41
+ // session_start re-fetches (when cachedModels is empty) use the override
42
+ // rather than the broken default <baseUrl>/models path.
43
+ modelsUrl?: string;
44
+ modelsIdField?: string;
45
+ modelsKeepTask?: string;
38
46
  }
39
47
 
40
48
  interface ExtensionConfig {
@@ -48,6 +56,15 @@ interface OpenAIModelsResponse {
48
56
  data: Array<{ id: string; context_window?: number; max_tokens?: number }>;
49
57
  }
50
58
 
59
+ /** Loose shape for a single entry in any /models response. */
60
+ type RawModel = {
61
+ id?: string;
62
+ name?: string;
63
+ context_window?: number;
64
+ max_tokens?: number;
65
+ task?: { name?: string };
66
+ };
67
+
51
68
  // ─────────────────────────────────────────────────────────────────────────────
52
69
  // Provider templates
53
70
  // ─────────────────────────────────────────────────────────────────────────────
@@ -64,6 +81,17 @@ const TEMPLATES: Record<string, {
64
81
  fallbackModels?: string[];
65
82
  /** Where to obtain the API key; presence implies the key is required. */
66
83
  keyHint?: string;
84
+ /**
85
+ * Optional model-discovery overrides for providers whose /models endpoint
86
+ * lives at a non-standard path / shape. May contain the same placeholders
87
+ * as baseUrl (YOUR_ACCOUNT_ID, YOUR_GATEWAY_SLUG, YOUR_PROVIDER); they are
88
+ * substituted alongside the baseUrl substitution in /compat-login.
89
+ */
90
+ modelsUrl?: string;
91
+ /** Field on each model entry that carries the upstream id (default "id"). */
92
+ modelsIdField?: string;
93
+ /** Keep only models whose task.name matches this string (case-insensitive). */
94
+ modelsKeepTask?: string;
67
95
  }> = {
68
96
  openrouter: {
69
97
  displayName: "OpenRouter",
@@ -97,9 +125,13 @@ const TEMPLATES: Record<string, {
97
125
  },
98
126
  github_models: {
99
127
  displayName: "GitHub Models",
100
- baseUrl: "https://models.inference.ai.azure.com",
128
+ // The legacy Azure endpoint (models.inference.ai.azure.com) was retired.
129
+ // Chat now lives under /inference, and the catalog under /catalog/models
130
+ // on the same host — different path entirely, hence the modelsUrl override.
131
+ baseUrl: "https://models.github.ai/inference",
101
132
  keyless: false,
102
- keyHint: "github.com/settings/tokens",
133
+ keyHint: "github.com/settings/tokens (fine-grained: Models → read)",
134
+ modelsUrl: "https://models.github.ai/catalog/models",
103
135
  },
104
136
  sambanova: {
105
137
  displayName: "SambaNova",
@@ -124,8 +156,14 @@ const TEMPLATES: Record<string, {
124
156
  baseUrl: "https://api.cloudflare.com/client/v4/accounts/YOUR_ACCOUNT_ID/ai/v1",
125
157
  keyless: false,
126
158
  promptUrl: true,
127
- keyHint: "dash.cloudflare.com → My Profile → API Tokens",
128
- // Cloudflare Workers AI returns 405 for GET /v1/models; use a curated list.
159
+ keyHint: "dash.cloudflare.com → My Profile → API Tokens (`Workers AI: Read`)",
160
+ // The OpenAI-compat base /ai/v1 returns 405 for GET /models. The real
161
+ // catalog lives at /ai/models/search, keys ids in "name" (reserving "id"
162
+ // for an internal UUID), and mixes Text Generation with embeddings and
163
+ // image tasks — so we filter by task.name.
164
+ modelsUrl: "https://api.cloudflare.com/client/v4/accounts/YOUR_ACCOUNT_ID/ai/models/search?per_page=100",
165
+ modelsIdField: "name",
166
+ modelsKeepTask: "Text Generation",
129
167
  fallbackModels: [
130
168
  "@cf/meta/llama-4-scout-17b-16e-instruct",
131
169
  "@cf/meta/llama-3.3-70b-instruct-fp8-fast",
@@ -138,10 +176,16 @@ const TEMPLATES: Record<string, {
138
176
  displayName: "Cloudflare AI Gateway",
139
177
  // YOUR_PROVIDER is the upstream slug (e.g. "workers-ai", "openai"). /v1 is
140
178
  // appended so that fetchModels and chat completions hit the correct path.
179
+ // Common failure modes:
180
+ // - 401 "Unauthorized": the token lacks `AI Gateway: Run` permission,
181
+ // or the gateway has Authenticated Gateway enabled (requires a
182
+ // separate cf-aig-authorization header, not yet supported here).
183
+ // - 400 "Please configure AI Gateway": the gateway slug doesn't exist
184
+ // under this account, or the upstream provider isn't configured on it.
141
185
  baseUrl: "https://gateway.ai.cloudflare.com/v1/YOUR_ACCOUNT_ID/YOUR_GATEWAY_SLUG/YOUR_PROVIDER/v1",
142
186
  keyless: false,
143
187
  promptUrl: true,
144
- keyHint: "dash.cloudflare.com → My Profile → API Tokens",
188
+ keyHint: "dash.cloudflare.com → My Profile → API Tokens (`AI Gateway: Run` + `Workers AI: Read`)",
145
189
  fallbackModels: [
146
190
  "@cf/meta/llama-4-scout-17b-16e-instruct",
147
191
  "@cf/meta/llama-3.3-70b-instruct-fp8-fast",
@@ -323,22 +367,69 @@ function isLocalUrl(url: string): boolean {
323
367
  }
324
368
  }
325
369
 
326
- async function fetchModels(baseUrl: string, apiKey: string | null): Promise<CachedModel[]> {
327
- const url = `${baseUrl.replace(/\/+$/, "")}/models`;
370
+ /** Optional per-provider overrides controlling how /models is fetched. */
371
+ interface FetchOverrides {
372
+ /** Full URL to fetch instead of `<baseUrl>/models`. */
373
+ url?: string;
374
+ /** Field on each entry that holds the upstream id (default `id`). */
375
+ idField?: string;
376
+ /** Keep only entries whose `task.name` matches (case-insensitive). */
377
+ keepTask?: string;
378
+ }
379
+
380
+ async function fetchModels(
381
+ baseUrl: string,
382
+ apiKey: string | null,
383
+ overrides: FetchOverrides = {},
384
+ ): Promise<CachedModel[]> {
385
+ // Honor a per-provider override (e.g. GitHub Models' /catalog/models lives
386
+ // on a different path than its inference endpoint; Cloudflare Workers AI's
387
+ // catalog is at /ai/models/search). Fall back to <baseUrl>/models otherwise.
388
+ const url = overrides.url ?? `${baseUrl.replace(/\/+$/, "")}/models`;
389
+ const idField = overrides.idField ?? "id";
390
+ const keepTask = overrides.keepTask;
391
+
328
392
  const headers: Record<string, string> = { Accept: "application/json" };
329
393
  if (apiKey) headers["Authorization"] = `Bearer ${apiKey}`;
330
394
 
331
395
  const resp = await fetch(url, { headers });
332
396
  if (!resp.ok) {
397
+ // The Authorization header is never echoed here, so the error body is
398
+ // safe to surface even though we include the upstream's full response.
333
399
  const body = await resp.text().catch(() => "");
334
400
  throw new Error(`HTTP ${resp.status} from ${url}: ${body}`);
335
401
  }
336
402
 
337
- const json = (await resp.json()) as OpenAIModelsResponse;
338
- if (!Array.isArray(json?.data)) throw new Error(`Unexpected response from ${url}`);
403
+ // Normalize the various shapes /models can return:
404
+ // - OpenAI style: {"data": [...]}
405
+ // - Cloudflare / some gateways: {"result": [...]}
406
+ // - Together (and a few others): [...] (bare JSON array)
407
+ const json = (await resp.json()) as unknown;
408
+ let raw: RawModel[] | undefined;
409
+ if (Array.isArray(json)) {
410
+ raw = json as RawModel[];
411
+ } else if (json && typeof json === "object") {
412
+ const obj = json as { data?: unknown; result?: unknown };
413
+ if (Array.isArray(obj.data)) raw = obj.data as RawModel[];
414
+ else if (Array.isArray(obj.result)) raw = obj.result as RawModel[];
415
+ }
416
+ if (!raw) {
417
+ throw new Error(
418
+ `Unexpected /models payload shape from ${url} ` +
419
+ `(expected an array or an object with a "data" or "result" array).`
420
+ );
421
+ }
339
422
 
340
- return json.data
341
- .map((m) => ({ id: m.id, contextWindow: m.context_window, maxTokens: m.max_tokens }))
423
+ return raw
424
+ .filter((m) => {
425
+ if (keepTask === undefined) return true;
426
+ const taskName = m.task?.name ?? "";
427
+ return taskName.toLowerCase() === keepTask.toLowerCase();
428
+ })
429
+ .map((m) => {
430
+ const id = (m as Record<string, unknown>)[idField] as string | undefined;
431
+ return { id: id ?? "", contextWindow: m.context_window, maxTokens: m.max_tokens };
432
+ })
342
433
  .filter((m) => Boolean(m.id))
343
434
  .sort((a, b) => a.id.localeCompare(b.id));
344
435
  }
@@ -404,9 +495,14 @@ export default async function (pi: ExtensionAPI) {
404
495
  registerProvider(pi, key, p);
405
496
  registered.push(p.displayName);
406
497
  } else {
407
- // Cache is empty (e.g. migrated from older config). Try a live fetch.
498
+ // Cache is empty (e.g. migrated from older config). Try a live fetch,
499
+ // honoring any per-provider discovery overrides stored on the config.
408
500
  try {
409
- const models = await fetchModels(p.baseUrl, p.apiKey);
501
+ const models = await fetchModels(p.baseUrl, p.apiKey, {
502
+ url: p.modelsUrl,
503
+ idField: p.modelsIdField,
504
+ keepTask: p.modelsKeepTask,
505
+ });
410
506
  if (models.length > 0) {
411
507
  p.cachedModels = models;
412
508
  saveConfig(config);
@@ -456,6 +552,11 @@ export default async function (pi: ExtensionAPI) {
456
552
 
457
553
  // Step 2 — base URL
458
554
  let baseUrl = tpl.baseUrl;
555
+ // modelsUrl tracks the discovery URL through the same placeholder
556
+ // substitutions baseUrl goes through, so providers that put the model
557
+ // catalog on a different path (GitHub Models, Cloudflare Workers AI)
558
+ // get a fully-resolved URL by the time we call fetchModels.
559
+ let modelsUrl: string | undefined = tpl.modelsUrl;
459
560
  if (key === "cloudflare_workers") {
460
561
  const entered = await ctx.ui.input(
461
562
  "Account ID",
@@ -466,6 +567,7 @@ export default async function (pi: ExtensionAPI) {
466
567
  const accountId = entered.trim();
467
568
  if (!accountId) { ctx.ui.notify("Account ID cannot be empty.", "error"); return; }
468
569
  baseUrl = tpl.baseUrl.replace("YOUR_ACCOUNT_ID", accountId);
570
+ if (modelsUrl) modelsUrl = modelsUrl.replace("YOUR_ACCOUNT_ID", accountId);
469
571
  } else if (key === "cloudflare_ai_gateway") {
470
572
  const accountIdInput = await ctx.ui.input(
471
573
  "Account ID",
@@ -497,6 +599,12 @@ export default async function (pi: ExtensionAPI) {
497
599
  .replace("YOUR_ACCOUNT_ID", accountId)
498
600
  .replace("YOUR_GATEWAY_SLUG", gatewaySlug)
499
601
  .replace("YOUR_PROVIDER", provider);
602
+ if (modelsUrl) {
603
+ modelsUrl = modelsUrl
604
+ .replace("YOUR_ACCOUNT_ID", accountId)
605
+ .replace("YOUR_GATEWAY_SLUG", gatewaySlug)
606
+ .replace("YOUR_PROVIDER", provider);
607
+ }
500
608
  } else if (tpl.promptUrl) {
501
609
  const defaultUrl = tpl.baseUrl;
502
610
  const prompt = isLocalUrl(defaultUrl)
@@ -523,7 +631,11 @@ export default async function (pi: ExtensionAPI) {
523
631
  ctx.ui.notify(`Connecting to ${baseUrl} …`, "info");
524
632
  let models: CachedModel[];
525
633
  try {
526
- models = await fetchModels(baseUrl, apiKey);
634
+ models = await fetchModels(baseUrl, apiKey, {
635
+ url: modelsUrl,
636
+ idField: tpl.modelsIdField,
637
+ keepTask: tpl.modelsKeepTask,
638
+ });
527
639
  } catch (err) {
528
640
  if (tpl.fallbackModels && tpl.fallbackModels.length > 0) {
529
641
  ctx.ui.notify(
@@ -555,12 +667,18 @@ export default async function (pi: ExtensionAPI) {
555
667
  }
556
668
  }
557
669
 
558
- // Step 5 — save to config and register with pi
670
+ // Step 5 — save to config and register with pi.
671
+ // The discovery overrides are persisted so session_start can re-fetch
672
+ // correctly when cachedModels is empty (without them, the rehydrate
673
+ // path would hit <baseUrl>/models and 404 for these providers).
559
674
  config.providers[key] = {
560
675
  displayName: tpl.displayName,
561
676
  baseUrl,
562
677
  apiKey,
563
678
  cachedModels: models,
679
+ modelsUrl,
680
+ modelsIdField: tpl.modelsIdField,
681
+ modelsKeepTask: tpl.modelsKeepTask,
564
682
  };
565
683
  saveConfig(config);
566
684
  registerProvider(pi, key, config.providers[key]);
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@billjr99/pi-openai-compat",
3
- "version": "1.1.11",
3
+ "version": "1.1.12",
4
4
  "description": "pi-coding-agent extension: OpenAI-compatible endpoint support (OpenRouter, NVIDIA NIM, Nous Portal, Ollama, custom)",
5
5
  "author": "Bill Mongan <https://github.com/BillJr99>",
6
6
  "license": "MIT",