pi-tarmis-provider 1.0.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +82 -0
- package/LICENSE +21 -0
- package/README.md +184 -0
- package/custom-models.json +1 -0
- package/deprecated-models.json +1 -0
- package/index.ts +463 -0
- package/models.json +143 -0
- package/package.json +61 -0
- package/patch.json +34 -0
- package/scripts/update-models.js +501 -0
package/index.ts
ADDED
|
@@ -0,0 +1,463 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Tarmis Provider Extension
|
|
3
|
+
*
|
|
4
|
+
* Registers Tarmis (tarmis.ai) as a custom provider using the openai-completions API.
|
|
5
|
+
* Base URL: https://api.tarmis.ai/v1
|
|
6
|
+
*
|
|
7
|
+
* Tarmis hosts open-weight models on its own GPUs behind a single
|
|
8
|
+
* OpenAI-compatible Chat Completions endpoint. The /v1/models endpoint returns
|
|
9
|
+
* rich metadata per model — name, per-token pricing (prompt, completion, and an
|
|
10
|
+
* optional cached-input rate), context_length, max_output_length,
|
|
11
|
+
* input_modalities, and supported_features ("tools", "reasoning"). This extension
|
|
12
|
+
* derives models.json directly from that metadata, so pricing, context windows,
|
|
13
|
+
* vision support, and the reasoning flag self-heal on every model sync.
|
|
14
|
+
*
|
|
15
|
+
* Key API characteristics:
|
|
16
|
+
* - OpenAI Chat Completions compatible (/v1/chat/completions)
|
|
17
|
+
* - /v1/models is PUBLIC — the model catalog is fetched without an API key,
|
|
18
|
+
* so pi lists Tarmis models (and the SWR revalidation runs) even before a
|
|
19
|
+
* key is configured. A key is only required to actually call a model.
|
|
20
|
+
* - Pricing is per-token in the API; converted to per-million tokens for pi.
|
|
21
|
+
* - supported_features drives reasoning + tool support automatically.
|
|
22
|
+
*
|
|
23
|
+
* Reasoning compatibility:
|
|
24
|
+
* The API reports which models support reasoning, but not the wire format
|
|
25
|
+
* each model expects. Reasoning models are given the standard OpenAI-
|
|
26
|
+
* compatible defaults (thinkingFormat: "openai", supportsReasoningEffort:
|
|
27
|
+
* true, supportsDeveloperRole: false). If a model on Tarmis needs a different
|
|
28
|
+
* format (e.g. DeepSeek's native thinking field), override it in patch.json —
|
|
29
|
+
* see AGENTS.md.
|
|
30
|
+
*
|
|
31
|
+
* Model resolution strategy: Stale-While-Revalidate
|
|
32
|
+
* 1. Serve stale immediately: disk cache -> embedded models.json (zero-latency)
|
|
33
|
+
* 2. Revalidate in background: live API /models -> merge with embedded -> cache -> hot-swap
|
|
34
|
+
* 3. patch.json + custom-models.json applied on top of whichever source won
|
|
35
|
+
*
|
|
36
|
+
* Merge order: [live|cache|embedded] -> apply patch.json -> merge custom-models.json
|
|
37
|
+
*
|
|
38
|
+
* Usage:
|
|
39
|
+
* # Option 1: Store in auth.json (recommended)
|
|
40
|
+
* # Add to ~/.pi/agent/auth.json:
|
|
41
|
+
* # "tarmis": { "type": "api_key", "key": "your-api-key" }
|
|
42
|
+
*
|
|
43
|
+
* # Option 2: Set as environment variable
|
|
44
|
+
* export TARMIS_API_KEY=your-api-key
|
|
45
|
+
*
|
|
46
|
+
* # Run pi with the extension
|
|
47
|
+
* pi -e /path/to/pi-tarmis-provider
|
|
48
|
+
*
|
|
49
|
+
* Then use /model to select from available models like DeepSeek V4 Flash,
|
|
50
|
+
* GLM-5.2, MiniMax M3, Kimi K3, and more.
|
|
51
|
+
*
|
|
52
|
+
* @see https://docs.tarmis.ai
|
|
53
|
+
*/
|
|
54
|
+
|
|
55
|
+
import { getAgentDir, type ExtensionAPI, type ModelRegistry } from "@earendil-works/pi-coding-agent";
|
|
56
|
+
import modelsData from "./models.json" with { type: "json" };
|
|
57
|
+
import customModelsData from "./custom-models.json" with { type: "json" };
|
|
58
|
+
import patchData from "./patch.json" with { type: "json" };
|
|
59
|
+
import deprecatedData from "./deprecated-models.json" with { type: "json" };
|
|
60
|
+
import fs from "fs";
|
|
61
|
+
import path from "path";
|
|
62
|
+
|
|
63
|
+
// ─── Types ────────────────────────────────────────────────────────────────────
|
|
64
|
+
|
|
65
|
+
interface JsonModel {
|
|
66
|
+
id: string;
|
|
67
|
+
name: string;
|
|
68
|
+
reasoning: boolean;
|
|
69
|
+
input: ("text" | "image")[];
|
|
70
|
+
cost: {
|
|
71
|
+
input: number;
|
|
72
|
+
output: number;
|
|
73
|
+
cacheRead: number;
|
|
74
|
+
cacheWrite: number;
|
|
75
|
+
};
|
|
76
|
+
contextWindow: number;
|
|
77
|
+
maxTokens: number;
|
|
78
|
+
thinkingLevelMap?: Record<string, string | null>;
|
|
79
|
+
compat?: {
|
|
80
|
+
supportsDeveloperRole?: boolean;
|
|
81
|
+
supportsStore?: boolean;
|
|
82
|
+
maxTokensField?: "max_completion_tokens" | "max_tokens";
|
|
83
|
+
thinkingFormat?: "openai" | "openrouter" | "together" | "deepseek" | "zai" | "qwen" | "qwen-chat-template" | "string-thinking";
|
|
84
|
+
supportsReasoningEffort?: boolean;
|
|
85
|
+
requiresReasoningContentOnAssistantMessages?: boolean;
|
|
86
|
+
};
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
interface PatchEntry {
|
|
90
|
+
name?: string;
|
|
91
|
+
reasoning?: boolean;
|
|
92
|
+
input?: ("text" | "image")[];
|
|
93
|
+
cost?: {
|
|
94
|
+
input?: number;
|
|
95
|
+
output?: number;
|
|
96
|
+
cacheRead?: number;
|
|
97
|
+
cacheWrite?: number;
|
|
98
|
+
};
|
|
99
|
+
contextWindow?: number;
|
|
100
|
+
maxTokens?: number;
|
|
101
|
+
thinkingLevelMap?: Record<string, string | null>;
|
|
102
|
+
compat?: Record<string, unknown>;
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
type PatchData = Record<string, PatchEntry>;
|
|
106
|
+
|
|
107
|
+
// ─── Patch Application ────────────────────────────────────────────────────────
|
|
108
|
+
|
|
109
|
+
function applyPatch(model: JsonModel, patch: PatchEntry): JsonModel {
|
|
110
|
+
const result = { ...model };
|
|
111
|
+
|
|
112
|
+
if (patch.name !== undefined) result.name = patch.name;
|
|
113
|
+
if (patch.reasoning !== undefined) result.reasoning = patch.reasoning;
|
|
114
|
+
if (patch.input !== undefined) result.input = patch.input;
|
|
115
|
+
if (patch.contextWindow !== undefined) result.contextWindow = patch.contextWindow;
|
|
116
|
+
if (patch.maxTokens !== undefined) result.maxTokens = patch.maxTokens;
|
|
117
|
+
if (patch.thinkingLevelMap !== undefined) result.thinkingLevelMap = { ...patch.thinkingLevelMap };
|
|
118
|
+
|
|
119
|
+
if (patch.cost) {
|
|
120
|
+
result.cost = {
|
|
121
|
+
input: patch.cost.input ?? result.cost.input,
|
|
122
|
+
output: patch.cost.output ?? result.cost.output,
|
|
123
|
+
cacheRead: patch.cost.cacheRead ?? result.cost.cacheRead,
|
|
124
|
+
cacheWrite: patch.cost.cacheWrite ?? result.cost.cacheWrite,
|
|
125
|
+
};
|
|
126
|
+
}
|
|
127
|
+
if (patch.compat) {
|
|
128
|
+
result.compat = { ...(result.compat || {}), ...patch.compat };
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
if (!result.reasoning && result.compat?.thinkingFormat) {
|
|
132
|
+
delete result.compat.thinkingFormat;
|
|
133
|
+
}
|
|
134
|
+
if (result.compat && Object.keys(result.compat).length === 0) {
|
|
135
|
+
delete result.compat;
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
return result;
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
/** Full pipeline: base models -> patch -> custom -> result */
|
|
142
|
+
function buildModels(base: JsonModel[], custom: JsonModel[], patch: PatchData): JsonModel[] {
|
|
143
|
+
const modelMap = new Map<string, JsonModel>();
|
|
144
|
+
|
|
145
|
+
// Seed with the base list plus grace-period deprecated models so patch.json
|
|
146
|
+
// entries apply to deprecated models exactly as while the model was live
|
|
147
|
+
// (withDeprecated keeps live data on id conflicts).
|
|
148
|
+
for (const model of withDeprecated(base)) {
|
|
149
|
+
modelMap.set(model.id, model);
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
for (const [id, patchEntry] of Object.entries(patch)) {
|
|
153
|
+
const existing = modelMap.get(id);
|
|
154
|
+
if (existing) {
|
|
155
|
+
modelMap.set(id, applyPatch(existing, patchEntry));
|
|
156
|
+
}
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
for (const model of custom) {
|
|
160
|
+
const existing = modelMap.get(model.id);
|
|
161
|
+
const patchEntry = patch[model.id];
|
|
162
|
+
if (existing && patchEntry) {
|
|
163
|
+
modelMap.set(model.id, applyPatch(model, patchEntry));
|
|
164
|
+
} else if (existing) {
|
|
165
|
+
modelMap.set(model.id, model);
|
|
166
|
+
} else if (patchEntry) {
|
|
167
|
+
modelMap.set(model.id, applyPatch(model, patchEntry));
|
|
168
|
+
} else {
|
|
169
|
+
modelMap.set(model.id, model);
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
return Array.from(modelMap.values());
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
// ─── Tarmis /v1/models transform ──────────────────────────────────────────────
|
|
177
|
+
//
|
|
178
|
+
// Tarmis returns rich per-model metadata, so we derive as much as possible
|
|
179
|
+
// directly from the API instead of curating it by hand:
|
|
180
|
+
// - name <- apiModel.name (trailing parentheticals stripped)
|
|
181
|
+
// - reasoning <- supported_features includes "reasoning"
|
|
182
|
+
// - input/vision <- input_modalities includes "image"
|
|
183
|
+
// - cost <- pricing[0].prompt / completion (per-token -> per-million)
|
|
184
|
+
// - cacheRead <- pricing[0].input_cache_read (per-token -> per-million), if present
|
|
185
|
+
// - contextWindow <- context_length
|
|
186
|
+
// - maxTokens <- max_output_length
|
|
187
|
+
//
|
|
188
|
+
// Reasoning models get the standard OpenAI-compatible defaults (thinkingFormat
|
|
189
|
+
// "openai", supportsReasoningEffort true, supportsDeveloperRole false). The API
|
|
190
|
+
// reports *that* a model supports reasoning but not *how* it expects it on the
|
|
191
|
+
// wire; patch.json overrides the format for any model that needs something else
|
|
192
|
+
// (e.g. a DeepSeek-native thinking field).
|
|
193
|
+
|
|
194
|
+
const PER_TOKEN_TO_PER_MILLION = 1_000_000;
|
|
195
|
+
// Clean float noise (e.g. 0.0000003 * 1e6 -> 0.30000000000000004) without losing
|
|
196
|
+
// micro-dollar precision on small cache rates.
|
|
197
|
+
const COST_ROUND_FACTOR = 1_000_000;
|
|
198
|
+
|
|
199
|
+
/** Convert a per-token price (string or number from the API) to per-million. */
|
|
200
|
+
function toPerMillion(perToken: string | number | undefined | null): number {
|
|
201
|
+
if (perToken === undefined || perToken === null) return 0;
|
|
202
|
+
const n = typeof perToken === "number" ? perToken : parseFloat(String(perToken));
|
|
203
|
+
if (!Number.isFinite(n) || n <= 0) return 0;
|
|
204
|
+
const perMillion = n * PER_TOKEN_TO_PER_MILLION;
|
|
205
|
+
return Math.round(perMillion * COST_ROUND_FACTOR) / COST_ROUND_FACTOR;
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
function cleanDisplayName(name: unknown, id: string): string {
|
|
209
|
+
if (typeof name === "string" && name.trim().length > 0) {
|
|
210
|
+
// Strip a trailing parenthetical (e.g. "GLM-5.2 (default: FP4 — fast + cheap)")
|
|
211
|
+
const stripped = name.replace(/\s*\([^)]*\)\s*$/, "").trim();
|
|
212
|
+
if (stripped.length > 0) return stripped;
|
|
213
|
+
}
|
|
214
|
+
const parts = id.split("/");
|
|
215
|
+
const rawName = parts.length > 1 ? parts.slice(1).join("/") : id;
|
|
216
|
+
return rawName.replace(/-/g, " ").replace(/\b\w/g, (c: string) => c.toUpperCase());
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
function transformApiModel(apiModel: any): JsonModel | null {
|
|
220
|
+
if (!apiModel || typeof apiModel.id !== "string" || apiModel.id.length === 0) return null;
|
|
221
|
+
|
|
222
|
+
const features: string[] = Array.isArray(apiModel.supported_features) ? apiModel.supported_features : [];
|
|
223
|
+
const reasoning = features.some((f) => typeof f === "string" && f.toLowerCase() === "reasoning");
|
|
224
|
+
|
|
225
|
+
const inputMods: string[] = Array.isArray(apiModel.input_modalities) ? apiModel.input_modalities : ["text"];
|
|
226
|
+
const hasImage = inputMods.some((m) => typeof m === "string" && m.toLowerCase().includes("image"));
|
|
227
|
+
|
|
228
|
+
// pricing is an array of per-token rate cards; take the first (standard tier).
|
|
229
|
+
const pricingEntry =
|
|
230
|
+
Array.isArray(apiModel.pricing) && apiModel.pricing.length > 0 && typeof apiModel.pricing[0] === "object"
|
|
231
|
+
? apiModel.pricing[0]
|
|
232
|
+
: {};
|
|
233
|
+
|
|
234
|
+
// Accept either OpenAI-style (prompt/completion) or Tarmis alt spellings.
|
|
235
|
+
const inputCost = toPerMillion(pricingEntry.prompt ?? pricingEntry.input);
|
|
236
|
+
const outputCost = toPerMillion(pricingEntry.completion ?? pricingEntry.output);
|
|
237
|
+
const cacheRead = toPerMillion(pricingEntry.input_cache_read ?? pricingEntry.cache_read);
|
|
238
|
+
|
|
239
|
+
const contextLength = Number(apiModel.context_length);
|
|
240
|
+
const maxOutputLength = Number(apiModel.max_output_length);
|
|
241
|
+
const contextWindow = Number.isFinite(contextLength) && contextLength > 0 ? contextLength : 131072;
|
|
242
|
+
const maxTokens =
|
|
243
|
+
Number.isFinite(maxOutputLength) && maxOutputLength > 0 ? maxOutputLength : contextWindow;
|
|
244
|
+
|
|
245
|
+
const model: JsonModel = {
|
|
246
|
+
id: apiModel.id,
|
|
247
|
+
name: cleanDisplayName(apiModel.name, apiModel.id),
|
|
248
|
+
reasoning,
|
|
249
|
+
input: hasImage ? ["text", "image"] : ["text"],
|
|
250
|
+
cost: {
|
|
251
|
+
input: inputCost,
|
|
252
|
+
output: outputCost,
|
|
253
|
+
cacheRead,
|
|
254
|
+
cacheWrite: 0,
|
|
255
|
+
},
|
|
256
|
+
contextWindow,
|
|
257
|
+
maxTokens,
|
|
258
|
+
};
|
|
259
|
+
|
|
260
|
+
if (reasoning) {
|
|
261
|
+
model.compat = {
|
|
262
|
+
thinkingFormat: "openai",
|
|
263
|
+
supportsReasoningEffort: true,
|
|
264
|
+
supportsDeveloperRole: false,
|
|
265
|
+
};
|
|
266
|
+
model.thinkingLevelMap = {
|
|
267
|
+
minimal: null,
|
|
268
|
+
low: null,
|
|
269
|
+
medium: null,
|
|
270
|
+
high: "high",
|
|
271
|
+
max: "max",
|
|
272
|
+
};
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
return model;
|
|
276
|
+
}
|
|
277
|
+
|
|
278
|
+
// ─── Stale-While-Revalidate Model Sync ────────────────────────────────────────
|
|
279
|
+
|
|
280
|
+
const PROVIDER_ID = "tarmis";
|
|
281
|
+
const BASE_URL = "https://api.tarmis.ai/v1";
|
|
282
|
+
const MODELS_URL = `${BASE_URL}/models`;
|
|
283
|
+
const CACHE_DIR = path.join(getAgentDir(), "cache");
|
|
284
|
+
const CACHE_PATH = path.join(CACHE_DIR, `${PROVIDER_ID}-models.json`);
|
|
285
|
+
const LIVE_FETCH_TIMEOUT_MS = 8000;
|
|
286
|
+
|
|
287
|
+
async function fetchLiveModels(apiKey: string | undefined, signal?: AbortSignal): Promise<JsonModel[] | null> {
|
|
288
|
+
try {
|
|
289
|
+
const response = await fetch(MODELS_URL, {
|
|
290
|
+
headers: apiKey ? { Authorization: `Bearer ${apiKey}` } : {},
|
|
291
|
+
signal: signal ? AbortSignal.any([AbortSignal.timeout(LIVE_FETCH_TIMEOUT_MS), signal]) : AbortSignal.timeout(LIVE_FETCH_TIMEOUT_MS),
|
|
292
|
+
});
|
|
293
|
+
if (!response.ok) return null;
|
|
294
|
+
const data = await response.json();
|
|
295
|
+
const apiModels = Array.isArray(data) ? data : (data.data || []);
|
|
296
|
+
if (!Array.isArray(apiModels) || apiModels.length === 0) return null;
|
|
297
|
+
return apiModels.map(transformApiModel).filter((m): m is JsonModel => m !== null);
|
|
298
|
+
} catch {
|
|
299
|
+
return null;
|
|
300
|
+
}
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
function loadCachedModels(): JsonModel[] | null {
|
|
304
|
+
try {
|
|
305
|
+
const data = JSON.parse(fs.readFileSync(CACHE_PATH, "utf8"));
|
|
306
|
+
return Array.isArray(data) ? data : null;
|
|
307
|
+
} catch {
|
|
308
|
+
return null;
|
|
309
|
+
}
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
function cacheModels(models: JsonModel[]): void {
|
|
313
|
+
try {
|
|
314
|
+
fs.mkdirSync(CACHE_DIR, { recursive: true });
|
|
315
|
+
fs.writeFileSync(CACHE_PATH, JSON.stringify(models, null, 2) + "\n");
|
|
316
|
+
} catch {
|
|
317
|
+
// Cache write failure is non-fatal
|
|
318
|
+
}
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
function mergeWithEmbedded(liveModels: JsonModel[], embeddedModels: JsonModel[]): JsonModel[] {
|
|
322
|
+
const embeddedMap = new Map(embeddedModels.map(m => [m.id, m]));
|
|
323
|
+
const seen = new Set<string>();
|
|
324
|
+
const result: JsonModel[] = [];
|
|
325
|
+
for (const liveModel of liveModels) {
|
|
326
|
+
const embedded = embeddedMap.get(liveModel.id);
|
|
327
|
+
seen.add(liveModel.id);
|
|
328
|
+
if (embedded) {
|
|
329
|
+
// Self-heal: live API pricing is authoritative field-by-field. Prefer the
|
|
330
|
+
// live cost when the API reports it (non-zero); fall back to embedded when
|
|
331
|
+
// the API is silent (0) so curated cacheRead/cacheWrite isn't clobbered and
|
|
332
|
+
// providers whose /models endpoint exposes no pricing keep their curated
|
|
333
|
+
// cost. Curation (reasoning/input/compat/name) still wins via ...embedded.
|
|
334
|
+
result.push({
|
|
335
|
+
...liveModel,
|
|
336
|
+
...embedded,
|
|
337
|
+
cost: {
|
|
338
|
+
input: liveModel.cost.input || embedded.cost.input,
|
|
339
|
+
output: liveModel.cost.output || embedded.cost.output,
|
|
340
|
+
cacheRead: liveModel.cost.cacheRead || embedded.cost.cacheRead,
|
|
341
|
+
cacheWrite: liveModel.cost.cacheWrite || embedded.cost.cacheWrite,
|
|
342
|
+
},
|
|
343
|
+
contextWindow: liveModel.contextWindow || embedded.contextWindow,
|
|
344
|
+
});
|
|
345
|
+
} else {
|
|
346
|
+
result.push(liveModel);
|
|
347
|
+
}
|
|
348
|
+
}
|
|
349
|
+
for (const em of embeddedModels) {
|
|
350
|
+
if (!seen.has(em.id)) {
|
|
351
|
+
result.push(em);
|
|
352
|
+
}
|
|
353
|
+
}
|
|
354
|
+
return result;
|
|
355
|
+
}
|
|
356
|
+
|
|
357
|
+
// Grace period for delisted models. When the provider API stops listing a
|
|
358
|
+
// model, update-models.js moves its last-known definition into
|
|
359
|
+
// deprecated-models.json (stamped with deprecatedAt) instead of dropping it.
|
|
360
|
+
// For 14 days the model keeps working here so in-flight sessions and saved
|
|
361
|
+
// model settings do not break; afterwards it is evicted permanently.
|
|
362
|
+
const DEPRECATED_MODEL_TTL_MS = 14 * 24 * 60 * 60 * 1000;
|
|
363
|
+
|
|
364
|
+
// Grace-period deprecated models with deprecation metadata stripped.
|
|
365
|
+
function activeDeprecatedModels(): JsonModel[] {
|
|
366
|
+
const now = Date.now();
|
|
367
|
+
const result: JsonModel[] = [];
|
|
368
|
+
for (const entry of Object.values(deprecatedData as Record<string, JsonModel & { deprecatedAt?: string }>)) {
|
|
369
|
+
if (!entry?.id) continue;
|
|
370
|
+
const removedAt = Date.parse(entry.deprecatedAt ?? "");
|
|
371
|
+
if (Number.isNaN(removedAt) || now - removedAt > DEPRECATED_MODEL_TTL_MS) continue;
|
|
372
|
+
const model = { ...entry } as JsonModel & { deprecatedAt?: string };
|
|
373
|
+
delete model.deprecatedAt;
|
|
374
|
+
result.push(model);
|
|
375
|
+
}
|
|
376
|
+
return result;
|
|
377
|
+
}
|
|
378
|
+
|
|
379
|
+
// Append grace-period deprecated models the list does not already have (live data wins).
|
|
380
|
+
function withDeprecated(models: JsonModel[]): JsonModel[] {
|
|
381
|
+
const seen = new Set(models.map((m) => m.id));
|
|
382
|
+
const extras = activeDeprecatedModels().filter((m) => !seen.has(m.id));
|
|
383
|
+
return extras.length > 0 ? [...models, ...extras] : models;
|
|
384
|
+
}
|
|
385
|
+
|
|
386
|
+
function loadStaleModels(embeddedModels: JsonModel[]): JsonModel[] {
|
|
387
|
+
const cached = loadCachedModels();
|
|
388
|
+
if (!cached || cached.length === 0) return embeddedModels;
|
|
389
|
+
|
|
390
|
+
const cachedMap = new Map(cached.map(m => [m.id, m]));
|
|
391
|
+
for (const em of embeddedModels) {
|
|
392
|
+
if (!cachedMap.has(em.id)) {
|
|
393
|
+
cached.push(em);
|
|
394
|
+
}
|
|
395
|
+
}
|
|
396
|
+
return cached;
|
|
397
|
+
}
|
|
398
|
+
|
|
399
|
+
async function revalidateModels(apiKey: string | undefined, embeddedModels: JsonModel[], signal?: AbortSignal): Promise<JsonModel[] | null> {
|
|
400
|
+
// /v1/models is public on Tarmis — revalidate even without an API key so the
|
|
401
|
+
// catalogue stays fresh before a key is configured. A key is only needed to
|
|
402
|
+
// actually call a model. The Authorization header is still sent when a key is
|
|
403
|
+
// available, in case the endpoint becomes auth-gated later.
|
|
404
|
+
const liveModels = await fetchLiveModels(apiKey, signal);
|
|
405
|
+
if (!liveModels || liveModels.length === 0) return null;
|
|
406
|
+
const merged = mergeWithEmbedded(liveModels, embeddedModels);
|
|
407
|
+
cacheModels(merged);
|
|
408
|
+
return merged;
|
|
409
|
+
}
|
|
410
|
+
|
|
411
|
+
// ─── API Key Resolution (via ModelRegistry) ────────────────────────────────────
|
|
412
|
+
|
|
413
|
+
let revalidateAbort: AbortController | null = null;
|
|
414
|
+
|
|
415
|
+
// Resolved so it can be threaded into the SWR fetch (the Bearer header is sent
|
|
416
|
+
// when available — harmless while /v1/models stays public). The provider's own
|
|
417
|
+
// apiKey is the "$TARMIS_API_KEY" placeholder, resolved by pi core at request
|
|
418
|
+
// time; the SWR fetch needs no key while /v1/models is public.
|
|
419
|
+
async function resolveApiKey(modelRegistry: ModelRegistry): Promise<string | undefined> {
|
|
420
|
+
return (await modelRegistry.getApiKeyForProvider("tarmis")) ?? undefined;
|
|
421
|
+
}
|
|
422
|
+
|
|
423
|
+
// ─── Extension Entry Point ────────────────────────────────────────────────────
|
|
424
|
+
|
|
425
|
+
export default function (pi: ExtensionAPI) {
|
|
426
|
+
const embeddedModels = modelsData as JsonModel[];
|
|
427
|
+
const customModels = customModelsData as JsonModel[];
|
|
428
|
+
const patches = patchData as PatchData;
|
|
429
|
+
|
|
430
|
+
const staleBase = loadStaleModels(embeddedModels);
|
|
431
|
+
const staleModels = buildModels(staleBase, customModels, patches);
|
|
432
|
+
|
|
433
|
+
pi.registerProvider("tarmis", {
|
|
434
|
+
baseUrl: BASE_URL,
|
|
435
|
+
apiKey: "$TARMIS_API_KEY",
|
|
436
|
+
api: "openai-completions",
|
|
437
|
+
models: staleModels,
|
|
438
|
+
});
|
|
439
|
+
|
|
440
|
+
pi.on("session_start", async (_event, ctx) => {
|
|
441
|
+
revalidateAbort?.abort();
|
|
442
|
+
revalidateAbort = new AbortController();
|
|
443
|
+
const signal = revalidateAbort.signal;
|
|
444
|
+
// /v1/models is public, so revalidation proceeds with or without a key; the
|
|
445
|
+
// key is still threaded through so the Bearer header is sent when available.
|
|
446
|
+
resolveApiKey(ctx.modelRegistry).then((apiKey) => {
|
|
447
|
+
revalidateModels(apiKey, embeddedModels, signal).then((freshBase) => {
|
|
448
|
+
if (freshBase && !signal.aborted) {
|
|
449
|
+
pi.registerProvider("tarmis", {
|
|
450
|
+
baseUrl: BASE_URL,
|
|
451
|
+
apiKey: "$TARMIS_API_KEY",
|
|
452
|
+
api: "openai-completions",
|
|
453
|
+
models: buildModels(freshBase, customModels, patches),
|
|
454
|
+
});
|
|
455
|
+
}
|
|
456
|
+
});
|
|
457
|
+
});
|
|
458
|
+
});
|
|
459
|
+
|
|
460
|
+
pi.on("session_shutdown", () => {
|
|
461
|
+
revalidateAbort?.abort();
|
|
462
|
+
});
|
|
463
|
+
}
|
package/models.json
ADDED
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
[
|
|
2
|
+
{
|
|
3
|
+
"id": "deepseek-ai/DeepSeek-V4-Flash",
|
|
4
|
+
"name": "DeepSeek V4 Flash",
|
|
5
|
+
"reasoning": true,
|
|
6
|
+
"input": [
|
|
7
|
+
"text"
|
|
8
|
+
],
|
|
9
|
+
"cost": {
|
|
10
|
+
"input": 0.09,
|
|
11
|
+
"output": 0.18,
|
|
12
|
+
"cacheRead": 0.02,
|
|
13
|
+
"cacheWrite": 0
|
|
14
|
+
},
|
|
15
|
+
"contextWindow": 1048576,
|
|
16
|
+
"maxTokens": 384000,
|
|
17
|
+
"compat": {
|
|
18
|
+
"thinkingFormat": "openai",
|
|
19
|
+
"supportsReasoningEffort": true,
|
|
20
|
+
"supportsDeveloperRole": false
|
|
21
|
+
},
|
|
22
|
+
"thinkingLevelMap": {
|
|
23
|
+
"minimal": null,
|
|
24
|
+
"low": null,
|
|
25
|
+
"medium": null,
|
|
26
|
+
"high": "high",
|
|
27
|
+
"max": "max"
|
|
28
|
+
}
|
|
29
|
+
},
|
|
30
|
+
{
|
|
31
|
+
"id": "deepseek-ai/DeepSeek-V4-Pro",
|
|
32
|
+
"name": "DeepSeek V4 Pro",
|
|
33
|
+
"reasoning": true,
|
|
34
|
+
"input": [
|
|
35
|
+
"text"
|
|
36
|
+
],
|
|
37
|
+
"cost": {
|
|
38
|
+
"input": 0.435,
|
|
39
|
+
"output": 0.87,
|
|
40
|
+
"cacheRead": 0.09,
|
|
41
|
+
"cacheWrite": 0
|
|
42
|
+
},
|
|
43
|
+
"contextWindow": 1048576,
|
|
44
|
+
"maxTokens": 384000,
|
|
45
|
+
"compat": {
|
|
46
|
+
"thinkingFormat": "openai",
|
|
47
|
+
"supportsReasoningEffort": true,
|
|
48
|
+
"supportsDeveloperRole": false
|
|
49
|
+
},
|
|
50
|
+
"thinkingLevelMap": {
|
|
51
|
+
"minimal": null,
|
|
52
|
+
"low": null,
|
|
53
|
+
"medium": null,
|
|
54
|
+
"high": "high",
|
|
55
|
+
"max": "max"
|
|
56
|
+
}
|
|
57
|
+
},
|
|
58
|
+
{
|
|
59
|
+
"id": "minimax/minimax-m3",
|
|
60
|
+
"name": "MiniMax M3",
|
|
61
|
+
"reasoning": true,
|
|
62
|
+
"input": [
|
|
63
|
+
"text"
|
|
64
|
+
],
|
|
65
|
+
"cost": {
|
|
66
|
+
"input": 0.3,
|
|
67
|
+
"output": 1.2,
|
|
68
|
+
"cacheRead": 0.06,
|
|
69
|
+
"cacheWrite": 0
|
|
70
|
+
},
|
|
71
|
+
"contextWindow": 524288,
|
|
72
|
+
"maxTokens": 131072,
|
|
73
|
+
"compat": {
|
|
74
|
+
"thinkingFormat": "openai",
|
|
75
|
+
"supportsReasoningEffort": true,
|
|
76
|
+
"supportsDeveloperRole": false
|
|
77
|
+
},
|
|
78
|
+
"thinkingLevelMap": {
|
|
79
|
+
"minimal": null,
|
|
80
|
+
"low": null,
|
|
81
|
+
"medium": null,
|
|
82
|
+
"high": "high",
|
|
83
|
+
"max": "max"
|
|
84
|
+
}
|
|
85
|
+
},
|
|
86
|
+
{
|
|
87
|
+
"id": "moonshotai/Kimi-K3",
|
|
88
|
+
"name": "Kimi K3",
|
|
89
|
+
"reasoning": true,
|
|
90
|
+
"input": [
|
|
91
|
+
"text",
|
|
92
|
+
"image"
|
|
93
|
+
],
|
|
94
|
+
"cost": {
|
|
95
|
+
"input": 2.5,
|
|
96
|
+
"output": 10,
|
|
97
|
+
"cacheRead": 0.25,
|
|
98
|
+
"cacheWrite": 0
|
|
99
|
+
},
|
|
100
|
+
"contextWindow": 1048576,
|
|
101
|
+
"maxTokens": 131072,
|
|
102
|
+
"compat": {
|
|
103
|
+
"thinkingFormat": "openai",
|
|
104
|
+
"supportsReasoningEffort": true,
|
|
105
|
+
"supportsDeveloperRole": false
|
|
106
|
+
},
|
|
107
|
+
"thinkingLevelMap": {
|
|
108
|
+
"minimal": null,
|
|
109
|
+
"low": null,
|
|
110
|
+
"medium": null,
|
|
111
|
+
"high": "high",
|
|
112
|
+
"max": "max"
|
|
113
|
+
}
|
|
114
|
+
},
|
|
115
|
+
{
|
|
116
|
+
"id": "z-ai/glm-5.2",
|
|
117
|
+
"name": "GLM-5.2",
|
|
118
|
+
"reasoning": true,
|
|
119
|
+
"input": [
|
|
120
|
+
"text"
|
|
121
|
+
],
|
|
122
|
+
"cost": {
|
|
123
|
+
"input": 0.6,
|
|
124
|
+
"output": 2.2,
|
|
125
|
+
"cacheRead": 0.14,
|
|
126
|
+
"cacheWrite": 0
|
|
127
|
+
},
|
|
128
|
+
"contextWindow": 1048576,
|
|
129
|
+
"maxTokens": 128000,
|
|
130
|
+
"compat": {
|
|
131
|
+
"thinkingFormat": "openai",
|
|
132
|
+
"supportsReasoningEffort": true,
|
|
133
|
+
"supportsDeveloperRole": false
|
|
134
|
+
},
|
|
135
|
+
"thinkingLevelMap": {
|
|
136
|
+
"minimal": null,
|
|
137
|
+
"low": null,
|
|
138
|
+
"medium": null,
|
|
139
|
+
"high": "high",
|
|
140
|
+
"max": "max"
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
]
|
package/package.json
ADDED
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "pi-tarmis-provider",
|
|
3
|
+
"version": "1.0.7",
|
|
4
|
+
"description": "Tarmis provider extension for pi - Open models on Tarmis's own GPUs via an OpenAI-compatible API",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"main": "index.ts",
|
|
7
|
+
"scripts": {
|
|
8
|
+
"clean": "echo 'nothing to clean'",
|
|
9
|
+
"build": "echo 'nothing to build'",
|
|
10
|
+
"check": "echo 'nothing to check'",
|
|
11
|
+
"update-models": "node scripts/update-models.js"
|
|
12
|
+
},
|
|
13
|
+
"keywords": [
|
|
14
|
+
"pi",
|
|
15
|
+
"extension",
|
|
16
|
+
"provider",
|
|
17
|
+
"tarmis",
|
|
18
|
+
"ai",
|
|
19
|
+
"llm",
|
|
20
|
+
"openai-compatible",
|
|
21
|
+
"deepseek",
|
|
22
|
+
"glm",
|
|
23
|
+
"kimi",
|
|
24
|
+
"minimax"
|
|
25
|
+
],
|
|
26
|
+
"author": "monotykamary",
|
|
27
|
+
"license": "MIT",
|
|
28
|
+
"homepage": "https://github.com/monotykamary/pi-tarmis-provider#readme",
|
|
29
|
+
"repository": {
|
|
30
|
+
"type": "git",
|
|
31
|
+
"url": "git+https://github.com/monotykamary/pi-tarmis-provider.git"
|
|
32
|
+
},
|
|
33
|
+
"bugs": {
|
|
34
|
+
"url": "https://github.com/monotykamary/pi-tarmis-provider/issues"
|
|
35
|
+
},
|
|
36
|
+
"files": [
|
|
37
|
+
"index.ts",
|
|
38
|
+
"models.json",
|
|
39
|
+
"deprecated-models.json",
|
|
40
|
+
"custom-models.json",
|
|
41
|
+
"patch.json",
|
|
42
|
+
"scripts/update-models.js",
|
|
43
|
+
"README.md",
|
|
44
|
+
"LICENSE",
|
|
45
|
+
"AGENTS.md"
|
|
46
|
+
],
|
|
47
|
+
"publishConfig": {
|
|
48
|
+
"access": "public"
|
|
49
|
+
},
|
|
50
|
+
"pi": {
|
|
51
|
+
"extensions": [
|
|
52
|
+
"./index.ts"
|
|
53
|
+
]
|
|
54
|
+
},
|
|
55
|
+
"devDependencies": {
|
|
56
|
+
"@earendil-works/pi-coding-agent": "^0.84.2",
|
|
57
|
+
"@types/node": "^25.9.1",
|
|
58
|
+
"typescript": "^6.0.3"
|
|
59
|
+
},
|
|
60
|
+
"packageManager": "pnpm@11.20.0"
|
|
61
|
+
}
|
package/patch.json
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
{
|
|
2
|
+
"z-ai/glm-5.2": {
|
|
3
|
+
"thinkingLevelMap": {
|
|
4
|
+
"off": "none",
|
|
5
|
+
"minimal": "high",
|
|
6
|
+
"low": "high",
|
|
7
|
+
"medium": "high",
|
|
8
|
+
"high": "high",
|
|
9
|
+
"max": "max"
|
|
10
|
+
}
|
|
11
|
+
},
|
|
12
|
+
"minimax/minimax-m3": {
|
|
13
|
+
"compat": {
|
|
14
|
+
"thinkingFormat": "zai"
|
|
15
|
+
}
|
|
16
|
+
},
|
|
17
|
+
"deepseek-ai/DeepSeek-V4-Flash": {
|
|
18
|
+
"compat": {
|
|
19
|
+
"thinkingFormat": "deepseek",
|
|
20
|
+
"requiresReasoningContentOnAssistantMessages": true
|
|
21
|
+
}
|
|
22
|
+
},
|
|
23
|
+
"moonshotai/Kimi-K3": {
|
|
24
|
+
"input": ["text", "image"],
|
|
25
|
+
"thinkingLevelMap": {
|
|
26
|
+
"off": "none",
|
|
27
|
+
"minimal": null,
|
|
28
|
+
"low": null,
|
|
29
|
+
"medium": null,
|
|
30
|
+
"high": "high",
|
|
31
|
+
"max": "max"
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
}
|