pi-zro-provider 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/FUNDING.yml +4 -0
- package/AGENTS.md +58 -0
- package/LICENSE +21 -0
- package/README.md +232 -0
- package/custom-models.json +1 -0
- package/deprecated-models.json +1 -0
- package/index.ts +1110 -0
- package/models.json +89 -0
- package/package.json +40 -0
- package/patch.json +1 -0
- package/pnpm-workspace.yaml +15 -0
- package/scripts/update-models.js +387 -0
- package/status.ts +298 -0
- package/tests/status.smoke.ts +164 -0
- package/tsconfig.json +15 -0
package/index.ts
ADDED
|
@@ -0,0 +1,1110 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Zro Provider Extension
|
|
3
|
+
*
|
|
4
|
+
* Registers Zro (zro.moonmath.ai) as a custom provider using the
|
|
5
|
+
* openai-completions API. Base URL: https://zro.moonmath.ai/v1
|
|
6
|
+
*
|
|
7
|
+
* Model metadata comes from Zro's CLI model catalog, GET /api/cli/models —
|
|
8
|
+
* the same endpoint `zro models` uses. It provides canonical ids, display
|
|
9
|
+
* names, context/output limits, and per-model reasoning effort levels (each
|
|
10
|
+
* level pairs an id with a piLevel). patch.json remains available for
|
|
11
|
+
* verified endpoint regressions, but currently contains no overrides.
|
|
12
|
+
*
|
|
13
|
+
* Model resolution strategy: Stale-While-Revalidate
|
|
14
|
+
* 1. Serve stale immediately: disk cache → embedded models.json (zero-latency)
|
|
15
|
+
* 2. Revalidate in background: live API /api/cli/models → merge with embedded → cache → hot-swap
|
|
16
|
+
* 3. patch.json + custom-models.json applied on top of whichever source won
|
|
17
|
+
*
|
|
18
|
+
* Merge order: [live|cache|embedded] → apply patch.json → merge custom-models.json
|
|
19
|
+
*
|
|
20
|
+
* Footer Status Widget:
|
|
21
|
+
* A below-editor line shows Zro session + account state:
|
|
22
|
+
*
|
|
23
|
+
* ⚡ $0.42 · 7 req Pro ◆ $12.34 avail · $2.50 pack · 1.2k req/30d
|
|
24
|
+
* └─ session activity ──┘ └──────────────── account / quota ─────────────┘
|
|
25
|
+
*
|
|
26
|
+
* The left side reports what this session has spent/sent: request count
|
|
27
|
+
* always, plus token totals and, when the response exposes one, a cost
|
|
28
|
+
* extension on each chat completion — pi requests
|
|
29
|
+
* stream_options.include_usage, so the final SSE chunk carries usage and we
|
|
30
|
+
* read it off a teed response stream, no polling). The right side reports
|
|
31
|
+
* the plan name, total available spend, usage-pack balance, and 30-day
|
|
32
|
+
* request/token activity from Zro's /api/cli/status endpoint (the same
|
|
33
|
+
* view `zro status` prints), plus a request-rate atom captured from
|
|
34
|
+
* response headers when present. The right side compresses across
|
|
35
|
+
* progressive tiers as the terminal narrows. The balance flips to a ⚠
|
|
36
|
+
* warning at/below lowBalanceUsd.
|
|
37
|
+
*
|
|
38
|
+
* Lifecycle (mirrors pi-neuralwatt-provider): nothing renders before this
|
|
39
|
+
* session's first Zro turn completes, so fresh sessions and other
|
|
40
|
+
* providers' sessions see no half-empty line. Account status is prefetched
|
|
41
|
+
* on session start or model select when a Zro model is active, so the
|
|
42
|
+
* first turn ends with data already cached. It is polled again on pi's
|
|
43
|
+
* agent_settled event (fires only once no automatic retry, compaction, or
|
|
44
|
+
* queued continuation can follow) — and nowhere else, so sessions without
|
|
45
|
+
* Zro turns make zero status-related API calls. Between polls the balance
|
|
46
|
+
* moves optimistically: each turn's captured spend is deducted from the
|
|
47
|
+
* last available-spend value at turn_end so the account line tracks spend
|
|
48
|
+
* live; the agent_settled poll reconciles any drift.
|
|
49
|
+
*
|
|
50
|
+
* Display Configuration:
|
|
51
|
+
* Create ~/.pi/agent/extensions/zro.json:
|
|
52
|
+
* {
|
|
53
|
+
* "session": "widget", // "widget" | "statusbar" | "off"
|
|
54
|
+
* "account": "widget", // "widget" | "statusbar" | "off"
|
|
55
|
+
* "hideOnOtherProvider": true, // hide when a non-Zro model is active
|
|
56
|
+
* "lowBalanceUsd": 10 // warn threshold, null/false disables
|
|
57
|
+
* }
|
|
58
|
+
*
|
|
59
|
+
* - "widget" (default): rendered in the below-editor status line
|
|
60
|
+
* - "statusbar": rendered in the built-in pi status bar
|
|
61
|
+
* - "off": hidden entirely (account=off also skips status fetches)
|
|
62
|
+
*
|
|
63
|
+
* Manage interactively with /zro-status, or non-interactively:
|
|
64
|
+
* /zro-status session widget|statusbar|off
|
|
65
|
+
* /zro-status account widget|statusbar|off
|
|
66
|
+
* /zro-status hide true|false
|
|
67
|
+
* /zro-status lowBalance <usd>|off
|
|
68
|
+
* /zro-status refresh (re-fetch account status now)
|
|
69
|
+
* /zro-status reset
|
|
70
|
+
*
|
|
71
|
+
* Usage:
|
|
72
|
+
* # Option 1: Use your existing zro CLI login (no extra key needed)
|
|
73
|
+
* # Run `zro login` once; this extension reads ~/.config/zro/credentials.json
|
|
74
|
+
* # (XDG_CONFIG_HOME aware) as a fallback.
|
|
75
|
+
*
|
|
76
|
+
* # Option 2: Store in auth.json (recommended)
|
|
77
|
+
* # Add to ~/.pi/agent/auth.json:
|
|
78
|
+
* # "zro": { "type": "api_key", "key": "your-api-key" }
|
|
79
|
+
*
|
|
80
|
+
* # Option 3: Set as environment variable
|
|
81
|
+
* export ZRO_API_KEY=your-api-key
|
|
82
|
+
*
|
|
83
|
+
* # Run pi with the extension
|
|
84
|
+
* pi -e /path/to/pi-zro-provider
|
|
85
|
+
*
|
|
86
|
+
* Then use /model to select from available models.
|
|
87
|
+
*
|
|
88
|
+
* @see https://zro.moonmath.ai
|
|
89
|
+
*/
|
|
90
|
+
|
|
91
|
+
import { clampThinkingLevel, streamOpenAICompletions } from "@earendil-works/pi-ai/compat";
|
|
92
|
+
import type { AssistantMessageEventStream, SimpleStreamOptions } from "@earendil-works/pi-ai/compat";
|
|
93
|
+
import { getAgentDir, type ExtensionAPI, type ExtensionContext, type ModelRegistry } from "@earendil-works/pi-coding-agent";
|
|
94
|
+
import modelsData from "./models.json" with { type: "json" };
|
|
95
|
+
import customModelsData from "./custom-models.json" with { type: "json" };
|
|
96
|
+
import patchData from "./patch.json" with { type: "json" };
|
|
97
|
+
import deprecatedData from "./deprecated-models.json" with { type: "json" };
|
|
98
|
+
import {
|
|
99
|
+
applyOptimisticSpend,
|
|
100
|
+
buildAccountTiers,
|
|
101
|
+
buildSessionLine,
|
|
102
|
+
coerceStatusConfig,
|
|
103
|
+
DEFAULT_STATUS_CONFIG,
|
|
104
|
+
EMPTY_ACCOUNT,
|
|
105
|
+
EMPTY_SESSION_STATS,
|
|
106
|
+
StatusLineWidget,
|
|
107
|
+
accountHasData,
|
|
108
|
+
type AccountState,
|
|
109
|
+
type SessionStats,
|
|
110
|
+
type StatusConfig,
|
|
111
|
+
} from "./status";
|
|
112
|
+
import fs from "fs";
|
|
113
|
+
import { homedir } from "os";
|
|
114
|
+
import path from "path";
|
|
115
|
+
|
|
116
|
+
// ─── Types ────────────────────────────────────────────────────────────────────
|
|
117
|
+
|
|
118
|
+
interface JsonModel {
|
|
119
|
+
id: string;
|
|
120
|
+
name: string;
|
|
121
|
+
reasoning: boolean;
|
|
122
|
+
input: ("text" | "image")[];
|
|
123
|
+
cost: {
|
|
124
|
+
input: number;
|
|
125
|
+
output: number;
|
|
126
|
+
cacheRead: number;
|
|
127
|
+
cacheWrite: number;
|
|
128
|
+
};
|
|
129
|
+
contextWindow: number;
|
|
130
|
+
maxTokens: number;
|
|
131
|
+
thinkingLevelMap?: Record<string, string | null>;
|
|
132
|
+
compat?: {
|
|
133
|
+
supportsDeveloperRole?: boolean;
|
|
134
|
+
supportsStore?: boolean;
|
|
135
|
+
maxTokensField?: "max_completion_tokens" | "max_tokens";
|
|
136
|
+
thinkingFormat?: "openai" | "zai" | "qwen" | "qwen-chat-template" | "deepseek" | "openrouter" | "baseten" | "string-thinking" | "together" | "ant-ling" | "chat-template";
|
|
137
|
+
supportsReasoningEffort?: boolean;
|
|
138
|
+
requiresReasoningContentOnAssistantMessages?: boolean;
|
|
139
|
+
};
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
interface PatchEntry {
|
|
143
|
+
name?: string;
|
|
144
|
+
reasoning?: boolean;
|
|
145
|
+
input?: ("text" | "image")[];
|
|
146
|
+
cost?: {
|
|
147
|
+
input?: number;
|
|
148
|
+
output?: number;
|
|
149
|
+
cacheRead?: number;
|
|
150
|
+
cacheWrite?: number;
|
|
151
|
+
};
|
|
152
|
+
contextWindow?: number;
|
|
153
|
+
maxTokens?: number;
|
|
154
|
+
thinkingLevelMap?: Record<string, string | null>;
|
|
155
|
+
compat?: Record<string, unknown>;
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
type PatchData = Record<string, PatchEntry>;
|
|
159
|
+
|
|
160
|
+
// ─── Patch Application ────────────────────────────────────────────────────────
|
|
161
|
+
|
|
162
|
+
function applyPatch(model: JsonModel, patch: PatchEntry): JsonModel {
|
|
163
|
+
const result = { ...model };
|
|
164
|
+
|
|
165
|
+
if (patch.name !== undefined) result.name = patch.name;
|
|
166
|
+
if (patch.reasoning !== undefined) result.reasoning = patch.reasoning;
|
|
167
|
+
if (patch.input !== undefined) result.input = patch.input;
|
|
168
|
+
if (patch.contextWindow !== undefined) result.contextWindow = patch.contextWindow;
|
|
169
|
+
if (patch.maxTokens !== undefined) result.maxTokens = patch.maxTokens;
|
|
170
|
+
if (patch.thinkingLevelMap !== undefined) result.thinkingLevelMap = { ...patch.thinkingLevelMap };
|
|
171
|
+
|
|
172
|
+
if (patch.cost) {
|
|
173
|
+
result.cost = {
|
|
174
|
+
input: patch.cost.input ?? result.cost.input,
|
|
175
|
+
output: patch.cost.output ?? result.cost.output,
|
|
176
|
+
cacheRead: patch.cost.cacheRead ?? result.cost.cacheRead,
|
|
177
|
+
cacheWrite: patch.cost.cacheWrite ?? result.cost.cacheWrite,
|
|
178
|
+
};
|
|
179
|
+
}
|
|
180
|
+
if (patch.compat) {
|
|
181
|
+
result.compat = { ...(result.compat || {}), ...patch.compat };
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
if (!result.reasoning && result.compat?.thinkingFormat) {
|
|
185
|
+
delete result.compat.thinkingFormat;
|
|
186
|
+
}
|
|
187
|
+
if (!result.reasoning && result.thinkingLevelMap) {
|
|
188
|
+
delete result.thinkingLevelMap;
|
|
189
|
+
}
|
|
190
|
+
if (result.compat && Object.keys(result.compat).length === 0) {
|
|
191
|
+
delete result.compat;
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
return result;
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
/** Full pipeline: base models → patch → custom → result */
|
|
198
|
+
function buildModels(base: JsonModel[], custom: JsonModel[], patch: PatchData): JsonModel[] {
|
|
199
|
+
const modelMap = new Map<string, JsonModel>();
|
|
200
|
+
|
|
201
|
+
// Seed with the base list plus grace-period deprecated models so patch.json
|
|
202
|
+
// entries apply to deprecated models exactly as while the model was live
|
|
203
|
+
// (withDeprecated keeps live data on id conflicts).
|
|
204
|
+
for (const model of withDeprecated(base)) {
|
|
205
|
+
modelMap.set(model.id, model);
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
for (const [id, patchEntry] of Object.entries(patch)) {
|
|
209
|
+
const existing = modelMap.get(id);
|
|
210
|
+
if (existing) {
|
|
211
|
+
modelMap.set(id, applyPatch(existing, patchEntry));
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
for (const model of custom) {
|
|
216
|
+
const existing = modelMap.get(model.id);
|
|
217
|
+
const patchEntry = patch[model.id];
|
|
218
|
+
if (existing && patchEntry) {
|
|
219
|
+
modelMap.set(model.id, applyPatch(model, patchEntry));
|
|
220
|
+
} else if (existing) {
|
|
221
|
+
modelMap.set(model.id, model);
|
|
222
|
+
} else if (patchEntry) {
|
|
223
|
+
modelMap.set(model.id, applyPatch(model, patchEntry));
|
|
224
|
+
} else {
|
|
225
|
+
modelMap.set(model.id, model);
|
|
226
|
+
}
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
return Array.from(modelMap.values());
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
// ─── Stale-While-Revalidate Model Sync ────────────────────────────────────────
|
|
233
|
+
|
|
234
|
+
const PROVIDER_ID = "zro";
|
|
235
|
+
// Endpoint root is overridable like zro's own CLI (ZRO_ENDPOINT_ROOT).
|
|
236
|
+
const ENDPOINT_ROOT = (process.env.ZRO_ENDPOINT_ROOT || "https://zro.moonmath.ai").replace(/\/+$/, "");
|
|
237
|
+
const BASE_URL = `${ENDPOINT_ROOT}/v1`;
|
|
238
|
+
const MODELS_URL = `${ENDPOINT_ROOT}/api/cli/models`;
|
|
239
|
+
const STATUS_URL = `${ENDPOINT_ROOT}/api/cli/status`;
|
|
240
|
+
const CACHE_DIR = path.join(getAgentDir(), "cache");
|
|
241
|
+
const CACHE_PATH = path.join(CACHE_DIR, `${PROVIDER_ID}-models.json`);
|
|
242
|
+
const LIVE_FETCH_TIMEOUT_MS = 8000;
|
|
243
|
+
|
|
244
|
+
// pi thinking levels → provider level ids. Matches the transform zro's own
|
|
245
|
+
// pi adapter applies: each catalog reasoning level publishes a piLevel
|
|
246
|
+
// (off|minimal|low|medium|high|xhigh) and the id the inference proxy
|
|
247
|
+
// expects, e.g. glm-5.2 → { off: "none", high: "high", xhigh: "max" }.
|
|
248
|
+
const PI_THINKING_LEVELS = ["off", "minimal", "low", "medium", "high", "xhigh"] as const;
|
|
249
|
+
|
|
250
|
+
function buildZroThinkingLevelMap(levels: any[]): Record<string, string | null> {
|
|
251
|
+
const result: Record<string, string | null> = {
|
|
252
|
+
off: null,
|
|
253
|
+
minimal: null,
|
|
254
|
+
low: null,
|
|
255
|
+
medium: null,
|
|
256
|
+
high: null,
|
|
257
|
+
xhigh: null,
|
|
258
|
+
};
|
|
259
|
+
for (const level of levels) {
|
|
260
|
+
if (
|
|
261
|
+
level &&
|
|
262
|
+
typeof level === "object" &&
|
|
263
|
+
typeof level.id === "string" &&
|
|
264
|
+
level.id.trim() &&
|
|
265
|
+
typeof level.piLevel === "string" &&
|
|
266
|
+
Object.prototype.hasOwnProperty.call(result, level.piLevel)
|
|
267
|
+
) {
|
|
268
|
+
result[level.piLevel] = level.id;
|
|
269
|
+
}
|
|
270
|
+
}
|
|
271
|
+
return result;
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
/** Transform a model from Zro's CLI model catalog (/api/cli/models). */
|
|
275
|
+
function transformApiModel(apiModel: any): JsonModel | null {
|
|
276
|
+
if (typeof apiModel.id !== "string" || apiModel.id.length === 0) return null;
|
|
277
|
+
|
|
278
|
+
const levels = Array.isArray(apiModel?.reasoning?.levels) ? apiModel.reasoning.levels : [];
|
|
279
|
+
|
|
280
|
+
return {
|
|
281
|
+
id: apiModel.id,
|
|
282
|
+
name: apiModel.displayName || apiModel.id,
|
|
283
|
+
reasoning: true,
|
|
284
|
+
thinkingLevelMap: buildZroThinkingLevelMap(levels),
|
|
285
|
+
input: ["text"],
|
|
286
|
+
cost: {
|
|
287
|
+
input: 0,
|
|
288
|
+
output: 0,
|
|
289
|
+
cacheRead: 0,
|
|
290
|
+
cacheWrite: 0,
|
|
291
|
+
},
|
|
292
|
+
contextWindow: apiModel.contextWindow || 0,
|
|
293
|
+
maxTokens: apiModel.maxOutputTokens || apiModel.contextWindow || 0,
|
|
294
|
+
compat: {
|
|
295
|
+
supportsDeveloperRole: false,
|
|
296
|
+
supportsReasoningEffort: true,
|
|
297
|
+
maxTokensField: "max_tokens",
|
|
298
|
+
},
|
|
299
|
+
};
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
async function fetchLiveModels(apiKey: string, signal?: AbortSignal): Promise<JsonModel[] | null> {
|
|
303
|
+
try {
|
|
304
|
+
const response = await fetch(MODELS_URL, {
|
|
305
|
+
headers: { Authorization: `Bearer ${apiKey}` },
|
|
306
|
+
signal: signal ? AbortSignal.any([AbortSignal.timeout(LIVE_FETCH_TIMEOUT_MS), signal]) : AbortSignal.timeout(LIVE_FETCH_TIMEOUT_MS),
|
|
307
|
+
});
|
|
308
|
+
if (!response.ok) return null;
|
|
309
|
+
const data = await response.json();
|
|
310
|
+
const apiModels = Array.isArray(data) ? data : (data.models || data.data || []);
|
|
311
|
+
if (!Array.isArray(apiModels) || apiModels.length === 0) return null;
|
|
312
|
+
return apiModels.map(transformApiModel).filter((m): m is JsonModel => m !== null);
|
|
313
|
+
} catch {
|
|
314
|
+
return null;
|
|
315
|
+
}
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
function loadCachedModels(): JsonModel[] | null {
|
|
319
|
+
try {
|
|
320
|
+
const data = JSON.parse(fs.readFileSync(CACHE_PATH, "utf8"));
|
|
321
|
+
return Array.isArray(data) ? data : null;
|
|
322
|
+
} catch {
|
|
323
|
+
return null;
|
|
324
|
+
}
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
function cacheModels(models: JsonModel[]): void {
|
|
328
|
+
try {
|
|
329
|
+
fs.mkdirSync(CACHE_DIR, { recursive: true });
|
|
330
|
+
fs.writeFileSync(CACHE_PATH, JSON.stringify(models, null, 2) + "\n");
|
|
331
|
+
} catch {
|
|
332
|
+
// Cache write failure is non-fatal
|
|
333
|
+
}
|
|
334
|
+
}
|
|
335
|
+
|
|
336
|
+
function mergeWithEmbedded(liveModels: JsonModel[], embeddedModels: JsonModel[]): JsonModel[] {
|
|
337
|
+
const embeddedMap = new Map(embeddedModels.map(m => [m.id, m]));
|
|
338
|
+
const seen = new Set<string>();
|
|
339
|
+
const result: JsonModel[] = [];
|
|
340
|
+
for (const liveModel of liveModels) {
|
|
341
|
+
const embedded = embeddedMap.get(liveModel.id);
|
|
342
|
+
seen.add(liveModel.id);
|
|
343
|
+
if (embedded) {
|
|
344
|
+
// The live catalog is authoritative for context/output limits;
|
|
345
|
+
// curation (name/thinkingLevelMap/compat) still wins via ...embedded.
|
|
346
|
+
result.push({
|
|
347
|
+
...liveModel,
|
|
348
|
+
...embedded,
|
|
349
|
+
cost: liveModel.cost,
|
|
350
|
+
contextWindow: liveModel.contextWindow || embedded.contextWindow,
|
|
351
|
+
maxTokens: liveModel.maxTokens || embedded.maxTokens,
|
|
352
|
+
});
|
|
353
|
+
} else {
|
|
354
|
+
result.push(liveModel);
|
|
355
|
+
}
|
|
356
|
+
}
|
|
357
|
+
// Append any embedded models that the live API didn't return
|
|
358
|
+
for (const em of embeddedModels) {
|
|
359
|
+
if (!seen.has(em.id)) {
|
|
360
|
+
result.push(em);
|
|
361
|
+
}
|
|
362
|
+
}
|
|
363
|
+
return result;
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
// Grace period for delisted models. When the provider API stops listing a
|
|
367
|
+
// model, update-models.js moves its last-known definition into
|
|
368
|
+
// deprecated-models.json (stamped with deprecatedAt) instead of dropping it.
|
|
369
|
+
// For 14 days the model keeps working here so in-flight sessions and saved
|
|
370
|
+
// model settings do not break; afterwards it is evicted permanently.
|
|
371
|
+
const DEPRECATED_MODEL_TTL_MS = 14 * 24 * 60 * 60 * 1000;
|
|
372
|
+
|
|
373
|
+
// Grace-period deprecated models with deprecation metadata stripped.
|
|
374
|
+
function activeDeprecatedModels(): JsonModel[] {
|
|
375
|
+
const now = Date.now();
|
|
376
|
+
const result: JsonModel[] = [];
|
|
377
|
+
for (const entry of Object.values(deprecatedData as Record<string, JsonModel & { deprecatedAt?: string }>)) {
|
|
378
|
+
if (!entry?.id) continue;
|
|
379
|
+
const removedAt = Date.parse(entry.deprecatedAt ?? "");
|
|
380
|
+
if (Number.isNaN(removedAt) || now - removedAt > DEPRECATED_MODEL_TTL_MS) continue;
|
|
381
|
+
const model = { ...entry } as JsonModel & { deprecatedAt?: string };
|
|
382
|
+
delete model.deprecatedAt;
|
|
383
|
+
result.push(model);
|
|
384
|
+
}
|
|
385
|
+
return result;
|
|
386
|
+
}
|
|
387
|
+
|
|
388
|
+
// Append grace-period deprecated models the list does not already have (live data wins).
|
|
389
|
+
function withDeprecated(models: JsonModel[]): JsonModel[] {
|
|
390
|
+
const seen = new Set(models.map((m) => m.id));
|
|
391
|
+
const extras = activeDeprecatedModels().filter((m) => !seen.has(m.id));
|
|
392
|
+
return extras.length > 0 ? [...models, ...extras] : models;
|
|
393
|
+
}
|
|
394
|
+
|
|
395
|
+
function loadStaleModels(embeddedModels: JsonModel[]): JsonModel[] {
|
|
396
|
+
const cached = loadCachedModels();
|
|
397
|
+
if (!cached || cached.length === 0) return embeddedModels;
|
|
398
|
+
|
|
399
|
+
// Merge embedded models that are missing from cache (newly added models)
|
|
400
|
+
const cachedMap = new Map(cached.map(m => [m.id, m]));
|
|
401
|
+
for (const em of embeddedModels) {
|
|
402
|
+
if (!cachedMap.has(em.id)) {
|
|
403
|
+
cached.push(em);
|
|
404
|
+
}
|
|
405
|
+
}
|
|
406
|
+
return cached;
|
|
407
|
+
}
|
|
408
|
+
|
|
409
|
+
async function revalidateModels(apiKey: string | undefined, embeddedModels: JsonModel[], signal?: AbortSignal): Promise<JsonModel[] | null> {
|
|
410
|
+
if (!apiKey) return null;
|
|
411
|
+
const liveModels = await fetchLiveModels(apiKey, signal);
|
|
412
|
+
if (!liveModels || liveModels.length === 0) return null;
|
|
413
|
+
const merged = mergeWithEmbedded(liveModels, embeddedModels);
|
|
414
|
+
cacheModels(merged);
|
|
415
|
+
return merged;
|
|
416
|
+
}
|
|
417
|
+
|
|
418
|
+
// ─── API Key Resolution (via ModelRegistry) ────────────────────────────────────
|
|
419
|
+
|
|
420
|
+
let cachedApiKey: string | undefined;
|
|
421
|
+
let revalidateAbort: AbortController | null = null;
|
|
422
|
+
|
|
423
|
+
/** The zro CLI stores its login at ~/.config/zro/credentials.json (XDG aware). */
|
|
424
|
+
function storedZroCliKey(): string | undefined {
|
|
425
|
+
try {
|
|
426
|
+
const configRoot = process.env.XDG_CONFIG_HOME || path.join(homedir(), ".config");
|
|
427
|
+
const parsed = JSON.parse(fs.readFileSync(path.join(configRoot, "zro", "credentials.json"), "utf8"));
|
|
428
|
+
return typeof parsed?.apiKey === "string" && parsed.apiKey.trim() ? parsed.apiKey : undefined;
|
|
429
|
+
} catch {
|
|
430
|
+
return undefined;
|
|
431
|
+
}
|
|
432
|
+
}
|
|
433
|
+
|
|
434
|
+
async function resolveApiKey(modelRegistry: ModelRegistry): Promise<void> {
|
|
435
|
+
cachedApiKey =
|
|
436
|
+
(await modelRegistry.getApiKeyForProvider(PROVIDER_ID) ?? undefined) ||
|
|
437
|
+
(process.env.ZRO_API_KEY ?? undefined) ||
|
|
438
|
+
storedZroCliKey();
|
|
439
|
+
}
|
|
440
|
+
|
|
441
|
+
// ─── Status Display Configuration ──────────────────────────────────────────────
|
|
442
|
+
|
|
443
|
+
const CONFIG_PATH = path.join(getAgentDir(), "extensions", "zro.json");
|
|
444
|
+
|
|
445
|
+
let statusConfig: StatusConfig = { ...DEFAULT_STATUS_CONFIG };
|
|
446
|
+
|
|
447
|
+
function loadStatusConfig(): StatusConfig {
|
|
448
|
+
try {
|
|
449
|
+
const raw = JSON.parse(fs.readFileSync(CONFIG_PATH, "utf8"));
|
|
450
|
+
statusConfig = coerceStatusConfig(raw);
|
|
451
|
+
} catch {
|
|
452
|
+
// Missing or unreadable file → defaults
|
|
453
|
+
}
|
|
454
|
+
return statusConfig;
|
|
455
|
+
}
|
|
456
|
+
|
|
457
|
+
function writeStatusConfig(): void {
|
|
458
|
+
try {
|
|
459
|
+
let raw: Record<string, unknown> = {};
|
|
460
|
+
try {
|
|
461
|
+
const existing = JSON.parse(fs.readFileSync(CONFIG_PATH, "utf8"));
|
|
462
|
+
if (existing && typeof existing === "object" && !Array.isArray(existing)) raw = existing;
|
|
463
|
+
} catch {
|
|
464
|
+
// No existing file — start fresh
|
|
465
|
+
}
|
|
466
|
+
raw.session = statusConfig.session;
|
|
467
|
+
raw.account = statusConfig.account;
|
|
468
|
+
raw.hideOnOtherProvider = statusConfig.hideOnOtherProvider;
|
|
469
|
+
raw.lowBalanceUsd = statusConfig.lowBalanceUsd;
|
|
470
|
+
fs.mkdirSync(path.dirname(CONFIG_PATH), { recursive: true });
|
|
471
|
+
fs.writeFileSync(CONFIG_PATH, JSON.stringify(raw, null, 2) + "\n");
|
|
472
|
+
} catch {
|
|
473
|
+
// Config write failure is non-fatal — the in-memory config still applies
|
|
474
|
+
}
|
|
475
|
+
}
|
|
476
|
+
|
|
477
|
+
loadStatusConfig();
|
|
478
|
+
|
|
479
|
+
// ─── Response Metadata Capture ────────────────────────────────────────────────
|
|
480
|
+
// The custom streamSimple below wraps fetch per request (never globalThis —
|
|
481
|
+
// concurrent main/helper requests would clobber a global patch). For every
|
|
482
|
+
// /chat/completions response we capture x-ratelimit-* headers and tee the
|
|
483
|
+
// body: one copy goes to pi's OpenAI streaming layer, the other is scanned
|
|
484
|
+
// for the final usage chunk (tokens, spend extension) that closes the stream.
|
|
485
|
+
|
|
486
|
+
const sessionStats: SessionStats = { ...EMPTY_SESSION_STATS };
|
|
487
|
+
const account: AccountState = { ...EMPTY_ACCOUNT };
|
|
488
|
+
|
|
489
|
+
// Per-turn pending state — teed streams settle asynchronously, so capture
|
|
490
|
+
// lands in pending* and is committed at turn_end.
|
|
491
|
+
let pendingRequests = 0;
|
|
492
|
+
let pendingTokens = 0;
|
|
493
|
+
let pendingSpend = 0;
|
|
494
|
+
let pendingSawUsage = false;
|
|
495
|
+
let pendingSawOutOfCredits = false;
|
|
496
|
+
let outOfCreditsNotified = false;
|
|
497
|
+
|
|
498
|
+
const teeReaders = new Set<Promise<void>>();
|
|
499
|
+
|
|
500
|
+
function trackTeeReader(promise: Promise<void>): void {
|
|
501
|
+
teeReaders.add(promise);
|
|
502
|
+
const release = () => { teeReaders.delete(promise); };
|
|
503
|
+
promise.then(release, release);
|
|
504
|
+
}
|
|
505
|
+
|
|
506
|
+
function settleTeeReaders(): Promise<void> {
|
|
507
|
+
if (teeReaders.size === 0) return Promise.resolve();
|
|
508
|
+
const pending = Array.from(teeReaders);
|
|
509
|
+
return Promise.allSettled(pending).then(() => undefined);
|
|
510
|
+
}
|
|
511
|
+
|
|
512
|
+
function captureRateLimitHeaders(headers: Headers): void {
|
|
513
|
+
const limit = Number(headers.get("x-ratelimit-limit"));
|
|
514
|
+
const remaining = Number(headers.get("x-ratelimit-remaining"));
|
|
515
|
+
if (Number.isFinite(limit) && Number.isFinite(remaining)) {
|
|
516
|
+
account.rate = { limit, remaining, capturedAt: Date.now() };
|
|
517
|
+
return;
|
|
518
|
+
}
|
|
519
|
+
// Fallback: hourly budget headers some proxies forward
|
|
520
|
+
const limitHour = Number(headers.get("x-ratelimit-limit-hour"));
|
|
521
|
+
const remainingHour = Number(headers.get("x-ratelimit-remaining-hour"));
|
|
522
|
+
if (Number.isFinite(limitHour) && Number.isFinite(remainingHour)) {
|
|
523
|
+
account.rate = { limit: limitHour, remaining: remainingHour, capturedAt: Date.now() };
|
|
524
|
+
}
|
|
525
|
+
}
|
|
526
|
+
|
|
527
|
+
/** Extract spend/token data from a parsed completion chunk/body's usage object. */
|
|
528
|
+
function captureUsage(obj: any): void {
|
|
529
|
+
const usage = obj?.usage;
|
|
530
|
+
if (typeof usage !== "object" || usage === null) return;
|
|
531
|
+
const input = usage.prompt_tokens;
|
|
532
|
+
const output = usage.completion_tokens;
|
|
533
|
+
if (typeof input === "number" && Number.isFinite(input)) pendingTokens += input;
|
|
534
|
+
if (typeof output === "number" && Number.isFinite(output)) pendingTokens += output;
|
|
535
|
+
// Metered proxies expose a cost extension (usage.total_cost / usage.cost.usd);
|
|
536
|
+
// Zro may or may not — token totals are the always-available fallback.
|
|
537
|
+
const total = usage.total_cost ?? usage.cost?.usd ?? usage.cost?.total;
|
|
538
|
+
if (typeof total === "number" && Number.isFinite(total)) {
|
|
539
|
+
pendingSpend += Math.max(0, total);
|
|
540
|
+
}
|
|
541
|
+
pendingSawUsage = true;
|
|
542
|
+
}
|
|
543
|
+
|
|
544
|
+
/** Scan a teed response for the final usage chunk (SSE) or JSON body usage. */
|
|
545
|
+
async function readUsageFromTee(body: ReadableStream<Uint8Array>): Promise<void> {
|
|
546
|
+
const reader = body.getReader();
|
|
547
|
+
const decoder = new TextDecoder();
|
|
548
|
+
let buffer = "";
|
|
549
|
+
|
|
550
|
+
const processLine = (line: string): void => {
|
|
551
|
+
const trimmed = line.trim();
|
|
552
|
+
if (!trimmed.startsWith("data: ")) return;
|
|
553
|
+
const payload = trimmed.slice(6);
|
|
554
|
+
if (payload === "[DONE]") return;
|
|
555
|
+
try {
|
|
556
|
+
captureUsage(JSON.parse(payload));
|
|
557
|
+
} catch {
|
|
558
|
+
// Not JSON or no usage — benign
|
|
559
|
+
}
|
|
560
|
+
};
|
|
561
|
+
|
|
562
|
+
try {
|
|
563
|
+
while (true) {
|
|
564
|
+
const { done, value } = await reader.read();
|
|
565
|
+
if (done) break;
|
|
566
|
+
buffer += decoder.decode(value, { stream: true });
|
|
567
|
+
const lines = buffer.split("\n");
|
|
568
|
+
buffer = lines.pop() || "";
|
|
569
|
+
for (const line of lines) processLine(line);
|
|
570
|
+
}
|
|
571
|
+
} catch {
|
|
572
|
+
// Tee stream may error if the main stream is aborted — that's fine
|
|
573
|
+
}
|
|
574
|
+
|
|
575
|
+
const trailing = (buffer + decoder.decode(new Uint8Array(0), { stream: false })).trim();
|
|
576
|
+
if (trailing) {
|
|
577
|
+
if (trailing.startsWith("data: ")) {
|
|
578
|
+
processLine(trailing);
|
|
579
|
+
} else if (trailing.startsWith("{")) {
|
|
580
|
+
try {
|
|
581
|
+
captureUsage(JSON.parse(trailing));
|
|
582
|
+
} catch {
|
|
583
|
+
// Partial non-SSE body — ignore
|
|
584
|
+
}
|
|
585
|
+
}
|
|
586
|
+
}
|
|
587
|
+
|
|
588
|
+
try {
|
|
589
|
+
reader.releaseLock();
|
|
590
|
+
} catch {
|
|
591
|
+
// Ignore
|
|
592
|
+
}
|
|
593
|
+
}
|
|
594
|
+
|
|
595
|
+
// ─── Custom Streaming Provider ────────────────────────────────────────────────
|
|
596
|
+
|
|
597
|
+
function streamZro(
|
|
598
|
+
model: any,
|
|
599
|
+
context: any,
|
|
600
|
+
options?: SimpleStreamOptions,
|
|
601
|
+
): AssistantMessageEventStream {
|
|
602
|
+
const apiKey = (options as any)?.apiKey || cachedApiKey || storedZroCliKey() || "";
|
|
603
|
+
if (!apiKey) {
|
|
604
|
+
throw new Error(
|
|
605
|
+
`No API key for Zro. Run \`zro login\`, add it to ~/.pi/agent/auth.json, ` +
|
|
606
|
+
`set ZRO_API_KEY env var, or use --api-key.`,
|
|
607
|
+
);
|
|
608
|
+
}
|
|
609
|
+
|
|
610
|
+
const zroModel = { ...model, api: "openai-completions", baseUrl: model.baseUrl || BASE_URL };
|
|
611
|
+
|
|
612
|
+
// pi hands the user's thinking selection as options.reasoning (a raw
|
|
613
|
+
// ThinkingLevel); streamOpenAICompletions only reads reasoningEffort.
|
|
614
|
+
// Replicate pi-ai's clamp+convert so levels reach the request body.
|
|
615
|
+
const clampedReasoning = options?.reasoning ? clampThinkingLevel(zroModel, options.reasoning) : undefined;
|
|
616
|
+
const reasoningEffort = clampedReasoning === "off" ? undefined : clampedReasoning;
|
|
617
|
+
const { reasoning: _reasoning, ...streamOptions } = (options ?? {}) as any;
|
|
618
|
+
|
|
619
|
+
// Per-request fetch wrapper: owns its interceptor, safe under concurrency.
|
|
620
|
+
const upstreamFetch = (streamOptions as any).fetch ?? globalThis.fetch;
|
|
621
|
+
const metaFetch = async (input: RequestInfo | URL, init?: RequestInit) => {
|
|
622
|
+
const response = await upstreamFetch(input as any, init);
|
|
623
|
+
const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
|
|
624
|
+
if (!url.includes("/chat/completions")) return response;
|
|
625
|
+
|
|
626
|
+
pendingRequests += 1;
|
|
627
|
+
captureRateLimitHeaders(response.headers);
|
|
628
|
+
if (response.status === 402) pendingSawOutOfCredits = true;
|
|
629
|
+
if (!response.ok || !response.body) return response;
|
|
630
|
+
|
|
631
|
+
const [bodyForSdk, bodyForMeta] = response.body.tee();
|
|
632
|
+
trackTeeReader(readUsageFromTee(bodyForMeta));
|
|
633
|
+
return new Response(bodyForSdk, {
|
|
634
|
+
headers: response.headers,
|
|
635
|
+
status: response.status,
|
|
636
|
+
statusText: response.statusText,
|
|
637
|
+
});
|
|
638
|
+
};
|
|
639
|
+
|
|
640
|
+
return streamOpenAICompletions(zroModel, context, {
|
|
641
|
+
...streamOptions,
|
|
642
|
+
fetch: metaFetch,
|
|
643
|
+
reasoningEffort,
|
|
644
|
+
apiKey,
|
|
645
|
+
} as any);
|
|
646
|
+
}
|
|
647
|
+
|
|
648
|
+
// ─── Account Metadata Fetching ────────────────────────────────────────────────
|
|
649
|
+
|
|
650
|
+
const STATUS_MIN_INTERVAL_MS = 15_000;
|
|
651
|
+
const ACCOUNT_FETCH_TIMEOUT_MS = 8_000;
|
|
652
|
+
|
|
653
|
+
let statusAbort: AbortController | null = null;
|
|
654
|
+
// Bumped on every session_start; async continuations compare against this to
|
|
655
|
+
// drop work belonging to a replaced session (its ctx is stale and throws).
|
|
656
|
+
let statusEpoch = 0;
|
|
657
|
+
let lastStatusFetchAt = 0;
|
|
658
|
+
let statusInFlight: Promise<void> | null = null;
|
|
659
|
+
let metaFetched = false;
|
|
660
|
+
|
|
661
|
+
async function fetchJsonGet(url: string, apiKey: string, signal?: AbortSignal): Promise<any | null> {
|
|
662
|
+
try {
|
|
663
|
+
const response = await fetch(url, {
|
|
664
|
+
headers: { Authorization: `Bearer ${apiKey}` },
|
|
665
|
+
signal: signal
|
|
666
|
+
? AbortSignal.any([AbortSignal.timeout(ACCOUNT_FETCH_TIMEOUT_MS), signal])
|
|
667
|
+
: AbortSignal.timeout(ACCOUNT_FETCH_TIMEOUT_MS),
|
|
668
|
+
});
|
|
669
|
+
if (!response.ok) return null;
|
|
670
|
+
return await response.json();
|
|
671
|
+
} catch {
|
|
672
|
+
return null;
|
|
673
|
+
}
|
|
674
|
+
}
|
|
675
|
+
|
|
676
|
+
/** Plan/spend/pack/activity from /api/cli/status — the `zro status` view. Throttled unless forced. */
|
|
677
|
+
function refreshAccountStatus(apiKey: string | undefined, signal: AbortSignal | undefined, force: boolean): Promise<void> {
|
|
678
|
+
if (!apiKey) return Promise.resolve();
|
|
679
|
+
if (!force && Date.now() - lastStatusFetchAt < STATUS_MIN_INTERVAL_MS) return Promise.resolve();
|
|
680
|
+
lastStatusFetchAt = Date.now();
|
|
681
|
+
if (statusInFlight) return statusInFlight;
|
|
682
|
+
statusInFlight = (async () => {
|
|
683
|
+
try {
|
|
684
|
+
const data = await fetchJsonGet(STATUS_URL, apiKey, signal);
|
|
685
|
+
if (data === null) return;
|
|
686
|
+
const billing = data?.billing;
|
|
687
|
+
if (billing && typeof billing === "object") {
|
|
688
|
+
if (typeof billing.totalRemaining === "number") account.availableUsd = billing.totalRemaining;
|
|
689
|
+
if (typeof billing.usagePacks?.remaining === "number") account.usagePackUsd = billing.usagePacks.remaining;
|
|
690
|
+
const planName = billing.plan?.name;
|
|
691
|
+
if (typeof planName === "string" && planName.trim()) account.planName = planName.trim();
|
|
692
|
+
}
|
|
693
|
+
const activity = data?.activity30d;
|
|
694
|
+
if (activity && typeof activity === "object") {
|
|
695
|
+
if (typeof activity.requests === "number") account.activity30dRequests = activity.requests;
|
|
696
|
+
if (typeof activity.totalTokens === "number") account.activity30dTokens = activity.totalTokens;
|
|
697
|
+
}
|
|
698
|
+
if (typeof data?.key?.alias === "string" && data.key.alias.trim()) account.keyAlias = data.key.alias.trim();
|
|
699
|
+
metaFetched = true;
|
|
700
|
+
} finally {
|
|
701
|
+
statusInFlight = null;
|
|
702
|
+
}
|
|
703
|
+
})();
|
|
704
|
+
return statusInFlight;
|
|
705
|
+
}
|
|
706
|
+
|
|
707
|
+
// ─── Status Rendering ─────────────────────────────────────────────────────────
|
|
708
|
+
|
|
709
|
+
const WIDGET_KEY = "zro";
|
|
710
|
+
const STATUS_KEY_SESSION = "zro-session";
|
|
711
|
+
const STATUS_KEY_ACCOUNT = "zro-account";
|
|
712
|
+
|
|
713
|
+
function currentProviderId(ctx: ExtensionContext): string | undefined {
|
|
714
|
+
// ctx.model is a getter that can throw on stale contexts
|
|
715
|
+
try {
|
|
716
|
+
return (ctx.model as any)?.provider as string | undefined;
|
|
717
|
+
} catch {
|
|
718
|
+
return undefined;
|
|
719
|
+
}
|
|
720
|
+
}
|
|
721
|
+
|
|
722
|
+
function isStaleCtxError(err: unknown): boolean {
|
|
723
|
+
return err instanceof Error && err.message.includes("This extension ctx is stale");
|
|
724
|
+
}
|
|
725
|
+
|
|
726
|
+
// Render entry point: swallows the stale-ctx throw so a refresh racing a
|
|
727
|
+
// session replacement (newSession/fork/switchSession/reload) can't crash pi.
|
|
728
|
+
function updateStatus(ctx: ExtensionContext): void {
|
|
729
|
+
try {
|
|
730
|
+
renderStatus(ctx);
|
|
731
|
+
} catch (err) {
|
|
732
|
+
if (!isStaleCtxError(err)) throw err;
|
|
733
|
+
}
|
|
734
|
+
}
|
|
735
|
+
|
|
736
|
+
// Re-render once an async refresh lands, unless the session was replaced
|
|
737
|
+
// meanwhile (epoch bump) — its ctx is stale and the render is obsolete anyway.
|
|
738
|
+
function updateStatusAfter(promise: Promise<void>, ctx: ExtensionContext): void {
|
|
739
|
+
const epoch = statusEpoch;
|
|
740
|
+
void promise.then(() => {
|
|
741
|
+
if (epoch === statusEpoch) updateStatus(ctx);
|
|
742
|
+
});
|
|
743
|
+
}
|
|
744
|
+
|
|
745
|
+
function renderStatus(ctx: ExtensionContext): void {
|
|
746
|
+
const provider = currentProviderId(ctx);
|
|
747
|
+
const hiddenByOtherProvider =
|
|
748
|
+
statusConfig.hideOnOtherProvider && provider !== undefined && provider !== PROVIDER_ID;
|
|
749
|
+
|
|
750
|
+
const clearAll = () => {
|
|
751
|
+
ctx.ui.setStatus(STATUS_KEY_SESSION, undefined);
|
|
752
|
+
ctx.ui.setStatus(STATUS_KEY_ACCOUNT, undefined);
|
|
753
|
+
ctx.ui.setWidget(WIDGET_KEY, undefined);
|
|
754
|
+
};
|
|
755
|
+
|
|
756
|
+
if (hiddenByOtherProvider) {
|
|
757
|
+
clearAll();
|
|
758
|
+
return;
|
|
759
|
+
}
|
|
760
|
+
|
|
761
|
+
const hasActivity = sessionStats.requests > 0 || sessionStats.tokens > 0 || sessionStats.spend > 0;
|
|
762
|
+
const sessionLine = statusConfig.session !== "off" ? buildSessionLine(sessionStats) : undefined;
|
|
763
|
+
// Show only after Zro activity this session (like pi-neuralwatt):
|
|
764
|
+
// no empty-gap line on fresh sessions, no stale account glare on other
|
|
765
|
+
// providers' sessions.
|
|
766
|
+
const accountVisible = statusConfig.account !== "off" && accountHasData(account) && hasActivity;
|
|
767
|
+
const lowBalance =
|
|
768
|
+
statusConfig.lowBalanceUsd !== null && account.availableUsd !== null && account.availableUsd <= statusConfig.lowBalanceUsd;
|
|
769
|
+
const accTiers = accountVisible ? buildAccountTiers(account, lowBalance) : [];
|
|
770
|
+
|
|
771
|
+
// Status bar (built-in footer slots)
|
|
772
|
+
const sBar = statusConfig.session === "statusbar" ? sessionLine : undefined;
|
|
773
|
+
const aBar = statusConfig.account === "statusbar" && accountVisible ? accTiers[0] : undefined;
|
|
774
|
+
if (sBar && aBar) {
|
|
775
|
+
// Combined to avoid eating two footer slots
|
|
776
|
+
ctx.ui.setStatus(STATUS_KEY_SESSION, ctx.ui.theme.fg(lowBalance ? "warning" : "dim", `${sBar} · ${aBar}`));
|
|
777
|
+
ctx.ui.setStatus(STATUS_KEY_ACCOUNT, undefined);
|
|
778
|
+
} else {
|
|
779
|
+
ctx.ui.setStatus(STATUS_KEY_SESSION, sBar ? ctx.ui.theme.fg("dim", sBar) : undefined);
|
|
780
|
+
ctx.ui.setStatus(STATUS_KEY_ACCOUNT, aBar ? ctx.ui.theme.fg(lowBalance ? "warning" : "dim", aBar) : undefined);
|
|
781
|
+
}
|
|
782
|
+
|
|
783
|
+
// Below-editor widget (two-zone, width-aware)
|
|
784
|
+
const leftW = statusConfig.session === "widget" ? sessionLine : undefined;
|
|
785
|
+
const rightW = statusConfig.account === "widget" && accountVisible ? accTiers : undefined;
|
|
786
|
+
if (leftW !== undefined || (rightW !== undefined && rightW.length > 0)) {
|
|
787
|
+
ctx.ui.setWidget(
|
|
788
|
+
WIDGET_KEY,
|
|
789
|
+
(_tui: any, theme: any) => new StatusLineWidget(theme, leftW ?? "", rightW ?? [], lowBalance),
|
|
790
|
+
{ placement: "belowEditor" },
|
|
791
|
+
);
|
|
792
|
+
} else {
|
|
793
|
+
ctx.ui.setWidget(WIDGET_KEY, undefined);
|
|
794
|
+
}
|
|
795
|
+
}
|
|
796
|
+
|
|
797
|
+
function resetStatusState(): void {
|
|
798
|
+
sessionStats.requests = 0;
|
|
799
|
+
sessionStats.tokens = 0;
|
|
800
|
+
sessionStats.spend = 0;
|
|
801
|
+
Object.assign(account, EMPTY_ACCOUNT);
|
|
802
|
+
pendingRequests = 0;
|
|
803
|
+
pendingTokens = 0;
|
|
804
|
+
pendingSpend = 0;
|
|
805
|
+
pendingSawUsage = false;
|
|
806
|
+
pendingSawOutOfCredits = false;
|
|
807
|
+
outOfCreditsNotified = false;
|
|
808
|
+
lastStatusFetchAt = 0;
|
|
809
|
+
metaFetched = false;
|
|
810
|
+
}
|
|
811
|
+
|
|
812
|
+
/** Commit per-turn pending capture into session state (after tees settle). */
|
|
813
|
+
function commitPending(ctx: ExtensionContext): void {
|
|
814
|
+
if (!pendingSawUsage && pendingRequests === 0) return;
|
|
815
|
+
sessionStats.requests += pendingRequests;
|
|
816
|
+
sessionStats.tokens += pendingTokens;
|
|
817
|
+
sessionStats.spend += pendingSpend;
|
|
818
|
+
|
|
819
|
+
// Optimistic balance: deduct this turn's observed spend so the account
|
|
820
|
+
// line ticks down per turn with zero extra API calls. Every status poll
|
|
821
|
+
// overwrites account.availableUsd (never adjusts), so this cannot
|
|
822
|
+
// double-count; the agent_settled poll reconciles any drift.
|
|
823
|
+
applyOptimisticSpend(account, pendingSpend);
|
|
824
|
+
|
|
825
|
+
pendingRequests = 0;
|
|
826
|
+
pendingTokens = 0;
|
|
827
|
+
pendingSpend = 0;
|
|
828
|
+
pendingSawUsage = false;
|
|
829
|
+
|
|
830
|
+
if (pendingSawOutOfCredits) {
|
|
831
|
+
pendingSawOutOfCredits = false;
|
|
832
|
+
// Re-fetch now so the balance reflects exhaustion immediately
|
|
833
|
+
updateStatusAfter(refreshAccountStatus(cachedApiKey, statusAbort?.signal ?? undefined, true), ctx);
|
|
834
|
+
if (!outOfCreditsNotified && ctx.hasUI) {
|
|
835
|
+
outOfCreditsNotified = true;
|
|
836
|
+
ctx.ui.notify("Zro has run out of available spend — top up at zro.moonmath.ai", "error");
|
|
837
|
+
}
|
|
838
|
+
}
|
|
839
|
+
}
|
|
840
|
+
|
|
841
|
+
// ─── Status Command ────────────────────────────────────────────────────────────
|
|
842
|
+
|
|
843
|
+
function statusSummary(): string {
|
|
844
|
+
const lb = statusConfig.lowBalanceUsd === null ? "off" : `${statusConfig.lowBalanceUsd}`;
|
|
845
|
+
return `session=${statusConfig.session}, account=${statusConfig.account}, hideOnOtherProvider=${statusConfig.hideOnOtherProvider}, lowBalanceUsd=${lb}`;
|
|
846
|
+
}
|
|
847
|
+
|
|
848
|
+
const STATUS_USAGE =
|
|
849
|
+
"Usage: /zro-status [session|account widget|statusbar|off · hide true|false · lowBalance <usd>|off · refresh · reset]";
|
|
850
|
+
|
|
851
|
+
async function handleStatusCommand(args: string, ctx: ExtensionContext): Promise<void> {
|
|
852
|
+
const tokens = args.trim().split(/\s+/).filter(Boolean);
|
|
853
|
+
|
|
854
|
+
if (tokens.length === 0) {
|
|
855
|
+
if (!ctx.hasUI) {
|
|
856
|
+
ctx.ui.notify(statusSummary(), "info");
|
|
857
|
+
return;
|
|
858
|
+
}
|
|
859
|
+
await configureStatusInteractive(ctx);
|
|
860
|
+
return;
|
|
861
|
+
}
|
|
862
|
+
|
|
863
|
+
const [rawKey, rawValue] = tokens;
|
|
864
|
+
const key = rawKey.toLowerCase();
|
|
865
|
+
const value = rawValue?.toLowerCase();
|
|
866
|
+
|
|
867
|
+
if (key === "refresh") {
|
|
868
|
+
metaFetched = false;
|
|
869
|
+
await refreshAccountStatus(cachedApiKey, statusAbort?.signal ?? undefined, true);
|
|
870
|
+
updateStatus(ctx);
|
|
871
|
+
const bal = account.availableUsd !== null ? `$${account.availableUsd} avail` : "unknown";
|
|
872
|
+
ctx.ui.notify(`Zro account: ${bal}. ${statusSummary()}`, "info");
|
|
873
|
+
return;
|
|
874
|
+
}
|
|
875
|
+
|
|
876
|
+
if (key === "reset" && tokens.length === 1) {
|
|
877
|
+
statusConfig = { ...DEFAULT_STATUS_CONFIG };
|
|
878
|
+
writeStatusConfig();
|
|
879
|
+
updateStatus(ctx);
|
|
880
|
+
ctx.ui.notify(`Zro status reset. ${statusSummary()}`, "info");
|
|
881
|
+
return;
|
|
882
|
+
}
|
|
883
|
+
|
|
884
|
+
if ((key === "session" || key === "account") && tokens.length === 2) {
|
|
885
|
+
if (value !== "widget" && value !== "statusbar" && value !== "off") {
|
|
886
|
+
ctx.ui.notify(STATUS_USAGE, "error");
|
|
887
|
+
return;
|
|
888
|
+
}
|
|
889
|
+
statusConfig[key] = value;
|
|
890
|
+
writeStatusConfig();
|
|
891
|
+
if (value !== "off" && key === "account") {
|
|
892
|
+
// Turning account on: make sure we have data to show
|
|
893
|
+
updateStatusAfter(refreshAccountStatus(cachedApiKey, statusAbort?.signal ?? undefined, true), ctx);
|
|
894
|
+
}
|
|
895
|
+
updateStatus(ctx);
|
|
896
|
+
ctx.ui.notify(`Zro ${key} line: ${value}. ${statusSummary()}`, "info");
|
|
897
|
+
return;
|
|
898
|
+
}
|
|
899
|
+
|
|
900
|
+
if ((key === "hide" || key === "hideonotherprovider") && tokens.length === 2) {
|
|
901
|
+
if (value !== "true" && value !== "false") {
|
|
902
|
+
ctx.ui.notify(STATUS_USAGE, "error");
|
|
903
|
+
return;
|
|
904
|
+
}
|
|
905
|
+
statusConfig.hideOnOtherProvider = value === "true";
|
|
906
|
+
writeStatusConfig();
|
|
907
|
+
updateStatus(ctx);
|
|
908
|
+
ctx.ui.notify(`Zro status. ${statusSummary()}`, "info");
|
|
909
|
+
return;
|
|
910
|
+
}
|
|
911
|
+
|
|
912
|
+
if (key === "lowbalance" && tokens.length === 2) {
|
|
913
|
+
if (value === "off") {
|
|
914
|
+
statusConfig.lowBalanceUsd = null;
|
|
915
|
+
} else {
|
|
916
|
+
const n = Number(value);
|
|
917
|
+
if (!Number.isFinite(n) || n <= 0) {
|
|
918
|
+
ctx.ui.notify(STATUS_USAGE, "error");
|
|
919
|
+
return;
|
|
920
|
+
}
|
|
921
|
+
statusConfig.lowBalanceUsd = n;
|
|
922
|
+
}
|
|
923
|
+
writeStatusConfig();
|
|
924
|
+
updateStatus(ctx);
|
|
925
|
+
ctx.ui.notify(`Zro status. ${statusSummary()}`, "info");
|
|
926
|
+
return;
|
|
927
|
+
}
|
|
928
|
+
|
|
929
|
+
ctx.ui.notify(STATUS_USAGE, "error");
|
|
930
|
+
}
|
|
931
|
+
|
|
932
|
+
async function configureStatusInteractive(ctx: ExtensionContext): Promise<void> {
|
|
933
|
+
const modes = ["widget", "statusbar", "off"] as const;
|
|
934
|
+
const nextMode = (m: string) => modes[(modes.indexOf(m as any) + 1) % modes.length];
|
|
935
|
+
|
|
936
|
+
for (;;) {
|
|
937
|
+
const lb = statusConfig.lowBalanceUsd === null ? "off" : `$${statusConfig.lowBalanceUsd}`;
|
|
938
|
+
const sessionOpt = `Session line (spend/tokens/requests): ${statusConfig.session}`;
|
|
939
|
+
const accountOpt = `Account line (plan/balance/packs/activity): ${statusConfig.account}`;
|
|
940
|
+
const hideOpt = `Hide on other providers: ${statusConfig.hideOnOtherProvider ? "on" : "off"}`;
|
|
941
|
+
const lbOpt = `Low-balance warning: ${lb}`;
|
|
942
|
+
const refreshOpt = "Refresh account status now";
|
|
943
|
+
const doneOpt = "Done";
|
|
944
|
+
|
|
945
|
+
const choice = await ctx.ui.select("Zro footer status", [
|
|
946
|
+
sessionOpt,
|
|
947
|
+
accountOpt,
|
|
948
|
+
hideOpt,
|
|
949
|
+
lbOpt,
|
|
950
|
+
refreshOpt,
|
|
951
|
+
doneOpt,
|
|
952
|
+
]);
|
|
953
|
+
|
|
954
|
+
if (choice === undefined || choice === doneOpt) {
|
|
955
|
+
updateStatus(ctx);
|
|
956
|
+
return;
|
|
957
|
+
}
|
|
958
|
+
if (choice === sessionOpt) {
|
|
959
|
+
statusConfig.session = nextMode(statusConfig.session);
|
|
960
|
+
writeStatusConfig();
|
|
961
|
+
continue;
|
|
962
|
+
}
|
|
963
|
+
if (choice === accountOpt) {
|
|
964
|
+
statusConfig.account = nextMode(statusConfig.account);
|
|
965
|
+
writeStatusConfig();
|
|
966
|
+
if (statusConfig.account !== "off") {
|
|
967
|
+
updateStatusAfter(refreshAccountStatus(cachedApiKey, statusAbort?.signal ?? undefined, true), ctx);
|
|
968
|
+
}
|
|
969
|
+
continue;
|
|
970
|
+
}
|
|
971
|
+
if (choice === hideOpt) {
|
|
972
|
+
statusConfig.hideOnOtherProvider = !statusConfig.hideOnOtherProvider;
|
|
973
|
+
writeStatusConfig();
|
|
974
|
+
updateStatus(ctx);
|
|
975
|
+
continue;
|
|
976
|
+
}
|
|
977
|
+
if (choice === lbOpt) {
|
|
978
|
+
const presets = ["off", "1", "5", "10", "25", "50", "100"];
|
|
979
|
+
const current = statusConfig.lowBalanceUsd === null ? "off" : String(statusConfig.lowBalanceUsd);
|
|
980
|
+
const ordered = presets.includes(current) ? presets : [current, ...presets];
|
|
981
|
+
const pick = await ctx.ui.select("Warn at/below balance (USD)", ordered);
|
|
982
|
+
if (pick !== undefined) {
|
|
983
|
+
statusConfig.lowBalanceUsd = pick === "off" ? null : Number(pick);
|
|
984
|
+
writeStatusConfig();
|
|
985
|
+
updateStatus(ctx);
|
|
986
|
+
}
|
|
987
|
+
continue;
|
|
988
|
+
}
|
|
989
|
+
if (choice === refreshOpt) {
|
|
990
|
+
metaFetched = false;
|
|
991
|
+
await refreshAccountStatus(cachedApiKey, statusAbort?.signal ?? undefined, true);
|
|
992
|
+
updateStatus(ctx);
|
|
993
|
+
continue;
|
|
994
|
+
}
|
|
995
|
+
}
|
|
996
|
+
}
|
|
997
|
+
|
|
998
|
+
// ─── Extension Entry Point ────────────────────────────────────────────────────
|
|
999
|
+
|
|
1000
|
+
// The currently-registered model list — starts stale, hot-swapped when the
|
|
1001
|
+
// live catalog lands. Provider identity funnels through makeProviderConfig so
|
|
1002
|
+
// the stream handler and models never desync.
|
|
1003
|
+
let currentModels: JsonModel[] = [];
|
|
1004
|
+
|
|
1005
|
+
function makeProviderConfig(models: JsonModel[] = currentModels) {
|
|
1006
|
+
return {
|
|
1007
|
+
baseUrl: BASE_URL,
|
|
1008
|
+
apiKey: "$ZRO_API_KEY",
|
|
1009
|
+
// Custom API name so our streamSimple registers as its own handler and
|
|
1010
|
+
// never shadows pi's built-in openai-completions pipeline for other
|
|
1011
|
+
// providers. streamZro delegates to pi-ai's OpenAI-compat streamer.
|
|
1012
|
+
api: "zro",
|
|
1013
|
+
headers: { "User-Agent": "pi-coding-agent" },
|
|
1014
|
+
models,
|
|
1015
|
+
streamSimple: streamZro,
|
|
1016
|
+
};
|
|
1017
|
+
}
|
|
1018
|
+
|
|
1019
|
+
export default function (pi: ExtensionAPI) {
|
|
1020
|
+
const embeddedModels = modelsData as JsonModel[];
|
|
1021
|
+
const customModels = customModelsData as JsonModel[];
|
|
1022
|
+
const patches = patchData as PatchData;
|
|
1023
|
+
|
|
1024
|
+
const staleBase = loadStaleModels(embeddedModels);
|
|
1025
|
+
const staleModels = buildModels(staleBase, customModels, patches);
|
|
1026
|
+
currentModels = staleModels;
|
|
1027
|
+
|
|
1028
|
+
pi.registerProvider(PROVIDER_ID, makeProviderConfig(staleModels));
|
|
1029
|
+
|
|
1030
|
+
pi.registerCommand("zro-status", {
|
|
1031
|
+
description: "Configure the Zro footer status (session spend, account balance, packs, activity)",
|
|
1032
|
+
handler: async (args, ctx) => {
|
|
1033
|
+
await handleStatusCommand(args, ctx);
|
|
1034
|
+
},
|
|
1035
|
+
});
|
|
1036
|
+
|
|
1037
|
+
pi.on("session_start", async (_event, ctx) => {
|
|
1038
|
+
const epoch = ++statusEpoch;
|
|
1039
|
+
revalidateAbort?.abort();
|
|
1040
|
+
revalidateAbort = new AbortController();
|
|
1041
|
+
const signal = revalidateAbort.signal;
|
|
1042
|
+
statusAbort?.abort();
|
|
1043
|
+
statusAbort = new AbortController();
|
|
1044
|
+
const statusSignal = statusAbort.signal;
|
|
1045
|
+
|
|
1046
|
+
loadStatusConfig();
|
|
1047
|
+
resetStatusState();
|
|
1048
|
+
updateStatus(ctx); // clears any carryover; activity-gated, renders nothing yet
|
|
1049
|
+
// Re-register so our identity (custom api + streamSimple) always wins
|
|
1050
|
+
// over anything that touched provider registration during load.
|
|
1051
|
+
pi.registerProvider(PROVIDER_ID, makeProviderConfig());
|
|
1052
|
+
|
|
1053
|
+
resolveApiKey(ctx.modelRegistry).then(() => {
|
|
1054
|
+
// A session replacement while the key resolved invalidated the
|
|
1055
|
+
// captured ctx (fast-resume, /new, /fork); nothing below may touch it.
|
|
1056
|
+
if (epoch !== statusEpoch) return;
|
|
1057
|
+
// Prefetch account status only when a Zro model is active
|
|
1058
|
+
// (pi-neuralwatt also prefetches so the first turn ends with data, but
|
|
1059
|
+
// gating here avoids API calls in sessions that never use the provider).
|
|
1060
|
+
if (currentProviderId(ctx) === PROVIDER_ID) {
|
|
1061
|
+
updateStatusAfter(refreshAccountStatus(cachedApiKey, statusSignal, true), ctx);
|
|
1062
|
+
}
|
|
1063
|
+
revalidateModels(cachedApiKey, embeddedModels, signal).then((freshBase) => {
|
|
1064
|
+
if (freshBase && epoch === statusEpoch && !signal.aborted) {
|
|
1065
|
+
currentModels = buildModels(freshBase, customModels, patches);
|
|
1066
|
+
pi.registerProvider(PROVIDER_ID, makeProviderConfig());
|
|
1067
|
+
}
|
|
1068
|
+
});
|
|
1069
|
+
});
|
|
1070
|
+
});
|
|
1071
|
+
|
|
1072
|
+
pi.on("model_select", (event, ctx) => {
|
|
1073
|
+
updateStatus(ctx);
|
|
1074
|
+
const model: any = (event as any).model;
|
|
1075
|
+
if (model?.provider === PROVIDER_ID && cachedApiKey) {
|
|
1076
|
+
updateStatusAfter(refreshAccountStatus(cachedApiKey, statusAbort?.signal ?? undefined, false), ctx);
|
|
1077
|
+
}
|
|
1078
|
+
});
|
|
1079
|
+
|
|
1080
|
+
pi.on("turn_end", async (_event, ctx) => {
|
|
1081
|
+
// Ensure every concurrent response tee has landed before committing.
|
|
1082
|
+
await settleTeeReaders();
|
|
1083
|
+
commitPending(ctx);
|
|
1084
|
+
// If the session_start/model_select status fetch raced or failed, retry
|
|
1085
|
+
// once we have real activity so the very first turn shows the balance.
|
|
1086
|
+
if (sessionStats.requests > 0 && account.availableUsd === null && !metaFetched) {
|
|
1087
|
+
await refreshAccountStatus(cachedApiKey, statusAbort?.signal ?? undefined, false);
|
|
1088
|
+
}
|
|
1089
|
+
updateStatus(ctx);
|
|
1090
|
+
});
|
|
1091
|
+
|
|
1092
|
+
// agent_settled (not agent_end): fires only when no automatic retry,
|
|
1093
|
+
// compaction, or queued continuation can follow — the one moment polling
|
|
1094
|
+
// /api/cli/status is both fresh and not redundant. Gated on session activity
|
|
1095
|
+
// so sessions without Zro turns make zero API calls here.
|
|
1096
|
+
pi.on("agent_settled", async (_event, ctx) => {
|
|
1097
|
+
if (sessionStats.requests > 0 || sessionStats.tokens > 0 || sessionStats.spend > 0) {
|
|
1098
|
+
await refreshAccountStatus(cachedApiKey, statusAbort?.signal ?? undefined, false);
|
|
1099
|
+
updateStatus(ctx);
|
|
1100
|
+
}
|
|
1101
|
+
});
|
|
1102
|
+
|
|
1103
|
+
pi.on("session_shutdown", (_event, ctx) => {
|
|
1104
|
+
revalidateAbort?.abort();
|
|
1105
|
+
statusAbort?.abort();
|
|
1106
|
+
ctx.ui.setStatus(STATUS_KEY_SESSION, undefined);
|
|
1107
|
+
ctx.ui.setStatus(STATUS_KEY_ACCOUNT, undefined);
|
|
1108
|
+
ctx.ui.setWidget(WIDGET_KEY, undefined);
|
|
1109
|
+
});
|
|
1110
|
+
}
|