@oh-my-pi/pi-ai 18.0.7 → 18.0.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,17 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [18.0.8] - 2026-08-27
6
+
7
+ ### Added
8
+
9
+ - Added Z.AI GLM Coding Plan usage tracking: credit-based `CREDIT_LIMIT` windows (5h + weekly) now surface in `omp usage` and the status line with the plan tier (`plan: lite/pro/max`).
10
+
11
+ ### Fixed
12
+
13
+ - Fixed Amazon Bedrock requests to OpenAI-schema models (the `gpt-5.x` SKUs) failing with HTTP 400 `unknown_parameter: 'thinking'` when reasoning was enabled, by sending `reasoning.effort` instead of Anthropic's `thinking` budget block for models the catalog marks as effort-controlled.
14
+ - Fixed Cursor replay rejecting sessions with orphaned tool results while preserving their output as assistant context.
15
+
5
16
  ## [18.0.7] - 2026-08-26
6
17
 
7
18
  ### Added
@@ -698,6 +698,10 @@ export interface DeveloperMessage {
698
698
  content: string | (TextContent | ImageContent)[];
699
699
  /** Who initiated this message for billing/attribution semantics. */
700
700
  attribution?: MessageAttribution;
701
+ /** True if the message was injected by the system (e.g., auto-continue) and initiates a fresh run rather than continuing the current one. */
702
+ synthetic?: boolean;
703
+ /** True when the synthetic prompt was a deliberate operator action (`.`, `c` continue shortcut) rather than an automatic continuation — its timestamp is the turn's prompt time. */
704
+ userInitiated?: boolean;
701
705
  /** Provider-specific opaque payload used to reconstruct transport-native history. */
702
706
  providerPayload?: ProviderPayload;
703
707
  timestamp: number;
@@ -781,6 +785,8 @@ export interface AssistantMessage {
781
785
  timestamp: number;
782
786
  duration?: number;
783
787
  ttft?: number;
788
+ /** Local wall-clock time the response finished streaming (ms since epoch); stamped by the session at message_end so prompt→yield timing never depends on provider-reported duration. */
789
+ completedAt?: number;
784
790
  }
785
791
  export interface ToolResultMessage<TDetails = unknown> {
786
792
  role: "toolResult";
@@ -1,5 +1,5 @@
1
1
  import type { FetchImpl, Provider } from "./types.js";
2
- export type UsageUnit = "percent" | "tokens" | "requests" | "usd" | "minutes" | "bytes" | "unknown";
2
+ export type UsageUnit = "percent" | "tokens" | "requests" | "credits" | "usd" | "minutes" | "bytes" | "unknown";
3
3
  export type UsageStatus = "ok" | "warning" | "exhausted" | "unknown";
4
4
  /** Time window for a limit (e.g. 5h, 7d, monthly). */
5
5
  export interface UsageWindow {
@@ -204,7 +204,7 @@ export interface ClientUsageClientSummary {
204
204
  export interface ClientUsageSummary {
205
205
  clients: ClientUsageClientSummary[];
206
206
  }
207
- export declare const usageUnitSchema: import("@oh-my-pi/omptype").FluentType<"bytes" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd", "bytes" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd">;
207
+ export declare const usageUnitSchema: import("@oh-my-pi/omptype").FluentType<"bytes" | "credits" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd", "bytes" | "credits" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd">;
208
208
  export declare const usageStatusSchema: import("@oh-my-pi/omptype").FluentType<"exhausted" | "ok" | "unknown" | "warning", "exhausted" | "ok" | "unknown" | "warning">;
209
209
  export declare const usageWindowSchema: import("@oh-my-pi/omptype").FluentType<{
210
210
  durationMs?: number | undefined;
@@ -223,14 +223,14 @@ export declare const usageAmountSchema: import("@oh-my-pi/omptype").FluentType<{
223
223
  limit?: number | undefined;
224
224
  remaining?: number | undefined;
225
225
  remainingFraction?: number | undefined;
226
- unit: "bytes" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
226
+ unit: "bytes" | "credits" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
227
227
  used?: number | undefined;
228
228
  usedFraction?: number | undefined;
229
229
  }, {
230
230
  limit?: number | undefined;
231
231
  remaining?: number | undefined;
232
232
  remainingFraction?: number | undefined;
233
- unit: "bytes" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
233
+ unit: "bytes" | "credits" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
234
234
  used?: number | undefined;
235
235
  usedFraction?: number | undefined;
236
236
  }>;
@@ -258,7 +258,7 @@ export declare const usageLimitSchema: import("@oh-my-pi/omptype").FluentType<{
258
258
  limit?: number | undefined;
259
259
  remaining?: number | undefined;
260
260
  remainingFraction?: number | undefined;
261
- unit: "bytes" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
261
+ unit: "bytes" | "credits" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
262
262
  used?: number | undefined;
263
263
  usedFraction?: number | undefined;
264
264
  };
@@ -288,7 +288,7 @@ export declare const usageLimitSchema: import("@oh-my-pi/omptype").FluentType<{
288
288
  limit?: number | undefined;
289
289
  remaining?: number | undefined;
290
290
  remainingFraction?: number | undefined;
291
- unit: "bytes" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
291
+ unit: "bytes" | "credits" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
292
292
  used?: number | undefined;
293
293
  usedFraction?: number | undefined;
294
294
  };
@@ -345,7 +345,7 @@ export declare const usageReportSchema: import("@oh-my-pi/omptype").FluentType<{
345
345
  limit?: number | undefined;
346
346
  remaining?: number | undefined;
347
347
  remainingFraction?: number | undefined;
348
- unit: "bytes" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
348
+ unit: "bytes" | "credits" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
349
349
  used?: number | undefined;
350
350
  usedFraction?: number | undefined;
351
351
  };
@@ -390,7 +390,7 @@ export declare const usageReportSchema: import("@oh-my-pi/omptype").FluentType<{
390
390
  limit?: number | undefined;
391
391
  remaining?: number | undefined;
392
392
  remainingFraction?: number | undefined;
393
- unit: "bytes" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
393
+ unit: "bytes" | "credits" | "minutes" | "percent" | "requests" | "tokens" | "unknown" | "usd";
394
394
  used?: number | undefined;
395
395
  usedFraction?: number | undefined;
396
396
  };
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@oh-my-pi/pi-ai",
4
- "version": "18.0.7",
4
+ "version": "18.0.8",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://omp.sh",
7
7
  "author": "Stencil Labs, Inc.",
@@ -37,10 +37,10 @@
37
37
  "fmt": "biome format --write ."
38
38
  },
39
39
  "dependencies": {
40
- "@oh-my-pi/omptype": "18.0.7",
41
- "@oh-my-pi/pi-catalog": "18.0.7",
42
- "@oh-my-pi/pi-utils": "18.0.7",
43
- "@oh-my-pi/pi-wire": "18.0.7"
40
+ "@oh-my-pi/omptype": "18.0.8",
41
+ "@oh-my-pi/pi-catalog": "18.0.8",
42
+ "@oh-my-pi/pi-utils": "18.0.8",
43
+ "@oh-my-pi/pi-wire": "18.0.8"
44
44
  },
45
45
  "devDependencies": {
46
46
  "@types/bun": "^1.3.14"
@@ -208,7 +208,7 @@ const usageAmountSchema = type({
208
208
  "remaining?": "number",
209
209
  "usedFraction?": "number",
210
210
  "remainingFraction?": "number",
211
- unit: "'percent' | 'tokens' | 'requests' | 'usd' | 'minutes' | 'bytes' | 'unknown'",
211
+ unit: "'percent' | 'tokens' | 'requests' | 'credits' | 'usd' | 'minutes' | 'bytes' | 'unknown'",
212
212
  });
213
213
 
214
214
  const usageScopeSchema = type({
@@ -1089,6 +1089,15 @@ function buildAdditionalModelRequestFields(
1089
1089
  };
1090
1090
  }
1091
1091
 
1092
+ if (mode === "effort") {
1093
+ // OpenAI-schema models on Bedrock (the GPT-5.x SKUs) reject the
1094
+ // Anthropic budget block with `unknown_parameter: 'thinking'` and take
1095
+ // `reasoning.effort` instead — same effort vocabulary the catalog
1096
+ // already bakes (low/medium/high/xhigh/max).
1097
+ const level = requireSupportedEffort(model, reasoning);
1098
+ return { reasoning: { effort: model.thinking?.effortMap?.[level] ?? level } };
1099
+ }
1100
+
1092
1101
  const level = requireSupportedEffort(model, reasoning);
1093
1102
  const defaultBudgets: Record<Effort, number> = {
1094
1103
  minimal: 1024,
@@ -4844,6 +4844,27 @@ export function buildCursorSystemPromptJsons(systemPrompt: readonly string[] | u
4844
4844
  return systemPrompts.map(content => JSON.stringify({ role: "system", content }));
4845
4845
  }
4846
4846
 
4847
+ function collectCursorToolHistory(messages: Message[], historyEnd: number) {
4848
+ const toolResults = new Map<string, ToolResultMessage>();
4849
+ const pairedToolCallIds = new Set<string>();
4850
+ for (let index = 0; index < historyEnd; index++) {
4851
+ const message = messages[index];
4852
+ if (message.role === "toolResult") {
4853
+ toolResults.set(message.toolCallId, message);
4854
+ } else if (message.role === "assistant") {
4855
+ for (const item of message.content) {
4856
+ if (item.type === "toolCall") pairedToolCallIds.add(item.id);
4857
+ }
4858
+ }
4859
+ }
4860
+ return { toolResults, pairedToolCallIds };
4861
+ }
4862
+
4863
+ function cursorOrphanToolResultText(result: ToolResultMessage): string {
4864
+ const prefix = result.isError ? "[Tool Error]" : "[Tool Result]";
4865
+ return `${prefix}\n${toolResultToText(result) || "(empty result)"}`;
4866
+ }
4867
+
4847
4868
  function buildRootPromptMessagesJson(
4848
4869
  messages: Message[],
4849
4870
  systemPromptIds: Uint8Array[],
@@ -4852,6 +4873,8 @@ function buildRootPromptMessagesJson(
4852
4873
  targetModelId?: string,
4853
4874
  ): Uint8Array[] {
4854
4875
  assertCursorKimiK3HistoryReplayable(messages, activeUserMessageIndex, targetModelId);
4876
+ const historyEnd = activeUserMessageIndex >= 0 ? activeUserMessageIndex : messages.length;
4877
+ const { pairedToolCallIds } = collectCursorToolHistory(messages, historyEnd);
4855
4878
  const entries: Uint8Array[] = [...systemPromptIds];
4856
4879
  const pushJson = (obj: unknown) => {
4857
4880
  const bytes = new TextEncoder().encode(JSON.stringify(obj));
@@ -4870,6 +4893,13 @@ function buildRootPromptMessagesJson(
4870
4893
  if (content.length === 0) continue;
4871
4894
  pushJson({ role: "assistant", content });
4872
4895
  } else if (msg.role === "toolResult") {
4896
+ if (!pairedToolCallIds.has(msg.toolCallId)) {
4897
+ pushJson({
4898
+ role: "assistant",
4899
+ content: [{ type: "text", text: cursorOrphanToolResultText(msg) }],
4900
+ });
4901
+ continue;
4902
+ }
4873
4903
  // Emit even when the result text is empty: the assistant `tool-call` is
4874
4904
  // already in history, so dropping the pair would replay an orphaned call.
4875
4905
  const toolCallId = normalizeToolCallId(msg.toolCallId);
@@ -5008,18 +5038,7 @@ function buildConversationTurns(
5008
5038
  ): Uint8Array[] {
5009
5039
  const turns: Uint8Array[] = [];
5010
5040
  const historyEnd = activeUserMessageIndex >= 0 ? activeUserMessageIndex : messages.length;
5011
- const toolResults = new Map<string, ToolResultMessage>();
5012
- const pairedToolCallIds = new Set<string>();
5013
- for (let index = 0; index < historyEnd; index++) {
5014
- const message = messages[index];
5015
- if (message.role === "toolResult") {
5016
- toolResults.set(message.toolCallId, message);
5017
- } else if (message.role === "assistant") {
5018
- for (const item of message.content) {
5019
- if (item.type === "toolCall") pairedToolCallIds.add(item.id);
5020
- }
5021
- }
5022
- }
5041
+ const { toolResults, pairedToolCallIds } = collectCursorToolHistory(messages, historyEnd);
5023
5042
 
5024
5043
  let i = 0;
5025
5044
  while (i < messages.length) {
@@ -5077,17 +5096,13 @@ function buildConversationTurns(
5077
5096
  stepBlobIds.push(storeCursorBlob(blobStore, toBinary(ConversationStepSchema, step)));
5078
5097
  }
5079
5098
  } else if (stepMsg.role === "toolResult" && !pairedToolCallIds.has(stepMsg.toolCallId)) {
5080
- const text = toolResultToText(stepMsg);
5081
- if (text) {
5082
- const prefix = stepMsg.isError ? "[Tool Error]" : "[Tool Result]";
5083
- const step = create(ConversationStepSchema, {
5084
- message: {
5085
- case: "assistantMessage",
5086
- value: create(AssistantMessageSchema, { text: `${prefix}\n${text}` }),
5087
- },
5088
- });
5089
- stepBlobIds.push(storeCursorBlob(blobStore, toBinary(ConversationStepSchema, step)));
5090
- }
5099
+ const step = create(ConversationStepSchema, {
5100
+ message: {
5101
+ case: "assistantMessage",
5102
+ value: create(AssistantMessageSchema, { text: cursorOrphanToolResultText(stepMsg) }),
5103
+ },
5104
+ });
5105
+ stepBlobIds.push(storeCursorBlob(blobStore, toBinary(ConversationStepSchema, step)));
5091
5106
  }
5092
5107
  i++;
5093
5108
  }
package/src/stream.ts CHANGED
@@ -2080,8 +2080,8 @@ function mapOptionsForApi<TApi extends Api>(
2080
2080
  guardrailVersion: model.guardrailVersion ?? options?.guardrailVersion,
2081
2081
  guardrailTrace: model.guardrailTrace ?? options?.guardrailTrace,
2082
2082
  };
2083
- // Adaptive mode sends effort directly, no budget_tokens — skip budget inflation.
2084
- if (model.thinking?.mode === "anthropic-adaptive") {
2083
+ // Effort modes send effort directly, no budget_tokens — skip budget inflation.
2084
+ if (model.thinking?.mode === "effort" || model.thinking?.mode === "anthropic-adaptive") {
2085
2085
  return castApi<"bedrock-converse-stream">(bedrockBase);
2086
2086
  }
2087
2087
  const budgetInfo = resolveBedrockThinkingBudget(model as Model<"bedrock-converse-stream">, options);
package/src/types.ts CHANGED
@@ -879,6 +879,10 @@ export interface DeveloperMessage {
879
879
  content: string | (TextContent | ImageContent)[];
880
880
  /** Who initiated this message for billing/attribution semantics. */
881
881
  attribution?: MessageAttribution;
882
+ /** True if the message was injected by the system (e.g., auto-continue) and initiates a fresh run rather than continuing the current one. */
883
+ synthetic?: boolean;
884
+ /** True when the synthetic prompt was a deliberate operator action (`.`, `c` continue shortcut) rather than an automatic continuation — its timestamp is the turn's prompt time. */
885
+ userInitiated?: boolean;
882
886
  /** Provider-specific opaque payload used to reconstruct transport-native history. */
883
887
  providerPayload?: ProviderPayload;
884
888
  timestamp: number; // Unix timestamp in milliseconds
@@ -976,6 +980,8 @@ export interface AssistantMessage {
976
980
  timestamp: number; // Unix timestamp in milliseconds
977
981
  duration?: number; // Request duration in milliseconds
978
982
  ttft?: number; // Time to first token in milliseconds
983
+ /** Local wall-clock time the response finished streaming (ms since epoch); stamped by the session at message_end so prompt→yield timing never depends on provider-reported duration. */
984
+ completedAt?: number;
979
985
  }
980
986
 
981
987
  export interface ToolResultMessage<TDetails = unknown> {
package/src/usage/zai.ts CHANGED
@@ -51,6 +51,8 @@ interface ZaiQuotaPayload {
51
51
  msg?: string;
52
52
  data?: {
53
53
  limits?: ZaiUsageLimitItem[];
54
+ /** Coding-plan tier (e.g. "lite", "pro", "max") surfaced as the plan label. */
55
+ level?: string;
54
56
  };
55
57
  }
56
58
 
@@ -193,17 +195,33 @@ function buildModelUsageUrl(baseUrl: string, now: Date): string {
193
195
  }
194
196
 
195
197
  function getZaiCredentialLimits(report: UsageReport): UsageLimit[] {
196
- const limits = report.limits.filter(
197
- limit => limit.id.startsWith("zai:requests:") || limit.id.startsWith("zai:tokens:"),
198
+ return report.limits.filter(
199
+ limit =>
200
+ limit.id.startsWith("zai:requests:") ||
201
+ limit.id.startsWith("zai:tokens:") ||
202
+ limit.id.startsWith("zai:credits:"),
198
203
  );
199
- return limits;
204
+ }
205
+
206
+ function zaiLimitPressure(limit: UsageLimit): number {
207
+ const fraction = limit.amount.usedFraction;
208
+ return typeof fraction === "number" && Number.isFinite(fraction) ? fraction : -1;
200
209
  }
201
210
 
202
211
  function rankZaiRequestLimits(report: UsageReport): UsageLimit[] {
203
212
  const requestLimits = report.limits.filter(limit => limit.id.startsWith("zai:requests:"));
204
213
  const credentialLimits = getZaiCredentialLimits(report);
205
214
  const limits = requestLimits.length > 0 ? requestLimits : credentialLimits;
206
- const ranked = [...limits];
215
+ // Mixed-meter payloads (tokens + credits on the same plan) can repeat a
216
+ // window; keep the most-binding limit per window so a second 5h row never
217
+ // displaces the weekly window when primary/secondary are picked positionally.
218
+ const byWindow = new Map<number, UsageLimit>();
219
+ for (const limit of limits) {
220
+ const durationMs = limit.window?.durationMs ?? Number.POSITIVE_INFINITY;
221
+ const current = byWindow.get(durationMs);
222
+ if (!current || zaiLimitPressure(limit) > zaiLimitPressure(current)) byWindow.set(durationMs, limit);
223
+ }
224
+ const ranked = [...byWindow.values()];
207
225
  ranked.sort((left, right) => {
208
226
  const leftDuration = left.window?.durationMs ?? Number.POSITIVE_INFINITY;
209
227
  const rightDuration = right.window?.durationMs ?? Number.POSITIVE_INFINITY;
@@ -306,6 +324,33 @@ async function fetchZaiUsage(params: UsageFetchParams, ctx: UsageFetchContext):
306
324
  status: getUsageStatus(amount.usedFraction),
307
325
  });
308
326
  }
327
+ if (parsed.type === "CREDIT_LIMIT") {
328
+ // GLM Coding Plan windows (e.g. 12k credits / 5h + 60k credits / week):
329
+ // `usage` is the plan's credit allotment, `currentValue` the spend.
330
+ // `percentage` is a server-rounded integer (11 for 1438/12000 ≈ 11.98%),
331
+ // so prefer the exact ratio and fall back to it only without absolutes.
332
+ const window = buildZaiWindow(parsed);
333
+ const hasAbsoluteMeter = parsed.currentValue !== undefined && parsed.usage !== undefined && parsed.usage > 0;
334
+ const amount = buildUsageAmount({
335
+ used: parsed.currentValue,
336
+ limit: parsed.usage,
337
+ remaining: parsed.remaining,
338
+ percentage: hasAbsoluteMeter ? undefined : parsed.percentage,
339
+ unit: "credits",
340
+ });
341
+ limits.push({
342
+ id: `zai:credits:${window.id}`,
343
+ label: `ZAI ${window.label} Credit Quota`,
344
+ scope: {
345
+ provider: params.provider,
346
+ windowId: window.id,
347
+ shared: true,
348
+ },
349
+ window,
350
+ amount,
351
+ status: getUsageStatus(amount.usedFraction),
352
+ });
353
+ }
309
354
  }
310
355
 
311
356
  if (limits.length === 0) return null;
@@ -318,6 +363,7 @@ async function fetchZaiUsage(params: UsageFetchParams, ctx: UsageFetchContext):
318
363
  endpoint: url,
319
364
  accountId: credential.accountId,
320
365
  email: credential.email,
366
+ ...(typeof payload.data?.level === "string" && payload.data.level ? { planType: payload.data.level } : {}),
321
367
  },
322
368
  raw: payload,
323
369
  };
package/src/usage.ts CHANGED
@@ -6,7 +6,7 @@
6
6
  */
7
7
  import { type } from "@oh-my-pi/omptype";
8
8
  import type { FetchImpl, Provider } from "./types";
9
- export type UsageUnit = "percent" | "tokens" | "requests" | "usd" | "minutes" | "bytes" | "unknown";
9
+ export type UsageUnit = "percent" | "tokens" | "requests" | "credits" | "usd" | "minutes" | "bytes" | "unknown";
10
10
 
11
11
  export type UsageStatus = "ok" | "warning" | "exhausted" | "unknown";
12
12
 
@@ -240,7 +240,9 @@ export interface ClientUsageSummary {
240
240
 
241
241
  // ─── Zod schemas (wire-shape validation for the broker `/v1/usage` endpoint) ─
242
242
 
243
- export const usageUnitSchema = type("'percent' | 'tokens' | 'requests' | 'usd' | 'minutes' | 'bytes' | 'unknown'");
243
+ export const usageUnitSchema = type(
244
+ "'percent' | 'tokens' | 'requests' | 'credits' | 'usd' | 'minutes' | 'bytes' | 'unknown'",
245
+ );
244
246
  export const usageStatusSchema = type("'ok' | 'warning' | 'exhausted' | 'unknown'");
245
247
 
246
248
  export const usageWindowSchema = type({