@prestyj/ai 5.10.0 → 5.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -77,19 +77,31 @@ interface RawContent {
77
77
  data: Record<string, unknown>;
78
78
  }
79
79
  type ContentPart = TextContent | ThinkingContent | ImageContent | VideoContent | ToolCall | ServerToolCall | ServerToolResult | RawContent;
80
- interface SystemMessage {
80
+ type MessageProvenanceSource = "human" | "agent" | "runtime";
81
+ type MessageProvenanceKind = "prompt" | "steering" | "notification" | "completion_gate" | "review_follow_up" | "continuation" | "model_switch" | "automation" | "compaction_summary" | "compaction_ack";
82
+ type MessageProvenanceVisibility = "transcript" | "hidden" | "summary";
83
+ /** Internal message metadata. `stream()` removes it before provider dispatch. */
84
+ interface MessageProvenance {
85
+ source: MessageProvenanceSource;
86
+ kind: MessageProvenanceKind;
87
+ visibility: MessageProvenanceVisibility;
88
+ }
89
+ interface MessageMetadata {
90
+ provenance?: MessageProvenance;
91
+ }
92
+ interface SystemMessage extends MessageMetadata {
81
93
  role: "system";
82
94
  content: string;
83
95
  }
84
- interface UserMessage {
96
+ interface UserMessage extends MessageMetadata {
85
97
  role: "user";
86
98
  content: string | (TextContent | ImageContent | VideoContent)[];
87
99
  }
88
- interface AssistantMessage {
100
+ interface AssistantMessage extends MessageMetadata {
89
101
  role: "assistant";
90
102
  content: string | ContentPart[];
91
103
  }
92
- interface ToolResultMessage {
104
+ interface ToolResultMessage extends MessageMetadata {
93
105
  role: "tool";
94
106
  content: ToolResult[];
95
107
  }
@@ -161,7 +173,10 @@ interface StreamResponse {
161
173
  }
162
174
  interface Usage {
163
175
  inputTokens: number;
176
+ /** Total billed output tokens, including reasoning tokens when the provider reports them separately. */
164
177
  outputTokens: number;
178
+ /** Reasoning/thinking-token subset of outputTokens. */
179
+ reasoningTokens?: number;
165
180
  cacheRead?: number;
166
181
  cacheWrite?: number;
167
182
  serverToolUse?: {
@@ -299,7 +314,7 @@ declare class StreamResult implements AsyncIterable<StreamEvent> {
299
314
  * Local model ids are namespaced by endpoint (`local/<endpointId>/<rawId>`) so
300
315
  * the same model name served by two machines stays distinct in the registry.
301
316
  * The server only knows the raw id, so strip the routing prefix here — at the
302
- * one place that talks to the wire. Counterpart to gg-core's
317
+ * one place that talks to the wire. Counterpart to @prestyj/core's
303
318
  * `formatLocalModelId`/`parseLocalModelId`.
304
319
  */
305
320
  declare function localWireModelId(id: string): string;
@@ -368,7 +383,7 @@ declare class ProviderRegistryImpl {
368
383
  declare const providerRegistry: ProviderRegistryImpl;
369
384
 
370
385
  /**
371
- * Error model for gg-ai and downstream consumers.
386
+ * Error model for @prestyj/ai and downstream consumers.
372
387
  *
373
388
  * Every error users see should answer one question: "is this me or them?"
374
389
  * That answer drives whether they retry, switch model, log in, or report a
@@ -435,7 +450,7 @@ declare class ProviderError extends EZCoderAIError {
435
450
  * transient per-minute throttle)? These don't clear with a quick retry — the
436
451
  * user has to wait for the window to reset — so callers must surface them as a
437
452
  * hard stop, not silently retry for minutes. Detected from the canonical
438
- * "usage limit reached" message gg-ai stamps onto the ProviderError.
453
+ * "usage limit reached" message @prestyj/ai stamps onto the ProviderError.
439
454
  */
440
455
  declare function isUsageLimitError(err: unknown): boolean;
441
456
  /**
@@ -490,8 +505,23 @@ declare function redactText(text: string, options?: RedactionOptions): string;
490
505
  */
491
506
  declare function redactValue<T>(value: T, options?: RedactionOptions): T;
492
507
 
508
+ /** True when the string contains at least one unpaired surrogate. */
509
+ declare function hasLoneSurrogate(text: string): boolean;
510
+ /** Replace unpaired surrogates with U+FFFD; returns the input when already valid. */
511
+ declare function toWellFormedText(text: string): string;
512
+ /** `text.slice(0, chars)` that never cuts an astral character in half. */
513
+ declare function sliceHead(text: string, chars: number): string;
514
+ /** `text.slice(-chars)` that never cuts an astral character in half. */
515
+ declare function sliceTail(text: string, chars: number): string;
493
516
  /**
494
- * Provider-level diagnostic hook. Mirrors the pattern used by gg-agent's
517
+ * Strip unpaired surrogates from everything headed for the wire. Returns the
518
+ * same array (and same message objects) when the history is already valid, so
519
+ * the clean path stays allocation-free.
520
+ */
521
+ declare function sanitizeMessagesForWire(messages: Message[]): Message[];
522
+
523
+ /**
524
+ * Provider-level diagnostic hook. Mirrors the pattern used by @prestyj/agent's
495
525
  * setStreamDiagnostic — the host app wires a callback (typically writing to
496
526
  * a debug log) and providers call `providerDiag(...)` to record interesting
497
527
  * lifecycle events (e.g. raw SSE event types and timings).
@@ -612,4 +642,4 @@ interface PalsuProviderConfig {
612
642
  */
613
643
  declare function registerPalsuProvider(config?: PalsuProviderConfig): PalsuProviderHandle;
614
644
 
615
- export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, EZCoderAIError, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, localWireModelId, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
645
+ export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, EZCoderAIError, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, type ImageContent, type Message, type MessageProvenance, type MessageProvenanceKind, type MessageProvenanceSource, type MessageProvenanceVisibility, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, hasLoneSurrogate, isHardBillingMessage, isUsageLimitError, localWireModelId, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, sanitizeMessagesForWire, setProviderDiagnostic, sliceHead, sliceTail, stream, toAnthropicMessages, toOpenAIMessages, toWellFormedText };
package/dist/index.d.ts CHANGED
@@ -77,19 +77,31 @@ interface RawContent {
77
77
  data: Record<string, unknown>;
78
78
  }
79
79
  type ContentPart = TextContent | ThinkingContent | ImageContent | VideoContent | ToolCall | ServerToolCall | ServerToolResult | RawContent;
80
- interface SystemMessage {
80
+ type MessageProvenanceSource = "human" | "agent" | "runtime";
81
+ type MessageProvenanceKind = "prompt" | "steering" | "notification" | "completion_gate" | "review_follow_up" | "continuation" | "model_switch" | "automation" | "compaction_summary" | "compaction_ack";
82
+ type MessageProvenanceVisibility = "transcript" | "hidden" | "summary";
83
+ /** Internal message metadata. `stream()` removes it before provider dispatch. */
84
+ interface MessageProvenance {
85
+ source: MessageProvenanceSource;
86
+ kind: MessageProvenanceKind;
87
+ visibility: MessageProvenanceVisibility;
88
+ }
89
+ interface MessageMetadata {
90
+ provenance?: MessageProvenance;
91
+ }
92
+ interface SystemMessage extends MessageMetadata {
81
93
  role: "system";
82
94
  content: string;
83
95
  }
84
- interface UserMessage {
96
+ interface UserMessage extends MessageMetadata {
85
97
  role: "user";
86
98
  content: string | (TextContent | ImageContent | VideoContent)[];
87
99
  }
88
- interface AssistantMessage {
100
+ interface AssistantMessage extends MessageMetadata {
89
101
  role: "assistant";
90
102
  content: string | ContentPart[];
91
103
  }
92
- interface ToolResultMessage {
104
+ interface ToolResultMessage extends MessageMetadata {
93
105
  role: "tool";
94
106
  content: ToolResult[];
95
107
  }
@@ -161,7 +173,10 @@ interface StreamResponse {
161
173
  }
162
174
  interface Usage {
163
175
  inputTokens: number;
176
+ /** Total billed output tokens, including reasoning tokens when the provider reports them separately. */
164
177
  outputTokens: number;
178
+ /** Reasoning/thinking-token subset of outputTokens. */
179
+ reasoningTokens?: number;
165
180
  cacheRead?: number;
166
181
  cacheWrite?: number;
167
182
  serverToolUse?: {
@@ -299,7 +314,7 @@ declare class StreamResult implements AsyncIterable<StreamEvent> {
299
314
  * Local model ids are namespaced by endpoint (`local/<endpointId>/<rawId>`) so
300
315
  * the same model name served by two machines stays distinct in the registry.
301
316
  * The server only knows the raw id, so strip the routing prefix here — at the
302
- * one place that talks to the wire. Counterpart to gg-core's
317
+ * one place that talks to the wire. Counterpart to @prestyj/core's
303
318
  * `formatLocalModelId`/`parseLocalModelId`.
304
319
  */
305
320
  declare function localWireModelId(id: string): string;
@@ -368,7 +383,7 @@ declare class ProviderRegistryImpl {
368
383
  declare const providerRegistry: ProviderRegistryImpl;
369
384
 
370
385
  /**
371
- * Error model for gg-ai and downstream consumers.
386
+ * Error model for @prestyj/ai and downstream consumers.
372
387
  *
373
388
  * Every error users see should answer one question: "is this me or them?"
374
389
  * That answer drives whether they retry, switch model, log in, or report a
@@ -435,7 +450,7 @@ declare class ProviderError extends EZCoderAIError {
435
450
  * transient per-minute throttle)? These don't clear with a quick retry — the
436
451
  * user has to wait for the window to reset — so callers must surface them as a
437
452
  * hard stop, not silently retry for minutes. Detected from the canonical
438
- * "usage limit reached" message gg-ai stamps onto the ProviderError.
453
+ * "usage limit reached" message @prestyj/ai stamps onto the ProviderError.
439
454
  */
440
455
  declare function isUsageLimitError(err: unknown): boolean;
441
456
  /**
@@ -490,8 +505,23 @@ declare function redactText(text: string, options?: RedactionOptions): string;
490
505
  */
491
506
  declare function redactValue<T>(value: T, options?: RedactionOptions): T;
492
507
 
508
+ /** True when the string contains at least one unpaired surrogate. */
509
+ declare function hasLoneSurrogate(text: string): boolean;
510
+ /** Replace unpaired surrogates with U+FFFD; returns the input when already valid. */
511
+ declare function toWellFormedText(text: string): string;
512
+ /** `text.slice(0, chars)` that never cuts an astral character in half. */
513
+ declare function sliceHead(text: string, chars: number): string;
514
+ /** `text.slice(-chars)` that never cuts an astral character in half. */
515
+ declare function sliceTail(text: string, chars: number): string;
493
516
  /**
494
- * Provider-level diagnostic hook. Mirrors the pattern used by gg-agent's
517
+ * Strip unpaired surrogates from everything headed for the wire. Returns the
518
+ * same array (and same message objects) when the history is already valid, so
519
+ * the clean path stays allocation-free.
520
+ */
521
+ declare function sanitizeMessagesForWire(messages: Message[]): Message[];
522
+
523
+ /**
524
+ * Provider-level diagnostic hook. Mirrors the pattern used by @prestyj/agent's
495
525
  * setStreamDiagnostic — the host app wires a callback (typically writing to
496
526
  * a debug log) and providers call `providerDiag(...)` to record interesting
497
527
  * lifecycle events (e.g. raw SSE event types and timings).
@@ -612,4 +642,4 @@ interface PalsuProviderConfig {
612
642
  */
613
643
  declare function registerPalsuProvider(config?: PalsuProviderConfig): PalsuProviderHandle;
614
644
 
615
- export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, EZCoderAIError, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, localWireModelId, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
645
+ export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, EZCoderAIError, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, type ImageContent, type Message, type MessageProvenance, type MessageProvenanceKind, type MessageProvenanceSource, type MessageProvenanceVisibility, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, hasLoneSurrogate, isHardBillingMessage, isUsageLimitError, localWireModelId, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, sanitizeMessagesForWire, setProviderDiagnostic, sliceHead, sliceTail, stream, toAnthropicMessages, toOpenAIMessages, toWellFormedText };
package/dist/index.js CHANGED
@@ -1160,7 +1160,7 @@ function parseToolArguments(argsJson) {
1160
1160
  var NON_STREAMING_TIMEOUT_MS = 60 * 60 * 1e3;
1161
1161
  var anthropicClientCache = /* @__PURE__ */ new Map();
1162
1162
  function fineGrainedToolStreamingEnabled() {
1163
- const raw = process.env.GG_FINE_GRAINED_TOOL_STREAMING ?? process.env.CLAUDE_CODE_ENABLE_FINE_GRAINED_TOOL_STREAMING;
1163
+ const raw = process.env.EZ_FINE_GRAINED_TOOL_STREAMING ?? process.env.CLAUDE_CODE_ENABLE_FINE_GRAINED_TOOL_STREAMING;
1164
1164
  if (!raw) return false;
1165
1165
  const v = raw.trim().toLowerCase();
1166
1166
  return v === "1" || v === "true" || v === "yes" || v === "on";
@@ -2025,7 +2025,15 @@ async function* runStream2(options) {
2025
2025
  if (chunk.usage) {
2026
2026
  ({ inputTokens, outputTokens, cacheRead, cacheWrite } = extractOpenAIUsage(chunk.usage));
2027
2027
  }
2028
- if (!choice) continue;
2028
+ if (!choice) {
2029
+ const gatewayError = classifyChoicelessFrame(chunk);
2030
+ if (gatewayError) {
2031
+ throw new ProviderError(providerName, gatewayError.message, {
2032
+ statusCode: gatewayError.statusCode
2033
+ });
2034
+ }
2035
+ continue;
2036
+ }
2029
2037
  if (choice.finish_reason) {
2030
2038
  finishReason = choice.finish_reason;
2031
2039
  }
@@ -2209,6 +2217,31 @@ function completionToResponse(completion, endpointKey) {
2209
2217
  }
2210
2218
  };
2211
2219
  }
2220
+ function classifyChoicelessFrame(frame) {
2221
+ if (!frame || typeof frame !== "object" || Array.isArray(frame)) return null;
2222
+ const rec = frame;
2223
+ if (Array.isArray(rec.choices)) return null;
2224
+ const statusOf = (value) => {
2225
+ const n = typeof value === "string" ? Number(value) : value;
2226
+ return typeof n === "number" && Number.isFinite(n) && n >= 400 && n <= 599 ? n : void 0;
2227
+ };
2228
+ const statusCode = statusOf(rec.status) ?? statusOf(rec.statusCode) ?? statusOf(rec.code);
2229
+ const typeIsError = typeof rec.type === "string" && rec.type.toLowerCase() === "error";
2230
+ let detailText;
2231
+ const detail = rec.detail;
2232
+ if (typeof detail === "string" && detail.trim()) {
2233
+ detailText = detail.trim();
2234
+ } else if (Array.isArray(detail)) {
2235
+ const parts = detail.map(
2236
+ (d) => d && typeof d === "object" && typeof d.msg === "string" ? d.msg : typeof d === "string" ? d : ""
2237
+ ).filter(Boolean);
2238
+ if (parts.length) detailText = parts.join("; ");
2239
+ }
2240
+ if (statusCode === void 0 && !typeIsError && !detailText) return null;
2241
+ const rawMessage = (typeof rec.message === "string" && rec.message.trim() ? rec.message.trim() : void 0) ?? detailText ?? (typeof rec.error === "string" && rec.error.trim() ? rec.error.trim() : void 0) ?? (statusCode !== void 0 ? `Gateway returned status ${statusCode}.` : "Gateway error.");
2242
+ const message = rawMessage.slice(0, 500);
2243
+ return { message, statusCode };
2244
+ }
2212
2245
  function classifyOpenAICompatLimit(args) {
2213
2246
  const { status, code, type, message } = args;
2214
2247
  const codeType = `${code ?? ""} ${type ?? ""}`.toLowerCase();
@@ -2220,6 +2253,7 @@ function classifyOpenAICompatLimit(args) {
2220
2253
  return null;
2221
2254
  }
2222
2255
  function toError2(err, provider = "openai") {
2256
+ if (err instanceof ProviderError) return err;
2223
2257
  if (err instanceof OpenAI.APIError) {
2224
2258
  const body = err.error;
2225
2259
  const bodyMessage = typeof body?.message === "string" && body.message.trim() ? body.message.trim() : void 0;
@@ -3319,14 +3353,16 @@ async function* runStream4(options) {
3319
3353
  let thinkingAccum = "";
3320
3354
  let stopReason = "end_turn";
3321
3355
  let inputTokens = 0;
3322
- let outputTokens = 0;
3356
+ let candidateTokens = 0;
3357
+ let reasoningTokens = 0;
3323
3358
  let cacheRead = 0;
3324
3359
  let toolIndex = 0;
3325
3360
  const handleResponse = function* (chunk) {
3326
3361
  const usage = usageFromResponse(chunk);
3327
3362
  if (usage) {
3328
3363
  inputTokens = usage.promptTokenCount ?? inputTokens;
3329
- outputTokens = usage.candidatesTokenCount ?? outputTokens;
3364
+ candidateTokens = usage.candidatesTokenCount ?? candidateTokens;
3365
+ reasoningTokens = usage.thoughtsTokenCount ?? reasoningTokens;
3330
3366
  cacheRead = usage.cachedContentTokenCount ?? cacheRead;
3331
3367
  }
3332
3368
  const reason = finishReasonFromResponse(chunk);
@@ -3382,6 +3418,7 @@ async function* runStream4(options) {
3382
3418
  }
3383
3419
  if (pendingToolCalls.length > 0) stopReason = "tool_use";
3384
3420
  const adjustedInputTokens = Math.max(0, inputTokens - cacheRead);
3421
+ const outputTokens = candidateTokens + reasoningTokens;
3385
3422
  const streamResponse = {
3386
3423
  message: {
3387
3424
  role: "assistant",
@@ -3391,6 +3428,7 @@ async function* runStream4(options) {
3391
3428
  usage: {
3392
3429
  inputTokens: adjustedInputTokens,
3393
3430
  outputTokens,
3431
+ ...reasoningTokens > 0 ? { reasoningTokens } : {},
3394
3432
  ...cacheRead > 0 ? { cacheRead } : {}
3395
3433
  }
3396
3434
  };
@@ -3439,9 +3477,145 @@ var ProviderRegistryImpl = class {
3439
3477
  };
3440
3478
  var providerRegistry = new ProviderRegistryImpl();
3441
3479
 
3480
+ // src/utils/well-formed.ts
3481
+ var LONE_SURROGATE = /[\uD800-\uDBFF](?![\uDC00-\uDFFF])|(?<![\uD800-\uDBFF])[\uDC00-\uDFFF]/;
3482
+ var LONE_SURROGATE_GLOBAL = new RegExp(LONE_SURROGATE, "g");
3483
+ var REPLACEMENT = "\uFFFD";
3484
+ function hasLoneSurrogate(text) {
3485
+ const isWellFormed = text.isWellFormed;
3486
+ if (typeof isWellFormed === "function") return !isWellFormed.call(text);
3487
+ return LONE_SURROGATE.test(text);
3488
+ }
3489
+ function toWellFormedText(text) {
3490
+ if (!hasLoneSurrogate(text)) return text;
3491
+ const toWellFormed = text.toWellFormed;
3492
+ if (typeof toWellFormed === "function") return toWellFormed.call(text);
3493
+ return text.replace(LONE_SURROGATE_GLOBAL, REPLACEMENT);
3494
+ }
3495
+ function isHighSurrogate(code) {
3496
+ return code !== void 0 && code >= 55296 && code <= 56319;
3497
+ }
3498
+ function isLowSurrogate(code) {
3499
+ return code !== void 0 && code >= 56320 && code <= 57343;
3500
+ }
3501
+ function sliceHead(text, chars) {
3502
+ if (chars <= 0) return "";
3503
+ if (chars >= text.length) return text;
3504
+ const end = isHighSurrogate(text.charCodeAt(chars - 1)) ? chars - 1 : chars;
3505
+ return text.slice(0, end);
3506
+ }
3507
+ function sliceTail(text, chars) {
3508
+ if (chars <= 0) return "";
3509
+ if (chars >= text.length) return text;
3510
+ const start = text.length - chars;
3511
+ return text.slice(isLowSurrogate(text.charCodeAt(start)) ? start + 1 : start);
3512
+ }
3513
+ function sanitizeJsonValue(value) {
3514
+ if (typeof value === "string") return toWellFormedText(value);
3515
+ if (Array.isArray(value)) {
3516
+ let changed = false;
3517
+ const next = value.map((item) => {
3518
+ const sanitized = sanitizeJsonValue(item);
3519
+ if (sanitized !== item) changed = true;
3520
+ return sanitized;
3521
+ });
3522
+ return changed ? next : value;
3523
+ }
3524
+ if (value !== null && typeof value === "object") {
3525
+ let changed = false;
3526
+ const next = {};
3527
+ for (const [key, item] of Object.entries(value)) {
3528
+ const sanitizedKey = toWellFormedText(key);
3529
+ const sanitized = sanitizeJsonValue(item);
3530
+ if (sanitizedKey !== key || sanitized !== item) changed = true;
3531
+ next[sanitizedKey] = sanitized;
3532
+ }
3533
+ return changed ? next : value;
3534
+ }
3535
+ return value;
3536
+ }
3537
+ function sanitizeRecord(value) {
3538
+ return sanitizeJsonValue(value);
3539
+ }
3540
+ function sanitizePart(part) {
3541
+ switch (part.type) {
3542
+ case "text":
3543
+ case "thinking": {
3544
+ const text = toWellFormedText(part.text);
3545
+ return text === part.text ? part : { ...part, text };
3546
+ }
3547
+ case "tool_call": {
3548
+ const args = sanitizeRecord(part.args);
3549
+ return args === part.args ? part : { ...part, args };
3550
+ }
3551
+ case "server_tool_call": {
3552
+ const input = sanitizeJsonValue(part.input);
3553
+ return input === part.input ? part : { ...part, input };
3554
+ }
3555
+ case "server_tool_result": {
3556
+ const data = sanitizeJsonValue(part.data);
3557
+ return data === part.data ? part : { ...part, data };
3558
+ }
3559
+ case "raw": {
3560
+ const data = sanitizeRecord(part.data);
3561
+ return data === part.data ? part : { ...part, data };
3562
+ }
3563
+ default:
3564
+ return part;
3565
+ }
3566
+ }
3567
+ function sanitizeParts(parts) {
3568
+ let changed = false;
3569
+ const next = parts.map((part) => {
3570
+ const sanitized = sanitizePart(part);
3571
+ if (sanitized !== part) changed = true;
3572
+ return sanitized;
3573
+ });
3574
+ return changed ? next : parts;
3575
+ }
3576
+ function sanitizeToolResultContent(content) {
3577
+ if (typeof content === "string") return toWellFormedText(content);
3578
+ return sanitizeParts(content);
3579
+ }
3580
+ function sanitizeToolResults(results) {
3581
+ let changed = false;
3582
+ const next = results.map((result) => {
3583
+ const content = sanitizeToolResultContent(result.content);
3584
+ if (content === result.content) return result;
3585
+ changed = true;
3586
+ return { ...result, content };
3587
+ });
3588
+ return changed ? next : results;
3589
+ }
3590
+ function sanitizeMessage(message) {
3591
+ if (message.role === "tool") {
3592
+ const content2 = sanitizeToolResults(message.content);
3593
+ return content2 === message.content ? message : { ...message, content: content2 };
3594
+ }
3595
+ if (typeof message.content === "string") {
3596
+ const content2 = toWellFormedText(message.content);
3597
+ return content2 === message.content ? message : { ...message, content: content2 };
3598
+ }
3599
+ const content = sanitizeParts(message.content);
3600
+ return content === message.content ? message : { ...message, content };
3601
+ }
3602
+ function sanitizeMessagesForWire(messages) {
3603
+ let sanitized;
3604
+ for (let index = 0; index < messages.length; index++) {
3605
+ const message = messages[index];
3606
+ const next = sanitizeMessage(message);
3607
+ if (next === message) continue;
3608
+ sanitized ??= messages.slice();
3609
+ sanitized[index] = next;
3610
+ }
3611
+ return sanitized ?? messages;
3612
+ }
3613
+
3442
3614
  // src/stream.ts
3443
3615
  var GLM_CODING_BASE_URL = "https://api.z.ai/api/coding/paas/v4";
3444
3616
  var KIMI_CODE_USER_AGENT = `kimi-code-cli/${process.env.KIMI_CODE_VERSION ?? "1.0.11"}`;
3617
+ var GROK_CLI_PROXY_HOST = "cli-chat-proxy.grok.com";
3618
+ var GROK_CLI_VERSION = process.env.GROK_CLI_VERSION ?? "0.2.101";
3445
3619
  providerRegistry.register("anthropic", {
3446
3620
  stream: (options) => streamAnthropic(options)
3447
3621
  });
@@ -3508,13 +3682,25 @@ providerRegistry.register("xai", {
3508
3682
  // xAI's public API (console.x.ai key) is OpenAI-compatible — ride the Chat
3509
3683
  // Completions transport like Moonshot/DeepSeek. Grok reasoning models take
3510
3684
  // top-level `reasoning_effort` (low/medium/high), which the shared thinking
3511
- // path already sends. xAI's OAuth path exists but only via the Grok CLI's
3512
- // private Responses proxy (cli-chat-proxy.grok.com) with reverse-engineered
3513
- // attribution headers and account-tier gating — intentionally not wired.
3514
- stream: (options) => streamOpenAI({
3515
- ...options,
3516
- baseUrl: options.baseUrl ?? "https://api.x.ai/v1"
3517
- })
3685
+ // path already sends.
3686
+ //
3687
+ // Subscription OAuth (SuperGrok / X Premium) routes to the Grok CLI chat proxy
3688
+ // instead, which speaks the same Chat Completions wire but gates on Grok-CLI
3689
+ // client identity. Inject those headers centrally here — exactly as the Kimi
3690
+ // endpoint above — so EVERY stream (agent loop, compaction, title-gen,
3691
+ // sub-agents) is accepted rather than depending on each call site to thread
3692
+ // headers. Caller-provided headers still win on collision.
3693
+ stream: (options) => {
3694
+ const baseUrl = options.baseUrl ?? "https://api.x.ai/v1";
3695
+ const defaultHeaders = baseUrl.includes(GROK_CLI_PROXY_HOST) ? {
3696
+ "X-XAI-Token-Auth": "xai-grok-cli",
3697
+ "x-grok-client-version": GROK_CLI_VERSION,
3698
+ "x-grok-client-identifier": "ezcoder",
3699
+ "x-grok-model-override": options.model,
3700
+ ...options.defaultHeaders
3701
+ } : options.defaultHeaders;
3702
+ return streamOpenAI({ ...options, baseUrl, defaultHeaders });
3703
+ }
3518
3704
  });
3519
3705
  providerRegistry.register("minimax", {
3520
3706
  stream: (options) => streamAnthropic({
@@ -3560,13 +3746,25 @@ function stream(options) {
3560
3746
  if (options.supportsVideo !== true && messagesContainVideo(options.messages)) {
3561
3747
  throw new VideoUnsupportedError();
3562
3748
  }
3749
+ const wireMessages = stripMessageProvenance(options.messages);
3563
3750
  const messages = clampProviderContextImages(
3564
- options.messages,
3751
+ sanitizeMessagesForWire(wireMessages),
3565
3752
  options.provider,
3566
3753
  options.supportsImages
3567
3754
  );
3568
3755
  return entry.stream(messages === options.messages ? options : { ...options, messages });
3569
3756
  }
3757
+ function stripMessageProvenance(messages) {
3758
+ let stripped;
3759
+ for (let index = 0; index < messages.length; index++) {
3760
+ const message = messages[index];
3761
+ if (!message.provenance) continue;
3762
+ stripped ??= messages.slice();
3763
+ const { provenance: _provenance, ...wireMessage } = message;
3764
+ stripped[index] = wireMessage;
3765
+ }
3766
+ return stripped ?? messages;
3767
+ }
3570
3768
  function messagesContainVideo(messages) {
3571
3769
  for (const msg of messages) {
3572
3770
  if (typeof msg.content === "string" || !Array.isArray(msg.content)) continue;
@@ -3951,6 +4149,7 @@ export {
3951
4149
  environmentSecrets,
3952
4150
  formatError,
3953
4151
  formatErrorForDisplay,
4152
+ hasLoneSurrogate,
3954
4153
  isHardBillingMessage,
3955
4154
  isUsageLimitError,
3956
4155
  localWireModelId,
@@ -3963,9 +4162,13 @@ export {
3963
4162
  redactText,
3964
4163
  redactValue,
3965
4164
  registerPalsuProvider,
4165
+ sanitizeMessagesForWire,
3966
4166
  setProviderDiagnostic,
4167
+ sliceHead,
4168
+ sliceTail,
3967
4169
  stream,
3968
4170
  toAnthropicMessages,
3969
- toOpenAIMessages
4171
+ toOpenAIMessages,
4172
+ toWellFormedText
3970
4173
  };
3971
4174
  //# sourceMappingURL=index.js.map