@sayknow-cli/ai 0.3.0 → 0.3.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/CHANGELOG.md +36 -0
  2. package/dist/types/auth-gateway/http.d.ts +1 -0
  3. package/dist/types/model-thinking.d.ts +1 -1
  4. package/dist/types/provider-models/openai-compat.d.ts +5 -0
  5. package/dist/types/providers/google-gemini-cli.d.ts +4 -1
  6. package/dist/types/providers/google-gemini-headers.d.ts +27 -5
  7. package/dist/types/providers/mock.d.ts +2 -0
  8. package/dist/types/providers/openai-codex/request-transformer.d.ts +1 -0
  9. package/dist/types/providers/openai-responses-shared.d.ts +16 -1
  10. package/dist/types/types.d.ts +9 -1
  11. package/dist/types/utils/json-parse.d.ts +8 -0
  12. package/dist/types/utils/oauth/fugu.d.ts +1 -0
  13. package/dist/types/utils/oauth/index.d.ts +1 -0
  14. package/dist/types/utils/oauth/types.d.ts +1 -1
  15. package/dist/types/utils/overflow.d.ts +11 -1
  16. package/package.json +2 -2
  17. package/src/auth-gateway/http.ts +5 -1
  18. package/src/auth-gateway/server.ts +18 -1
  19. package/src/auth-storage.ts +41 -27
  20. package/src/model-thinking.ts +2 -9
  21. package/src/models.json +67 -1
  22. package/src/provider-models/descriptors.ts +7 -0
  23. package/src/provider-models/openai-compat.ts +9 -0
  24. package/src/providers/google-gemini-cli.ts +64 -18
  25. package/src/providers/google-gemini-headers.ts +72 -18
  26. package/src/providers/mock.ts +3 -0
  27. package/src/providers/openai-codex/request-transformer.ts +53 -0
  28. package/src/providers/openai-codex-responses.ts +24 -11
  29. package/src/providers/openai-completions-compat.ts +2 -1
  30. package/src/providers/openai-completions.ts +12 -1
  31. package/src/providers/openai-responses-shared.ts +43 -1
  32. package/src/stream.ts +1 -0
  33. package/src/types.ts +9 -0
  34. package/src/utils/json-parse.ts +18 -0
  35. package/src/utils/oauth/fugu.ts +15 -0
  36. package/src/utils/oauth/index.ts +9 -0
  37. package/src/utils/oauth/types.ts +1 -0
  38. package/src/utils/overflow.ts +35 -1
  39. package/src/utils.ts +87 -0
@@ -52,14 +52,26 @@ const OVERFLOW_PATTERNS = [
52
52
  /\b413\b.*\b(request|payload|entity)\b.*\btoo large\b/i, // "413 Request Entity Too Large" variants
53
53
  /model_context_window_exceeded/i, // z.ai non-standard finish_reason surfaced as error text
54
54
  ];
55
+ /**
56
+ * Threshold below which a "successful" (stopReason "stop") response with empty
57
+ * content is considered anomalous. Some proxies (notably LiteLLM) return an
58
+ * empty `choices[0].message.content` with a near-zero `usage` (e.g. input: 1,
59
+ * output: 1) when the upstream model context window is exceeded, instead of
60
+ * surfacing a proper error. The total token count for such a response is well
61
+ * below any realistic turn, so we treat it as a proxy-level overflow signal.
62
+ */
63
+ const EMPTY_RESPONSE_USAGE_THRESHOLD = 5;
55
64
  /**
56
65
  * Check if an assistant message represents a context overflow error.
57
66
  *
58
- * This handles two cases:
67
+ * This handles three cases:
59
68
  * 1. Error-based overflow: Most providers return stopReason "error" with a
60
69
  * specific error message pattern.
61
70
  * 2. Silent overflow: Some providers accept overflow requests and return
62
71
  * successfully. For these, we check if usage.input exceeds the context window.
72
+ * 3. Proxy-level overflow: Some proxies (e.g. LiteLLM) return a "successful"
73
+ * response with empty content and a fabricated near-zero usage when the
74
+ * upstream model's context window is exceeded.
63
75
  *
64
76
  * ## Reliability by Provider
65
77
  *
@@ -83,6 +95,13 @@ const OVERFLOW_PATTERNS = [
83
95
  * - z.ai: Sometimes accepts overflow silently (detectable via usage.input > contextWindow),
84
96
  * sometimes returns rate limit errors. Pass contextWindow param to detect silent overflow.
85
97
  * - Ollama: Silently truncates input without error. Cannot be detected via this function.
98
+ * - LiteLLM proxy: Returns a "successful" response with empty content and a
99
+ * fabricated near-zero usage (e.g. input: 1, output: 1) when the upstream
100
+ * model's context window is exceeded. Detected via Case 3 (empty content +
101
+ * anomalously low usage). Note: the LiteLLM proxy's context limit may differ
102
+ * from the underlying model's advertised contextWindow (e.g. configured via
103
+ * `model_info.max_tokens` in LiteLLM's config.yaml), so Case 2 (which compares
104
+ * usage.input against contextWindow) may not catch it.
86
105
  * The response will have usage.input < expected, but we don't know the expected value.
87
106
  *
88
107
  * ## Custom Providers
@@ -126,6 +145,21 @@ export function isContextOverflow(message: AssistantMessage, contextWindow?: num
126
145
  }
127
146
  }
128
147
 
148
+ // Case 3: Empty response with anomalously low usage (proxy-level overflow)
149
+ // Some proxies (e.g. LiteLLM) return a "successful" response (stopReason "stop")
150
+ // with empty content and a near-zero token count when the upstream model's
151
+ // context window is exceeded. This is distinct from silent overflow (Case 2),
152
+ // where the provider reports the real input token count. Here the proxy
153
+ // fabricates a bogus usage (input: 1, output: 1) that is far below any
154
+ // realistic turn, so we detect it heuristically.
155
+ if (
156
+ message.stopReason === "stop" &&
157
+ message.content.length === 0 &&
158
+ message.usage.input + message.usage.output <= EMPTY_RESPONSE_USAGE_THRESHOLD
159
+ ) {
160
+ return true;
161
+ }
162
+
129
163
  return false;
130
164
  }
131
165
 
package/src/utils.ts CHANGED
@@ -91,6 +91,92 @@ export function sanitizeOpenAIResponsesHistoryItemsForReplay(items: Array<Record
91
91
  return sanitized ? [sanitized] : [];
92
92
  });
93
93
  }
94
+ function stringifyResponsesStringParamForReplay(value: unknown): string {
95
+ if (typeof value === "string") return value.toWellFormed();
96
+ try {
97
+ const encoded = JSON.stringify(value);
98
+ if (typeof encoded === "string") return encoded.toWellFormed();
99
+ } catch {
100
+ // Fall through to String().
101
+ }
102
+ return String(value ?? "").toWellFormed();
103
+ }
104
+
105
+ function normalizeResponsesMessageTextForReplay(value: unknown): string {
106
+ if (typeof value === "string") return value.toWellFormed();
107
+ if (value && typeof value === "object") {
108
+ const nestedText = (value as { text?: unknown }).text;
109
+ if (typeof nestedText === "string") return nestedText.toWellFormed();
110
+ }
111
+ return stringifyResponsesStringParamForReplay(value);
112
+ }
113
+
114
+ type ResponsesImageDetail = "auto" | "low" | "high";
115
+
116
+ interface NormalizedResponsesImageUrl {
117
+ readonly imageUrl: string;
118
+ readonly detail?: ResponsesImageDetail;
119
+ }
120
+
121
+ function isResponsesImageDetail(value: unknown): value is ResponsesImageDetail {
122
+ return value === "auto" || value === "low" || value === "high";
123
+ }
124
+
125
+ function normalizeResponsesImageUrlForReplay(value: unknown): NormalizedResponsesImageUrl {
126
+ if (typeof value === "string") return { imageUrl: value.toWellFormed() };
127
+ if (value && typeof value === "object" && "url" in value && typeof value.url === "string") {
128
+ const detail = "detail" in value && isResponsesImageDetail(value.detail) ? value.detail : undefined;
129
+ return {
130
+ imageUrl: value.url.toWellFormed(),
131
+ ...(detail ? { detail } : {}),
132
+ };
133
+ }
134
+ return { imageUrl: stringifyResponsesStringParamForReplay(value) };
135
+ }
136
+
137
+ function sanitizeResponsesMessageContentForReplay(content: unknown): unknown {
138
+ if (typeof content === "string") return content.toWellFormed();
139
+ if (!Array.isArray(content)) return content;
140
+ return content.map(part => {
141
+ if (!part || typeof part !== "object") return part;
142
+ const sanitizedPart = { ...(part as Record<string, unknown>) };
143
+ if ("text" in sanitizedPart) {
144
+ sanitizedPart.text = normalizeResponsesMessageTextForReplay(sanitizedPart.text);
145
+ }
146
+ if ("image_url" in sanitizedPart) {
147
+ const normalizedImageUrl = normalizeResponsesImageUrlForReplay(sanitizedPart.image_url);
148
+ sanitizedPart.image_url = normalizedImageUrl.imageUrl;
149
+ if (sanitizedPart.type === "image_url") {
150
+ sanitizedPart.type = "input_image";
151
+ }
152
+ if (normalizedImageUrl.detail) {
153
+ sanitizedPart.detail = normalizedImageUrl.detail;
154
+ } else if ("detail" in sanitizedPart && !isResponsesImageDetail(sanitizedPart.detail)) {
155
+ delete sanitizedPart.detail;
156
+ }
157
+ }
158
+ return sanitizedPart;
159
+ });
160
+ }
161
+
162
+ function sanitizeResponsesStringFieldsForReplay(item: Record<string, unknown>): void {
163
+ if (item.type === "message") {
164
+ item.content = sanitizeResponsesMessageContentForReplay(item.content);
165
+ }
166
+ if (item.type === "function_call" && "arguments" in item && typeof item.arguments !== "string") {
167
+ item.arguments = stringifyResponsesStringParamForReplay(item.arguments);
168
+ }
169
+ if (item.type === "custom_tool_call" && "input" in item && typeof item.input !== "string") {
170
+ item.input = stringifyResponsesStringParamForReplay(item.input);
171
+ }
172
+ if (
173
+ (item.type === "function_call_output" || item.type === "custom_tool_call_output") &&
174
+ "output" in item &&
175
+ typeof item.output !== "string"
176
+ ) {
177
+ item.output = stringifyResponsesStringParamForReplay(item.output);
178
+ }
179
+ }
94
180
 
95
181
  function sanitizeOpenAIResponsesHistoryItemForReplay(
96
182
  item: Record<string, unknown>,
@@ -105,6 +191,7 @@ function sanitizeOpenAIResponsesHistoryItemForReplay(
105
191
  if (typeof item.call_id === "string") {
106
192
  sanitizedItem.call_id = normalizeReplayedResponsesHistoryCallId(item.call_id, normalizedCallIds);
107
193
  }
194
+ sanitizeResponsesStringFieldsForReplay(sanitizedItem);
108
195
 
109
196
  return sanitizedItem as unknown as OpenAIResponsesReplayItem;
110
197
  }