@sayknow-cli/ai 0.3.0 → 0.3.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +36 -0
- package/dist/types/auth-gateway/http.d.ts +1 -0
- package/dist/types/model-thinking.d.ts +1 -1
- package/dist/types/provider-models/openai-compat.d.ts +5 -0
- package/dist/types/providers/google-gemini-cli.d.ts +4 -1
- package/dist/types/providers/google-gemini-headers.d.ts +27 -5
- package/dist/types/providers/mock.d.ts +2 -0
- package/dist/types/providers/openai-codex/request-transformer.d.ts +1 -0
- package/dist/types/providers/openai-responses-shared.d.ts +16 -1
- package/dist/types/types.d.ts +9 -1
- package/dist/types/utils/json-parse.d.ts +8 -0
- package/dist/types/utils/oauth/fugu.d.ts +1 -0
- package/dist/types/utils/oauth/index.d.ts +1 -0
- package/dist/types/utils/oauth/types.d.ts +1 -1
- package/dist/types/utils/overflow.d.ts +11 -1
- package/package.json +2 -2
- package/src/auth-gateway/http.ts +5 -1
- package/src/auth-gateway/server.ts +18 -1
- package/src/auth-storage.ts +41 -27
- package/src/model-thinking.ts +2 -9
- package/src/models.json +67 -1
- package/src/provider-models/descriptors.ts +7 -0
- package/src/provider-models/openai-compat.ts +9 -0
- package/src/providers/google-gemini-cli.ts +64 -18
- package/src/providers/google-gemini-headers.ts +72 -18
- package/src/providers/mock.ts +3 -0
- package/src/providers/openai-codex/request-transformer.ts +53 -0
- package/src/providers/openai-codex-responses.ts +24 -11
- package/src/providers/openai-completions-compat.ts +2 -1
- package/src/providers/openai-completions.ts +12 -1
- package/src/providers/openai-responses-shared.ts +43 -1
- package/src/stream.ts +1 -0
- package/src/types.ts +9 -0
- package/src/utils/json-parse.ts +18 -0
- package/src/utils/oauth/fugu.ts +15 -0
- package/src/utils/oauth/index.ts +9 -0
- package/src/utils/oauth/types.ts +1 -0
- package/src/utils/overflow.ts +35 -1
- package/src/utils.ts +87 -0
package/src/utils/overflow.ts
CHANGED
|
@@ -52,14 +52,26 @@ const OVERFLOW_PATTERNS = [
|
|
|
52
52
|
/\b413\b.*\b(request|payload|entity)\b.*\btoo large\b/i, // "413 Request Entity Too Large" variants
|
|
53
53
|
/model_context_window_exceeded/i, // z.ai non-standard finish_reason surfaced as error text
|
|
54
54
|
];
|
|
55
|
+
/**
|
|
56
|
+
* Threshold below which a "successful" (stopReason "stop") response with empty
|
|
57
|
+
* content is considered anomalous. Some proxies (notably LiteLLM) return an
|
|
58
|
+
* empty `choices[0].message.content` with a near-zero `usage` (e.g. input: 1,
|
|
59
|
+
* output: 1) when the upstream model context window is exceeded, instead of
|
|
60
|
+
* surfacing a proper error. The total token count for such a response is well
|
|
61
|
+
* below any realistic turn, so we treat it as a proxy-level overflow signal.
|
|
62
|
+
*/
|
|
63
|
+
const EMPTY_RESPONSE_USAGE_THRESHOLD = 5;
|
|
55
64
|
/**
|
|
56
65
|
* Check if an assistant message represents a context overflow error.
|
|
57
66
|
*
|
|
58
|
-
* This handles
|
|
67
|
+
* This handles three cases:
|
|
59
68
|
* 1. Error-based overflow: Most providers return stopReason "error" with a
|
|
60
69
|
* specific error message pattern.
|
|
61
70
|
* 2. Silent overflow: Some providers accept overflow requests and return
|
|
62
71
|
* successfully. For these, we check if usage.input exceeds the context window.
|
|
72
|
+
* 3. Proxy-level overflow: Some proxies (e.g. LiteLLM) return a "successful"
|
|
73
|
+
* response with empty content and a fabricated near-zero usage when the
|
|
74
|
+
* upstream model's context window is exceeded.
|
|
63
75
|
*
|
|
64
76
|
* ## Reliability by Provider
|
|
65
77
|
*
|
|
@@ -83,6 +95,13 @@ const OVERFLOW_PATTERNS = [
|
|
|
83
95
|
* - z.ai: Sometimes accepts overflow silently (detectable via usage.input > contextWindow),
|
|
84
96
|
* sometimes returns rate limit errors. Pass contextWindow param to detect silent overflow.
|
|
85
97
|
* - Ollama: Silently truncates input without error. Cannot be detected via this function.
|
|
98
|
+
* - LiteLLM proxy: Returns a "successful" response with empty content and a
|
|
99
|
+
* fabricated near-zero usage (e.g. input: 1, output: 1) when the upstream
|
|
100
|
+
* model's context window is exceeded. Detected via Case 3 (empty content +
|
|
101
|
+
* anomalously low usage). Note: the LiteLLM proxy's context limit may differ
|
|
102
|
+
* from the underlying model's advertised contextWindow (e.g. configured via
|
|
103
|
+
* `model_info.max_tokens` in LiteLLM's config.yaml), so Case 2 (which compares
|
|
104
|
+
* usage.input against contextWindow) may not catch it.
|
|
86
105
|
* The response will have usage.input < expected, but we don't know the expected value.
|
|
87
106
|
*
|
|
88
107
|
* ## Custom Providers
|
|
@@ -126,6 +145,21 @@ export function isContextOverflow(message: AssistantMessage, contextWindow?: num
|
|
|
126
145
|
}
|
|
127
146
|
}
|
|
128
147
|
|
|
148
|
+
// Case 3: Empty response with anomalously low usage (proxy-level overflow)
|
|
149
|
+
// Some proxies (e.g. LiteLLM) return a "successful" response (stopReason "stop")
|
|
150
|
+
// with empty content and a near-zero token count when the upstream model's
|
|
151
|
+
// context window is exceeded. This is distinct from silent overflow (Case 2),
|
|
152
|
+
// where the provider reports the real input token count. Here the proxy
|
|
153
|
+
// fabricates a bogus usage (input: 1, output: 1) that is far below any
|
|
154
|
+
// realistic turn, so we detect it heuristically.
|
|
155
|
+
if (
|
|
156
|
+
message.stopReason === "stop" &&
|
|
157
|
+
message.content.length === 0 &&
|
|
158
|
+
message.usage.input + message.usage.output <= EMPTY_RESPONSE_USAGE_THRESHOLD
|
|
159
|
+
) {
|
|
160
|
+
return true;
|
|
161
|
+
}
|
|
162
|
+
|
|
129
163
|
return false;
|
|
130
164
|
}
|
|
131
165
|
|
package/src/utils.ts
CHANGED
|
@@ -91,6 +91,92 @@ export function sanitizeOpenAIResponsesHistoryItemsForReplay(items: Array<Record
|
|
|
91
91
|
return sanitized ? [sanitized] : [];
|
|
92
92
|
});
|
|
93
93
|
}
|
|
94
|
+
function stringifyResponsesStringParamForReplay(value: unknown): string {
|
|
95
|
+
if (typeof value === "string") return value.toWellFormed();
|
|
96
|
+
try {
|
|
97
|
+
const encoded = JSON.stringify(value);
|
|
98
|
+
if (typeof encoded === "string") return encoded.toWellFormed();
|
|
99
|
+
} catch {
|
|
100
|
+
// Fall through to String().
|
|
101
|
+
}
|
|
102
|
+
return String(value ?? "").toWellFormed();
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
function normalizeResponsesMessageTextForReplay(value: unknown): string {
|
|
106
|
+
if (typeof value === "string") return value.toWellFormed();
|
|
107
|
+
if (value && typeof value === "object") {
|
|
108
|
+
const nestedText = (value as { text?: unknown }).text;
|
|
109
|
+
if (typeof nestedText === "string") return nestedText.toWellFormed();
|
|
110
|
+
}
|
|
111
|
+
return stringifyResponsesStringParamForReplay(value);
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
type ResponsesImageDetail = "auto" | "low" | "high";
|
|
115
|
+
|
|
116
|
+
interface NormalizedResponsesImageUrl {
|
|
117
|
+
readonly imageUrl: string;
|
|
118
|
+
readonly detail?: ResponsesImageDetail;
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
function isResponsesImageDetail(value: unknown): value is ResponsesImageDetail {
|
|
122
|
+
return value === "auto" || value === "low" || value === "high";
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
function normalizeResponsesImageUrlForReplay(value: unknown): NormalizedResponsesImageUrl {
|
|
126
|
+
if (typeof value === "string") return { imageUrl: value.toWellFormed() };
|
|
127
|
+
if (value && typeof value === "object" && "url" in value && typeof value.url === "string") {
|
|
128
|
+
const detail = "detail" in value && isResponsesImageDetail(value.detail) ? value.detail : undefined;
|
|
129
|
+
return {
|
|
130
|
+
imageUrl: value.url.toWellFormed(),
|
|
131
|
+
...(detail ? { detail } : {}),
|
|
132
|
+
};
|
|
133
|
+
}
|
|
134
|
+
return { imageUrl: stringifyResponsesStringParamForReplay(value) };
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
function sanitizeResponsesMessageContentForReplay(content: unknown): unknown {
|
|
138
|
+
if (typeof content === "string") return content.toWellFormed();
|
|
139
|
+
if (!Array.isArray(content)) return content;
|
|
140
|
+
return content.map(part => {
|
|
141
|
+
if (!part || typeof part !== "object") return part;
|
|
142
|
+
const sanitizedPart = { ...(part as Record<string, unknown>) };
|
|
143
|
+
if ("text" in sanitizedPart) {
|
|
144
|
+
sanitizedPart.text = normalizeResponsesMessageTextForReplay(sanitizedPart.text);
|
|
145
|
+
}
|
|
146
|
+
if ("image_url" in sanitizedPart) {
|
|
147
|
+
const normalizedImageUrl = normalizeResponsesImageUrlForReplay(sanitizedPart.image_url);
|
|
148
|
+
sanitizedPart.image_url = normalizedImageUrl.imageUrl;
|
|
149
|
+
if (sanitizedPart.type === "image_url") {
|
|
150
|
+
sanitizedPart.type = "input_image";
|
|
151
|
+
}
|
|
152
|
+
if (normalizedImageUrl.detail) {
|
|
153
|
+
sanitizedPart.detail = normalizedImageUrl.detail;
|
|
154
|
+
} else if ("detail" in sanitizedPart && !isResponsesImageDetail(sanitizedPart.detail)) {
|
|
155
|
+
delete sanitizedPart.detail;
|
|
156
|
+
}
|
|
157
|
+
}
|
|
158
|
+
return sanitizedPart;
|
|
159
|
+
});
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
function sanitizeResponsesStringFieldsForReplay(item: Record<string, unknown>): void {
|
|
163
|
+
if (item.type === "message") {
|
|
164
|
+
item.content = sanitizeResponsesMessageContentForReplay(item.content);
|
|
165
|
+
}
|
|
166
|
+
if (item.type === "function_call" && "arguments" in item && typeof item.arguments !== "string") {
|
|
167
|
+
item.arguments = stringifyResponsesStringParamForReplay(item.arguments);
|
|
168
|
+
}
|
|
169
|
+
if (item.type === "custom_tool_call" && "input" in item && typeof item.input !== "string") {
|
|
170
|
+
item.input = stringifyResponsesStringParamForReplay(item.input);
|
|
171
|
+
}
|
|
172
|
+
if (
|
|
173
|
+
(item.type === "function_call_output" || item.type === "custom_tool_call_output") &&
|
|
174
|
+
"output" in item &&
|
|
175
|
+
typeof item.output !== "string"
|
|
176
|
+
) {
|
|
177
|
+
item.output = stringifyResponsesStringParamForReplay(item.output);
|
|
178
|
+
}
|
|
179
|
+
}
|
|
94
180
|
|
|
95
181
|
function sanitizeOpenAIResponsesHistoryItemForReplay(
|
|
96
182
|
item: Record<string, unknown>,
|
|
@@ -105,6 +191,7 @@ function sanitizeOpenAIResponsesHistoryItemForReplay(
|
|
|
105
191
|
if (typeof item.call_id === "string") {
|
|
106
192
|
sanitizedItem.call_id = normalizeReplayedResponsesHistoryCallId(item.call_id, normalizedCallIds);
|
|
107
193
|
}
|
|
194
|
+
sanitizeResponsesStringFieldsForReplay(sanitizedItem);
|
|
108
195
|
|
|
109
196
|
return sanitizedItem as unknown as OpenAIResponsesReplayItem;
|
|
110
197
|
}
|