@sayknow-cli/ai 0.2.4 → 0.2.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +23 -0
- package/README.md +1 -0
- package/dist/types/provider-models/openai-compat.d.ts +2 -0
- package/dist/types/providers/anthropic.d.ts +0 -6
- package/dist/types/types.d.ts +11 -0
- package/dist/types/utils/schema/index.d.ts +1 -0
- package/dist/types/utils/schema/root-combinator.d.ts +12 -0
- package/package.json +2 -2
- package/src/model-thinking.ts +1 -1
- package/src/models.json +4974 -701
- package/src/models.ts +12 -5
- package/src/provider-models/openai-compat.ts +349 -60
- package/src/providers/amazon-bedrock.ts +2 -2
- package/src/providers/anthropic.ts +8 -123
- package/src/providers/azure-openai-responses.ts +2 -2
- package/src/providers/cursor.ts +2 -2
- package/src/providers/ollama.ts +2 -2
- package/src/providers/openai-codex-responses.ts +25 -5
- package/src/providers/openai-completions-compat.ts +2 -0
- package/src/providers/openai-completions.ts +12 -2
- package/src/providers/openai-responses-shared.ts +165 -85
- package/src/providers/openai-responses.ts +8 -2
- package/src/types.ts +11 -0
- package/src/utils/schema/index.ts +1 -0
- package/src/utils/schema/root-combinator.ts +143 -0
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,29 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [0.6.4] - 2026-06-20
|
|
6
|
+
|
|
7
|
+
### Fixed
|
|
8
|
+
|
|
9
|
+
- Fixed argument mis-attribution in the OpenAI-compatible Responses API streaming decoder when a single response emits multiple tool-call items. The decoder buffered streamed argument deltas against a single most-recent item/block slot, so interleaved or back-to-back `function_call`/`custom_tool_call` argument deltas could be applied to the wrong item — finalizing a tool call with another call's arguments (e.g. one tool's payload landing on a different tool's schema and tripping validation). Streamed deltas now accumulate against a per-item buffer keyed on stable item identity (`item_id` primary; positional `output_index` only when finite), each block records its content index at registration time, and finalization writes onto the same block stored in the message content while reading only the matching item's buffer. Single-tool-call streams, reasoning/text streaming, and the Chat Completions path are unchanged.
|
|
10
|
+
|
|
11
|
+
## [0.6.2] - 2026-06-19
|
|
12
|
+
### Added
|
|
13
|
+
|
|
14
|
+
- Added opt-in `compat.sendSessionHeaders` for the `openai-completions` provider. When enabled (default off), the agent session id is forwarded as vendor-neutral `session_id` and `x-session-id` request headers to any OpenAI-compatible endpoint, letting relays/proxies do session-affinity routing and reuse a server-side prompt cache keyed by the session. Previously only the `openai-responses` provider injected session headers, and only against a first-party OpenAI base URL. Injection runs after the caller's `headers`/`extraHeaders` are merged and before `requestTransform`, and uses `??=` so any header the caller already set always wins; it is skipped entirely when the flag is off or no session id is available, leaving existing provider behavior byte-identical. The first-party `openai-responses` gating is unchanged.
|
|
15
|
+
|
|
16
|
+
### Fixed
|
|
17
|
+
|
|
18
|
+
- Prevented OpenAI Codex Responses `invalid_function_parameters` / tool-schema validation error events from being treated as retryable `server_error`s, so malformed request schemas fail fast instead of burning the full retry budget.
|
|
19
|
+
- Read LM Studio `/v1/models` nested metadata such as `meta.n_ctx`, `meta.n_ctx_train`, and `details.max_tokens` when normalizing dynamically discovered GGUF-backed local models.
|
|
20
|
+
- Corrected the bundled `openai-codex/gpt-5.5` context window from an overstated 400K back to its true 272K (272,000-token) window, so context-cap / auto-compaction thresholds no longer let gpt-5.5 sessions overrun the model's real limit before compaction (#873).
|
|
21
|
+
|
|
22
|
+
## [0.6.1] - 2026-06-18
|
|
23
|
+
|
|
24
|
+
### Fixed
|
|
25
|
+
|
|
26
|
+
- Generalized tool `input_schema` root-combinator flattening across providers so discriminated-union tool inputs (e.g. the `computer` tool, a `z.union`) no longer ship a bare top-level `anyOf`/`oneOf`/`allOf` root that strict validators reject. The Anthropic-only fix from 0.5.4 is now the shared, provider-agnostic `flattenToolRootCombinators` (in `utils/schema`) and is applied by Amazon Bedrock, OpenAI Chat Completions / Responses / Codex-Responses / Azure-Responses, Ollama, and Cursor. Previously only Anthropic flattened the root, so those providers forwarded the union root verbatim and a union-root tool failed upstream — Bedrock Converse (including via Kiro/CodeWhisperer relays) returned `400 TOOL_SCHEMA_INVALID: The value at toolConfig.tools.N.toolSpec.inputSchema.json.type must be one of the following: object`. Anthropic behavior is unchanged (it now calls the shared util), Google / Cloud Code Assist keep their own object-merge, and object-root tools, nested combinators, and runtime Zod validation are all untouched.
|
|
27
|
+
|
|
5
28
|
## [0.6.0] - 2026-06-18
|
|
6
29
|
### Fixed
|
|
7
30
|
|
package/README.md
CHANGED
|
@@ -778,6 +778,7 @@ The `openai-completions` API is implemented by many providers with minor differe
|
|
|
778
778
|
interface OpenAICompat {
|
|
779
779
|
supportsStore?: boolean; // Whether provider supports the `store` field (default: true)
|
|
780
780
|
supportsDeveloperRole?: boolean; // Whether provider supports `developer` role vs `system` (default: true)
|
|
781
|
+
sendSessionHeaders?: boolean; // Forward the session id as `session_id`/`x-session-id` headers for relay session-affinity & prompt-cache reuse (default: false)
|
|
781
782
|
supportsReasoningEffort?: boolean; // Whether provider supports `reasoning_effort` (default: true)
|
|
782
783
|
maxTokensField?: "max_completion_tokens" | "max_tokens"; // Which field name to use (default: max_completion_tokens)
|
|
783
784
|
extraBody?: Record<string, unknown>; // Extra request-body fields for custom proxy routing or provider-specific options
|
|
@@ -228,6 +228,8 @@ export interface ModelsDevProviderDescriptor {
|
|
|
228
228
|
* Can return null to skip the model, or an array to emit multiple models.
|
|
229
229
|
*/
|
|
230
230
|
transformModel?: (model: Model<Api>, modelId: string, raw: ModelsDevModel) => Model<Api> | Model<Api>[] | null;
|
|
231
|
+
/** Optional static rows appended after mapped models. Used only for official provider catalogs missing from models.dev. */
|
|
232
|
+
appendModels?: readonly Model<Api>[];
|
|
231
233
|
/**
|
|
232
234
|
* Optional: override the API type per-model.
|
|
233
235
|
* Called with (modelId, raw). Return the API type to use.
|
|
@@ -189,10 +189,4 @@ export declare function convertAnthropicMessages(messages: Message[], model: Mod
|
|
|
189
189
|
* object, so callers like the resolve tool keep working open-map semantics.
|
|
190
190
|
*/
|
|
191
191
|
export declare function normalizeAnthropicToolSchema(schema: unknown): unknown;
|
|
192
|
-
/**
|
|
193
|
-
* Anthropic rejects tool `input_schema` roots containing top-level oneOf/anyOf/allOf.
|
|
194
|
-
* Keep the generic normalizer schema-preserving, then flatten only provider-emitted
|
|
195
|
-
* tool roots into one object while leaving nested combinators untouched.
|
|
196
|
-
*/
|
|
197
|
-
export declare function normalizeAnthropicToolRootInputSchema(schema: Record<string, unknown>): Record<string, unknown>;
|
|
198
192
|
export {};
|
package/dist/types/types.d.ts
CHANGED
|
@@ -606,6 +606,17 @@ export interface OpenAICompat extends ToolChoiceCompat {
|
|
|
606
606
|
supportsStore?: boolean;
|
|
607
607
|
/** Whether the provider supports the `developer` role (vs `system`). Default: auto-detected from URL. */
|
|
608
608
|
supportsDeveloperRole?: boolean;
|
|
609
|
+
/**
|
|
610
|
+
* Whether to forward the agent session id as vendor-neutral session-identity
|
|
611
|
+
* headers (`session_id`, `x-session-id`) on every chat-completions request.
|
|
612
|
+
* Off by default. Opt in for OpenAI-compatible proxies/relays that route on
|
|
613
|
+
* session affinity or reuse a server-side prompt cache keyed by session.
|
|
614
|
+
* First-party OpenAI does not need this (it has its own gated injection in
|
|
615
|
+
* the openai-responses provider). Headers are only added when a non-empty
|
|
616
|
+
* session id is available and are never allowed to overwrite a header the
|
|
617
|
+
* caller already set via `headers`/`requestTransform`.
|
|
618
|
+
*/
|
|
619
|
+
sendSessionHeaders?: boolean;
|
|
609
620
|
/**
|
|
610
621
|
* Whether the provider's chat-completions endpoint accepts multiple
|
|
611
622
|
* leading `system`/`developer` messages. When false, ordered system
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
/** True when a JSON Schema node describes an object (explicit `type` or `properties`). */
|
|
2
|
+
export declare function isJsonSchemaObjectNode(schema: Record<string, unknown>): boolean;
|
|
3
|
+
/**
|
|
4
|
+
* Flatten a provider-emitted tool ROOT whose top level is a `oneOf`/`anyOf`/`allOf`
|
|
5
|
+
* combinator into one `type: "object"` schema: merge object-branch properties,
|
|
6
|
+
* derive the discriminant (`action`) enum, keep the common required set, and demote
|
|
7
|
+
* leftover combinators plus per-branch guidance into the description. Nested
|
|
8
|
+
* combinators (inside individual properties) are left untouched.
|
|
9
|
+
*
|
|
10
|
+
* Idempotent: a root that already lacks top-level combinators is returned unchanged.
|
|
11
|
+
*/
|
|
12
|
+
export declare function flattenToolRootCombinators(schema: Record<string, unknown>): Record<string, unknown>;
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@sayknow-cli/ai",
|
|
4
|
-
"version": "0.2.
|
|
4
|
+
"version": "0.2.6",
|
|
5
5
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
6
6
|
"homepage": "https://github.com/jaybeyond/Sayknow_CLI",
|
|
7
7
|
"author": "jaybeyond",
|
|
@@ -43,7 +43,7 @@
|
|
|
43
43
|
"dependencies": {
|
|
44
44
|
"@anthropic-ai/sdk": "^0.94.0",
|
|
45
45
|
"@bufbuild/protobuf": "^2.12.0",
|
|
46
|
-
"@sayknow-cli/utils": "0.2.
|
|
46
|
+
"@sayknow-cli/utils": "0.2.6",
|
|
47
47
|
"openai": "^6.36.0",
|
|
48
48
|
"partial-json": "^0.1.7",
|
|
49
49
|
"zod": "4.4.3"
|
package/src/model-thinking.ts
CHANGED
|
@@ -409,7 +409,7 @@ function applyGeneratedModelPolicy(model: ApiModel<Api>): void {
|
|
|
409
409
|
// 512K. The stale 512K survives generate-models (provider-scoped models bypass the
|
|
410
410
|
// models.dev refresh in applyGlobalModelsDevFallback), tripping auto-compaction /
|
|
411
411
|
// context-cap thresholds 2x early on MiniMax sessions. Pin to the true 1M.
|
|
412
|
-
if (model.id === "minimax-m3") {
|
|
412
|
+
if (model.provider !== "opencode-go" && model.id === "minimax-m3") {
|
|
413
413
|
model.contextWindow = 1_000_000;
|
|
414
414
|
}
|
|
415
415
|
}
|