@sayknow-cli/ai 0.2.4 → 0.2.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,29 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [0.6.4] - 2026-06-20
6
+
7
+ ### Fixed
8
+
9
+ - Fixed argument mis-attribution in the OpenAI-compatible Responses API streaming decoder when a single response emits multiple tool-call items. The decoder buffered streamed argument deltas against a single most-recent item/block slot, so interleaved or back-to-back `function_call`/`custom_tool_call` argument deltas could be applied to the wrong item — finalizing a tool call with another call's arguments (e.g. one tool's payload landing on a different tool's schema and tripping validation). Streamed deltas now accumulate against a per-item buffer keyed on stable item identity (`item_id` primary; positional `output_index` only when finite), each block records its content index at registration time, and finalization writes onto the same block stored in the message content while reading only the matching item's buffer. Single-tool-call streams, reasoning/text streaming, and the Chat Completions path are unchanged.
10
+
11
+ ## [0.6.2] - 2026-06-19
12
+ ### Added
13
+
14
+ - Added opt-in `compat.sendSessionHeaders` for the `openai-completions` provider. When enabled (default off), the agent session id is forwarded as vendor-neutral `session_id` and `x-session-id` request headers to any OpenAI-compatible endpoint, letting relays/proxies do session-affinity routing and reuse a server-side prompt cache keyed by the session. Previously only the `openai-responses` provider injected session headers, and only against a first-party OpenAI base URL. Injection runs after the caller's `headers`/`extraHeaders` are merged and before `requestTransform`, and uses `??=` so any header the caller already set always wins; it is skipped entirely when the flag is off or no session id is available, leaving existing provider behavior byte-identical. The first-party `openai-responses` gating is unchanged.
15
+
16
+ ### Fixed
17
+
18
+ - Prevented OpenAI Codex Responses `invalid_function_parameters` / tool-schema validation error events from being treated as retryable `server_error`s, so malformed request schemas fail fast instead of burning the full retry budget.
19
+ - Read LM Studio `/v1/models` nested metadata such as `meta.n_ctx`, `meta.n_ctx_train`, and `details.max_tokens` when normalizing dynamically discovered GGUF-backed local models.
20
+ - Corrected the bundled `openai-codex/gpt-5.5` context window from an overstated 400K back to its true 272K (272,000-token) window, so context-cap / auto-compaction thresholds no longer let gpt-5.5 sessions overrun the model's real limit before compaction (#873).
21
+
22
+ ## [0.6.1] - 2026-06-18
23
+
24
+ ### Fixed
25
+
26
+ - Generalized tool `input_schema` root-combinator flattening across providers so discriminated-union tool inputs (e.g. the `computer` tool, a `z.union`) no longer ship a bare top-level `anyOf`/`oneOf`/`allOf` root that strict validators reject. The Anthropic-only fix from 0.5.4 is now the shared, provider-agnostic `flattenToolRootCombinators` (in `utils/schema`) and is applied by Amazon Bedrock, OpenAI Chat Completions / Responses / Codex-Responses / Azure-Responses, Ollama, and Cursor. Previously only Anthropic flattened the root, so those providers forwarded the union root verbatim and a union-root tool failed upstream — Bedrock Converse (including via Kiro/CodeWhisperer relays) returned `400 TOOL_SCHEMA_INVALID: The value at toolConfig.tools.N.toolSpec.inputSchema.json.type must be one of the following: object`. Anthropic behavior is unchanged (it now calls the shared util), Google / Cloud Code Assist keep their own object-merge, and object-root tools, nested combinators, and runtime Zod validation are all untouched.
27
+
5
28
  ## [0.6.0] - 2026-06-18
6
29
  ### Fixed
7
30
 
package/README.md CHANGED
@@ -778,6 +778,7 @@ The `openai-completions` API is implemented by many providers with minor differe
778
778
  interface OpenAICompat {
779
779
  supportsStore?: boolean; // Whether provider supports the `store` field (default: true)
780
780
  supportsDeveloperRole?: boolean; // Whether provider supports `developer` role vs `system` (default: true)
781
+ sendSessionHeaders?: boolean; // Forward the session id as `session_id`/`x-session-id` headers for relay session-affinity & prompt-cache reuse (default: false)
781
782
  supportsReasoningEffort?: boolean; // Whether provider supports `reasoning_effort` (default: true)
782
783
  maxTokensField?: "max_completion_tokens" | "max_tokens"; // Which field name to use (default: max_completion_tokens)
783
784
  extraBody?: Record<string, unknown>; // Extra request-body fields for custom proxy routing or provider-specific options
@@ -228,6 +228,8 @@ export interface ModelsDevProviderDescriptor {
228
228
  * Can return null to skip the model, or an array to emit multiple models.
229
229
  */
230
230
  transformModel?: (model: Model<Api>, modelId: string, raw: ModelsDevModel) => Model<Api> | Model<Api>[] | null;
231
+ /** Optional static rows appended after mapped models. Used only for official provider catalogs missing from models.dev. */
232
+ appendModels?: readonly Model<Api>[];
231
233
  /**
232
234
  * Optional: override the API type per-model.
233
235
  * Called with (modelId, raw). Return the API type to use.
@@ -189,10 +189,4 @@ export declare function convertAnthropicMessages(messages: Message[], model: Mod
189
189
  * object, so callers like the resolve tool keep working open-map semantics.
190
190
  */
191
191
  export declare function normalizeAnthropicToolSchema(schema: unknown): unknown;
192
- /**
193
- * Anthropic rejects tool `input_schema` roots containing top-level oneOf/anyOf/allOf.
194
- * Keep the generic normalizer schema-preserving, then flatten only provider-emitted
195
- * tool roots into one object while leaving nested combinators untouched.
196
- */
197
- export declare function normalizeAnthropicToolRootInputSchema(schema: Record<string, unknown>): Record<string, unknown>;
198
192
  export {};
@@ -606,6 +606,17 @@ export interface OpenAICompat extends ToolChoiceCompat {
606
606
  supportsStore?: boolean;
607
607
  /** Whether the provider supports the `developer` role (vs `system`). Default: auto-detected from URL. */
608
608
  supportsDeveloperRole?: boolean;
609
+ /**
610
+ * Whether to forward the agent session id as vendor-neutral session-identity
611
+ * headers (`session_id`, `x-session-id`) on every chat-completions request.
612
+ * Off by default. Opt in for OpenAI-compatible proxies/relays that route on
613
+ * session affinity or reuse a server-side prompt cache keyed by session.
614
+ * First-party OpenAI does not need this (it has its own gated injection in
615
+ * the openai-responses provider). Headers are only added when a non-empty
616
+ * session id is available and are never allowed to overwrite a header the
617
+ * caller already set via `headers`/`requestTransform`.
618
+ */
619
+ sendSessionHeaders?: boolean;
609
620
  /**
610
621
  * Whether the provider's chat-completions endpoint accepts multiple
611
622
  * leading `system`/`developer` messages. When false, ordered system
@@ -7,6 +7,7 @@ export * from "./fields";
7
7
  export * from "./json-schema-validator";
8
8
  export * from "./meta-validator";
9
9
  export * from "./normalize";
10
+ export * from "./root-combinator";
10
11
  export * from "./spill";
11
12
  export * from "./types";
12
13
  export * from "./wire";
@@ -0,0 +1,12 @@
1
+ /** True when a JSON Schema node describes an object (explicit `type` or `properties`). */
2
+ export declare function isJsonSchemaObjectNode(schema: Record<string, unknown>): boolean;
3
+ /**
4
+ * Flatten a provider-emitted tool ROOT whose top level is a `oneOf`/`anyOf`/`allOf`
5
+ * combinator into one `type: "object"` schema: merge object-branch properties,
6
+ * derive the discriminant (`action`) enum, keep the common required set, and demote
7
+ * leftover combinators plus per-branch guidance into the description. Nested
8
+ * combinators (inside individual properties) are left untouched.
9
+ *
10
+ * Idempotent: a root that already lacks top-level combinators is returned unchanged.
11
+ */
12
+ export declare function flattenToolRootCombinators(schema: Record<string, unknown>): Record<string, unknown>;
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@sayknow-cli/ai",
4
- "version": "0.2.4",
4
+ "version": "0.2.6",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://github.com/jaybeyond/Sayknow_CLI",
7
7
  "author": "jaybeyond",
@@ -43,7 +43,7 @@
43
43
  "dependencies": {
44
44
  "@anthropic-ai/sdk": "^0.94.0",
45
45
  "@bufbuild/protobuf": "^2.12.0",
46
- "@sayknow-cli/utils": "0.2.4",
46
+ "@sayknow-cli/utils": "0.2.6",
47
47
  "openai": "^6.36.0",
48
48
  "partial-json": "^0.1.7",
49
49
  "zod": "4.4.3"
@@ -409,7 +409,7 @@ function applyGeneratedModelPolicy(model: ApiModel<Api>): void {
409
409
  // 512K. The stale 512K survives generate-models (provider-scoped models bypass the
410
410
  // models.dev refresh in applyGlobalModelsDevFallback), tripping auto-compaction /
411
411
  // context-cap thresholds 2x early on MiniMax sessions. Pin to the true 1M.
412
- if (model.id === "minimax-m3") {
412
+ if (model.provider !== "opencode-go" && model.id === "minimax-m3") {
413
413
  model.contextWindow = 1_000_000;
414
414
  }
415
415
  }