@gajae-code/ai 0.9.2 → 0.9.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +8 -0
- package/dist/types/model-retirements.d.ts +6 -0
- package/dist/types/models.d.ts +11 -2
- package/dist/types/utils/discovery/antigravity.d.ts +6 -0
- package/package.json +2 -2
- package/src/model-cache.ts +1 -1
- package/src/model-manager.ts +14 -3
- package/src/model-retirements.ts +13 -0
- package/src/models.json +0 -29
- package/src/models.ts +26 -5
- package/src/provider-models/google.ts +1 -0
- package/src/providers/anthropic.ts +35 -15
- package/src/stream.ts +85 -36
- package/src/utils/discovery/antigravity.ts +16 -2
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,11 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [0.9.4] - 2026-07-09
|
|
6
|
+
### Fixed
|
|
7
|
+
|
|
8
|
+
- Preserved Anthropic OAuth tool-call names and streamed arguments across interleaved tool-use blocks, preventing prefixed tool names and partial JSON deltas from being dropped or misattributed.
|
|
9
|
+
|
|
5
10
|
## [0.9.2] - 2026-07-09
|
|
6
11
|
### Added
|
|
7
12
|
|
|
@@ -9,6 +14,9 @@
|
|
|
9
14
|
|
|
10
15
|
### Fixed
|
|
11
16
|
|
|
17
|
+
- Refreshed the default Gemini CLI impersonation version to 0.50.0 so the spoofed User-Agent freshness gate passes for the 0.9.2 release.
|
|
18
|
+
- Hid the non-callable `google-antigravity/gemini-3.1-pro-high` selector from bundled, dynamic, and cached Antigravity catalogs after live Cloud Code Assist calls returned HTTP 400; `google-antigravity/gemini-3.1-pro-low:high` remains the working high-thinking path.
|
|
19
|
+
- Preserved Anthropic tool-use arguments supplied on `content_block_start` when no `input_json_delta` chunks follow, preventing finished tool calls from collapsing back to `{}`.
|
|
12
20
|
- Refreshed the default Gemini CLI impersonation version to 0.50.0 so the spoofed User-Agent freshness gate passes for the 0.9.2 release.
|
|
13
21
|
|
|
14
22
|
## [0.9.1] - 2026-07-08
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
export declare const RETIRED_MODEL_KEYS: readonly ["google-antigravity/gemini-3.1-pro-high"];
|
|
2
|
+
export declare function isRetiredModelKey(provider: string, modelId: string): boolean;
|
|
3
|
+
export declare function isRetiredModel(model: {
|
|
4
|
+
provider: string;
|
|
5
|
+
id: string;
|
|
6
|
+
}): boolean;
|
package/dist/types/models.d.ts
CHANGED
|
@@ -1,6 +1,14 @@
|
|
|
1
|
-
import MODELS from "./models.json";
|
|
2
1
|
import type { Api, KnownProvider, Model, Usage } from "./types";
|
|
3
|
-
|
|
2
|
+
/**
|
|
3
|
+
* Static bundled model registry loaded lazily from `models.json`.
|
|
4
|
+
*
|
|
5
|
+
* This module intentionally exposes compile-time defaults only.
|
|
6
|
+
* It does not include runtime discovery, models.dev overlays, or on-disk cache state.
|
|
7
|
+
*
|
|
8
|
+
* For runtime-aware resolution, use `createModelManager()` / `resolveProviderModels()`.
|
|
9
|
+
*/
|
|
10
|
+
type BundledCatalog = typeof import("./models.json");
|
|
11
|
+
export type GeneratedProvider = keyof BundledCatalog;
|
|
4
12
|
export declare function getBundledModel<TApi extends Api = Api>(provider: GeneratedProvider, modelId: string): Model<TApi>;
|
|
5
13
|
export declare function getBundledProviders(): KnownProvider[];
|
|
6
14
|
export declare function getBundledModels(provider: GeneratedProvider): Model<Api>[];
|
|
@@ -10,3 +18,4 @@ export declare function calculateCost<TApi extends Api>(model: Model<TApi>, usag
|
|
|
10
18
|
* Returns false if either model is null or undefined.
|
|
11
19
|
*/
|
|
12
20
|
export declare function modelsAreEqual<TApi extends Api>(a: Model<TApi> | null | undefined, b: Model<TApi> | null | undefined): boolean;
|
|
21
|
+
export {};
|
|
@@ -51,6 +51,12 @@ export interface FetchAntigravityDiscoveryModelsOptions {
|
|
|
51
51
|
signal?: AbortSignal;
|
|
52
52
|
/** Optional fetch implementation override for tests. */
|
|
53
53
|
fetcher?: typeof fetch;
|
|
54
|
+
/**
|
|
55
|
+
* Provider id the caller assigns to returned models. Scopes retired-selector
|
|
56
|
+
* filtering (e.g. `google-gemini-cli` reuses this helper and remaps rows).
|
|
57
|
+
* Default: `google-antigravity`.
|
|
58
|
+
*/
|
|
59
|
+
targetProvider?: "google-antigravity" | "google-gemini-cli";
|
|
54
60
|
}
|
|
55
61
|
/**
|
|
56
62
|
* Fetches discoverable Antigravity models and normalizes them into canonical model entries.
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@gajae-code/ai",
|
|
4
|
-
"version": "0.9.
|
|
4
|
+
"version": "0.9.4",
|
|
5
5
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
6
6
|
"homepage": "https://gajae-code.com",
|
|
7
7
|
"author": "Yeachan-Heo and Gajae Code Contributors",
|
|
@@ -40,7 +40,7 @@
|
|
|
40
40
|
"dependencies": {
|
|
41
41
|
"@anthropic-ai/sdk": "^0.94.0",
|
|
42
42
|
"@bufbuild/protobuf": "^2.12.0",
|
|
43
|
-
"@gajae-code/utils": "0.9.
|
|
43
|
+
"@gajae-code/utils": "0.9.4",
|
|
44
44
|
"openai": "^6.36.0",
|
|
45
45
|
"partial-json": "^0.1.7",
|
|
46
46
|
"zod": "4.4.3"
|
package/src/model-cache.ts
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
* Replaces per-provider JSON files with a single cache.db.
|
|
4
4
|
*/
|
|
5
5
|
import { Database } from "bun:sqlite";
|
|
6
|
-
import { getModelDbPath } from "@gajae-code/utils";
|
|
6
|
+
import { getModelDbPath } from "@gajae-code/utils/dirs";
|
|
7
7
|
import type { Api, Model } from "./types";
|
|
8
8
|
|
|
9
9
|
const CACHE_SCHEMA_VERSION = 3;
|
package/src/model-manager.ts
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
import { readModelCache, writeModelCache } from "./model-cache";
|
|
2
|
+
import { isRetiredModel, isRetiredModelKey } from "./model-retirements";
|
|
2
3
|
import { applyGeneratedModelPolicies, enrichModelThinking } from "./model-thinking";
|
|
3
4
|
import { type GeneratedProvider, getBundledModels } from "./models";
|
|
4
5
|
import type { Api, Model, Provider } from "./types";
|
|
5
|
-
import { isRecord } from "./utils";
|
|
6
6
|
|
|
7
7
|
const DEFAULT_CACHE_TTL_MS = 2 * 60 * 60 * 1000;
|
|
8
8
|
const NON_AUTHORITATIVE_RETRY_MS = 5 * 60 * 1000;
|
|
@@ -85,7 +85,14 @@ function passModelList<TApi extends Api>(value: unknown): Model<TApi>[] {
|
|
|
85
85
|
}
|
|
86
86
|
const out: Model<TApi>[] = [];
|
|
87
87
|
for (const item of value) {
|
|
88
|
-
if (item === null || typeof item !== "object"
|
|
88
|
+
if (item === null || typeof item !== "object") {
|
|
89
|
+
continue;
|
|
90
|
+
}
|
|
91
|
+
const candidate = item as { id?: unknown; provider?: unknown };
|
|
92
|
+
if (typeof candidate.id !== "string") {
|
|
93
|
+
continue;
|
|
94
|
+
}
|
|
95
|
+
if (typeof candidate.provider === "string" && isRetiredModelKey(candidate.provider, candidate.id)) {
|
|
89
96
|
continue;
|
|
90
97
|
}
|
|
91
98
|
out.push(enrichModelThinking(item as Model<TApi>));
|
|
@@ -382,7 +389,7 @@ function normalizeModelList<TApi extends Api>(value: unknown): Model<TApi>[] {
|
|
|
382
389
|
}
|
|
383
390
|
const models: Model<TApi>[] = [];
|
|
384
391
|
for (const item of value) {
|
|
385
|
-
if (isModelLike(item)) {
|
|
392
|
+
if (isModelLike(item) && !isRetiredModel(item)) {
|
|
386
393
|
models.push(enrichModelThinking(item as Model<TApi>));
|
|
387
394
|
}
|
|
388
395
|
}
|
|
@@ -441,6 +448,10 @@ function isModelLike(value: unknown): value is Model<Api> {
|
|
|
441
448
|
return true;
|
|
442
449
|
}
|
|
443
450
|
|
|
451
|
+
function isRecord(value: unknown): value is Record<string, unknown> {
|
|
452
|
+
return typeof value === "object" && value !== null;
|
|
453
|
+
}
|
|
454
|
+
|
|
444
455
|
function isModelInputArray(value: unknown): value is ("text" | "image")[] {
|
|
445
456
|
if (!Array.isArray(value) || value.length === 0) {
|
|
446
457
|
return false;
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
// Retired from advertised catalogs because Cloud Code Assist rejects live calls
|
|
2
|
+
// with HTTP 400. The callable high-thinking path is gemini-3.1-pro-low:high.
|
|
3
|
+
export const RETIRED_MODEL_KEYS = ["google-antigravity/gemini-3.1-pro-high"] as const;
|
|
4
|
+
|
|
5
|
+
const RETIRED_MODEL_KEY_SET = new Set<string>(RETIRED_MODEL_KEYS);
|
|
6
|
+
|
|
7
|
+
export function isRetiredModelKey(provider: string, modelId: string): boolean {
|
|
8
|
+
return RETIRED_MODEL_KEY_SET.has(`${provider}/${modelId}`);
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
export function isRetiredModel(model: { provider: string; id: string }): boolean {
|
|
12
|
+
return isRetiredModelKey(model.provider, model.id);
|
|
13
|
+
}
|
package/src/models.json
CHANGED
|
@@ -10978,35 +10978,6 @@
|
|
|
10978
10978
|
]
|
|
10979
10979
|
}
|
|
10980
10980
|
},
|
|
10981
|
-
"gemini-3.1-pro-high": {
|
|
10982
|
-
"id": "gemini-3.1-pro-high",
|
|
10983
|
-
"name": "Gemini 3.1 Pro (High) (Antigravity)",
|
|
10984
|
-
"api": "google-gemini-cli",
|
|
10985
|
-
"provider": "google-antigravity",
|
|
10986
|
-
"baseUrl": "https://daily-cloudcode-pa.sandbox.googleapis.com",
|
|
10987
|
-
"reasoning": true,
|
|
10988
|
-
"input": [
|
|
10989
|
-
"text",
|
|
10990
|
-
"image"
|
|
10991
|
-
],
|
|
10992
|
-
"cost": {
|
|
10993
|
-
"input": 0,
|
|
10994
|
-
"output": 0,
|
|
10995
|
-
"cacheRead": 0,
|
|
10996
|
-
"cacheWrite": 0
|
|
10997
|
-
},
|
|
10998
|
-
"contextWindow": 1048576,
|
|
10999
|
-
"maxTokens": 65535,
|
|
11000
|
-
"thinking": {
|
|
11001
|
-
"mode": "google-level",
|
|
11002
|
-
"minLevel": "low",
|
|
11003
|
-
"maxLevel": "high",
|
|
11004
|
-
"levels": [
|
|
11005
|
-
"low",
|
|
11006
|
-
"high"
|
|
11007
|
-
]
|
|
11008
|
-
}
|
|
11009
|
-
},
|
|
11010
10981
|
"gemini-3.1-pro-low": {
|
|
11011
10982
|
"id": "gemini-3.1-pro-low",
|
|
11012
10983
|
"name": "Gemini 3.1 Pro (Low) (Antigravity)",
|
package/src/models.ts
CHANGED
|
@@ -1,26 +1,46 @@
|
|
|
1
|
+
import { readFileSync } from "node:fs";
|
|
2
|
+
import { isRetiredModelKey } from "./model-retirements";
|
|
1
3
|
import { applyGeneratedModelPolicies, enrichModelThinking } from "./model-thinking";
|
|
2
|
-
|
|
4
|
+
// `with { type: "file" }` is embedded by `bun build --compile` and resolves to
|
|
5
|
+
// the bunfs path inside standalone binaries (and to the on-disk path in dev).
|
|
6
|
+
// A plain `createRequire` of a `.json` listed as an extra compile entrypoint is
|
|
7
|
+
// NOT emitted into the bunfs, and its cwd-fallback masks the failure whenever
|
|
8
|
+
// the process runs inside a repo checkout — see PR body for the minimal repro.
|
|
9
|
+
import modelsJsonPath from "./models.json" with { type: "file" };
|
|
3
10
|
import type { Api, KnownProvider, Model, Usage } from "./types";
|
|
4
11
|
import { isClaudeForcedToolChoiceIncapableModelId } from "./utils/tool-choice-capability";
|
|
5
12
|
|
|
6
13
|
/**
|
|
7
|
-
* Static bundled model registry loaded from `models.json`.
|
|
14
|
+
* Static bundled model registry loaded lazily from `models.json`.
|
|
8
15
|
*
|
|
9
16
|
* This module intentionally exposes compile-time defaults only.
|
|
10
17
|
* It does not include runtime discovery, models.dev overlays, or on-disk cache state.
|
|
11
18
|
*
|
|
12
19
|
* For runtime-aware resolution, use `createModelManager()` / `resolveProviderModels()`.
|
|
13
20
|
*/
|
|
14
|
-
|
|
21
|
+
type BundledCatalog = typeof import("./models.json");
|
|
22
|
+
|
|
23
|
+
let bundledCatalog: BundledCatalog | undefined;
|
|
24
|
+
let providerNames: KnownProvider[] | undefined;
|
|
15
25
|
const providerModelRegistry: Map<string, Map<string, Model<Api>>> = new Map();
|
|
16
26
|
|
|
27
|
+
function getBundledCatalog(): BundledCatalog {
|
|
28
|
+
// TS types a .json import as its contents; at runtime `with { type: "file" }`
|
|
29
|
+
// yields the file path (bunfs path in compiled binaries, disk path in dev).
|
|
30
|
+
bundledCatalog ??= JSON.parse(readFileSync(modelsJsonPath as unknown as string, "utf8")) as BundledCatalog;
|
|
31
|
+
return bundledCatalog;
|
|
32
|
+
}
|
|
33
|
+
|
|
17
34
|
function getProviderModels(provider: GeneratedProvider): Map<string, Model<Api>> | undefined {
|
|
18
35
|
const cached = providerModelRegistry.get(provider);
|
|
19
36
|
if (cached) return cached;
|
|
20
|
-
const models =
|
|
37
|
+
const models = getBundledCatalog()[provider];
|
|
21
38
|
if (!models) return undefined;
|
|
22
39
|
const providerModels = new Map<string, Model<Api>>();
|
|
23
40
|
for (const [id, model] of Object.entries(models)) {
|
|
41
|
+
if (isRetiredModelKey(provider, id)) {
|
|
42
|
+
continue;
|
|
43
|
+
}
|
|
24
44
|
providerModels.set(id, applyBundledCompatDefaults(enrichModelThinking(model as Model<Api>)));
|
|
25
45
|
}
|
|
26
46
|
providerModelRegistry.set(provider, providerModels);
|
|
@@ -52,7 +72,7 @@ function applyBundledCompatDefaults(model: Model<Api>): Model<Api> {
|
|
|
52
72
|
return policyModels[0] ?? normalized;
|
|
53
73
|
}
|
|
54
74
|
|
|
55
|
-
export type GeneratedProvider = keyof
|
|
75
|
+
export type GeneratedProvider = keyof BundledCatalog;
|
|
56
76
|
|
|
57
77
|
export function getBundledModel<TApi extends Api = Api>(provider: GeneratedProvider, modelId: string): Model<TApi> {
|
|
58
78
|
const providerModels = getProviderModels(provider);
|
|
@@ -62,6 +82,7 @@ export function getBundledModel<TApi extends Api = Api>(provider: GeneratedProvi
|
|
|
62
82
|
export function getBundledProviders(): KnownProvider[] {
|
|
63
83
|
// Defensive copy: the old eager path returned a fresh Array.from(...), so
|
|
64
84
|
// callers may freely mutate their result without corrupting enumeration.
|
|
85
|
+
providerNames ??= Object.keys(getBundledCatalog()) as KnownProvider[];
|
|
65
86
|
return providerNames.slice();
|
|
66
87
|
}
|
|
67
88
|
|
|
@@ -586,15 +586,12 @@ const ANTHROPIC_BUILTIN_TOOL_NAMES = new Set(["web_search", "code_execution", "t
|
|
|
586
586
|
export const applyClaudeToolPrefix = (name: string, prefixOverride: string = claudeToolPrefix) => {
|
|
587
587
|
if (!prefixOverride) return name;
|
|
588
588
|
if (ANTHROPIC_BUILTIN_TOOL_NAMES.has(name.toLowerCase())) return name;
|
|
589
|
-
const prefix = prefixOverride.toLowerCase();
|
|
590
|
-
if (name.toLowerCase().startsWith(prefix)) return name;
|
|
591
589
|
return `${prefixOverride}${name}`;
|
|
592
590
|
};
|
|
593
591
|
|
|
594
592
|
export const stripClaudeToolPrefix = (name: string, prefixOverride: string = claudeToolPrefix) => {
|
|
595
593
|
if (!prefixOverride) return name;
|
|
596
|
-
|
|
597
|
-
if (!name.toLowerCase().startsWith(prefix)) return name;
|
|
594
|
+
if (!name.startsWith(prefixOverride)) return name;
|
|
598
595
|
return name.slice(prefixOverride.length);
|
|
599
596
|
};
|
|
600
597
|
|
|
@@ -1325,6 +1322,25 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1325
1322
|
| (ToolCall & { partialJson: string })
|
|
1326
1323
|
) & { index: number };
|
|
1327
1324
|
const blocks = output.content as Block[];
|
|
1325
|
+
const blocksByAnthropicIndex = new Map<number, Block>();
|
|
1326
|
+
const getBlockByAnthropicIndex = (anthropicIndex: number) => {
|
|
1327
|
+
const block = blocksByAnthropicIndex.get(anthropicIndex);
|
|
1328
|
+
if (!block) return { block: undefined, contentIndex: -1 };
|
|
1329
|
+
return { block, contentIndex: blocks.indexOf(block) };
|
|
1330
|
+
};
|
|
1331
|
+
const trackBlockByAnthropicIndex = (anthropicIndex: number, block: Block) => {
|
|
1332
|
+
// A duplicate start for an active index is a provider-envelope violation;
|
|
1333
|
+
// finalize the orphaned block so no internal stream fields leak into output.
|
|
1334
|
+
const orphaned = blocksByAnthropicIndex.get(anthropicIndex);
|
|
1335
|
+
if (orphaned) {
|
|
1336
|
+
if (orphaned.type === "toolCall" && orphaned.partialJson.trim()) {
|
|
1337
|
+
orphaned.arguments = parseStreamingJson(orphaned.partialJson);
|
|
1338
|
+
}
|
|
1339
|
+
delete (orphaned as { index?: number }).index;
|
|
1340
|
+
delete (orphaned as { partialJson?: string }).partialJson;
|
|
1341
|
+
}
|
|
1342
|
+
blocksByAnthropicIndex.set(anthropicIndex, block);
|
|
1343
|
+
};
|
|
1328
1344
|
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getStreamIdleTimeoutMs();
|
|
1329
1345
|
const firstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs);
|
|
1330
1346
|
stream.push({ type: "start", partial: output });
|
|
@@ -1334,6 +1350,8 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1334
1350
|
let providerRetryAttempt = 0;
|
|
1335
1351
|
let thinkingRepairAttempted = false;
|
|
1336
1352
|
while (true) {
|
|
1353
|
+
// Retries reset output.content; drop stale block correlations from the aborted attempt.
|
|
1354
|
+
blocksByAnthropicIndex.clear();
|
|
1337
1355
|
activeAbortTracker = createAbortSourceTracker(options?.signal);
|
|
1338
1356
|
const firstEventTimeoutAbortError = new Error(
|
|
1339
1357
|
"Anthropic stream timed out while waiting for the first event",
|
|
@@ -1403,6 +1421,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1403
1421
|
index: event.index,
|
|
1404
1422
|
};
|
|
1405
1423
|
output.content.push(block);
|
|
1424
|
+
trackBlockByAnthropicIndex(event.index, block);
|
|
1406
1425
|
stream.push({
|
|
1407
1426
|
type: "text_start",
|
|
1408
1427
|
contentIndex: output.content.length - 1,
|
|
@@ -1416,6 +1435,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1416
1435
|
index: event.index,
|
|
1417
1436
|
};
|
|
1418
1437
|
output.content.push(block);
|
|
1438
|
+
trackBlockByAnthropicIndex(event.index, block);
|
|
1419
1439
|
stream.push({
|
|
1420
1440
|
type: "thinking_start",
|
|
1421
1441
|
contentIndex: output.content.length - 1,
|
|
@@ -1428,6 +1448,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1428
1448
|
index: event.index,
|
|
1429
1449
|
};
|
|
1430
1450
|
output.content.push(block);
|
|
1451
|
+
trackBlockByAnthropicIndex(event.index, block);
|
|
1431
1452
|
} else if (event.content_block.type === "tool_use") {
|
|
1432
1453
|
streamedReplayUnsafeContent = true;
|
|
1433
1454
|
const block: Block = {
|
|
@@ -1441,6 +1462,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1441
1462
|
index: event.index,
|
|
1442
1463
|
};
|
|
1443
1464
|
output.content.push(block);
|
|
1465
|
+
trackBlockByAnthropicIndex(event.index, block);
|
|
1444
1466
|
stream.push({
|
|
1445
1467
|
type: "toolcall_start",
|
|
1446
1468
|
contentIndex: output.content.length - 1,
|
|
@@ -1449,8 +1471,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1449
1471
|
}
|
|
1450
1472
|
} else if (event.type === "content_block_delta") {
|
|
1451
1473
|
if (event.delta.type === "text_delta") {
|
|
1452
|
-
const
|
|
1453
|
-
const block = blocks[index];
|
|
1474
|
+
const { block, contentIndex: index } = getBlockByAnthropicIndex(event.index);
|
|
1454
1475
|
if (block && block.type === "text") {
|
|
1455
1476
|
block.text += event.delta.text;
|
|
1456
1477
|
stream.push({
|
|
@@ -1461,8 +1482,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1461
1482
|
});
|
|
1462
1483
|
}
|
|
1463
1484
|
} else if (event.delta.type === "thinking_delta") {
|
|
1464
|
-
const
|
|
1465
|
-
const block = blocks[index];
|
|
1485
|
+
const { block, contentIndex: index } = getBlockByAnthropicIndex(event.index);
|
|
1466
1486
|
if (block && block.type === "thinking") {
|
|
1467
1487
|
block.thinking += event.delta.thinking;
|
|
1468
1488
|
stream.push({
|
|
@@ -1473,8 +1493,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1473
1493
|
});
|
|
1474
1494
|
}
|
|
1475
1495
|
} else if (event.delta.type === "input_json_delta") {
|
|
1476
|
-
const
|
|
1477
|
-
const block = blocks[index];
|
|
1496
|
+
const { block, contentIndex: index } = getBlockByAnthropicIndex(event.index);
|
|
1478
1497
|
if (block && block.type === "toolCall") {
|
|
1479
1498
|
block.partialJson += event.delta.partial_json;
|
|
1480
1499
|
block.arguments = parseStreamingJson(block.partialJson);
|
|
@@ -1486,17 +1505,16 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1486
1505
|
});
|
|
1487
1506
|
}
|
|
1488
1507
|
} else if (event.delta.type === "signature_delta") {
|
|
1489
|
-
const
|
|
1490
|
-
const block = blocks[index];
|
|
1508
|
+
const { block } = getBlockByAnthropicIndex(event.index);
|
|
1491
1509
|
if (block && block.type === "thinking") {
|
|
1492
1510
|
block.thinkingSignature = block.thinkingSignature || "";
|
|
1493
1511
|
block.thinkingSignature += event.delta.signature;
|
|
1494
1512
|
}
|
|
1495
1513
|
}
|
|
1496
1514
|
} else if (event.type === "content_block_stop") {
|
|
1497
|
-
const
|
|
1498
|
-
const block = blocks[index];
|
|
1515
|
+
const { block, contentIndex: index } = getBlockByAnthropicIndex(event.index);
|
|
1499
1516
|
if (block) {
|
|
1517
|
+
blocksByAnthropicIndex.delete(event.index);
|
|
1500
1518
|
delete (block as { index?: number }).index;
|
|
1501
1519
|
if (block.type === "text") {
|
|
1502
1520
|
stream.push({
|
|
@@ -1513,7 +1531,9 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1513
1531
|
partial: output,
|
|
1514
1532
|
});
|
|
1515
1533
|
} else if (block.type === "toolCall") {
|
|
1516
|
-
|
|
1534
|
+
if (block.partialJson.trim()) {
|
|
1535
|
+
block.arguments = parseStreamingJson(block.partialJson);
|
|
1536
|
+
}
|
|
1517
1537
|
delete (block as { partialJson?: string }).partialJson;
|
|
1518
1538
|
stream.push({
|
|
1519
1539
|
type: "toolcall_end",
|
package/src/stream.ts
CHANGED
|
@@ -12,22 +12,15 @@ import {
|
|
|
12
12
|
import type { BedrockOptions } from "./providers/amazon-bedrock";
|
|
13
13
|
import type { AnthropicOptions } from "./providers/anthropic";
|
|
14
14
|
import type { CursorOptions } from "./providers/cursor";
|
|
15
|
-
import { isGitLabDuoModel, streamGitLabDuo } from "./providers/gitlab-duo";
|
|
16
15
|
import type { GoogleOptions } from "./providers/google";
|
|
17
16
|
import type { GoogleGeminiCliOptions } from "./providers/google-gemini-cli";
|
|
18
17
|
import type { GoogleVertexOptions } from "./providers/google-vertex";
|
|
19
|
-
import { isKimiModel, streamKimi } from "./providers/kimi";
|
|
20
18
|
import type { OllamaChatOptions } from "./providers/ollama";
|
|
21
19
|
import type { OpenAICompletionsOptions } from "./providers/openai-completions";
|
|
22
|
-
import { streamPiNative } from "./providers/pi-native-client";
|
|
23
20
|
// Heavy provider stream functions are imported lazily via register-builtins,
|
|
24
|
-
// which wraps each provider module in a dynamic import.
|
|
25
|
-
//
|
|
26
|
-
//
|
|
27
|
-
// gitlab-duo / kimi / synthetic providers stay eager because their modules
|
|
28
|
-
// export routing predicates (isGitLabDuoModel, isKimiModel, isSyntheticModel)
|
|
29
|
-
// that must be callable synchronously before streaming begins, and their
|
|
30
|
-
// modules are thin wrappers with no heavy SDK dependencies.
|
|
21
|
+
// which wraps each provider module in a dynamic import. Thin provider routing
|
|
22
|
+
// modules are also loaded lazily below by returning an outer stream and piping
|
|
23
|
+
// the dynamically imported inner stream into it.
|
|
31
24
|
import {
|
|
32
25
|
streamAnthropic,
|
|
33
26
|
streamAzureOpenAIResponses,
|
|
@@ -41,7 +34,6 @@ import {
|
|
|
41
34
|
streamOpenAICompletions,
|
|
42
35
|
streamOpenAIResponses,
|
|
43
36
|
} from "./providers/register-builtins";
|
|
44
|
-
import { isSyntheticModel, streamSynthetic } from "./providers/synthetic";
|
|
45
37
|
import type {
|
|
46
38
|
Api,
|
|
47
39
|
AssistantMessage,
|
|
@@ -240,6 +232,45 @@ export function formatProviderCredentialHint(provider: string): string {
|
|
|
240
232
|
}
|
|
241
233
|
return parts.join(" ");
|
|
242
234
|
}
|
|
235
|
+
function pipeAssistantStream(
|
|
236
|
+
outer: AssistantMessageEventStream,
|
|
237
|
+
inner: AssistantMessageEventStream,
|
|
238
|
+
signal?: AbortSignal,
|
|
239
|
+
): void {
|
|
240
|
+
void (async () => {
|
|
241
|
+
try {
|
|
242
|
+
for await (const event of inner) {
|
|
243
|
+
outer.push(event);
|
|
244
|
+
// The inner provider stream owns abort semantics (it receives the
|
|
245
|
+
// same signal), but stop forwarding as soon as the consumer
|
|
246
|
+
// aborted so a misbehaving inner stream cannot keep the pipe
|
|
247
|
+
// buffering events indefinitely.
|
|
248
|
+
if (signal?.aborted && !outer.done) {
|
|
249
|
+
outer.end(await inner.result());
|
|
250
|
+
return;
|
|
251
|
+
}
|
|
252
|
+
}
|
|
253
|
+
if (!outer.done) outer.end(await inner.result());
|
|
254
|
+
} catch (error) {
|
|
255
|
+
outer.fail(error);
|
|
256
|
+
}
|
|
257
|
+
})();
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
function streamFromLazyImport(
|
|
261
|
+
createInner: () => Promise<AssistantMessageEventStream>,
|
|
262
|
+
signal?: AbortSignal,
|
|
263
|
+
): AssistantMessageEventStream {
|
|
264
|
+
const outer = new AssistantMessageEventStream();
|
|
265
|
+
void (async () => {
|
|
266
|
+
try {
|
|
267
|
+
pipeAssistantStream(outer, await createInner(), signal);
|
|
268
|
+
} catch (error) {
|
|
269
|
+
outer.fail(error);
|
|
270
|
+
}
|
|
271
|
+
})();
|
|
272
|
+
return outer;
|
|
273
|
+
}
|
|
243
274
|
|
|
244
275
|
/**
|
|
245
276
|
* Build an actionable "missing API key" error for a provider, used by the
|
|
@@ -262,15 +293,21 @@ export function stream<TApi extends Api>(
|
|
|
262
293
|
return customApiProvider.stream(model, context, options as StreamOptions);
|
|
263
294
|
}
|
|
264
295
|
|
|
265
|
-
if (
|
|
296
|
+
if (model.provider === "gitlab-duo") {
|
|
266
297
|
const apiKey = (options as StreamOptions | undefined)?.apiKey || getEnvApiKey(model.provider);
|
|
267
298
|
if (!apiKey) {
|
|
268
299
|
throw new Error(formatMissingApiKeyError(model.provider));
|
|
269
300
|
}
|
|
270
|
-
return
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
301
|
+
return streamFromLazyImport(
|
|
302
|
+
async () => {
|
|
303
|
+
const { streamGitLabDuo } = await import("./providers/gitlab-duo");
|
|
304
|
+
return streamGitLabDuo(model, context, {
|
|
305
|
+
...(options as SimpleStreamOptions | undefined),
|
|
306
|
+
apiKey,
|
|
307
|
+
});
|
|
308
|
+
},
|
|
309
|
+
(options as StreamOptions | undefined)?.signal,
|
|
310
|
+
);
|
|
274
311
|
}
|
|
275
312
|
|
|
276
313
|
// Vertex AI uses Application Default Credentials, not API keys
|
|
@@ -446,7 +483,10 @@ export function streamSimple<TApi extends Api>(
|
|
|
446
483
|
// extension-registered APIs can't accidentally override a configured
|
|
447
484
|
// pi-native transport.
|
|
448
485
|
if (model.transport === "pi-native") {
|
|
449
|
-
return
|
|
486
|
+
return streamFromLazyImport(async () => {
|
|
487
|
+
const { streamPiNative } = await import("./providers/pi-native-client");
|
|
488
|
+
return streamPiNative(model, context, options);
|
|
489
|
+
}, options?.signal);
|
|
450
490
|
}
|
|
451
491
|
|
|
452
492
|
// Check custom API registry (extension-provided APIs)
|
|
@@ -471,31 +511,40 @@ export function streamSimple<TApi extends Api>(
|
|
|
471
511
|
}
|
|
472
512
|
|
|
473
513
|
// GitLab Duo - wraps Anthropic/OpenAI behind GitLab AI Gateway direct access tokens
|
|
474
|
-
if (
|
|
475
|
-
return
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
514
|
+
if (model.provider === "gitlab-duo") {
|
|
515
|
+
return streamFromLazyImport(async () => {
|
|
516
|
+
const { streamGitLabDuo } = await import("./providers/gitlab-duo");
|
|
517
|
+
return streamGitLabDuo(model, context, {
|
|
518
|
+
...options,
|
|
519
|
+
apiKey,
|
|
520
|
+
});
|
|
521
|
+
}, options?.signal);
|
|
479
522
|
}
|
|
480
523
|
|
|
481
524
|
// Kimi Code - route to dedicated handler that wraps OpenAI or Anthropic API
|
|
482
|
-
if (
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
|
|
525
|
+
if (model.provider === "kimi-code") {
|
|
526
|
+
return streamFromLazyImport(async () => {
|
|
527
|
+
const { streamKimi } = await import("./providers/kimi");
|
|
528
|
+
// Pass raw SimpleStreamOptions - streamKimi handles mapping internally
|
|
529
|
+
return streamKimi(model as Model<"openai-completions">, context, {
|
|
530
|
+
...options,
|
|
531
|
+
apiKey,
|
|
532
|
+
format: options?.kimiApiFormat ?? "anthropic",
|
|
533
|
+
});
|
|
534
|
+
}, options?.signal);
|
|
489
535
|
}
|
|
490
536
|
|
|
491
537
|
// Synthetic - route to dedicated handler that wraps OpenAI or Anthropic API
|
|
492
|
-
if (
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
538
|
+
if (model.provider === "synthetic") {
|
|
539
|
+
return streamFromLazyImport(async () => {
|
|
540
|
+
const { streamSynthetic } = await import("./providers/synthetic");
|
|
541
|
+
// Pass raw SimpleStreamOptions - streamSynthetic handles mapping internally
|
|
542
|
+
return streamSynthetic(model as Model<"openai-completions">, context, {
|
|
543
|
+
...options,
|
|
544
|
+
apiKey,
|
|
545
|
+
format: options?.syntheticApiFormat ?? "openai", // Default to OpenAI format
|
|
546
|
+
});
|
|
547
|
+
}, options?.signal);
|
|
499
548
|
}
|
|
500
549
|
|
|
501
550
|
const providerOptions = mapOptionsForApi(model, options, apiKey);
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import * as z from "zod/v4";
|
|
2
|
+
import { isRetiredModelKey } from "../../model-retirements";
|
|
2
3
|
import { getAntigravityUserAgent } from "../../providers/google-gemini-headers";
|
|
3
4
|
import type { Model } from "../../types";
|
|
4
|
-
import { toPositiveNumber } from "../../utils";
|
|
5
5
|
|
|
6
6
|
const DEFAULT_ANTIGRAVITY_DISCOVERY_ENDPOINTS = [
|
|
7
7
|
"https://daily-cloudcode-pa.googleapis.com",
|
|
@@ -162,6 +162,12 @@ export interface FetchAntigravityDiscoveryModelsOptions {
|
|
|
162
162
|
signal?: AbortSignal;
|
|
163
163
|
/** Optional fetch implementation override for tests. */
|
|
164
164
|
fetcher?: typeof fetch;
|
|
165
|
+
/**
|
|
166
|
+
* Provider id the caller assigns to returned models. Scopes retired-selector
|
|
167
|
+
* filtering (e.g. `google-gemini-cli` reuses this helper and remaps rows).
|
|
168
|
+
* Default: `google-antigravity`.
|
|
169
|
+
*/
|
|
170
|
+
targetProvider?: "google-antigravity" | "google-gemini-cli";
|
|
165
171
|
}
|
|
166
172
|
|
|
167
173
|
/**
|
|
@@ -174,6 +180,7 @@ export async function fetchAntigravityDiscoveryModels(
|
|
|
174
180
|
options: FetchAntigravityDiscoveryModelsOptions,
|
|
175
181
|
): Promise<Model<"google-gemini-cli">[] | null> {
|
|
176
182
|
const fetcher = options.fetcher ?? fetch;
|
|
183
|
+
const targetProvider = options.targetProvider ?? "google-antigravity";
|
|
177
184
|
const endpoints = options.endpoint
|
|
178
185
|
? [trimTrailingSlashes(options.endpoint)]
|
|
179
186
|
: DEFAULT_ANTIGRAVITY_DISCOVERY_ENDPOINTS.map(trimTrailingSlashes);
|
|
@@ -214,7 +221,7 @@ export async function fetchAntigravityDiscoveryModels(
|
|
|
214
221
|
const models: Model<"google-gemini-cli">[] = [];
|
|
215
222
|
|
|
216
223
|
for (const [modelId, model] of Object.entries(parsed.models ?? {})) {
|
|
217
|
-
if (ANTIGRAVITY_DISCOVERY_DENYLIST.has(modelId)) {
|
|
224
|
+
if (ANTIGRAVITY_DISCOVERY_DENYLIST.has(modelId) || isRetiredModelKey(targetProvider, modelId)) {
|
|
218
225
|
continue;
|
|
219
226
|
}
|
|
220
227
|
if (model.isInternal === true) {
|
|
@@ -256,6 +263,13 @@ function parseAntigravityDiscoveryResponse(value: unknown): AntigravityDiscovery
|
|
|
256
263
|
return parsed.data;
|
|
257
264
|
}
|
|
258
265
|
|
|
266
|
+
function toPositiveNumber(value: unknown, fallback: number): number {
|
|
267
|
+
if (typeof value !== "number" || !Number.isFinite(value) || value <= 0) {
|
|
268
|
+
return fallback;
|
|
269
|
+
}
|
|
270
|
+
return value;
|
|
271
|
+
}
|
|
272
|
+
|
|
259
273
|
function trimTrailingSlashes(value: string): string {
|
|
260
274
|
return value.replace(/\/+$/, "");
|
|
261
275
|
}
|