agent-accelerator 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/README.md +83 -112
  2. package/{SYSTEM_PROMPT_AGENT.md → examples/prompts/SYSTEM_PROMPT_AGENT.md} +1 -1
  3. package/package.json +14 -9
  4. package/src/agent/agent.ts +191 -71
  5. package/src/agent/context.ts +1 -0
  6. package/src/agent/delegation.ts +61 -12
  7. package/src/agent/loop.ts +71 -48
  8. package/src/data/README.md +6 -6
  9. package/src/index.ts +130 -45
  10. package/src/models/catalog-cache.ts +60 -9
  11. package/src/models/catalog.ts +53 -7
  12. package/src/providers/google.ts +926 -0
  13. package/src/providers/openai-compat.ts +1147 -0
  14. package/src/providers/openai.ts +959 -0
  15. package/src/providers/openrouter-responses.ts +949 -0
  16. package/src/providers/openrouter.ts +1037 -0
  17. package/src/{ai-sdk → providers}/registry.ts +43 -55
  18. package/src/providers.ts +490 -0
  19. package/src/streaming/sse-parser.ts +6 -4
  20. package/src/tools/executor.ts +19 -6
  21. package/src/tools/schema.ts +21 -11
  22. package/src/types/agent.ts +8 -1
  23. package/src/types/core.ts +1 -1
  24. package/src/types/message.ts +5 -0
  25. package/src/types/model.ts +3 -7
  26. package/src/types/provider-payloads.ts +2 -84
  27. package/src/types/tool.ts +6 -0
  28. package/src/update-models.ts +56 -0
  29. package/src/utils/cache.ts +1 -1
  30. package/src/utils/documents.ts +517 -0
  31. package/src/utils/env.ts +0 -7
  32. package/src/{ai-sdk → utils}/errors.ts +61 -2
  33. package/src/utils/headers.ts +10 -20
  34. package/src/utils/media.ts +5 -2
  35. package/src/utils/retry.ts +89 -0
  36. package/src/utils/serialization.ts +15 -0
  37. package/src/ai-sdk/converters.ts +0 -342
  38. package/src/ai-sdk/executor.ts +0 -454
  39. package/src/ai-sdk/index.ts +0 -55
  40. package/src/ai-sdk/model-provider.ts +0 -303
  41. package/src/ai-sdk/options.ts +0 -306
  42. package/src/ai-sdk/provider.ts +0 -415
  43. package/src/tokens/counter.ts +0 -136
  44. /package/{SYSTEM_PROMPT.md → examples/prompts/SYSTEM_PROMPT.md} +0 -0
  45. /package/{SYSTEM_PROMPT_TOOLS.md → examples/prompts/SYSTEM_PROMPT_TOOLS.md} +0 -0
package/src/agent/loop.ts CHANGED
@@ -7,8 +7,9 @@ import { AssistantMessageEventStream } from "../streaming/event-stream.ts";
7
7
  import { AgentContext } from "./context.ts";
8
8
  import { toStandardToolDeclarations } from "../tools/tool.ts";
9
9
  import { executeToolCalls } from "../tools/executor.ts";
10
- import { getModelFromCatalog, ensureModelCatalogFresh } from "../models/catalog.ts";
11
- import { countTokens } from "../tokens/counter.ts";
10
+ import { getModelFromCatalog, ensureModelCatalogFresh, validateModelThinking } from "../models/catalog.ts";
11
+ import { preprocessFilePartsForBypass } from "../utils/documents.ts";
12
+ import { noteProviderTurn } from "../providers.ts";
12
13
 
13
14
  export interface AgentLoopConfig {
14
15
  agentName?: string;
@@ -19,6 +20,8 @@ export interface AgentLoopConfig {
19
20
  options?: ProviderRequestOptions;
20
21
  runOptions?: AgentRunOptions;
21
22
  maxTurns?: number;
23
+ /** Convert `file` parts client-side for pdf-incapable models (see Agent flag). */
24
+ bypassInputFileModality?: boolean;
22
25
  }
23
26
 
24
27
  export function computeCostFromPricing(usage: TokenUsage, spec?: ModelSpec): TokenUsage["cost"] {
@@ -67,6 +70,13 @@ function accumulateUsage(target: TokenUsage, source: TokenUsage, spec?: ModelSpe
67
70
  target.cacheWriteTokens = (target.cacheWriteTokens ?? 0) + (source.cacheWriteTokens ?? 0);
68
71
  target.thinkingTokens = (target.thinkingTokens ?? 0) + (source.thinkingTokens ?? 0);
69
72
 
73
+ // Canonical invariant (provider-agnostic): cache hits are a subset of
74
+ // input. Some providers occasionally report cached > input on long chained
75
+ // runs, which surfaces as >100% hit rates downstream. Clamp the aggregate
76
+ // so no consumer can observe an impossible ratio.
77
+ target.cachedTokens = Math.min(target.cachedTokens ?? 0, target.inputTokens);
78
+ target.cacheReadTokens = Math.min(target.cacheReadTokens ?? 0, target.inputTokens);
79
+
70
80
  const sourceCost = computeCostFromPricing(source, spec);
71
81
  if (sourceCost) {
72
82
  source.cost = sourceCost;
@@ -155,12 +165,26 @@ async function executeToolCallsWithRepeatGuard(options: {
155
165
  sessionId: options.sessionId,
156
166
  })
157
167
  : [];
158
- const byId = new Map<string, ToolResultRecord>();
159
- for (const result of executed) byId.set(result.id, result);
160
- for (const [id, result] of blocked) byId.set(id, result);
168
+ const byId = new Map<string, ToolResultRecord[]>();
169
+ for (const result of executed) {
170
+ const queue = byId.get(result.id) ?? [];
171
+ queue.push(result);
172
+ byId.set(result.id, queue);
173
+ }
174
+ for (const [id, result] of blocked) {
175
+ const queue = byId.get(id) ?? [];
176
+ queue.push(result);
177
+ byId.set(id, queue);
178
+ }
161
179
 
162
180
  return {
163
- results: toolCalls.map((call) => byId.get(call.id)!).filter(Boolean),
181
+ // Consume per-id queues in call order: some open-weights models reuse one
182
+ // id for several distinct calls in a turn, and a plain id->result map
183
+ // would hand every call the last result. Queues keep each result aligned
184
+ // with its own call.
185
+ results: toolCalls
186
+ .map((call) => byId.get(call.id)?.shift())
187
+ .filter((r): r is ToolResultRecord => Boolean(r)),
164
188
  fingerprints: currentFingerprints,
165
189
  };
166
190
  }
@@ -181,6 +205,14 @@ export async function runAgentLoop(config: AgentLoopConfig): Promise<AgentRespon
181
205
  maxTurns = 10,
182
206
  } = config;
183
207
 
208
+ if (config.bypassInputFileModality) {
209
+ await preprocessFilePartsForBypass(config.context.messages, {
210
+ providerId: provider.id,
211
+ modelId,
212
+ signal: runOptions?.signal,
213
+ });
214
+ }
215
+
184
216
  const standardTools = toStandardToolDeclarations(tools);
185
217
  const startTime = Date.now();
186
218
 
@@ -204,30 +236,7 @@ export async function runAgentLoop(config: AgentLoopConfig): Promise<AgentRespon
204
236
  while (turns < maxTurns) {
205
237
  turns++;
206
238
 
207
- // Battle-tested context window check — trim oldest history if needed, keep cached prefix (system+tools)
208
239
  const spec = getModelFromCatalog(provider.id, modelId);
209
- const contextWindow = spec?.limit?.context ?? spec?.contextWindow ?? 128000;
210
- const maxOutput = spec?.limit?.output ?? spec?.maxOutputTokens ?? 8192;
211
- // Reserve for output + 10% headroom, ensure first turn already cache-friendly
212
- const budgetForInput = Math.floor(contextWindow * 0.9) - maxOutput;
213
- let estimated = 0;
214
- try {
215
- estimated = countTokens({ systemPrompt: context.systemPrompt, messages: context.messages, tools: standardTools as any });
216
- } catch {}
217
- if (estimated > budgetForInput && context.messages.length > 2) {
218
- // Keep system + last 70% of history, drop oldest middle (preserve cached prefix stability)
219
- const keepCount = Math.max(2, Math.floor(context.messages.length * 0.7));
220
- const toKeep = context.messages.slice(-keepCount);
221
- // Preserve at least system and first user if possible
222
- const firstUserIdx = context.messages.findIndex((m) => m.role === "user");
223
- if (firstUserIdx >= 0 && firstUserIdx < context.messages.length - keepCount) {
224
- // Drop middle, keep head + tail for cache stability
225
- const head = context.messages.slice(0, 1);
226
- context.messages = [...head, ...toKeep];
227
- } else {
228
- context.messages = toKeep;
229
- }
230
- }
231
240
 
232
241
  const providerOptions: ProviderRequestOptions = {
233
242
  ...options,
@@ -250,6 +259,12 @@ export async function runAgentLoop(config: AgentLoopConfig): Promise<AgentRespon
250
259
 
251
260
  finalResult = genResult;
252
261
 
262
+ // Canonical session routing: lets adapters detect provider switches.
263
+ noteProviderTurn(
264
+ runOptions?.sessionId || options?.sessionId || options?.cache?.sessionId,
265
+ provider.id
266
+ );
267
+
253
268
  // Safety fallback: if no tool calls and text is empty, rescue answer from thinking
254
269
  if ((!genResult.text || genResult.text.trim() === "") && (!genResult.toolCalls || genResult.toolCalls.length === 0) && genResult.thinking) {
255
270
  if (genResult.thinking.includes("</think>")) {
@@ -340,7 +355,7 @@ export function streamAgentLoop(config: AgentLoopConfig): AssistantMessageEventS
340
355
  const startTime = Date.now();
341
356
 
342
357
  // Cancellation: merge user signal + outer cancel() into one linked controller.
343
- // Vercel doStream honors abortSignal for all providers, so aborting this
358
+ // Native fetch honors abortSignal on every provider, so aborting this
344
359
  // stops the HTTP request; we also cancel the active inner provider stream.
345
360
  const linked = new AbortController();
346
361
  const userSignal = config.runOptions?.signal;
@@ -365,6 +380,27 @@ export function streamAgentLoop(config: AgentLoopConfig): AssistantMessageEventS
365
380
  (async () => {
366
381
  try {
367
382
  await ensureModelCatalogFresh();
383
+ // Re-validate after refresh: Agent.stream() validates before the catalog
384
+ // is guaranteed fresh, so a cold cache validates permissively. A mismatch
385
+ // discovered here fails the stream instead of reaching the provider.
386
+ {
387
+ const lvl = config.runOptions?.thinkingLevel ?? config.options?.thinking?.level;
388
+ if (lvl) {
389
+ try {
390
+ validateModelThinking(config.provider.id, config.modelId, lvl);
391
+ } catch (err) {
392
+ outerStream.fail(err instanceof Error ? err : new Error(String(err)));
393
+ return;
394
+ }
395
+ }
396
+ }
397
+ if (config.bypassInputFileModality) {
398
+ await preprocessFilePartsForBypass(config.context.messages, {
399
+ providerId: config.provider.id,
400
+ modelId: config.modelId,
401
+ signal: config.runOptions?.signal,
402
+ });
403
+ }
368
404
  const {
369
405
  agentName,
370
406
  provider,
@@ -399,25 +435,7 @@ export function streamAgentLoop(config: AgentLoopConfig): AssistantMessageEventS
399
435
  throwIfCancelled();
400
436
  turns++;
401
437
 
402
- // Same context-window trim as non-stream (ensure cache prefix stable)
403
438
  const spec2 = getModelFromCatalog(provider.id, modelId);
404
- const cw2 = spec2?.limit?.context ?? spec2?.contextWindow ?? 128000;
405
- const mo2 = spec2?.limit?.output ?? spec2?.maxOutputTokens ?? 8192;
406
- const budget2 = Math.floor(cw2 * 0.9) - mo2;
407
- try {
408
- const est2 = countTokens({ systemPrompt: context.systemPrompt, messages: context.messages, tools: standardTools as any });
409
- if (est2 > budget2 && context.messages.length > 2) {
410
- const keep2 = Math.max(2, Math.floor(context.messages.length * 0.7));
411
- const toKeep = context.messages.slice(-keep2);
412
- const firstUserIdx = context.messages.findIndex((m) => m.role === "user");
413
- if (firstUserIdx >= 0 && firstUserIdx < context.messages.length - keep2) {
414
- const head = context.messages.slice(0, 1);
415
- context.messages = [...head, ...toKeep];
416
- } else {
417
- context.messages = toKeep;
418
- }
419
- }
420
- } catch {}
421
439
 
422
440
  const providerOptions: ProviderRequestOptions = {
423
441
  ...options,
@@ -449,6 +467,11 @@ export function streamAgentLoop(config: AgentLoopConfig): AssistantMessageEventS
449
467
  currentInner = null;
450
468
  throwIfCancelled();
451
469
  lastResponse = turnResponse;
470
+ // Canonical session routing: lets adapters detect provider switches.
471
+ noteProviderTurn(
472
+ runOptions?.sessionId || options?.sessionId || options?.cache?.sessionId,
473
+ provider.id
474
+ );
452
475
 
453
476
  // Safety fallback: if no tool calls and text is empty, rescue answer from thinking
454
477
  if ((!turnResponse.text || turnResponse.text.trim() === "") && (!turnResponse.toolCalls || turnResponse.toolCalls.length === 0) && turnResponse.thinking) {
@@ -11,13 +11,13 @@ Agent Accelerator references a catalog of **7,500+ models across 200+ providers*
11
11
  To keep the repository and published npm/Bun package lightweight and blazing fast:
12
12
 
13
13
  1. **Excluded from Production Package & Git**:
14
- * The ~4.5 MB full catalog snapshot (`models-cache.json` / `models.dev.json`) is **never** committed to git and is **excluded** from the npm/Bun distribution bundle.
15
- * Both files are tracked in `.gitignore` and `.npmignore`.
14
+ * The ~4.5 MB full catalog snapshot (`models.dev.json`) is **never** committed to git and is **excluded** from the npm/Bun distribution bundle.
15
+ * The file is tracked in `.gitignore` and `.npmignore`.
16
16
 
17
17
  2. **Dynamic 12-Hour Automated TTL Cache**:
18
- * On agent execution (`runAgentLoop` / `streamAgentLoop`), Agent Accelerator checks the status of `src/data/models-cache.json`.
18
+ * On agent execution (`runAgentLoop` / `streamAgentLoop`), Agent Accelerator checks the status of `src/data/models.dev.json`.
19
19
  * **Cache Hit (< 12h)**: Instant memory/disk load with zero network overhead.
20
- * **Cache Miss / Expired (≥ 12h)**: Automatically downloads the latest catalog directly from `https://models.dev/api.json`, caches it to `src/data/models-cache.json`, and loads the catalog into memory.
20
+ * **Cache Miss / Expired (≥ 12h)**: Automatically downloads the latest catalog directly from `https://models.dev/api.json`, caches it to `src/data/models.dev.json`, and loads the catalog into memory.
21
21
  * **Resilience**: If an existing cache exists on disk, temporary upstream network issues will safely continue using the local cached copy.
22
22
 
23
23
  ---
@@ -36,7 +36,7 @@ bun run update-models
36
36
  bun run update-models --force
37
37
 
38
38
  # Custom TTL (e.g., 24 hours)
39
- bun scripts/update-models.ts --ttl=24h
39
+ bun src/update-models.ts --ttl=24h
40
40
  ```
41
41
 
42
42
  ---
@@ -80,5 +80,5 @@ setCatalogTTL(6 * 60 * 60 * 1000);
80
80
  ```text
81
81
  src/data/
82
82
  ├── README.md # This documentation
83
- └── models-cache.json # Auto-generated on first run (gitignored, excluded from bundle)
83
+ └── models.dev.json # Auto-generated on first run (gitignored, excluded from bundle)
84
84
  ```
package/src/index.ts CHANGED
@@ -9,21 +9,43 @@ export type { CreateToolOptions } from "./tools/tool.ts";
9
9
  export { zodToJsonSchema, cleanJsonSchema } from "./tools/schema.ts";
10
10
  export { executeToolCalls } from "./tools/executor.ts";
11
11
 
12
+ // Client-side document conversion (optional anydoc peer — see utils/documents.ts)
13
+ export {
14
+ convertDocumentToMarkdown,
15
+ convertDocumentInput,
16
+ convert_document_to_markdown,
17
+ isAnydocAvailable,
18
+ resolveDocumentFormat,
19
+ truncateMarkdown,
20
+ buildDocumentXml,
21
+ modelSupportsFileInput,
22
+ preprocessFilePartsForBypass,
23
+ DocumentConversionError,
24
+ DEFAULT_DOCUMENT_MAX_CHARS,
25
+ MAX_DOCUMENT_FETCH_BYTES,
26
+ } from "./utils/documents.ts";
27
+ export type { ConvertDocumentOptions, ConvertedDocument } from "./utils/documents.ts";
28
+
12
29
  // Multi-Agent Delegation & Tools
13
30
  export {
14
31
  buildAgentTools,
32
+ agentToTool,
15
33
  createSubagentSpawnTool,
16
34
  } from "./agent/delegation.ts";
17
- export type { DynamicSubagentTask } from "./agent/delegation.ts";
35
+ export type { DynamicSubagentTask, AgentAsToolTarget } from "./agent/delegation.ts";
18
36
 
19
- // Providers & Registry — unified multi-provider layer powered by Vercel AI SDK
37
+ // Providers & Registry — unified multi-provider layer, all native REST
20
38
  export {
21
39
  getProvider,
22
40
  resolveModel,
23
41
  ModelProvider,
24
42
  ensureCustomProvider,
25
43
  normalizeProviderPrefix,
26
- } from "./ai-sdk/registry.ts";
44
+ } from "./providers/registry.ts";
45
+ export type {
46
+ ModelProviderInstance,
47
+ ModelProviderConfig,
48
+ } from "./providers/registry.ts";
27
49
  export {
28
50
  getModelFromCatalog,
29
51
  getModelThinkingInfo,
@@ -37,48 +59,78 @@ export {
37
59
  getCatalogTTL,
38
60
  getCacheDir,
39
61
  getCacheFilePath,
62
+ isValidCatalogPayload,
63
+ createGenericModelSpec,
40
64
  DEFAULT_CATALOG_TTL_MS,
41
65
  type CatalogStatus,
42
66
  type RefreshCatalogOptions,
43
67
  } from "./models/catalog.ts";
44
- export { AiSdkBaseProvider as BaseProvider } from "./ai-sdk/model-provider.ts";
45
- export {
46
- mapThinkingToReasoning,
47
- resolveEffectiveThinking,
48
- mapToolChoice,
49
- mapThinkingToProviderOptions,
50
- mapServiceTierToProviderOptions,
51
- mapCacheToProviderOptions,
52
- buildAiSdkCallOptions,
53
- withAiSdkRetries,
54
- isTransientAiSdkError,
55
- } from "./ai-sdk/options.ts";
56
- export { toConciseProviderError, assertModalitiesSupported } from "./ai-sdk/errors.ts";
57
-
58
- // Provider Implementations powered by Vercel AI SDK
68
+ export { resolveEffectiveThinking } from "./agent/agent.ts";
69
+ export { withRetries, isTransientError } from "./utils/retry.ts";
70
+ export { toConciseProviderError, assertModalitiesSupported, assertNoVideoPartsOnResponses } from "./utils/errors.ts";
71
+ // Native provider implementations. `openrouter-responses.ts` (discontinued
72
+ // beta-Responses transport) stays importable directly but is wired nowhere.
59
73
  export {
60
- GoogleAIStudioProvider,
61
- GoogleAiSdkProvider,
62
- OpenCodeProvider,
63
- OpenCodeAiSdkProvider,
64
- OpenRouterProvider,
65
- OpenRouterAiSdkProvider,
66
- OpenAIProvider,
67
- OpenAiAiSdkProvider,
68
- OpenAICompatibleProvider,
69
- CustomAiSdkProvider,
70
- AiSdkBaseProvider,
74
+ OpenAICompatibleChatProvider,
75
+ OpenAICompatibleChatProvider as OpenAICompatibleProvider,
71
76
  createOpenAICompatibleProvider,
72
77
  createCustomProvider,
73
78
  CustomProvider,
74
- createGenericModelSpec,
75
- } from "./ai-sdk/model-provider.ts";
76
- export type { CustomProviderOptions } from "./ai-sdk/model-provider.ts";
79
+ } from "./providers/openai-compat.ts";
80
+ export type { CustomProviderOptions } from "./providers/openai-compat.ts";
81
+
82
+ // Canonical provider contract (provider-agnostic) + native adapters.
83
+ export {
84
+ GoogleInteractionsProvider,
85
+ GoogleInteractionsProvider as GoogleAIStudioProvider,
86
+ clearInteractionChains,
87
+ } from "./providers/google.ts";
88
+ export {
89
+ OpenRouterChatCompletionsProvider,
90
+ OpenRouterChatCompletionsProvider as OpenRouterProvider,
91
+ } from "./providers/openrouter.ts";
92
+ // Discontinued beta-Responses transport, retained frozen as migration
93
+ // evidence (importable via `./providers/openrouter-responses.ts`, not wired
94
+ // anywhere): OpenRouterResponsesProvider.
95
+ export {
96
+ OpenAIResponsesProvider,
97
+ OpenAIResponsesProvider as OpenAIProvider,
98
+ } from "./providers/openai.ts";
99
+ export {
100
+ emitProviderWarning,
101
+ clearEmittedWarnings,
102
+ noteProviderTurn,
103
+ lastProviderFor,
104
+ isMixedProviderSession,
105
+ clearSessionRouting,
106
+ mapThinkingLevelToGoogle,
107
+ mapServiceTierToGoogle,
108
+ applyCacheForGoogle,
109
+ mapThinkingLevelToOpenRouter,
110
+ mapServiceTierToOpenRouter,
111
+ applyCacheForOpenRouter,
112
+ mapToolChoiceToOpenRouter,
113
+ mapThinkingLevelToOpenRouterChat,
114
+ mapToolChoiceToOpenRouterChat,
115
+ mapThinkingLevelToOpenAI,
116
+ mapServiceTierToOpenAI,
117
+ applyCacheForOpenAI,
118
+ applyCacheForCustom,
119
+ mapToolChoiceToOpenAI,
120
+ normalizeToolChoice,
121
+ parseStreamedToolArguments,
122
+ } from "./providers.ts";
123
+ export type {
124
+ ProviderCapabilityStatus,
125
+ CanonicalToolChoiceMode,
126
+ OpenRouterToolChoice,
127
+ OpenRouterChatToolChoice,
128
+ OpenAIToolChoice,
129
+ } from "./providers.ts";
77
130
 
78
131
  export {
79
132
  GOOGLE_MODELS,
80
133
  OPENAI_MODELS,
81
- OPENCODE_MODELS,
82
134
  OPENROUTER_MODELS,
83
135
  } from "./models/catalog.ts";
84
136
 
@@ -101,14 +153,6 @@ export type {
101
153
  CachedContentMetadata,
102
154
  } from "./utils/cache.ts";
103
155
 
104
-
105
- export {
106
- countTokens,
107
- estimateTokensFromText,
108
- estimateTokensFromMessage,
109
- estimateTokensFromPart,
110
- } from "./tokens/counter.ts";
111
-
112
156
  // Streaming & Events
113
157
  export { AssistantMessageEventStream } from "./streaming/event-stream.ts";
114
158
  export { SSEParser } from "./streaming/sse-parser.ts";
@@ -122,7 +166,7 @@ export { getApiKey, getEnv } from "./utils/env.ts";
122
166
  export { buildSessionHeaders } from "./utils/headers.ts";
123
167
  export { normalizeMediaInput, inferMimeType } from "./utils/media.ts";
124
168
  export { base64ToBytes, bytesToBase64 } from "./utils/base64.ts";
125
- export { toJsonSafe, safeStringify } from "./utils/serialization.ts";
169
+ export { toJsonSafe, safeStringify, escapeXml } from "./utils/serialization.ts";
126
170
 
127
171
  // Re-export Zod
128
172
  export { z };
@@ -184,7 +228,48 @@ export type {
184
228
  DynamicSubagentsConfig,
185
229
  } from "./types/agent.ts";
186
230
 
231
+ // Provider wire payload types (back-compat inspection of raw requests/responses,
232
+ // e.g. `res.raw.request.body as OpenAIChatCompletionRequest`).
233
+ // NOTE: wire `OpenAIToolChoice` intentionally omitted — it duplicates the
234
+ // canonical `OpenAIToolChoice` exported above (same shape); use that one.
187
235
  export type {
188
- ModelProviderInstance,
189
- ModelProviderConfig,
190
- } from "./ai-sdk/registry.ts";
236
+ GoogleThinkingLevel,
237
+ GoogleThinkingConfig,
238
+ GoogleFunctionCallingMode,
239
+ GoogleFunctionCallingConfig,
240
+ GoogleToolConfig,
241
+ GoogleFunctionDeclaration,
242
+ GoogleTool,
243
+ GoogleBlob,
244
+ GooglePart,
245
+ GoogleContent,
246
+ GoogleGenerationConfig,
247
+ GoogleCandidate,
248
+ GoogleUsageMetadata,
249
+ GoogleGenerateContentRequest,
250
+ GoogleGenerateContentResponse,
251
+ OpenRouterProviderRouting,
252
+ OpenRouterReasoning,
253
+ OpenRouterParameters,
254
+ OpenRouterUsage,
255
+ OpenRouterChatRequest,
256
+ OpenRouterResponse,
257
+ OpenAIMessageRole,
258
+ OpenAIReasoningEffort,
259
+ OpenAIServiceTier,
260
+ OpenAITextPart,
261
+ OpenAIImageUrlPart,
262
+ OpenAIInputAudioPart,
263
+ OpenAIVideoUrlPart,
264
+ OpenAIContentPart,
265
+ OpenAIToolCall,
266
+ OpenAIMessage,
267
+ OpenAITool,
268
+ OpenAIUsage,
269
+ OpenAIChoice,
270
+ OpenAIDelta,
271
+ OpenAIChunkChoice,
272
+ OpenAIChatCompletionRequest,
273
+ OpenAIChatCompletionResponse,
274
+ OpenAIChatCompletionChunk,
275
+ } from "./types/provider-payloads.ts";
@@ -1,5 +1,6 @@
1
1
  import * as path from "node:path";
2
2
  import * as fs from "node:fs";
3
+ import { getEnv } from "../utils/env.ts";
3
4
 
4
5
  /** Default TTL: 12 Hours (in milliseconds). */
5
6
  export const DEFAULT_CATALOG_TTL_MS = 12 * 60 * 60 * 1000;
@@ -65,14 +66,20 @@ function notifyUpdateListeners(): void {
65
66
  }
66
67
  }
67
68
 
68
- /** Non-configurable cache directory: src/data */
69
+ /**
70
+ * Catalog cache directory. Defaults to `src/data` (repo convention); override
71
+ * with `AGENT_CACHE_DIR` when embedding as a package or running on a
72
+ * read-only filesystem (serverless/containers).
73
+ */
69
74
  export function getCacheDir(): string {
75
+ const override = getEnv("AGENT_CACHE_DIR");
76
+ if (override && override.trim()) return path.resolve(override.trim());
70
77
  return path.resolve(process.cwd(), "src/data");
71
78
  }
72
79
 
73
- /** Non-configurable cache file path: src/data/models-cache.json */
80
+ /** Non-configurable cache file path: src/data/models.dev.json */
74
81
  export function getCacheFilePath(): string {
75
- return path.resolve(process.cwd(), "src/data/models-cache.json");
82
+ return path.resolve(process.cwd(), "src/data/models.dev.json");
76
83
  }
77
84
 
78
85
  function readJsonFileSync(filePath: string): any {
@@ -99,7 +106,7 @@ let activeFetchedAt: number | undefined = undefined;
99
106
  let activeTtlMs: number = globalCatalogTtlMs;
100
107
  let activeFromCache = false;
101
108
 
102
- // Synchronous bootstrap: load from src/data/models-cache.json if present
109
+ // Synchronous bootstrap: load from src/data/models.dev.json if present
103
110
  function initializeCatalogSync(): void {
104
111
  const cachePath = getCacheFilePath();
105
112
  const cached = readJsonFileSync(cachePath);
@@ -165,9 +172,30 @@ export function getCatalogStatus(): CatalogStatus {
165
172
  };
166
173
  }
167
174
 
175
+ /**
176
+ * Structural check for a models.dev catalog payload: a map of provider ids
177
+ * to entries that each carry a `models` map. Rejects chat-completion
178
+ * payloads, error envelopes, and other non-catalog JSON (e.g. from mocked
179
+ * `fetch` in tests) BEFORE they can overwrite the good on-disk cache.
180
+ */
181
+ export function isValidCatalogPayload(data: unknown): data is Record<string, any> {
182
+ if (!data || typeof data !== "object" || Array.isArray(data)) return false;
183
+ const entries = Object.entries(data as Record<string, unknown>);
184
+ if (entries.length < 3) return false;
185
+ let providersWithModels = 0;
186
+ for (const [, value] of entries) {
187
+ if (!value || typeof value !== "object" || Array.isArray(value)) return false;
188
+ const models = (value as Record<string, unknown>).models;
189
+ if (!models || typeof models !== "object" || Array.isArray(models)) return false;
190
+ if (Object.keys(models).length > 0) providersWithModels++;
191
+ }
192
+ // Guard against degenerate-but-shaped payloads (e.g. `{a:{models:{}}}`).
193
+ return providersWithModels >= 3;
194
+ }
195
+
168
196
  /**
169
197
  * Refreshes the model catalog by downloading from models.dev
170
- * directly into src/data/models-cache.json with a 12-hour TTL.
198
+ * directly into src/data/models.dev.json with a 12-hour TTL.
171
199
  */
172
200
  export async function refreshModelCatalog(options: RefreshCatalogOptions = {}): Promise<CatalogStatus> {
173
201
  const effectiveTtl = options.ttlMs ?? globalCatalogTtlMs;
@@ -211,6 +239,21 @@ export async function refreshModelCatalog(options: RefreshCatalogOptions = {}):
211
239
  }
212
240
 
213
241
  freshData = (await res.json()) as Record<string, any>;
242
+
243
+ // Never let a non-catalog payload (mocked fetch, proxy error page,
244
+ // chat-completion stub) overwrite the good on-disk cache.
245
+ if (!isValidCatalogPayload(freshData)) {
246
+ clearTimeout(timer);
247
+ const existing = readJsonFileSync(cachePath);
248
+ if (existing && existing.data) {
249
+ activeCatalog = existing.data;
250
+ activeFetchedAt = existing.fetchedAt;
251
+ activeTtlMs = existing.ttlMs || effectiveTtl;
252
+ activeFromCache = true;
253
+ notifyUpdateListeners();
254
+ }
255
+ return getCatalogStatus();
256
+ }
214
257
  } catch (err: any) {
215
258
  clearTimeout(timer);
216
259
  // If download fails, retain disk cache if available
@@ -236,15 +279,23 @@ export async function refreshModelCatalog(options: RefreshCatalogOptions = {}):
236
279
  data: freshData,
237
280
  };
238
281
 
239
- // Save to src/data/models-cache.json
240
- writeJsonFileSync(cachePath, cachePayload);
241
-
242
- // Update in-memory state
282
+ // Update in-memory state FIRST so a read-only filesystem still serves the
283
+ // fresh catalog for this process; disk persistence below is best-effort.
243
284
  activeCatalog = freshData;
244
285
  activeFetchedAt = fetchedAt;
245
286
  activeTtlMs = effectiveTtl;
246
287
  activeFromCache = true;
247
288
 
289
+ // Save to src/data/models.dev.json (or the AGENT_CACHE_DIR override)
290
+ try {
291
+ writeJsonFileSync(cachePath, cachePayload);
292
+ } catch (err) {
293
+ activeFromCache = false;
294
+ console.warn(
295
+ `[Agent Accelerator] Could not write model catalog cache to ${cachePath} (${err instanceof Error ? err.message : String(err)}). Continuing with the in-memory catalog.`
296
+ );
297
+ }
298
+
248
299
  notifyUpdateListeners();
249
300
  return getCatalogStatus();
250
301
  }