@almadar/llm 2.50.0 → 2.52.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{chunk-RBT22YSX.js → chunk-DLEZ7FGQ.js} +13 -4
- package/dist/chunk-DLEZ7FGQ.js.map +1 -0
- package/dist/chunk-MOIECDMB.js +172 -0
- package/dist/chunk-MOIECDMB.js.map +1 -0
- package/dist/chunk-OXFWONZP.js +369 -0
- package/dist/chunk-OXFWONZP.js.map +1 -0
- package/dist/{chunk-5AM54OAM.js → chunk-SJE3GTGZ.js} +5 -3
- package/dist/{chunk-5AM54OAM.js.map → chunk-SJE3GTGZ.js.map} +1 -1
- package/dist/{chunk-RPG3SUIB.js → chunk-T6AKOBX3.js} +25 -177
- package/dist/chunk-T6AKOBX3.js.map +1 -0
- package/dist/{client-SYMpnw-w.d.ts → client-DfzMDgkm.d.ts} +10 -1
- package/dist/client.d.ts +2 -2
- package/dist/client.js +3 -2
- package/dist/index.d.ts +12 -6
- package/dist/index.js +49 -12
- package/dist/index.js.map +1 -1
- package/dist/providers/index.d.ts +138 -1
- package/dist/providers/index.js +16 -1
- package/dist/{rate-limiter-CXaf8aAy.d.ts → rate-limiter-Bz0iSJgZ.d.ts} +20 -1
- package/dist/structured-output.d.ts +1 -1
- package/dist/structured-output.js +3 -2
- package/package.json +6 -4
- package/src/client.ts +22 -1
- package/src/image-client.ts +27 -4
- package/src/index.ts +25 -0
- package/src/providers/index.ts +26 -0
- package/src/providers/jev.ts +456 -0
- package/src/token-tracker.ts +67 -12
- package/dist/chunk-MUTXGY6D.js +0 -133
- package/dist/chunk-MUTXGY6D.js.map +0 -1
- package/dist/chunk-RBT22YSX.js.map +0 -1
- package/dist/chunk-RPG3SUIB.js.map +0 -1
|
@@ -1,169 +1,3 @@
|
|
|
1
|
-
// src/rate-limiter.ts
|
|
2
|
-
var RateLimiter = class {
|
|
3
|
-
constructor(options = {}) {
|
|
4
|
-
this.activeRequests = 0;
|
|
5
|
-
this.queue = [];
|
|
6
|
-
this.lastMinuteReset = Date.now();
|
|
7
|
-
this.lastSecondReset = Date.now();
|
|
8
|
-
this.pumping = false;
|
|
9
|
-
this.currentBackoffMs = 0;
|
|
10
|
-
this.wakeTimer = null;
|
|
11
|
-
this.requestsPerMinute = options.requestsPerMinute ?? 60;
|
|
12
|
-
this.requestsPerSecond = options.requestsPerSecond ?? 3;
|
|
13
|
-
this.maxConcurrent = options.maxConcurrent ?? 5;
|
|
14
|
-
this.baseBackoffMs = options.baseBackoffMs ?? 1e3;
|
|
15
|
-
this.maxBackoffMs = options.maxBackoffMs ?? 6e4;
|
|
16
|
-
this.minuteTokens = this.requestsPerMinute;
|
|
17
|
-
this.secondTokens = this.requestsPerSecond;
|
|
18
|
-
}
|
|
19
|
-
async execute(fn, _maxRetries = 3) {
|
|
20
|
-
return new Promise((resolve, reject) => {
|
|
21
|
-
this.queue.push({
|
|
22
|
-
execute: fn,
|
|
23
|
-
resolve,
|
|
24
|
-
reject,
|
|
25
|
-
retryCount: 0
|
|
26
|
-
});
|
|
27
|
-
this.pump();
|
|
28
|
-
});
|
|
29
|
-
}
|
|
30
|
-
getStatus() {
|
|
31
|
-
return {
|
|
32
|
-
queueLength: this.queue.length,
|
|
33
|
-
activeRequests: this.activeRequests,
|
|
34
|
-
minuteTokens: this.minuteTokens,
|
|
35
|
-
secondTokens: this.secondTokens,
|
|
36
|
-
backoffMs: this.currentBackoffMs
|
|
37
|
-
};
|
|
38
|
-
}
|
|
39
|
-
reset() {
|
|
40
|
-
this.minuteTokens = this.requestsPerMinute;
|
|
41
|
-
this.secondTokens = this.requestsPerSecond;
|
|
42
|
-
this.activeRequests = 0;
|
|
43
|
-
this.queue = [];
|
|
44
|
-
this.currentBackoffMs = 0;
|
|
45
|
-
this.lastMinuteReset = Date.now();
|
|
46
|
-
this.lastSecondReset = Date.now();
|
|
47
|
-
if (this.wakeTimer !== null) {
|
|
48
|
-
clearTimeout(this.wakeTimer);
|
|
49
|
-
this.wakeTimer = null;
|
|
50
|
-
}
|
|
51
|
-
}
|
|
52
|
-
/**
|
|
53
|
-
* Start queued requests up to the rate/concurrency budget, then return —
|
|
54
|
-
* in-flight requests re-enter the pump as they settle, and a wake timer
|
|
55
|
-
* covers token-bucket refills and 429 backoff. Never awaits a request.
|
|
56
|
-
*/
|
|
57
|
-
pump() {
|
|
58
|
-
if (this.pumping) return;
|
|
59
|
-
this.pumping = true;
|
|
60
|
-
try {
|
|
61
|
-
this.refillTokens();
|
|
62
|
-
while (this.queue.length > 0 && this.currentBackoffMs === 0 && this.canMakeRequest()) {
|
|
63
|
-
const request = this.queue.shift();
|
|
64
|
-
if (!request) break;
|
|
65
|
-
this.consumeTokens();
|
|
66
|
-
this.activeRequests++;
|
|
67
|
-
void this.settle(request);
|
|
68
|
-
}
|
|
69
|
-
} finally {
|
|
70
|
-
this.pumping = false;
|
|
71
|
-
}
|
|
72
|
-
if (this.queue.length === 0) return;
|
|
73
|
-
if (this.currentBackoffMs > 0) {
|
|
74
|
-
this.scheduleWake(this.currentBackoffMs, true);
|
|
75
|
-
} else if (this.activeRequests < this.maxConcurrent) {
|
|
76
|
-
this.scheduleWake(this.getWaitTime(), false);
|
|
77
|
-
}
|
|
78
|
-
}
|
|
79
|
-
scheduleWake(ms, clearsBackoff) {
|
|
80
|
-
if (this.wakeTimer !== null) return;
|
|
81
|
-
this.wakeTimer = setTimeout(() => {
|
|
82
|
-
this.wakeTimer = null;
|
|
83
|
-
if (clearsBackoff) this.currentBackoffMs = 0;
|
|
84
|
-
this.pump();
|
|
85
|
-
}, ms);
|
|
86
|
-
}
|
|
87
|
-
async settle(request) {
|
|
88
|
-
try {
|
|
89
|
-
const result = await request.execute();
|
|
90
|
-
request.resolve(result);
|
|
91
|
-
this.currentBackoffMs = 0;
|
|
92
|
-
} catch (error) {
|
|
93
|
-
const err = error instanceof Error ? error : new Error(String(error));
|
|
94
|
-
if (this.isRateLimitError(err)) {
|
|
95
|
-
this.currentBackoffMs = Math.min(
|
|
96
|
-
this.baseBackoffMs * Math.pow(2, request.retryCount),
|
|
97
|
-
this.maxBackoffMs
|
|
98
|
-
);
|
|
99
|
-
console.warn(
|
|
100
|
-
`[RateLimiter] Rate limited. Backing off for ${this.currentBackoffMs}ms (retry ${request.retryCount + 1})`
|
|
101
|
-
);
|
|
102
|
-
if (request.retryCount < 3) {
|
|
103
|
-
this.queue.unshift({
|
|
104
|
-
...request,
|
|
105
|
-
retryCount: request.retryCount + 1
|
|
106
|
-
});
|
|
107
|
-
} else {
|
|
108
|
-
request.reject(
|
|
109
|
-
new Error(
|
|
110
|
-
`Rate limit exceeded after ${request.retryCount + 1} retries: ${err.message}`
|
|
111
|
-
)
|
|
112
|
-
);
|
|
113
|
-
}
|
|
114
|
-
} else {
|
|
115
|
-
request.reject(err);
|
|
116
|
-
}
|
|
117
|
-
} finally {
|
|
118
|
-
this.activeRequests--;
|
|
119
|
-
this.pump();
|
|
120
|
-
}
|
|
121
|
-
}
|
|
122
|
-
refillTokens() {
|
|
123
|
-
const now = Date.now();
|
|
124
|
-
if (now - this.lastMinuteReset >= 6e4) {
|
|
125
|
-
this.minuteTokens = this.requestsPerMinute;
|
|
126
|
-
this.lastMinuteReset = now;
|
|
127
|
-
}
|
|
128
|
-
if (now - this.lastSecondReset >= 1e3) {
|
|
129
|
-
this.secondTokens = this.requestsPerSecond;
|
|
130
|
-
this.lastSecondReset = now;
|
|
131
|
-
}
|
|
132
|
-
}
|
|
133
|
-
canMakeRequest() {
|
|
134
|
-
return this.minuteTokens > 0 && this.secondTokens > 0 && this.activeRequests < this.maxConcurrent;
|
|
135
|
-
}
|
|
136
|
-
consumeTokens() {
|
|
137
|
-
this.minuteTokens--;
|
|
138
|
-
this.secondTokens--;
|
|
139
|
-
}
|
|
140
|
-
getWaitTime() {
|
|
141
|
-
const now = Date.now();
|
|
142
|
-
if (this.secondTokens <= 0) {
|
|
143
|
-
return Math.max(0, 1e3 - (now - this.lastSecondReset));
|
|
144
|
-
}
|
|
145
|
-
if (this.minuteTokens <= 0) {
|
|
146
|
-
return Math.max(0, 6e4 - (now - this.lastMinuteReset));
|
|
147
|
-
}
|
|
148
|
-
return 100;
|
|
149
|
-
}
|
|
150
|
-
isRateLimitError(error) {
|
|
151
|
-
const message = error.message.toLowerCase();
|
|
152
|
-
return message.includes("429") || message.includes("rate limit") || message.includes("too many requests") || message.includes("quota exceeded");
|
|
153
|
-
}
|
|
154
|
-
};
|
|
155
|
-
var globalRateLimiter = null;
|
|
156
|
-
function getGlobalRateLimiter(options) {
|
|
157
|
-
if (!globalRateLimiter) {
|
|
158
|
-
globalRateLimiter = new RateLimiter(options);
|
|
159
|
-
}
|
|
160
|
-
return globalRateLimiter;
|
|
161
|
-
}
|
|
162
|
-
function resetGlobalRateLimiter() {
|
|
163
|
-
globalRateLimiter?.reset();
|
|
164
|
-
globalRateLimiter = null;
|
|
165
|
-
}
|
|
166
|
-
|
|
167
1
|
// src/token-tracker.ts
|
|
168
2
|
import { appendFileSync, mkdirSync, readFileSync, writeFileSync } from "fs";
|
|
169
3
|
import { dirname, join } from "path";
|
|
@@ -243,7 +77,7 @@ function refreshPricingCache() {
|
|
|
243
77
|
}).catch(() => {
|
|
244
78
|
});
|
|
245
79
|
}
|
|
246
|
-
function
|
|
80
|
+
function getCostForModelIfKnown(model) {
|
|
247
81
|
const pricing = getPricing();
|
|
248
82
|
const orId = MODEL_ID_MAP[model];
|
|
249
83
|
if (orId && pricing[orId]) return pricing[orId];
|
|
@@ -251,7 +85,27 @@ function getCostForModel(model) {
|
|
|
251
85
|
for (const [key, cost] of Object.entries(pricing)) {
|
|
252
86
|
if (key.includes(model) || model.includes(key.split("/")[1] ?? "")) return cost;
|
|
253
87
|
}
|
|
254
|
-
return
|
|
88
|
+
return void 0;
|
|
89
|
+
}
|
|
90
|
+
function getCostForModel(model) {
|
|
91
|
+
return getCostForModelIfKnown(model) ?? { promptCostPer1K: 0, completionCostPer1K: 0 };
|
|
92
|
+
}
|
|
93
|
+
function priceTokens(costs, promptTokens, completionTokens, cached, written) {
|
|
94
|
+
const cacheReadRate = costs.cacheReadCostPer1K ?? costs.promptCostPer1K;
|
|
95
|
+
const cacheWriteRate = costs.cacheWriteCostPer1K ?? costs.promptCostPer1K;
|
|
96
|
+
const uncached = Math.max(0, promptTokens - cached - written);
|
|
97
|
+
return uncached / 1e3 * costs.promptCostPer1K + cached / 1e3 * cacheReadRate + written / 1e3 * cacheWriteRate + completionTokens / 1e3 * costs.completionCostPer1K;
|
|
98
|
+
}
|
|
99
|
+
function estimateCostUSD(model, tokens) {
|
|
100
|
+
const costs = getCostForModelIfKnown(model);
|
|
101
|
+
if (costs === void 0) return void 0;
|
|
102
|
+
return priceTokens(
|
|
103
|
+
costs,
|
|
104
|
+
tokens.promptTokens,
|
|
105
|
+
tokens.completionTokens,
|
|
106
|
+
Math.max(0, tokens.cachedPromptTokens ?? 0),
|
|
107
|
+
Math.max(0, tokens.cacheWriteTokens ?? 0)
|
|
108
|
+
);
|
|
255
109
|
}
|
|
256
110
|
var TokenTracker = class {
|
|
257
111
|
constructor(model = "claude-sonnet-4-5-20250929") {
|
|
@@ -278,11 +132,7 @@ var TokenTracker = class {
|
|
|
278
132
|
}
|
|
279
133
|
/** Cache-aware cost for one (or an aggregate of) call(s), in USD. */
|
|
280
134
|
costFor(model, promptTokens, completionTokens, cached, written) {
|
|
281
|
-
|
|
282
|
-
const cacheReadRate = costs.cacheReadCostPer1K ?? costs.promptCostPer1K;
|
|
283
|
-
const cacheWriteRate = costs.cacheWriteCostPer1K ?? costs.promptCostPer1K;
|
|
284
|
-
const uncached = Math.max(0, promptTokens - cached - written);
|
|
285
|
-
return uncached / 1e3 * costs.promptCostPer1K + cached / 1e3 * cacheReadRate + written / 1e3 * cacheWriteRate + completionTokens / 1e3 * costs.completionCostPer1K;
|
|
135
|
+
return priceTokens(getCostForModel(model), promptTokens, completionTokens, cached, written);
|
|
286
136
|
}
|
|
287
137
|
/**
|
|
288
138
|
* Record one LLM call's usage. `promptTokens` is the TOTAL input count
|
|
@@ -401,11 +251,9 @@ function resetGlobalTokenTracker() {
|
|
|
401
251
|
}
|
|
402
252
|
|
|
403
253
|
export {
|
|
404
|
-
|
|
405
|
-
getGlobalRateLimiter,
|
|
406
|
-
resetGlobalRateLimiter,
|
|
254
|
+
estimateCostUSD,
|
|
407
255
|
TokenTracker,
|
|
408
256
|
getGlobalTokenTracker,
|
|
409
257
|
resetGlobalTokenTracker
|
|
410
258
|
};
|
|
411
|
-
//# sourceMappingURL=chunk-
|
|
259
|
+
//# sourceMappingURL=chunk-T6AKOBX3.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"sources":["../src/token-tracker.ts"],"sourcesContent":["/**\n * Token Tracker for LLM Usage\n *\n * Tracks token usage across multiple LLM calls for:\n * - Cost estimation (pricing fetched from OpenRouter models API)\n * - Usage monitoring\n * - Quota management\n * - Per-call JSONL logging\n *\n * @packageDocumentation\n */\n\nimport { appendFileSync, mkdirSync, readFileSync, writeFileSync } from 'node:fs';\nimport { dirname, join } from 'node:path';\n\nexport interface TokenUsage {\n promptTokens: number;\n completionTokens: number;\n totalTokens: number;\n callCount: number;\n /** Subset of `promptTokens` that were cache reads (billed at the cache-read rate). */\n cachedPromptTokens: number;\n /** Subset of `promptTokens` written to cache (Anthropic; billed at the cache-write rate). */\n cacheWriteTokens: number;\n}\n\nexport interface TokenCost {\n promptCostPer1K: number;\n completionCostPer1K: number;\n /** Per-1K rate for cache-read (cache-hit) prompt tokens. Falls back to prompt rate when absent. */\n cacheReadCostPer1K?: number;\n /** Per-1K rate for cache-write (cache-creation) prompt tokens. Falls back to prompt rate when absent. */\n cacheWriteCostPer1K?: number;\n}\n\nexport interface CallLogEntry {\n timestamp: string;\n provider: string;\n model: string;\n promptTokens: number;\n completionTokens: number;\n totalTokens: number;\n /** Cache-read (cache-hit) subset of promptTokens, billed at the discounted rate. */\n cachedPromptTokens?: number;\n /** Cache-write subset of promptTokens (Anthropic). */\n cacheWriteTokens?: number;\n estimatedCost: number;\n durationMs?: number;\n source: 'local-log';\n}\n\n// ---------------------------------------------------------------------------\n// Pricing: fetched from OpenRouter /api/v1/models, cached to disk for 24h\n// ---------------------------------------------------------------------------\n\nconst ALMADAR_ROOT = process.env['ALMADAR_ROOT'] ?? process.cwd();\nconst PRICING_CACHE_PATH = join(ALMADAR_ROOT, '.llm-pricing-cache.json');\nconst CALL_LOG_PATH = join(ALMADAR_ROOT, '.llm-call-log.jsonl');\nconst CACHE_TTL_MS = 24 * 60 * 60 * 1000; // 24 hours\n// Bump when the cached TokenCost shape changes so stale on-disk caches are\n// invalidated on upgrade. v2 added cacheReadCostPer1K / cacheWriteCostPer1K.\nconst PRICING_CACHE_VERSION = 2;\n\n/** Map from our local model name to OpenRouter model ID */\nconst MODEL_ID_MAP: Record<string, string> = {\n // Anthropic\n 'claude-opus-4-5-20250929': 'anthropic/claude-opus-4.5',\n 'claude-sonnet-4-5-20250929': 'anthropic/claude-sonnet-4.5',\n 'claude-sonnet-4-20250514': 'anthropic/claude-sonnet-4',\n 'claude-3-5-haiku-20241022': 'anthropic/claude-3.5-haiku',\n // DeepSeek — map to current versions on OpenRouter\n 'deepseek-chat': 'deepseek/deepseek-v4-flash',\n 'deepseek-coder': 'deepseek/deepseek-v4-flash',\n 'deepseek-reasoner': 'deepseek/deepseek-v4-flash',\n 'deepseek-v4-pro': 'deepseek/deepseek-v4-pro',\n 'deepseek-v4-flash': 'deepseek/deepseek-v4-flash',\n // Kimi\n 'kimi-k2.5': 'moonshotai/kimi-k2.5',\n};\n\n// Fallback: zero cost — forces OpenRouter fetch for real pricing\nconst FALLBACK_COSTS: Record<string, TokenCost> = {};\n\ninterface PricingCache {\n version?: number;\n fetchedAt: number;\n models: Record<string, TokenCost>;\n}\n\nlet pricingCache: PricingCache | null = null;\n\nfunction loadCachedPricing(): PricingCache | null {\n try {\n const raw = readFileSync(PRICING_CACHE_PATH, 'utf-8');\n const parsed = JSON.parse(raw) as PricingCache;\n if (parsed.version === PRICING_CACHE_VERSION && Date.now() - parsed.fetchedAt < CACHE_TTL_MS) {\n return parsed;\n }\n } catch {\n // No cache or expired\n }\n return null;\n}\n\nasync function fetchPricingFromOpenRouter(): Promise<Record<string, TokenCost>> {\n const res = await fetch('https://openrouter.ai/api/v1/models');\n if (!res.ok) throw new Error(`OpenRouter models API: HTTP ${res.status}`);\n const json = await res.json() as {\n data?: Array<{\n id: string;\n pricing?: {\n prompt?: string;\n completion?: string;\n input_cache_read?: string;\n input_cache_write?: string;\n };\n }>;\n };\n const models: Record<string, TokenCost> = {};\n for (const m of json.data ?? []) {\n const promptPerToken = parseFloat(m.pricing?.prompt ?? '0');\n const completionPerToken = parseFloat(m.pricing?.completion ?? '0');\n const cacheReadPerToken = parseFloat(m.pricing?.input_cache_read ?? '0');\n const cacheWritePerToken = parseFloat(m.pricing?.input_cache_write ?? '0');\n if (promptPerToken > 0 || completionPerToken > 0) {\n models[m.id] = {\n promptCostPer1K: promptPerToken * 1000,\n completionCostPer1K: completionPerToken * 1000,\n // 0 (field absent) → leave undefined so cost math falls back to the prompt rate.\n ...(cacheReadPerToken > 0 ? { cacheReadCostPer1K: cacheReadPerToken * 1000 } : {}),\n ...(cacheWritePerToken > 0 ? { cacheWriteCostPer1K: cacheWritePerToken * 1000 } : {}),\n };\n }\n }\n return models;\n}\n\n/**\n * Get pricing for all models. Uses 24h disk cache, fetches from OpenRouter on miss.\n * Non-blocking: returns cached/fallback immediately, refreshes in background if stale.\n */\nfunction getPricing(): Record<string, TokenCost> {\n if (pricingCache) return pricingCache.models;\n\n const diskCache = loadCachedPricing();\n if (diskCache) {\n pricingCache = diskCache;\n return diskCache.models;\n }\n\n // Trigger background fetch, return fallback for now\n refreshPricingCache();\n return FALLBACK_COSTS;\n}\n\nfunction refreshPricingCache(): void {\n fetchPricingFromOpenRouter()\n .then((models) => {\n pricingCache = { version: PRICING_CACHE_VERSION, fetchedAt: Date.now(), models };\n try {\n mkdirSync(dirname(PRICING_CACHE_PATH), { recursive: true });\n writeFileSync(PRICING_CACHE_PATH, JSON.stringify(pricingCache));\n } catch {\n // Non-critical\n }\n })\n .catch(() => {\n // Silently fail, use fallback\n });\n}\n\n/**\n * Look up a model's pricing row without a zero-cost fallback — `undefined`\n * means \"no pricing known yet\" (OpenRouter fetch pending, or the model\n * isn't listed), distinct from a model that is genuinely free. Single\n * lookup path: `getCostForModel` (zero-fallback, for the tracker's own\n * running totals) and `estimateCostUSD` (pure, callers outside the\n * tracker) both resolve through this.\n */\nfunction getCostForModelIfKnown(model: string): TokenCost | undefined {\n const pricing = getPricing();\n // Try direct match on OpenRouter ID\n const orId = MODEL_ID_MAP[model];\n if (orId && pricing[orId]) return pricing[orId];\n // Try direct key match (e.g., user passed \"openai/gpt-4o\")\n if (pricing[model]) return pricing[model];\n // Fuzzy: find first key containing the model name\n for (const [key, cost] of Object.entries(pricing)) {\n if (key.includes(model) || model.includes(key.split('/')[1] ?? '')) return cost;\n }\n return undefined;\n}\n\nfunction getCostForModel(model: string): TokenCost {\n // No pricing available — return zero (OpenRouter fetch pending or model not listed)\n return getCostForModelIfKnown(model) ?? { promptCostPer1K: 0, completionCostPer1K: 0 };\n}\n\n/** Cache-aware cost formula shared by `TokenTracker.costFor` (instance,\n * zero-fallback pricing) and `estimateCostUSD` (pure, `undefined` pricing\n * propagates to the caller instead of silently pricing at $0). */\nfunction priceTokens(\n costs: TokenCost,\n promptTokens: number,\n completionTokens: number,\n cached: number,\n written: number,\n): number {\n const cacheReadRate = costs.cacheReadCostPer1K ?? costs.promptCostPer1K;\n const cacheWriteRate = costs.cacheWriteCostPer1K ?? costs.promptCostPer1K;\n const uncached = Math.max(0, promptTokens - cached - written);\n return (\n (uncached / 1000) * costs.promptCostPer1K +\n (cached / 1000) * cacheReadRate +\n (written / 1000) * cacheWriteRate +\n (completionTokens / 1000) * costs.completionCostPer1K\n );\n}\n\nexport interface EstimateCostTokens {\n promptTokens: number;\n completionTokens: number;\n /** Subset of `promptTokens` served from the provider's prefix cache. */\n cachedPromptTokens?: number;\n /** Subset of `promptTokens` written to cache (Anthropic cache-write). */\n cacheWriteTokens?: number;\n}\n\n/**\n * Pure per-call cost estimate from token counts alone, priced from the SAME\n * OpenRouter-fetched table `TokenTracker` uses for the whole-run total (the\n * 24h disk cache in `getPricing` — no second pricing source). Callers with\n * tokens + a model name but no `TokenTracker` instance (e.g. a trace-event\n * emitter) use this instead of reimplementing the cache-aware math.\n *\n * Returns `undefined` when the model has no known pricing row — the\n * caller's job to render that as \"unknown\", never as `$0`.\n */\nexport function estimateCostUSD(model: string, tokens: EstimateCostTokens): number | undefined {\n const costs = getCostForModelIfKnown(model);\n if (costs === undefined) return undefined;\n return priceTokens(\n costs,\n tokens.promptTokens,\n tokens.completionTokens,\n Math.max(0, tokens.cachedPromptTokens ?? 0),\n Math.max(0, tokens.cacheWriteTokens ?? 0),\n );\n}\n\n// ---------------------------------------------------------------------------\n// TokenTracker\n// ---------------------------------------------------------------------------\n\nexport class TokenTracker {\n private model: string;\n private usage: TokenUsage = {\n promptTokens: 0,\n completionTokens: 0,\n totalTokens: 0,\n callCount: 0,\n cachedPromptTokens: 0,\n cacheWriteTokens: 0,\n };\n\n /** Sum of provider-reported authoritative costs (e.g. OpenRouter `usage.cost`). */\n private authoritativeCostUSD = 0;\n /** Token buckets for calls WITHOUT an authoritative cost — priced cache-aware.\n * Keyed by the model that MADE the calls: `setModel` flips the tracker\n * between models mid-process (coordinator on pro, subagents on flash), and\n * pricing the aggregate at whichever model is current when someone reads\n * the estimate mispriced the whole history by up to the models' rate ratio\n * (snapshot/diff consumers saw phantom multi-dollar deltas — or $0 after a\n * negative-clamp — per battery spec). Per-model buckets keep the\n * retroactive late-pricing property without cross-model contamination. */\n private computedByModel = new Map<\n string,\n { promptTokens: number; completionTokens: number; cachedPromptTokens: number; cacheWriteTokens: number }\n >();\n\n constructor(model: string = 'claude-sonnet-4-5-20250929') {\n this.model = model;\n }\n\n /** Cache-aware cost for one (or an aggregate of) call(s), in USD. */\n private costFor(model: string, promptTokens: number, completionTokens: number, cached: number, written: number): number {\n return priceTokens(getCostForModel(model), promptTokens, completionTokens, cached, written);\n }\n\n /**\n * Record one LLM call's usage. `promptTokens` is the TOTAL input count\n * (cache reads + cache writes + uncached); `cachedPromptTokens` and\n * `cacheWriteTokens` are subsets of it, priced at their own (cheaper /\n * pricier) rates. Providers that don't report cache detail pass 0, which\n * reduces to the previous flat-rate behaviour.\n */\n addUsage(\n promptTokens: number,\n completionTokens: number,\n options?: {\n provider?: string;\n durationMs?: number;\n cachedPromptTokens?: number;\n cacheWriteTokens?: number;\n /** Provider-reported authoritative cost (e.g. OpenRouter `usage.cost`). When set, used verbatim. */\n costUSD?: number;\n },\n ): void {\n const cached = Math.min(promptTokens, Math.max(0, options?.cachedPromptTokens ?? 0));\n const written = Math.min(promptTokens - cached, Math.max(0, options?.cacheWriteTokens ?? 0));\n\n this.usage.promptTokens += promptTokens;\n this.usage.completionTokens += completionTokens;\n this.usage.totalTokens += promptTokens + completionTokens;\n this.usage.cachedPromptTokens += cached;\n this.usage.cacheWriteTokens += written;\n this.usage.callCount++;\n\n // Prefer the provider's authoritative cost (already cache- and routing-\n // adjusted). Otherwise bucket the tokens and price them cache-aware so\n // late-arriving pricing still applies retroactively to the estimate.\n const authoritative = options?.costUSD;\n let estimatedCost: number;\n if (authoritative != null && Number.isFinite(authoritative)) {\n this.authoritativeCostUSD += authoritative;\n estimatedCost = authoritative;\n } else {\n const bucket = this.computedByModel.get(this.model) ?? {\n promptTokens: 0,\n completionTokens: 0,\n cachedPromptTokens: 0,\n cacheWriteTokens: 0,\n };\n bucket.promptTokens += promptTokens;\n bucket.completionTokens += completionTokens;\n bucket.cachedPromptTokens += cached;\n bucket.cacheWriteTokens += written;\n this.computedByModel.set(this.model, bucket);\n estimatedCost = this.costFor(this.model, promptTokens, completionTokens, cached, written);\n }\n\n const entry: CallLogEntry = {\n timestamp: new Date().toISOString(),\n provider: options?.provider ?? 'unknown',\n model: this.model,\n promptTokens,\n completionTokens,\n totalTokens: promptTokens + completionTokens,\n cachedPromptTokens: cached,\n cacheWriteTokens: written,\n estimatedCost,\n durationMs: options?.durationMs,\n source: 'local-log',\n };\n\n try {\n mkdirSync(dirname(CALL_LOG_PATH), { recursive: true });\n appendFileSync(CALL_LOG_PATH, JSON.stringify(entry) + '\\n');\n } catch {\n // Non-critical: don't break LLM calls if logging fails\n }\n }\n\n getSummary(): TokenUsage {\n return { ...this.usage };\n }\n\n getEstimatedCost(): number {\n let computed = 0;\n for (const [model, b] of this.computedByModel) {\n computed += this.costFor(\n model,\n b.promptTokens,\n b.completionTokens,\n b.cachedPromptTokens,\n b.cacheWriteTokens,\n );\n }\n return this.authoritativeCostUSD + computed;\n }\n\n getFormattedCost(): string {\n const cost = this.getEstimatedCost();\n return `$${cost.toFixed(4)}`;\n }\n\n getReport(): string {\n const summary = this.getSummary();\n const cost = this.getEstimatedCost();\n return [\n `Token Usage Report (${this.model})`,\n `─────────────────────────────`,\n `Calls: ${summary.callCount}`,\n `Prompt Tokens: ${summary.promptTokens.toLocaleString()}`,\n `Completion Tokens: ${summary.completionTokens.toLocaleString()}`,\n `Total Tokens: ${summary.totalTokens.toLocaleString()}`,\n `Estimated Cost: $${cost.toFixed(4)}`,\n ].join('\\n');\n }\n\n reset(): void {\n this.usage = {\n promptTokens: 0,\n completionTokens: 0,\n totalTokens: 0,\n callCount: 0,\n cachedPromptTokens: 0,\n cacheWriteTokens: 0,\n };\n this.authoritativeCostUSD = 0;\n this.computedByModel.clear();\n }\n\n setModel(model: string): void {\n this.model = model;\n }\n}\n\n// Global tracker instance\nlet globalTracker: TokenTracker | null = null;\n\nexport function getGlobalTokenTracker(model?: string): TokenTracker {\n if (!globalTracker) {\n globalTracker = new TokenTracker(model);\n } else if (model) {\n globalTracker.setModel(model);\n }\n return globalTracker;\n}\n\nexport function resetGlobalTokenTracker(): void {\n globalTracker?.reset();\n}\n\nexport function getCallLogPath(): string {\n return CALL_LOG_PATH;\n}\n\n/** Force-refresh the pricing cache from OpenRouter. */\nexport async function refreshPricing(): Promise<void> {\n const models = await fetchPricingFromOpenRouter();\n pricingCache = { version: PRICING_CACHE_VERSION, fetchedAt: Date.now(), models };\n mkdirSync(dirname(PRICING_CACHE_PATH), { recursive: true });\n writeFileSync(PRICING_CACHE_PATH, JSON.stringify(pricingCache));\n}\n"],"mappings":";AAYA,SAAS,gBAAgB,WAAW,cAAc,qBAAqB;AACvE,SAAS,SAAS,YAAY;AA0C9B,IAAM,eAAe,QAAQ,IAAI,cAAc,KAAK,QAAQ,IAAI;AAChE,IAAM,qBAAqB,KAAK,cAAc,yBAAyB;AACvE,IAAM,gBAAgB,KAAK,cAAc,qBAAqB;AAC9D,IAAM,eAAe,KAAK,KAAK,KAAK;AAGpC,IAAM,wBAAwB;AAG9B,IAAM,eAAuC;AAAA;AAAA,EAE3C,4BAA4B;AAAA,EAC5B,8BAA8B;AAAA,EAC9B,4BAA4B;AAAA,EAC5B,6BAA6B;AAAA;AAAA,EAE7B,iBAAiB;AAAA,EACjB,kBAAkB;AAAA,EAClB,qBAAqB;AAAA,EACrB,mBAAmB;AAAA,EACnB,qBAAqB;AAAA;AAAA,EAErB,aAAa;AACf;AAGA,IAAM,iBAA4C,CAAC;AAQnD,IAAI,eAAoC;AAExC,SAAS,oBAAyC;AAChD,MAAI;AACF,UAAM,MAAM,aAAa,oBAAoB,OAAO;AACpD,UAAM,SAAS,KAAK,MAAM,GAAG;AAC7B,QAAI,OAAO,YAAY,yBAAyB,KAAK,IAAI,IAAI,OAAO,YAAY,cAAc;AAC5F,aAAO;AAAA,IACT;AAAA,EACF,QAAQ;AAAA,EAER;AACA,SAAO;AACT;AAEA,eAAe,6BAAiE;AAC9E,QAAM,MAAM,MAAM,MAAM,qCAAqC;AAC7D,MAAI,CAAC,IAAI,GAAI,OAAM,IAAI,MAAM,+BAA+B,IAAI,MAAM,EAAE;AACxE,QAAM,OAAO,MAAM,IAAI,KAAK;AAW5B,QAAM,SAAoC,CAAC;AAC3C,aAAW,KAAK,KAAK,QAAQ,CAAC,GAAG;AAC/B,UAAM,iBAAiB,WAAW,EAAE,SAAS,UAAU,GAAG;AAC1D,UAAM,qBAAqB,WAAW,EAAE,SAAS,cAAc,GAAG;AAClE,UAAM,oBAAoB,WAAW,EAAE,SAAS,oBAAoB,GAAG;AACvE,UAAM,qBAAqB,WAAW,EAAE,SAAS,qBAAqB,GAAG;AACzE,QAAI,iBAAiB,KAAK,qBAAqB,GAAG;AAChD,aAAO,EAAE,EAAE,IAAI;AAAA,QACb,iBAAiB,iBAAiB;AAAA,QAClC,qBAAqB,qBAAqB;AAAA;AAAA,QAE1C,GAAI,oBAAoB,IAAI,EAAE,oBAAoB,oBAAoB,IAAK,IAAI,CAAC;AAAA,QAChF,GAAI,qBAAqB,IAAI,EAAE,qBAAqB,qBAAqB,IAAK,IAAI,CAAC;AAAA,MACrF;AAAA,IACF;AAAA,EACF;AACA,SAAO;AACT;AAMA,SAAS,aAAwC;AAC/C,MAAI,aAAc,QAAO,aAAa;AAEtC,QAAM,YAAY,kBAAkB;AACpC,MAAI,WAAW;AACb,mBAAe;AACf,WAAO,UAAU;AAAA,EACnB;AAGA,sBAAoB;AACpB,SAAO;AACT;AAEA,SAAS,sBAA4B;AACnC,6BAA2B,EACxB,KAAK,CAAC,WAAW;AAChB,mBAAe,EAAE,SAAS,uBAAuB,WAAW,KAAK,IAAI,GAAG,OAAO;AAC/E,QAAI;AACF,gBAAU,QAAQ,kBAAkB,GAAG,EAAE,WAAW,KAAK,CAAC;AAC1D,oBAAc,oBAAoB,KAAK,UAAU,YAAY,CAAC;AAAA,IAChE,QAAQ;AAAA,IAER;AAAA,EACF,CAAC,EACA,MAAM,MAAM;AAAA,EAEb,CAAC;AACL;AAUA,SAAS,uBAAuB,OAAsC;AACpE,QAAM,UAAU,WAAW;AAE3B,QAAM,OAAO,aAAa,KAAK;AAC/B,MAAI,QAAQ,QAAQ,IAAI,EAAG,QAAO,QAAQ,IAAI;AAE9C,MAAI,QAAQ,KAAK,EAAG,QAAO,QAAQ,KAAK;AAExC,aAAW,CAAC,KAAK,IAAI,KAAK,OAAO,QAAQ,OAAO,GAAG;AACjD,QAAI,IAAI,SAAS,KAAK,KAAK,MAAM,SAAS,IAAI,MAAM,GAAG,EAAE,CAAC,KAAK,EAAE,EAAG,QAAO;AAAA,EAC7E;AACA,SAAO;AACT;AAEA,SAAS,gBAAgB,OAA0B;AAEjD,SAAO,uBAAuB,KAAK,KAAK,EAAE,iBAAiB,GAAG,qBAAqB,EAAE;AACvF;AAKA,SAAS,YACP,OACA,cACA,kBACA,QACA,SACQ;AACR,QAAM,gBAAgB,MAAM,sBAAsB,MAAM;AACxD,QAAM,iBAAiB,MAAM,uBAAuB,MAAM;AAC1D,QAAM,WAAW,KAAK,IAAI,GAAG,eAAe,SAAS,OAAO;AAC5D,SACG,WAAW,MAAQ,MAAM,kBACzB,SAAS,MAAQ,gBACjB,UAAU,MAAQ,iBAClB,mBAAmB,MAAQ,MAAM;AAEtC;AAqBO,SAAS,gBAAgB,OAAe,QAAgD;AAC7F,QAAM,QAAQ,uBAAuB,KAAK;AAC1C,MAAI,UAAU,OAAW,QAAO;AAChC,SAAO;AAAA,IACL;AAAA,IACA,OAAO;AAAA,IACP,OAAO;AAAA,IACP,KAAK,IAAI,GAAG,OAAO,sBAAsB,CAAC;AAAA,IAC1C,KAAK,IAAI,GAAG,OAAO,oBAAoB,CAAC;AAAA,EAC1C;AACF;AAMO,IAAM,eAAN,MAAmB;AAAA,EA0BxB,YAAY,QAAgB,8BAA8B;AAxB1D,SAAQ,QAAoB;AAAA,MAC1B,cAAc;AAAA,MACd,kBAAkB;AAAA,MAClB,aAAa;AAAA,MACb,WAAW;AAAA,MACX,oBAAoB;AAAA,MACpB,kBAAkB;AAAA,IACpB;AAGA;AAAA,SAAQ,uBAAuB;AAS/B;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,SAAQ,kBAAkB,oBAAI,IAG5B;AAGA,SAAK,QAAQ;AAAA,EACf;AAAA;AAAA,EAGQ,QAAQ,OAAe,cAAsB,kBAA0B,QAAgB,SAAyB;AACtH,WAAO,YAAY,gBAAgB,KAAK,GAAG,cAAc,kBAAkB,QAAQ,OAAO;AAAA,EAC5F;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EASA,SACE,cACA,kBACA,SAQM;AACN,UAAM,SAAS,KAAK,IAAI,cAAc,KAAK,IAAI,GAAG,SAAS,sBAAsB,CAAC,CAAC;AACnF,UAAM,UAAU,KAAK,IAAI,eAAe,QAAQ,KAAK,IAAI,GAAG,SAAS,oBAAoB,CAAC,CAAC;AAE3F,SAAK,MAAM,gBAAgB;AAC3B,SAAK,MAAM,oBAAoB;AAC/B,SAAK,MAAM,eAAe,eAAe;AACzC,SAAK,MAAM,sBAAsB;AACjC,SAAK,MAAM,oBAAoB;AAC/B,SAAK,MAAM;AAKX,UAAM,gBAAgB,SAAS;AAC/B,QAAI;AACJ,QAAI,iBAAiB,QAAQ,OAAO,SAAS,aAAa,GAAG;AAC3D,WAAK,wBAAwB;AAC7B,sBAAgB;AAAA,IAClB,OAAO;AACL,YAAM,SAAS,KAAK,gBAAgB,IAAI,KAAK,KAAK,KAAK;AAAA,QACrD,cAAc;AAAA,QACd,kBAAkB;AAAA,QAClB,oBAAoB;AAAA,QACpB,kBAAkB;AAAA,MACpB;AACA,aAAO,gBAAgB;AACvB,aAAO,oBAAoB;AAC3B,aAAO,sBAAsB;AAC7B,aAAO,oBAAoB;AAC3B,WAAK,gBAAgB,IAAI,KAAK,OAAO,MAAM;AAC3C,sBAAgB,KAAK,QAAQ,KAAK,OAAO,cAAc,kBAAkB,QAAQ,OAAO;AAAA,IAC1F;AAEA,UAAM,QAAsB;AAAA,MAC1B,YAAW,oBAAI,KAAK,GAAE,YAAY;AAAA,MAClC,UAAU,SAAS,YAAY;AAAA,MAC/B,OAAO,KAAK;AAAA,MACZ;AAAA,MACA;AAAA,MACA,aAAa,eAAe;AAAA,MAC5B,oBAAoB;AAAA,MACpB,kBAAkB;AAAA,MAClB;AAAA,MACA,YAAY,SAAS;AAAA,MACrB,QAAQ;AAAA,IACV;AAEA,QAAI;AACF,gBAAU,QAAQ,aAAa,GAAG,EAAE,WAAW,KAAK,CAAC;AACrD,qBAAe,eAAe,KAAK,UAAU,KAAK,IAAI,IAAI;AAAA,IAC5D,QAAQ;AAAA,IAER;AAAA,EACF;AAAA,EAEA,aAAyB;AACvB,WAAO,EAAE,GAAG,KAAK,MAAM;AAAA,EACzB;AAAA,EAEA,mBAA2B;AACzB,QAAI,WAAW;AACf,eAAW,CAAC,OAAO,CAAC,KAAK,KAAK,iBAAiB;AAC7C,kBAAY,KAAK;AAAA,QACf;AAAA,QACA,EAAE;AAAA,QACF,EAAE;AAAA,QACF,EAAE;AAAA,QACF,EAAE;AAAA,MACJ;AAAA,IACF;AACA,WAAO,KAAK,uBAAuB;AAAA,EACrC;AAAA,EAEA,mBAA2B;AACzB,UAAM,OAAO,KAAK,iBAAiB;AACnC,WAAO,IAAI,KAAK,QAAQ,CAAC,CAAC;AAAA,EAC5B;AAAA,EAEA,YAAoB;AAClB,UAAM,UAAU,KAAK,WAAW;AAChC,UAAM,OAAO,KAAK,iBAAiB;AACnC,WAAO;AAAA,MACL,uBAAuB,KAAK,KAAK;AAAA,MACjC;AAAA,MACA,uBAAuB,QAAQ,SAAS;AAAA,MACxC,uBAAuB,QAAQ,aAAa,eAAe,CAAC;AAAA,MAC5D,uBAAuB,QAAQ,iBAAiB,eAAe,CAAC;AAAA,MAChE,uBAAuB,QAAQ,YAAY,eAAe,CAAC;AAAA,MAC3D,wBAAwB,KAAK,QAAQ,CAAC,CAAC;AAAA,IACzC,EAAE,KAAK,IAAI;AAAA,EACb;AAAA,EAEA,QAAc;AACZ,SAAK,QAAQ;AAAA,MACX,cAAc;AAAA,MACd,kBAAkB;AAAA,MAClB,aAAa;AAAA,MACb,WAAW;AAAA,MACX,oBAAoB;AAAA,MACpB,kBAAkB;AAAA,IACpB;AACA,SAAK,uBAAuB;AAC5B,SAAK,gBAAgB,MAAM;AAAA,EAC7B;AAAA,EAEA,SAAS,OAAqB;AAC5B,SAAK,QAAQ;AAAA,EACf;AACF;AAGA,IAAI,gBAAqC;AAElC,SAAS,sBAAsB,OAA8B;AAClE,MAAI,CAAC,eAAe;AAClB,oBAAgB,IAAI,aAAa,KAAK;AAAA,EACxC,WAAW,OAAO;AAChB,kBAAc,SAAS,KAAK;AAAA,EAC9B;AACA,SAAO;AACT;AAEO,SAAS,0BAAgC;AAC9C,iBAAe,MAAM;AACvB;","names":[]}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { R as RateLimiterOptions, T as TokenUsage } from './rate-limiter-
|
|
1
|
+
import { R as RateLimiterOptions, T as TokenUsage } from './rate-limiter-Bz0iSJgZ.js';
|
|
2
2
|
import { ChatOpenAI } from '@langchain/openai';
|
|
3
3
|
import { ChatAnthropic } from '@langchain/anthropic';
|
|
4
4
|
import Anthropic from '@anthropic-ai/sdk';
|
|
@@ -133,6 +133,15 @@ interface LLMUsage {
|
|
|
133
133
|
* `prompt_tokens_details.cached_tokens`, deepseek-native
|
|
134
134
|
* `prompt_cache_hit_tokens`). Absent when the provider reports neither. */
|
|
135
135
|
cachedPromptTokens?: number;
|
|
136
|
+
/**
|
|
137
|
+
* USD cost of this one call. OpenRouter's authoritative `usage.cost`
|
|
138
|
+
* (real, routing+cache-adjusted charge) when the provider is
|
|
139
|
+
* `openrouter`; otherwise `estimateCostUSD` priced from the same
|
|
140
|
+
* OpenRouter-fetched table `TokenTracker` uses for the whole-run total.
|
|
141
|
+
* Absent — never `0` — when neither is available (pricing fetch
|
|
142
|
+
* pending, or the model isn't in OpenRouter's catalog).
|
|
143
|
+
*/
|
|
144
|
+
costUSD?: number;
|
|
136
145
|
}
|
|
137
146
|
type LLMFinishReason = 'stop' | 'length' | 'content_filter' | 'tool_calls' | null;
|
|
138
147
|
interface LLMResponse<T> {
|
package/dist/client.d.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
import './rate-limiter-
|
|
1
|
+
import './rate-limiter-Bz0iSJgZ.js';
|
|
2
2
|
import '@langchain/openai';
|
|
3
3
|
import '@langchain/anthropic';
|
|
4
4
|
import '@anthropic-ai/sdk';
|
|
5
5
|
import 'zod';
|
|
6
|
-
export { A as ANTHROPIC_MODELS, C as CacheAwareLLMCallOptions, b as CacheableBlock, D as DEEPSEEK_MODELS, K as KIMI_MODELS, j as LLMCallOptions, a as LLMClient, k as LLMClientOptions, L as LLMFinishReason, l as LLMProvider, m as LLMResponse, n as LLMStreamChunk, o as LLMStreamOptions, p as LLMUsage, O as OPENAI_MODELS, q as OPENROUTER_MODELS, P as PROVIDER_CATALOG, r as ProviderCatalogEntry, s as ProviderConfig, V as VisionCallOptions, t as VisionImageMediaType, u as VisionImagePart, v as buildVisionMessageContent, w as createAnthropicClient, x as createCreativeClient, y as createDeepSeekClient, z as createFixClient, B as createKimiClient, E as createOpenAIClient, F as createOpenRouterClient, G as createRequirementsClient, H as createZhipuClient, I as getAvailableProvider, J as getSharedLLMClient, M as isProviderAvailable, N as listLmstudioModels, Q as listSelectableProviders, S as resetSharedLLMClient } from './client-
|
|
6
|
+
export { A as ANTHROPIC_MODELS, C as CacheAwareLLMCallOptions, b as CacheableBlock, D as DEEPSEEK_MODELS, K as KIMI_MODELS, j as LLMCallOptions, a as LLMClient, k as LLMClientOptions, L as LLMFinishReason, l as LLMProvider, m as LLMResponse, n as LLMStreamChunk, o as LLMStreamOptions, p as LLMUsage, O as OPENAI_MODELS, q as OPENROUTER_MODELS, P as PROVIDER_CATALOG, r as ProviderCatalogEntry, s as ProviderConfig, V as VisionCallOptions, t as VisionImageMediaType, u as VisionImagePart, v as buildVisionMessageContent, w as createAnthropicClient, x as createCreativeClient, y as createDeepSeekClient, z as createFixClient, B as createKimiClient, E as createOpenAIClient, F as createOpenRouterClient, G as createRequirementsClient, H as createZhipuClient, I as getAvailableProvider, J as getSharedLLMClient, M as isProviderAvailable, N as listLmstudioModels, Q as listSelectableProviders, S as resetSharedLLMClient } from './client-DfzMDgkm.js';
|
|
7
7
|
import '@almadar/core';
|
package/dist/client.js
CHANGED
|
@@ -22,9 +22,10 @@ import {
|
|
|
22
22
|
listLmstudioModels,
|
|
23
23
|
listSelectableProviders,
|
|
24
24
|
resetSharedLLMClient
|
|
25
|
-
} from "./chunk-
|
|
25
|
+
} from "./chunk-DLEZ7FGQ.js";
|
|
26
26
|
import "./chunk-P4VCT25B.js";
|
|
27
|
-
import "./chunk-
|
|
27
|
+
import "./chunk-MOIECDMB.js";
|
|
28
|
+
import "./chunk-T6AKOBX3.js";
|
|
28
29
|
export {
|
|
29
30
|
ANTHROPIC_MODELS,
|
|
30
31
|
DEEPSEEK_MODELS,
|
package/dist/index.d.ts
CHANGED
|
@@ -1,11 +1,11 @@
|
|
|
1
|
-
import { L as LLMFinishReason, a as LLMClient } from './client-
|
|
2
|
-
export { A as ANTHROPIC_MODELS, C as CacheAwareLLMCallOptions, b as CacheableBlock, c as ChatCompletionChoice, d as ChatCompletionMessage, e as ChatCompletionResponse, f as ChatCompletionRole, g as ChatCompletionToolCall, h as ChatCompletionToolDef, i as ChatCompletionUsage, D as DEEPSEEK_MODELS, K as KIMI_MODELS, j as LLMCallOptions, k as LLMClientOptions, l as LLMProvider, m as LLMResponse, n as LLMStreamChunk, o as LLMStreamOptions, p as LLMUsage, O as OPENAI_MODELS, q as OPENROUTER_MODELS, P as PROVIDER_CATALOG, r as ProviderCatalogEntry, s as ProviderConfig, V as VisionCallOptions, t as VisionImageMediaType, u as VisionImagePart, v as buildVisionMessageContent, w as createAnthropicClient, x as createCreativeClient, y as createDeepSeekClient, z as createFixClient, B as createKimiClient, E as createOpenAIClient, F as createOpenRouterClient, G as createRequirementsClient, H as createZhipuClient, I as getAvailableProvider, J as getSharedLLMClient, M as isProviderAvailable, N as listLmstudioModels, Q as listSelectableProviders, R as parseChatCompletionResponse, S as resetSharedLLMClient } from './client-
|
|
3
|
-
export { a as RateLimiter, R as RateLimiterOptions, b as TokenTracker, T as TokenUsage, g as getGlobalRateLimiter, c as getGlobalTokenTracker, r as resetGlobalRateLimiter, d as resetGlobalTokenTracker } from './rate-limiter-
|
|
1
|
+
import { L as LLMFinishReason, a as LLMClient } from './client-DfzMDgkm.js';
|
|
2
|
+
export { A as ANTHROPIC_MODELS, C as CacheAwareLLMCallOptions, b as CacheableBlock, c as ChatCompletionChoice, d as ChatCompletionMessage, e as ChatCompletionResponse, f as ChatCompletionRole, g as ChatCompletionToolCall, h as ChatCompletionToolDef, i as ChatCompletionUsage, D as DEEPSEEK_MODELS, K as KIMI_MODELS, j as LLMCallOptions, k as LLMClientOptions, l as LLMProvider, m as LLMResponse, n as LLMStreamChunk, o as LLMStreamOptions, p as LLMUsage, O as OPENAI_MODELS, q as OPENROUTER_MODELS, P as PROVIDER_CATALOG, r as ProviderCatalogEntry, s as ProviderConfig, V as VisionCallOptions, t as VisionImageMediaType, u as VisionImagePart, v as buildVisionMessageContent, w as createAnthropicClient, x as createCreativeClient, y as createDeepSeekClient, z as createFixClient, B as createKimiClient, E as createOpenAIClient, F as createOpenRouterClient, G as createRequirementsClient, H as createZhipuClient, I as getAvailableProvider, J as getSharedLLMClient, M as isProviderAvailable, N as listLmstudioModels, Q as listSelectableProviders, R as parseChatCompletionResponse, S as resetSharedLLMClient } from './client-DfzMDgkm.js';
|
|
3
|
+
export { E as EstimateCostTokens, a as RateLimiter, R as RateLimiterOptions, b as TokenTracker, T as TokenUsage, e as estimateCostUSD, g as getGlobalRateLimiter, c as getGlobalTokenTracker, r as resetGlobalRateLimiter, d as resetGlobalTokenTracker } from './rate-limiter-Bz0iSJgZ.js';
|
|
4
4
|
import { JsonValue, ServiceParams, ServiceContract } from '@almadar/core';
|
|
5
5
|
export { autoCloseJson, extractJsonFromText, isValidJson, parseJsonResponse, safeParseJson } from './json-parser.js';
|
|
6
6
|
import { z } from 'zod';
|
|
7
7
|
export { JsonSchema, STRUCTURED_OUTPUT_MODELS, StructuredGenerationOptions, StructuredGenerationResult, StructuredOutputClient, StructuredOutputOptions, getStructuredOutputClient, isStructuredOutputAvailable, resetStructuredOutputClient } from './structured-output.js';
|
|
8
|
-
export { ErrorPrediction, GFlowNetResult, GoalSpec, MasarError, MasarGenerateOptions, MasarGenerateResult, MasarHealthResult, MasarProvider, MasarProviderOptions, PredictErrorsResult, RankEditsResult, RankedEdit, getMasarProvider, resetMasarProvider } from './providers/index.js';
|
|
8
|
+
export { ErrorPrediction, GFlowNetResult, GoalSpec, JEV_DECISIONS_URL, JEV_MODELS, JevAnswer, JevChoiceAnswer, JevChoiceQuestion, JevDecideRequest, JevDecideResult, JevError, JevHealthResult, JevJsonValue, JevModelId, JevNoulAnswer, JevNoulQuestion, JevProvider, JevProviderOptions, JevQuestion, JevScoreAnswer, JevScoreQuestion, JevState, JevUsage, MasarError, MasarGenerateOptions, MasarGenerateResult, MasarHealthResult, MasarProvider, MasarProviderOptions, PredictErrorsResult, RankEditsResult, RankedEdit, getJevProvider, getMasarProvider, isJevAvailable, resetJevProvider, resetMasarProvider } from './providers/index.js';
|
|
9
9
|
import '@langchain/openai';
|
|
10
10
|
import '@langchain/anthropic';
|
|
11
11
|
import '@anthropic-ai/sdk';
|
|
@@ -132,11 +132,13 @@ declare const LOCAL_IMAGE_MODELS: {
|
|
|
132
132
|
readonly Z_IMAGE_TURBO: "z-image-turbo";
|
|
133
133
|
readonly FLUX_SCHNELL: "flux-schnell";
|
|
134
134
|
readonly FLUX_DEV: "flux-dev";
|
|
135
|
+
readonly FLUX_KONTEXT: "flux-kontext";
|
|
136
|
+
readonly FLUX2_KLEIN: "flux2-klein";
|
|
135
137
|
};
|
|
136
138
|
declare const LOCAL_IMAGE_DEFAULTS: {
|
|
137
139
|
/** Z-Image-Turbo is the default: public on Hugging Face, no token required (FLUX is gated). */
|
|
138
|
-
|
|
139
|
-
|
|
140
|
+
sheet: string;
|
|
141
|
+
preview: string;
|
|
140
142
|
};
|
|
141
143
|
interface ImageClientOptions {
|
|
142
144
|
provider?: ImageProvider;
|
|
@@ -159,6 +161,10 @@ interface ImageGenerateOptions {
|
|
|
159
161
|
quality?: 'auto' | 'low' | 'medium' | 'high';
|
|
160
162
|
background?: 'auto' | 'transparent' | 'opaque';
|
|
161
163
|
seed?: number;
|
|
164
|
+
/** Diffusion step count (local shim; 1–100). */
|
|
165
|
+
steps?: number;
|
|
166
|
+
/** img2img denoise strength 0–1 (local shim `image_strength`), used with `references`. */
|
|
167
|
+
strength?: number;
|
|
162
168
|
references?: ReadonlyArray<ImageReference>;
|
|
163
169
|
provider?: {
|
|
164
170
|
order?: string[];
|
package/dist/index.js
CHANGED
|
@@ -23,7 +23,7 @@ import {
|
|
|
23
23
|
listSelectableProviders,
|
|
24
24
|
parseChatCompletionResponse,
|
|
25
25
|
resetSharedLLMClient
|
|
26
|
-
} from "./chunk-
|
|
26
|
+
} from "./chunk-DLEZ7FGQ.js";
|
|
27
27
|
import {
|
|
28
28
|
autoCloseJson,
|
|
29
29
|
extractJsonFromText,
|
|
@@ -37,21 +37,31 @@ import {
|
|
|
37
37
|
getStructuredOutputClient,
|
|
38
38
|
isStructuredOutputAvailable,
|
|
39
39
|
resetStructuredOutputClient
|
|
40
|
-
} from "./chunk-
|
|
40
|
+
} from "./chunk-SJE3GTGZ.js";
|
|
41
41
|
import {
|
|
42
42
|
RateLimiter,
|
|
43
|
-
TokenTracker,
|
|
44
43
|
getGlobalRateLimiter,
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
resetGlobalTokenTracker
|
|
48
|
-
} from "./chunk-RPG3SUIB.js";
|
|
44
|
+
resetGlobalRateLimiter
|
|
45
|
+
} from "./chunk-MOIECDMB.js";
|
|
49
46
|
import {
|
|
47
|
+
JEV_DECISIONS_URL,
|
|
48
|
+
JEV_MODELS,
|
|
49
|
+
JevError,
|
|
50
|
+
JevProvider,
|
|
50
51
|
MasarError,
|
|
51
52
|
MasarProvider,
|
|
53
|
+
getJevProvider,
|
|
52
54
|
getMasarProvider,
|
|
55
|
+
isJevAvailable,
|
|
56
|
+
resetJevProvider,
|
|
53
57
|
resetMasarProvider
|
|
54
|
-
} from "./chunk-
|
|
58
|
+
} from "./chunk-OXFWONZP.js";
|
|
59
|
+
import {
|
|
60
|
+
TokenTracker,
|
|
61
|
+
estimateCostUSD,
|
|
62
|
+
getGlobalTokenTracker,
|
|
63
|
+
resetGlobalTokenTracker
|
|
64
|
+
} from "./chunk-T6AKOBX3.js";
|
|
55
65
|
|
|
56
66
|
// src/embedding-client.ts
|
|
57
67
|
var EMBEDDING_PROVIDERS = ["openai", "openrouter", "lmstudio"];
|
|
@@ -219,12 +229,14 @@ var OPENROUTER_IMAGE_DEFAULTS = {
|
|
|
219
229
|
var LOCAL_IMAGE_MODELS = {
|
|
220
230
|
Z_IMAGE_TURBO: "z-image-turbo",
|
|
221
231
|
FLUX_SCHNELL: "flux-schnell",
|
|
222
|
-
FLUX_DEV: "flux-dev"
|
|
232
|
+
FLUX_DEV: "flux-dev",
|
|
233
|
+
FLUX_KONTEXT: "flux-kontext",
|
|
234
|
+
FLUX2_KLEIN: "flux2-klein"
|
|
223
235
|
};
|
|
224
236
|
var LOCAL_IMAGE_DEFAULTS = {
|
|
225
237
|
/** Z-Image-Turbo is the default: public on Hugging Face, no token required (FLUX is gated). */
|
|
226
|
-
sheet: LOCAL_IMAGE_MODELS.Z_IMAGE_TURBO,
|
|
227
|
-
preview: LOCAL_IMAGE_MODELS.Z_IMAGE_TURBO
|
|
238
|
+
sheet: process.env.LOCAL_IMAGE_SHEET_MODEL ?? LOCAL_IMAGE_MODELS.Z_IMAGE_TURBO,
|
|
239
|
+
preview: process.env.LOCAL_IMAGE_PREVIEW_MODEL ?? LOCAL_IMAGE_MODELS.Z_IMAGE_TURBO
|
|
228
240
|
};
|
|
229
241
|
var IMAGE_MAX_ATTEMPTS = 3;
|
|
230
242
|
var IMAGE_RETRY_BACKOFF_MS = [500, 2e3];
|
|
@@ -243,6 +255,8 @@ var API_KEY_ENV_VARS2 = {
|
|
|
243
255
|
var IMAGE_CHECKED_PARAM_NAMES = {
|
|
244
256
|
background: "background",
|
|
245
257
|
seed: "seed",
|
|
258
|
+
steps: "steps",
|
|
259
|
+
strength: "image_strength",
|
|
246
260
|
references: "input_references"
|
|
247
261
|
};
|
|
248
262
|
function isJsonObject(value) {
|
|
@@ -315,7 +329,11 @@ var _ImageClient = class _ImageClient {
|
|
|
315
329
|
const cached = this.capabilitiesCache.get(model);
|
|
316
330
|
if (cached) return cached;
|
|
317
331
|
if (this.provider === "local") {
|
|
318
|
-
const local = {
|
|
332
|
+
const local = {
|
|
333
|
+
model,
|
|
334
|
+
supportedParameters: /* @__PURE__ */ new Set(["seed", IMAGE_CHECKED_PARAM_NAMES.steps, IMAGE_CHECKED_PARAM_NAMES.strength, IMAGE_CHECKED_PARAM_NAMES.references]),
|
|
335
|
+
raw: {}
|
|
336
|
+
};
|
|
319
337
|
this.capabilitiesCache.set(model, local);
|
|
320
338
|
return local;
|
|
321
339
|
}
|
|
@@ -353,6 +371,14 @@ var _ImageClient = class _ImageClient {
|
|
|
353
371
|
delete filtered.seed;
|
|
354
372
|
dropped.push(IMAGE_CHECKED_PARAM_NAMES.seed);
|
|
355
373
|
}
|
|
374
|
+
if (filtered.steps !== void 0 && !caps.supportedParameters.has(IMAGE_CHECKED_PARAM_NAMES.steps)) {
|
|
375
|
+
delete filtered.steps;
|
|
376
|
+
dropped.push(IMAGE_CHECKED_PARAM_NAMES.steps);
|
|
377
|
+
}
|
|
378
|
+
if (filtered.strength !== void 0 && !caps.supportedParameters.has(IMAGE_CHECKED_PARAM_NAMES.strength)) {
|
|
379
|
+
delete filtered.strength;
|
|
380
|
+
dropped.push(IMAGE_CHECKED_PARAM_NAMES.strength);
|
|
381
|
+
}
|
|
356
382
|
if (filtered.references !== void 0 && filtered.references.length > 0 && !caps.supportedParameters.has(IMAGE_CHECKED_PARAM_NAMES.references)) {
|
|
357
383
|
delete filtered.references;
|
|
358
384
|
dropped.push(IMAGE_CHECKED_PARAM_NAMES.references);
|
|
@@ -375,6 +401,7 @@ var _ImageClient = class _ImageClient {
|
|
|
375
401
|
if (opts.quality !== void 0) requestBody.quality = opts.quality;
|
|
376
402
|
if (opts.background !== void 0) requestBody.background = opts.background;
|
|
377
403
|
if (opts.seed !== void 0) requestBody.seed = opts.seed;
|
|
404
|
+
if (opts.steps !== void 0) requestBody.steps = opts.steps;
|
|
378
405
|
if (opts.references !== void 0 && opts.references.length > 0) {
|
|
379
406
|
requestBody.input_references = opts.references.map((r) => ({ type: "image_url", image_url: { url: r.url } }));
|
|
380
407
|
}
|
|
@@ -398,6 +425,8 @@ var _ImageClient = class _ImageClient {
|
|
|
398
425
|
if (opts.n !== void 0) requestBody.n = opts.n;
|
|
399
426
|
if (opts.size !== void 0) requestBody.size = opts.size;
|
|
400
427
|
if (opts.seed !== void 0) requestBody.seed = opts.seed;
|
|
428
|
+
if (opts.steps !== void 0) requestBody.steps = opts.steps;
|
|
429
|
+
if (opts.strength !== void 0) requestBody.image_strength = opts.strength;
|
|
401
430
|
if (opts.references !== void 0 && opts.references.length > 0) {
|
|
402
431
|
requestBody.input_references = opts.references.map((r) => ({ type: "image_url", image_url: { url: r.url } }));
|
|
403
432
|
}
|
|
@@ -910,6 +939,10 @@ export {
|
|
|
910
939
|
EmbeddingClient,
|
|
911
940
|
IMAGE_PROVIDER_CATALOG,
|
|
912
941
|
ImageClient,
|
|
942
|
+
JEV_DECISIONS_URL,
|
|
943
|
+
JEV_MODELS,
|
|
944
|
+
JevError,
|
|
945
|
+
JevProvider,
|
|
913
946
|
KIMI_MODELS,
|
|
914
947
|
LLMClient,
|
|
915
948
|
LOCAL_IMAGE_DEFAULTS,
|
|
@@ -940,14 +973,17 @@ export {
|
|
|
940
973
|
createRequirementsClient,
|
|
941
974
|
createZhipuClient,
|
|
942
975
|
detectTruncation,
|
|
976
|
+
estimateCostUSD,
|
|
943
977
|
extractJsonFromText,
|
|
944
978
|
findLastCompleteElement,
|
|
945
979
|
getAvailableProvider,
|
|
946
980
|
getGlobalRateLimiter,
|
|
947
981
|
getGlobalTokenTracker,
|
|
982
|
+
getJevProvider,
|
|
948
983
|
getMasarProvider,
|
|
949
984
|
getSharedLLMClient,
|
|
950
985
|
getStructuredOutputClient,
|
|
986
|
+
isJevAvailable,
|
|
951
987
|
isLikelyTruncated,
|
|
952
988
|
isProviderAvailable,
|
|
953
989
|
isStructuredOutputAvailable,
|
|
@@ -959,6 +995,7 @@ export {
|
|
|
959
995
|
parseJsonResponse,
|
|
960
996
|
resetGlobalRateLimiter,
|
|
961
997
|
resetGlobalTokenTracker,
|
|
998
|
+
resetJevProvider,
|
|
962
999
|
resetMasarProvider,
|
|
963
1000
|
resetSharedLLMClient,
|
|
964
1001
|
resetStructuredOutputClient,
|