pyyol 1.8.0 → 1.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/pricing.js CHANGED
@@ -6,8 +6,9 @@
6
6
  // authoritative cost comes from the Pyyol Gateway). When a provider changes prices,
7
7
  // bump PRICING_VERSION and update the table — never edit silently, so a cost can
8
8
  // always be traced to the table that produced it.
9
+ import { isSelfHosted } from "./providers.js";
9
10
  // Bump whenever any rate below changes. Stamped onto every estimate.
10
- export const PRICING_VERSION = "2026-07-24";
11
+ export const PRICING_VERSION = "2026-08-06";
11
12
  // Canonical model id -> Rate. Lowercase, provider-agnostic.
12
13
  const TABLE = {
13
14
  // OpenAI
@@ -29,12 +30,32 @@ const TABLE = {
29
30
  // Google (Gemini)
30
31
  "gemini-flash": { input: 0.15, output: 0.6, cachedInput: 0.0375 },
31
32
  "gemini-pro": { input: 1.25, output: 5.0, cachedInput: 0.3125 },
33
+ // Open-weight served by a HOSTED provider — there IS a per-token bill.
34
+ //
35
+ // "Open weight" does not mean "free". Groq bills per token like anyone else, and
36
+ // recording $0 for it meant a Groq-backed agent reported no cost at all on a platform
37
+ // that advertises verified LLM cost tracking. (The Python SDK already scoped these by
38
+ // provider; this table did not, so the two disagreed about the same call.)
39
+ "groq-llama-8b": { input: 0.05, output: 0.08 },
40
+ "groq-llama-70b": { input: 0.59, output: 0.79 },
32
41
  // Open-weight / self-hosted (no per-token bill)
33
42
  llama: { input: 0.0, output: 0.0 },
34
43
  mistral: { input: 0.0, output: 0.0 },
35
44
  qwen: { input: 0.0, output: 0.0 },
36
45
  deepseek: { input: 0.27, output: 1.1 },
37
46
  };
47
+ // Provider-scoped rules, checked BEFORE the name rules. An open-weight model is $0
48
+ // when you run it yourself and very much not $0 when a hosted provider serves it — and
49
+ // the model id cannot tell you which, since "llama-3.3-70b" is the same string either
50
+ // way. Only an explicitly provider-attributed call gets a hosted rate.
51
+ const PROVIDER_RULES = {
52
+ groq: [
53
+ ["llama-3.1-8b", "groq-llama-8b"],
54
+ ["llama-3.1-70b", "groq-llama-70b"],
55
+ ["llama-3.3-70b", "groq-llama-70b"],
56
+ ["llama-4", "groq-llama-70b"],
57
+ ],
58
+ };
38
59
  // Last-resort rate for an unmapped model (never silently $0 unless open-weight).
39
60
  const FALLBACK = { input: 0.5, output: 1.5 };
40
61
  // Ordered [substring, canonical] rules; first match wins, most specific first.
@@ -72,11 +93,17 @@ const RULES = [
72
93
  ["qwen", "qwen"],
73
94
  ["deepseek", "deepseek"],
74
95
  ];
75
- /** Map a raw model string to a canonical table key, or null if unknown. */
76
- export function canonical(model) {
96
+ /** Map a raw model string to a canonical table key, or null if unknown.
97
+ * `provider` scopes the lookup: provider-specific rules win, because they are the
98
+ * only ones that know a hosted bill exists for a model that would otherwise be free. */
99
+ export function canonical(model, provider = "") {
77
100
  const m = (model ?? "").trim().toLowerCase();
78
101
  if (!m)
79
102
  return null;
103
+ for (const [needle, key] of PROVIDER_RULES[provider.trim().toLowerCase()] ?? []) {
104
+ if (m.includes(needle))
105
+ return key;
106
+ }
80
107
  if (m in TABLE)
81
108
  return m;
82
109
  for (const [needle, key] of RULES)
@@ -84,27 +111,84 @@ export function canonical(model) {
84
111
  return key;
85
112
  return null;
86
113
  }
87
- /** The Rate for a model (falls back to a mid-tier rate for unknown models). */
88
- export function rateFor(model) {
89
- const key = canonical(model);
114
+ // Rate for a model the developer serves themselves. Not a guess and not a fallback —
115
+ // there is no per-token bill, so any non-zero number here would be fiction.
116
+ const FREE = { input: 0, output: 0, cachedInput: 0 };
117
+ /** The Rate for a model (falls back to a mid-tier rate for unknown models).
118
+ * `provider` scopes the lookup so a self-hosted model stays at $0. */
119
+ export function rateFor(model, provider = "") {
120
+ // Self-hosted first, ahead of every name-based rule. The model id cannot tell you
121
+ // who served it — "llama-3.3-70b" is the same string on Groq's bill and on your own
122
+ // GPU — so without this an Ollama user is charged Groq's rates for electricity they
123
+ // already paid for, and the unknown-model fallback would invent a bill outright.
124
+ if (isSelfHosted(provider.trim().toLowerCase()))
125
+ return FREE;
126
+ const key = canonical(model, provider);
90
127
  return key !== null ? TABLE[key] : FALLBACK;
91
128
  }
92
129
  /** True if the model maps to an explicit table entry (not the fallback). */
93
130
  export function isKnown(model) {
94
131
  return canonical(model) !== null;
95
132
  }
133
+ // Cache-WRITE multipliers, applied to a model's input rate.
134
+ //
135
+ // Writing a prompt into a provider's cache is a separately-billed event from reading it
136
+ // back, and the two go in OPPOSITE directions: Anthropic surcharges a write to 1.25x
137
+ // input and discounts a read to 0.1x, while OpenAI does not bill writes at all. Recording
138
+ // only reads therefore does not merely lose a number — it prices the expensive half of
139
+ // caching at zero, and does so for the agents that cache hardest.
140
+ //
141
+ // Expressed as a multiplier rather than a per-model rate because that is how providers
142
+ // publish it: one ratio per model family. A multiplier also cannot drift out of step with
143
+ // a model's input rate the way a duplicated absolute number can.
144
+ const CACHE_WRITE_MULTIPLIER = [
145
+ ["claude-", 1.25], // Anthropic bills a cache write at 1.25x input
146
+ ["gpt-", 0.0], // OpenAI prompt caching is automatic; writes are not billed
147
+ ["o1", 0.0],
148
+ ["o3", 0.0],
149
+ ["o4", 0.0],
150
+ ["gemini-", 0.0], // implicit context caching is free
151
+ ];
152
+ // Multiplier for a family with no published cache-write behaviour: a write costs what an
153
+ // ordinary input token costs. Not 0.0, which would make an unrecognised model's caching
154
+ // silently free — the flattering direction.
155
+ const DEFAULT_CACHE_WRITE_MULTIPLIER = 1.0;
156
+ /** USD per 1M tokens for writing a prompt into the provider's cache. */
157
+ export function cacheWriteRate(model, provider = "") {
158
+ const rate = rateFor(model, provider);
159
+ const key = canonical(model, provider) ?? "";
160
+ for (const [prefix, mult] of CACHE_WRITE_MULTIPLIER) {
161
+ if (key.startsWith(prefix))
162
+ return rate.input * mult;
163
+ }
164
+ return rate.input * DEFAULT_CACHE_WRITE_MULTIPLIER;
165
+ }
96
166
  /**
97
- * USD cost estimate for one model call. `cachedTokens` are a subset of
98
- * `promptTokens` billed at the cached-input rate; `reasoningTokens` are output
99
- * tokens already counted in `completionTokens` (kept for reporting).
167
+ * USD cost estimate for one model call.
168
+ *
169
+ * `promptTokens` is the TOTAL billable input, and `cachedTokens` (reads) and
170
+ * `cachedWriteTokens` (creations) are SUBSETS of it — so the three partition the input
171
+ * into full-rate, read-rate and write-rate portions. Normalizing onto that convention is
172
+ * the caller's job (extractUsage does it): providers disagree about whether cache tokens
173
+ * sit inside their reported input count, and pricing must not have to know which.
174
+ *
175
+ * `reasoningTokens` are output tokens already counted in `completionTokens` (kept for
176
+ * reporting).
100
177
  */
101
178
  export function estimateCost(model, a = {}) {
102
- const rate = rateFor(model);
179
+ const rate = rateFor(model, a.provider ?? "");
103
180
  const prompt = Math.max(0, a.promptTokens ?? 0);
104
181
  const completion = Math.max(0, a.completionTokens ?? 0);
105
- const cached = Math.max(0, Math.min(a.cachedTokens ?? 0, prompt));
106
- const fullInput = prompt - cached;
107
- const cachedRate = rate.cachedInput ?? rate.input;
108
- const cost = (fullInput * rate.input + cached * cachedRate + completion * rate.output) / 1_000_000;
182
+ // Reads are taken out first, then writes from what remains, so the two subsets can
183
+ // never overlap and bill the same token twice.
184
+ const read = Math.max(0, Math.min(a.cachedTokens ?? 0, prompt));
185
+ const write = Math.max(0, Math.min(a.cachedWriteTokens ?? 0, prompt - read));
186
+ const fullInput = prompt - read - write;
187
+ const readRate = rate.cachedInput ?? rate.input;
188
+ const cost = (fullInput * rate.input +
189
+ read * readRate +
190
+ write * cacheWriteRate(model, a.provider ?? "") +
191
+ completion * rate.output) /
192
+ 1_000_000;
109
193
  return Math.round(cost * 1e8) / 1e8;
110
194
  }
@@ -0,0 +1,27 @@
1
+ export declare const OPENAI = "openai";
2
+ export declare const ANTHROPIC = "anthropic";
3
+ export declare const GOOGLE = "google";
4
+ export declare const GROQ = "groq";
5
+ export declare const OLLAMA = "ollama";
6
+ export declare const SELF_HOSTED = "self-hosted";
7
+ /** True for loopback, link-local and private-network addresses, and for the hostnames
8
+ * that conventionally mean "this machine". A model served from one of these has no
9
+ * per-token bill, which is why it is worth detecting even when unnamed. */
10
+ export declare function isLocalHost(host: string): boolean;
11
+ /** Provider key implied by a base URL, or "" when the host says nothing. A local
12
+ * address always resolves to SOMETHING — never "" — because "we could not tell" and
13
+ * "it runs on your own hardware for free" must not be the same answer. */
14
+ export declare function fromBaseUrl(baseUrl: string): string;
15
+ /** Provider key implied by a client's constructor or module name, or "". */
16
+ export declare function fromName(name: string): string;
17
+ /** The provider serving this call. baseURL wins over the name: the name only says
18
+ * which WIRE FORMAT the client speaks, while the URL says who is on the other end —
19
+ * and for every OpenAI-compatible endpoint those are different answers. */
20
+ export declare function resolve(o: {
21
+ name?: string;
22
+ baseUrl?: string;
23
+ fallback?: string;
24
+ }): string;
25
+ /** True when the provider runs on the developer's own hardware, so its tokens carry
26
+ * no per-token bill. */
27
+ export declare function isSelfHosted(provider: string): boolean;
@@ -0,0 +1,140 @@
1
+ // Which provider is actually serving this call?
2
+ //
3
+ // Identifying the provider by the client CLASS is not enough, and getting it wrong is
4
+ // not cosmetic: the provider decides whether a call is priced or free, and it is half
5
+ // of the model attribution the public benchmark ranks on.
6
+ //
7
+ // The problem is that most of the ecosystem speaks the OpenAI wire format. Ollama,
8
+ // vLLM, LM Studio, llama.cpp, OpenRouter, Together, Groq, DeepSeek, Azure and others
9
+ // are all routinely driven through the OpenAI SDK with nothing changed but `baseURL`.
10
+ // Classifying those by constructor name calls every one of them "openai" — pricing a
11
+ // locally-served Llama at OpenAI's rates and filing it under the wrong vendor.
12
+ //
13
+ // Resolution order: baseURL host → module/constructor name → caller's fallback.
14
+ // Anything on a loopback or private address is SELF-HOSTED even when the runtime is
15
+ // unrecognised: it has no per-token bill, and "unknown" would price it as if it did.
16
+ //
17
+ // Keep the provider keys here in step with `internal/rating/taxonomy.go` and the
18
+ // Python SDK's `providers.py` — the backend classifies exactly these strings.
19
+ export const OPENAI = "openai";
20
+ export const ANTHROPIC = "anthropic";
21
+ export const GOOGLE = "google";
22
+ export const GROQ = "groq";
23
+ export const OLLAMA = "ollama";
24
+ export const SELF_HOSTED = "self-hosted";
25
+ /** host substring -> provider key. Most specific first. */
26
+ const HOST_RULES = [
27
+ ["api.openai.com", OPENAI],
28
+ ["openai.azure.com", "azure"],
29
+ ["api.anthropic.com", ANTHROPIC],
30
+ ["bedrock-runtime", "bedrock"],
31
+ ["bedrock", "bedrock"],
32
+ ["generativelanguage.googleapis.com", GOOGLE],
33
+ ["aiplatform.googleapis.com", "vertex"],
34
+ ["api.groq.com", GROQ],
35
+ ["openrouter.ai", "openrouter"],
36
+ ["api.together.xyz", "together"],
37
+ ["together.ai", "together"],
38
+ ["api.fireworks.ai", "fireworks"],
39
+ ["api.deepinfra.com", "deepinfra"],
40
+ ["api.mistral.ai", "mistral"],
41
+ ["api.deepseek.com", "deepseek"],
42
+ ["api.cohere.ai", "cohere"],
43
+ ["api.cohere.com", "cohere"],
44
+ ["api.x.ai", "xai"],
45
+ ["api.perplexity.ai", "perplexity"],
46
+ ["api.cerebras.ai", "cerebras"],
47
+ ["api.sambanova.ai", "sambanova"],
48
+ ["api.studio.nebius", "nebius"],
49
+ ["api.hyperbolic.xyz", "hyperbolic"],
50
+ ["api.moonshot", "moonshot"],
51
+ ["dashscope.aliyuncs.com", "alibaba"],
52
+ ];
53
+ /** Default ports of the common local runtimes. */
54
+ const LOCAL_PORTS = {
55
+ "11434": OLLAMA,
56
+ "1234": "lmstudio",
57
+ "8000": "vllm",
58
+ "8080": "llamacpp",
59
+ "5000": "localai",
60
+ "3000": SELF_HOSTED,
61
+ "9997": SELF_HOSTED,
62
+ };
63
+ /** constructor/module-name substring -> provider. "openai" LAST: several packages
64
+ * embed it in their own names. */
65
+ const NAME_RULES = [
66
+ ["ollama", OLLAMA],
67
+ ["anthropic", ANTHROPIC],
68
+ ["groq", GROQ],
69
+ ["mistral", "mistral"],
70
+ ["cohere", "cohere"],
71
+ ["googlegenai", GOOGLE],
72
+ ["generativeai", GOOGLE],
73
+ ["google", GOOGLE],
74
+ ["openai", OPENAI],
75
+ ];
76
+ const SELF_HOSTED_KEYS = new Set([OLLAMA, SELF_HOSTED, "vllm", "lmstudio", "llamacpp", "localai", "tgi"]);
77
+ function parse(baseUrl) {
78
+ if (!baseUrl)
79
+ return { host: "", port: "" };
80
+ const withScheme = baseUrl.includes("://") ? baseUrl : `http://${baseUrl}`;
81
+ try {
82
+ const u = new URL(withScheme);
83
+ return { host: u.hostname.toLowerCase(), port: u.port };
84
+ }
85
+ catch {
86
+ return { host: "", port: "" };
87
+ }
88
+ }
89
+ /** True for loopback, link-local and private-network addresses, and for the hostnames
90
+ * that conventionally mean "this machine". A model served from one of these has no
91
+ * per-token bill, which is why it is worth detecting even when unnamed. */
92
+ export function isLocalHost(host) {
93
+ if (!host)
94
+ return false;
95
+ if (["localhost", "127.0.0.1", "::1", "0.0.0.0", "host.docker.internal"].includes(host))
96
+ return true;
97
+ if (host.endsWith(".local") || host.endsWith(".internal"))
98
+ return true;
99
+ // IPv4 private ranges: 10/8, 192.168/16, 172.16–31/12, 127/8, 169.254/16.
100
+ const m = /^(\d{1,3})\.(\d{1,3})\.(\d{1,3})\.(\d{1,3})$/.exec(host);
101
+ if (!m)
102
+ return false;
103
+ const [a, b] = [Number(m[1]), Number(m[2])];
104
+ return a === 10 || a === 127 || (a === 192 && b === 168) || (a === 172 && b >= 16 && b <= 31) || (a === 169 && b === 254);
105
+ }
106
+ /** Provider key implied by a base URL, or "" when the host says nothing. A local
107
+ * address always resolves to SOMETHING — never "" — because "we could not tell" and
108
+ * "it runs on your own hardware for free" must not be the same answer. */
109
+ export function fromBaseUrl(baseUrl) {
110
+ const { host, port } = parse(baseUrl);
111
+ if (!host)
112
+ return "";
113
+ for (const [needle, provider] of HOST_RULES) {
114
+ if (host.includes(needle))
115
+ return provider;
116
+ }
117
+ if (isLocalHost(host))
118
+ return LOCAL_PORTS[port] ?? SELF_HOSTED;
119
+ return "";
120
+ }
121
+ /** Provider key implied by a client's constructor or module name, or "". */
122
+ export function fromName(name) {
123
+ const n = (name || "").toLowerCase().replace(/[^a-z]/g, "");
124
+ for (const [needle, provider] of NAME_RULES) {
125
+ if (n.includes(needle))
126
+ return provider;
127
+ }
128
+ return "";
129
+ }
130
+ /** The provider serving this call. baseURL wins over the name: the name only says
131
+ * which WIRE FORMAT the client speaks, while the URL says who is on the other end —
132
+ * and for every OpenAI-compatible endpoint those are different answers. */
133
+ export function resolve(o) {
134
+ return fromBaseUrl(o.baseUrl ?? "") || fromName(o.name ?? "") || o.fallback || "";
135
+ }
136
+ /** True when the provider runs on the developer's own hardware, so its tokens carry
137
+ * no per-token bill. */
138
+ export function isSelfHosted(provider) {
139
+ return SELF_HOSTED_KEYS.has(provider);
140
+ }
@@ -0,0 +1,74 @@
1
+ /** Bump only for a change that intentionally invalidates existing fingerprints: it is part
2
+ * of the hashed payload, so a bump splits every agent's history into a new epoch. */
3
+ export declare const SCAFFOLD_VERSION = "pyyol-scaffold-v1";
4
+ /** Sampling parameters that change how a model behaves and are therefore part of the
5
+ * scaffold. Anything not listed is ignored, so a provider adding an unrelated field does
6
+ * not silently split every agent's history.
7
+ *
8
+ * `model` is absent on purpose. So are baseURL, apiKey, timeout and stream: the first two
9
+ * would make gateway routing look like a new scaffold, and the last two do not affect what
10
+ * the model decides. */
11
+ export declare const SAMPLING_KEYS: readonly string[];
12
+ type Any = Record<string, unknown> | unknown;
13
+ export type Components = Partial<Record<"client" | "roles" | "sampling" | "tools" | "system", string>>;
14
+ /** The fingerprint components observable in one provider request.
15
+ *
16
+ * Never throws: a fingerprinting problem must not break a developer's model call. */
17
+ export declare function extract(kwargs: Record<string, unknown>, endpoint?: string): Components;
18
+ /** The exact string that gets hashed.
19
+ *
20
+ * Specified rather than incidental: the Python SDK builds the same string and a shared
21
+ * conformance fixture checks both against the same expected fingerprints. Keys are emitted
22
+ * in a FIXED order (not sorted, not insertion order) so neither language's map iteration
23
+ * can affect the result, and an absent component is omitted rather than emitted empty — so
24
+ * adding a component later does not change the fingerprint of requests that never had one. */
25
+ export declare function canonical(c: Components): string;
26
+ /** Short, prefixed id for a scaffold. 16 hex chars of SHA-256 (64 bits).
27
+ *
28
+ * Short enough to read in a UI and group by in SQL. Collisions are irrelevant here in a way
29
+ * they would not be for a security token: fingerprints are only compared WITHIN one agent's
30
+ * own history, so the space that must stay distinct is a handful of harness versions. */
31
+ export declare function fingerprint(c: Components): string;
32
+ /** Why a request could not be fingerprinted, as a stable CODE rather than prose.
33
+ *
34
+ * A code on the wire and prose at the point of reading, deliberately. The reason repeats on
35
+ * every decision of every non-qualifying agent, so shipping and storing the sentence would
36
+ * duplicate it thousands of times per match. A code is also aggregatable — "how many agents
37
+ * are ineligible, and why" is a question worth being able to ask — and its wording can change
38
+ * later without a migration. */
39
+ export declare const ISSUE_NO_SYSTEM_PROMPT = "no_system_prompt";
40
+ export declare const ISSUE_NO_MESSAGES = "no_messages";
41
+ /** Prose for each code. Read by the CLI and the dev-facing trace; never stored. */
42
+ export declare const ISSUE_EXPLANATIONS: Readonly<Record<string, string>>;
43
+ /** Code for why this request yields no usable fingerprint, or "" when it does. */
44
+ export declare function issue(kwargs: Record<string, unknown>, endpoint?: string): string;
45
+ /** Human-readable reason for an issue code, or "" for no issue / an unknown code. */
46
+ export declare function explain(code: string): string;
47
+ /** Prose reason this request cannot be fingerprinted, or "" when it can.
48
+ *
49
+ * Convenience for local developer output; the wire carries `issue` codes. */
50
+ export declare function diagnose(kwargs: Record<string, unknown>, endpoint?: string): string;
51
+ /** Fingerprint one provider request, or "" when it cannot be fingerprinted.
52
+ *
53
+ * An empty result means "unknown", never a hash of nothing — a fingerprint shared by every
54
+ * request that failed to yield components would silently pool unrelated scaffolds into one
55
+ * bogus epoch and publish it as a controlled comparison.
56
+ *
57
+ * A SYSTEM PROMPT IS REQUIRED, and this is the sharp edge of the whole design. Without one,
58
+ * the hashable surface is `client` + `roles` + sampling — none of which move when the
59
+ * developer rewrites the instructions they actually steer the model with, because those
60
+ * instructions sit in a user message alongside the game state.
61
+ *
62
+ * That is worse than having no fingerprint. It is a FALSE CERTIFICATE: an agent could
63
+ * replace its entire strategy prompt mid-season, keep reporting the same scaffold id, and
64
+ * have the improvement attributed to whatever model it swapped to — the fingerprint would be
65
+ * manufacturing the confound it exists to remove. Hashing more cannot fix it, because the
66
+ * instructions and the game state are the same string. See `diagnose`. */
67
+ export declare function fromRequest(kwargs: Record<string, unknown>, endpoint?: string): string;
68
+ /** True when a set of observations can support a within-scaffold model comparison.
69
+ *
70
+ * Requires exactly one KNOWN scaffold. An unknown ("") fingerprint disqualifies rather than
71
+ * being ignored: treating "we could not tell" as "the same as the others" is how a
72
+ * confounded comparison gets published as a clean one. */
73
+ export declare function eligibleForPairing(fingerprints: ReadonlyArray<string | undefined>): boolean;
74
+ export type { Any };