metergraph-core 0.2.23__tar.gz → 0.2.24__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: metergraph-core
3
- Version: 0.2.23
3
+ Version: 0.2.24
4
4
  Summary: Reusable MeterGraph catalog and deterministic billing engine
5
5
  License-Expression: Apache-2.0
6
6
  Requires-Python: >=3.10
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "metergraph-core"
3
- version = "0.2.23"
3
+ version = "0.2.24"
4
4
  description = "Reusable MeterGraph catalog and deterministic billing engine"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -6,6 +6,7 @@ from .catalog import (
6
6
  CostResult,
7
7
  Price,
8
8
  ResolvedPrice,
9
+ counts_cache_read_in_input,
9
10
  direct_channel_for_provider,
10
11
  )
11
12
  from .billing import (
@@ -40,6 +41,7 @@ __all__ = [
40
41
  "RetrievalCatalog",
41
42
  "RetrievalCostResult",
42
43
  "RetrievalPrice",
44
+ "counts_cache_read_in_input",
43
45
  "direct_channel_for_provider",
44
46
  "load_catalog",
45
47
  "normalize_gateway_evidence",
@@ -37,6 +37,54 @@ _DIRECT_CHANNEL_BY_PROVIDER = {
37
37
  }
38
38
 
39
39
 
40
+ # How a call reports cached tokens is set by whoever serves the model, so the
41
+ # default is keyed on (publisher, channel) rather than the channel alone.
42
+ # OpenAI's `prompt_tokens`, Google's `promptTokenCount`, DeepSeek's and xAI's
43
+ # `prompt_tokens` all include the cached tokens they also report separately, so
44
+ # a row on these pairs must deduct them before applying the input rate. Without
45
+ # it a cached token is billed at the input rate and again at the cache-read
46
+ # rate -- several times the real cost on a cache-heavy workload, and enough to
47
+ # reorder a customer's spend ranking.
48
+ #
49
+ # A channel alone cannot decide this: google-vertex-ai serves Anthropic's
50
+ # models beside Google's, and Claude on Vertex keeps Anthropic's usage shape,
51
+ # where input_tokens already excludes cache reads. Deducting there overcharges
52
+ # in the opposite direction -- the same defect this default exists to remove.
53
+ # Anthropic and Bedrock are absent by design; a gateway channel depends on the
54
+ # provider behind it and states its own rules per row.
55
+ _INPUT_INCLUDES_CACHE_READ_PAIRS = frozenset({
56
+ ("openai", "openai-api"),
57
+ ("google", "google-api"),
58
+ ("google", "google-vertex-ai"),
59
+ ("deepseek", "deepseek-api"),
60
+ ("xai", "xai-api"),
61
+ })
62
+
63
+
64
+ def counts_cache_read_in_input(
65
+ publisher: Any, channel: Any, rules: Mapping[str, Any]
66
+ ) -> bool:
67
+ """Whether cache reads have to come out of billable input for this row.
68
+
69
+ A row states the answer when it differs from its publisher's behaviour on
70
+ that channel -- a gateway serving one of these providers, or a provider
71
+ that changes how it counts. Otherwise the (publisher, channel) pair
72
+ decides, because a catalog author has no way to know which providers report
73
+ cached tokens inside the input total.
74
+
75
+ An unknown publisher defaults to no deduction: over-billing a cached token
76
+ twice is the failure this exists to prevent, so a row we cannot place
77
+ keeps the arithmetic it had before.
78
+ """
79
+ stated = rules.get("input_includes_cache_read")
80
+ if stated is not None:
81
+ return bool(stated)
82
+ if not isinstance(publisher, str) or not isinstance(channel, str):
83
+ return False
84
+ pair = (publisher.strip().lower(), channel.strip().lower())
85
+ return pair in _INPUT_INCLUDES_CACHE_READ_PAIRS
86
+
87
+
40
88
  def _normalize_provider(provider: str) -> str:
41
89
  """Fold a provider spelling through metergraph-core's provider-alias map
42
90
  (e.g. ``aws``/``amazon-bedrock`` -> ``bedrock``, ``google-genai`` ->
@@ -94,6 +142,10 @@ class Price:
94
142
  effective_from: datetime
95
143
  effective_to: datetime | None
96
144
  source_url: str
145
+ # The catalog's publisher for the model this row prices. A channel does not
146
+ # imply one: google-vertex-ai serves Anthropic's models beside Google's,
147
+ # and the two report cache reads differently.
148
+ publisher: str | None = None
97
149
 
98
150
 
99
151
  @dataclass(frozen=True, slots=True)
@@ -167,7 +219,7 @@ def _price_tokens(
167
219
 
168
220
  billable_input = input_count
169
221
  deducted_input = 0
170
- if rules.get("input_includes_cache_read"):
222
+ if counts_cache_read_in_input(price.publisher, price.pricing_channel, rules):
171
223
  if cache_read_count > input_count:
172
224
  reasons.append("cache_read_exceeds_input")
173
225
  deducted_input = input_count
@@ -249,6 +249,7 @@ def parse_catalog(
249
249
  effective_from=effective_from,
250
250
  effective_to=effective_to,
251
251
  source_url=str(price["source_url"]).strip(),
252
+ publisher=publisher,
252
253
  )
253
254
  )
254
255
  return version, aliases, prices
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: metergraph-core
3
- Version: 0.2.23
3
+ Version: 0.2.24
4
4
  Summary: Reusable MeterGraph catalog and deterministic billing engine
5
5
  License-Expression: Apache-2.0
6
6
  Requires-Python: >=3.10