llmcachex-core 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
llmcachex/__init__.py ADDED
@@ -0,0 +1,7 @@
1
+ """LLMCacheX - Lightweight LLM/API request optimization library."""
2
+
3
+ from .decorator import cached_call
4
+
5
+ __version__ = "0.1.0"
6
+
7
+ __all__ = ["cached_call", "__version__"]
@@ -0,0 +1,61 @@
1
+ """Runtime analytics: events, usage metadata, cost estimation, storage.
2
+
3
+ This package is provider-independent. It records structured request
4
+ events (cache hits/misses, retries, rate-limit waits), aggregates them
5
+ for the dashboard API, and represents provider-supplied usage and
6
+ configurable cost estimates — never fabricating tokens or prices.
7
+
8
+ Typical usage::
9
+
10
+ from llmcachex.analytics import AnalyticsStore, RequestEvent, EventType
11
+
12
+ store = AnalyticsStore("llmcachex.db")
13
+ store.record(RequestEvent(
14
+ event_type=EventType.CACHE_MISS,
15
+ function_name="app.generate",
16
+ request_hash="abc...",
17
+ cache_status="miss",
18
+ success=True,
19
+ ))
20
+ """
21
+
22
+ from .cost import PricingConfig, coerce_decimal, estimate_cost
23
+ from .events import (
24
+ EVENT_TYPES,
25
+ HIT_TYPES,
26
+ REQUEST_EVENT_TYPES,
27
+ EventType,
28
+ HitType,
29
+ RequestEvent,
30
+ )
31
+ from .metrics import (
32
+ DEFAULT_MAX_EVENTS,
33
+ AnalyticsStore,
34
+ AnalyticsSummary,
35
+ PriorExecution,
36
+ ProviderAnalytics,
37
+ TimeSeriesPoint,
38
+ UsageTotals,
39
+ )
40
+ from .usage import Usage, coerce_usage
41
+
42
+ __all__ = [
43
+ "DEFAULT_MAX_EVENTS",
44
+ "EVENT_TYPES",
45
+ "HIT_TYPES",
46
+ "REQUEST_EVENT_TYPES",
47
+ "AnalyticsStore",
48
+ "AnalyticsSummary",
49
+ "EventType",
50
+ "HitType",
51
+ "PriorExecution",
52
+ "PricingConfig",
53
+ "ProviderAnalytics",
54
+ "RequestEvent",
55
+ "TimeSeriesPoint",
56
+ "Usage",
57
+ "UsageTotals",
58
+ "coerce_decimal",
59
+ "coerce_usage",
60
+ "estimate_cost",
61
+ ]
@@ -0,0 +1,158 @@
1
+ """Configurable cost estimation (no hardcoded provider pricing).
2
+
3
+ Prices are *data*, supplied by the developer or a provider adapter:
4
+ LLMCacheX never ships or claims current real-world pricing. All money
5
+ arithmetic uses :class:`decimal.Decimal` to avoid binary floating-point
6
+ rounding in cost-style calculations.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from collections.abc import Mapping
12
+ from dataclasses import dataclass
13
+ from decimal import Decimal, InvalidOperation
14
+ from typing import Any
15
+
16
+ from .usage import Usage
17
+
18
+ __all__ = ["PricingConfig", "estimate_cost", "coerce_decimal"]
19
+
20
+ _TOKENS_PER_1K = Decimal(1000)
21
+
22
+
23
+ def coerce_decimal(value: Any, *, field: str = "value") -> Decimal:
24
+ """Convert ``int | float | str | Decimal`` to an exact ``Decimal``.
25
+
26
+ Floats are routed through ``str()`` first so that e.g. ``0.0025``
27
+ becomes ``Decimal("0.0025")`` rather than the binary expansion of
28
+ the float.
29
+ """
30
+ if isinstance(value, Decimal):
31
+ return value
32
+ if isinstance(value, bool) or not isinstance(value, (int, float, str)):
33
+ raise ValueError(f"{field} must be a number, got {value!r}.")
34
+ try:
35
+ return Decimal(str(value))
36
+ except InvalidOperation as exc:
37
+ raise ValueError(f"{field} is not a valid number: {value!r}.") from exc
38
+
39
+
40
+ def _check_price(name: str, value: Decimal | None) -> Decimal | None:
41
+ if value is None:
42
+ return None
43
+ if value < 0:
44
+ raise ValueError(f"{name} must be >= 0, got {value}.")
45
+ return value
46
+
47
+
48
+ @dataclass(frozen=True)
49
+ class PricingConfig:
50
+ """Prices per 1000 tokens as configured data.
51
+
52
+ ``None`` means "price unknown". The configuration is intentionally
53
+ explicit: no default prices are bundled with the library, so an
54
+ estimate is only produced when the developer supplied pricing.
55
+
56
+ Args:
57
+ input_cost_per_1k_tokens: Price for 1000 input tokens, or None.
58
+ output_cost_per_1k_tokens: Price for 1000 output tokens, or None.
59
+ currency: Optional label (e.g. ``"USD"``) for display only.
60
+ """
61
+
62
+ input_cost_per_1k_tokens: Decimal | None = None
63
+ output_cost_per_1k_tokens: Decimal | None = None
64
+ currency: str | None = None
65
+
66
+ def __post_init__(self) -> None:
67
+ _check_price("input_cost_per_1k_tokens", self.input_cost_per_1k_tokens)
68
+ _check_price("output_cost_per_1k_tokens", self.output_cost_per_1k_tokens)
69
+
70
+ @classmethod
71
+ def from_values(
72
+ cls,
73
+ input_cost_per_1k_tokens: Any = None,
74
+ output_cost_per_1k_tokens: Any = None,
75
+ currency: str | None = None,
76
+ ) -> PricingConfig:
77
+ """Build from plain numbers/strings (converted to Decimal safely)."""
78
+ return cls(
79
+ input_cost_per_1k_tokens=(
80
+ None
81
+ if input_cost_per_1k_tokens is None
82
+ else coerce_decimal(
83
+ input_cost_per_1k_tokens,
84
+ field="input_cost_per_1k_tokens",
85
+ )
86
+ ),
87
+ output_cost_per_1k_tokens=(
88
+ None
89
+ if output_cost_per_1k_tokens is None
90
+ else coerce_decimal(
91
+ output_cost_per_1k_tokens,
92
+ field="output_cost_per_1k_tokens",
93
+ )
94
+ ),
95
+ currency=currency,
96
+ )
97
+
98
+ def estimate(self, usage: Usage | None) -> Decimal | None:
99
+ """Estimate cost for ``usage``; ``None`` when it cannot be derived.
100
+
101
+ Each side (input/output) contributes only when both its price and
102
+ its token count are known. If no side is computable the result is
103
+ ``None`` — never zero — so "unknown" stays distinguishable from
104
+ "free".
105
+ """
106
+ if usage is None:
107
+ return None
108
+ parts: list[Decimal] = []
109
+ if (
110
+ self.input_cost_per_1k_tokens is not None
111
+ and usage.input_tokens is not None
112
+ ):
113
+ parts.append(
114
+ Decimal(usage.input_tokens)
115
+ / _TOKENS_PER_1K
116
+ * self.input_cost_per_1k_tokens
117
+ )
118
+ if (
119
+ self.output_cost_per_1k_tokens is not None
120
+ and usage.output_tokens is not None
121
+ ):
122
+ parts.append(
123
+ Decimal(usage.output_tokens)
124
+ / _TOKENS_PER_1K
125
+ * self.output_cost_per_1k_tokens
126
+ )
127
+ if not parts:
128
+ return None
129
+ return sum(parts, Decimal(0))
130
+
131
+
132
+ def estimate_cost(
133
+ usage: Usage | None, pricing: PricingConfig | Mapping[str, Any] | None
134
+ ) -> Decimal | None:
135
+ """Estimate cost from usage + pricing, tolerating unknown inputs."""
136
+ if pricing is None:
137
+ return None
138
+ if isinstance(pricing, Mapping):
139
+ pricing = PricingConfig(
140
+ input_cost_per_1k_tokens=(
141
+ None
142
+ if pricing.get("input_cost_per_1k_tokens") is None
143
+ else coerce_decimal(
144
+ pricing["input_cost_per_1k_tokens"],
145
+ field="input_cost_per_1k_tokens",
146
+ )
147
+ ),
148
+ output_cost_per_1k_tokens=(
149
+ None
150
+ if pricing.get("output_cost_per_1k_tokens") is None
151
+ else coerce_decimal(
152
+ pricing["output_cost_per_1k_tokens"],
153
+ field="output_cost_per_1k_tokens",
154
+ )
155
+ ),
156
+ currency=pricing.get("currency"),
157
+ )
158
+ return pricing.estimate(usage)
@@ -0,0 +1,140 @@
1
+ """Structured request-activity events (metadata only).
2
+
3
+ One logical call produces at most one *request-level* event
4
+ (``cache_hit`` or ``cache_miss`` — the miss also carries the execution
5
+ outcome), plus one event per discrete ``retry`` attempt and per
6
+ ``rate_limit_wait``. The vocabulary is a small extensible set; unknown
7
+ event types are rejected when recording.
8
+
9
+ Provider/usage/cost fields are ``None`` until real provider metadata
10
+ exists. Nothing here ever stores prompts, responses or API keys.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ from dataclasses import dataclass, field
16
+ from datetime import datetime, timezone
17
+ from decimal import Decimal
18
+ from typing import Final
19
+
20
+ __all__ = [
21
+ "EventType",
22
+ "EVENT_TYPES",
23
+ "REQUEST_EVENT_TYPES",
24
+ "HitType",
25
+ "HIT_TYPES",
26
+ "RequestEvent",
27
+ "utc_now_iso",
28
+ ]
29
+
30
+
31
+ class EventType:
32
+ """Event vocabulary (plain string constants for easy storage)."""
33
+
34
+ CACHE_HIT: Final = "cache_hit"
35
+ CACHE_MISS: Final = "cache_miss"
36
+ RETRY: Final = "retry"
37
+ RATE_LIMIT_WAIT: Final = "rate_limit_wait"
38
+
39
+
40
+ class HitType:
41
+ """How a cache hit was produced (exact hash vs semantic similarity)."""
42
+
43
+ EXACT: Final = "exact"
44
+ SEMANTIC: Final = "semantic"
45
+
46
+
47
+ #: Valid ``hit_type`` values for hit events (``None`` elsewhere).
48
+ HIT_TYPES: Final[frozenset[str]] = frozenset(
49
+ {
50
+ HitType.EXACT,
51
+ HitType.SEMANTIC,
52
+ }
53
+ )
54
+
55
+
56
+ #: All currently valid event types. Extensible by adding a constant here.
57
+ EVENT_TYPES: Final[frozenset[str]] = frozenset(
58
+ {
59
+ EventType.CACHE_HIT,
60
+ EventType.CACHE_MISS,
61
+ EventType.RETRY,
62
+ EventType.RATE_LIMIT_WAIT,
63
+ }
64
+ )
65
+
66
+ #: Event types that count as one logical request.
67
+ REQUEST_EVENT_TYPES: Final[frozenset[str]] = frozenset(
68
+ {EventType.CACHE_HIT, EventType.CACHE_MISS}
69
+ )
70
+
71
+
72
+ def utc_now_iso() -> str:
73
+ """Current UTC time as a sortable ISO-8601 string."""
74
+ return datetime.now(timezone.utc).isoformat()
75
+
76
+
77
+ @dataclass
78
+ class RequestEvent:
79
+ """One analytics record.
80
+
81
+ Args:
82
+ event_type: One of :data:`EVENT_TYPES`.
83
+ function_name: Dotted function label (``module.qualname``).
84
+ request_hash: Deterministic request hash (no prompt content).
85
+ timestamp: ISO-8601 UTC timestamp.
86
+ cache_status: ``"hit"``, ``"miss"``, or ``None`` for events that
87
+ are not request-level (retry, rate-limit wait).
88
+ latency_ms: Measured latency (hit: lookup, miss: execution).
89
+ retry_count: Retries observed during this request.
90
+ success: For a miss: did the execution succeed. ``True`` for
91
+ hits, ``None`` where not applicable.
92
+ provider: Provider name if known, else ``None``.
93
+ model: Model name if known, else ``None``.
94
+ input_tokens / output_tokens / total_tokens: Provider-reported
95
+ usage, or ``None`` when unknown.
96
+ request_units: Optional provider-specific billing units.
97
+ estimated_cost: Cost of executing this request (miss only), or
98
+ ``None`` when usage/pricing are unavailable.
99
+ estimated_cost_saved: For a hit: the previously known execution
100
+ cost that was avoided, or ``None`` when unknown.
101
+ hit_type: For hit events: ``"exact"`` or ``"semantic"``;
102
+ ``None`` for misses and non-request events.
103
+ event_id: Assigned by the store after persistence.
104
+ """
105
+
106
+ event_type: str
107
+ function_name: str
108
+ request_hash: str
109
+ timestamp: str = field(default_factory=utc_now_iso)
110
+ cache_status: str | None = None
111
+ latency_ms: float | None = None
112
+ retry_count: int = 0
113
+ success: bool | None = None
114
+ provider: str | None = None
115
+ model: str | None = None
116
+ input_tokens: int | None = None
117
+ output_tokens: int | None = None
118
+ total_tokens: int | None = None
119
+ request_units: int | None = None
120
+ estimated_cost: Decimal | None = None
121
+ estimated_cost_saved: Decimal | None = None
122
+ hit_type: str | None = None
123
+ event_id: int | None = None
124
+
125
+ def __post_init__(self) -> None:
126
+ if self.event_type not in EVENT_TYPES:
127
+ raise ValueError(
128
+ f"Unknown event type {self.event_type!r}; "
129
+ f"expected one of {sorted(EVENT_TYPES)}."
130
+ )
131
+ if self.cache_status not in (None, "hit", "miss"):
132
+ raise ValueError(
133
+ f"cache_status must be 'hit', 'miss' or None, "
134
+ f"got {self.cache_status!r}."
135
+ )
136
+ if self.hit_type is not None and self.hit_type not in HIT_TYPES:
137
+ raise ValueError(
138
+ f"hit_type must be 'exact', 'semantic' or None, "
139
+ f"got {self.hit_type!r}."
140
+ )