tariffkit 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. tariffkit/__init__.py +52 -0
  2. tariffkit/account/__init__.py +41 -0
  3. tariffkit/account/cli.py +493 -0
  4. tariffkit/account/errors.py +25 -0
  5. tariffkit/account/model.py +641 -0
  6. tariffkit/account/rates.py +61 -0
  7. tariffkit/account/repository.py +329 -0
  8. tariffkit/billing/__init__.py +56 -0
  9. tariffkit/billing/engine.py +505 -0
  10. tariffkit/billing/ledger.py +365 -0
  11. tariffkit/billing/models.py +274 -0
  12. tariffkit/billing/netting.py +137 -0
  13. tariffkit/billing/trueup.py +498 -0
  14. tariffkit/cca.py +122 -0
  15. tariffkit/cli.py +809 -0
  16. tariffkit/config.py +296 -0
  17. tariffkit/data/__init__.py +40 -0
  18. tariffkit/data/cca/mce/2023-01-01.toml +61 -0
  19. tariffkit/data/cca/mce/2026-04-01.toml +152 -0
  20. tariffkit/data/export/pge/acc_plus/2023-04-15.toml +43 -0
  21. tariffkit/data/export/pge/nbt00.json.gz +0 -0
  22. tariffkit/data/export/pge/nbt23.json.gz +0 -0
  23. tariffkit/data/export/pge/nbt24.json.gz +0 -0
  24. tariffkit/data/export/pge/nbt25.json.gz +0 -0
  25. tariffkit/data/export/pge/nbt26.json.gz +0 -0
  26. tariffkit/data/holidays.toml +36 -0
  27. tariffkit/data/manifest.json +56 -0
  28. tariffkit/data/nsc/pge.toml +57 -0
  29. tariffkit/data/tariff/pge/eelec/2025-01-01.toml +151 -0
  30. tariffkit/data/tariff/pge/eelec/2025-03-01.toml +150 -0
  31. tariffkit/data/tariff/pge/eelec/2025-09-01.toml +150 -0
  32. tariffkit/data/tariff/pge/eelec/2026-01-01.toml +153 -0
  33. tariffkit/data/tariff/pge/eelec/2026-03-01.toml +157 -0
  34. tariffkit/data/tariff/pge/etouc/2025-01-01.toml +222 -0
  35. tariffkit/data/tariff/pge/etouc/2025-03-01.toml +221 -0
  36. tariffkit/data/tariff/pge/etouc/2025-09-01.toml +221 -0
  37. tariffkit/data/tariff/pge/etouc/2026-01-01.toml +224 -0
  38. tariffkit/data/tariff/pge/etouc/2026-03-01.toml +231 -0
  39. tariffkit/data/tariff/pge/ev2a/2025-01-01.toml +144 -0
  40. tariffkit/data/tariff/pge/ev2a/2025-03-01.toml +143 -0
  41. tariffkit/data/tariff/pge/ev2a/2025-09-01.toml +143 -0
  42. tariffkit/data/tariff/pge/ev2a/2026-01-01.toml +146 -0
  43. tariffkit/data/tariff/pge/ev2a/2026-03-01.toml +153 -0
  44. tariffkit/data/tax/ca_energy_resources/2025-01-01.toml +27 -0
  45. tariffkit/data/tax/ca_energy_resources/2026-01-01.toml +27 -0
  46. tariffkit/data/versioned.py +118 -0
  47. tariffkit/engine.py +82 -0
  48. tariffkit/errors.py +19 -0
  49. tariffkit/export/__init__.py +5 -0
  50. tariffkit/export/nbt.py +207 -0
  51. tariffkit/interop/__init__.py +21 -0
  52. tariffkit/interop/emhass.py +79 -0
  53. tariffkit/interop/predbat.py +102 -0
  54. tariffkit/interop/slots.py +63 -0
  55. tariffkit/models.py +164 -0
  56. tariffkit/mqtt/__init__.py +6 -0
  57. tariffkit/mqtt/discovery.py +84 -0
  58. tariffkit/mqtt/publisher.py +305 -0
  59. tariffkit/providers/__init__.py +1 -0
  60. tariffkit/providers/pge/__init__.py +33 -0
  61. tariffkit/providers/pge/reconcile.py +828 -0
  62. tariffkit/providers/pge/statements/__init__.py +26 -0
  63. tariffkit/providers/pge/statements/errors.py +20 -0
  64. tariffkit/providers/pge/statements/model.py +320 -0
  65. tariffkit/providers/pge/statements/ocr.py +193 -0
  66. tariffkit/providers/pge/statements/parse.py +813 -0
  67. tariffkit/py.typed +0 -0
  68. tariffkit/secrets.py +164 -0
  69. tariffkit/sources/__init__.py +71 -0
  70. tariffkit/sources/greenbutton.py +318 -0
  71. tariffkit/sources/homeassistant.py +342 -0
  72. tariffkit/sources/influx.py +359 -0
  73. tariffkit/sources/pge.py +1153 -0
  74. tariffkit/tariff/__init__.py +5 -0
  75. tariffkit/tariff/retail.py +271 -0
  76. tariffkit/timeutil.py +120 -0
  77. tariffkit/web/__init__.py +5 -0
  78. tariffkit/web/app.py +220 -0
  79. tariffkit-0.2.0.dist-info/METADATA +260 -0
  80. tariffkit-0.2.0.dist-info/RECORD +83 -0
  81. tariffkit-0.2.0.dist-info/WHEEL +4 -0
  82. tariffkit-0.2.0.dist-info/entry_points.txt +2 -0
  83. tariffkit-0.2.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,359 @@
1
+ """Interval readings from raw meter counters in InfluxDB 3.
2
+
3
+ Home Assistant writes each numeric sensor sample to InfluxDB, so the same
4
+ Rainforest Eagle-100 counters land here as a plain time series of readings
5
+ rather than the pre-aggregated buckets :mod:`tariffkit.sources.homeassistant`
6
+ returns. Two consequences, and they point in opposite directions.
7
+
8
+ **Totals are exact.** Energy over a window is the counter's endpoints, so it
9
+ does not matter how densely the counter was sampled in between. Measured against
10
+ a real statement, this reproduced 39.902 kWh imported against 39.906 billed and
11
+ 193.795 exported against 193.797 -- closer than PG&E's own CSV export, which
12
+ rounds every interval to two decimals and loses about 2% of a low-import month.
13
+
14
+ **Distribution across intervals is only as good as the sampling.** A sample
15
+ reports an advance since the previous one, not an instant, so energy is spread
16
+ pro rata over the span it accrued across (see :func:`_per_interval` for why
17
+ crediting it to the later sample instead is materially wrong at TOU boundaries).
18
+ Spacing on this data has a median of five minutes but a 90th percentile of
19
+ three quarters of an hour, so hourly is the honest default; anything finer is
20
+ interpolation, and should be checked against the density for that period.
21
+
22
+ The default entities are the **unfiltered** counters, which is the opposite of
23
+ the Home Assistant source's default and deliberate: those go back fourteen
24
+ months against the filtered pair's five, and the drop-to-zero behaviour that
25
+ makes them unusable raw is repaired here anyway.
26
+ """
27
+
28
+ from __future__ import annotations
29
+
30
+ import re
31
+ import tomllib
32
+ from dataclasses import dataclass
33
+ from datetime import UTC, datetime, timedelta
34
+ from pathlib import Path
35
+ from typing import Any
36
+
37
+ from ..account.model import MeterSource
38
+ from ..billing.models import IntervalReading
39
+ from ..config import default_config_path
40
+ from ..errors import ConfigError, DataError
41
+ from ..secrets import get_secret
42
+ from ..timeutil import to_pacific
43
+ from .homeassistant import load_dotenv
44
+
45
+ #: The raw Eagle-100 counters. Unfiltered on purpose -- see the module docstring.
46
+ DEFAULT_IMPORT_ENTITY = "eagle_100_total_energy_delivered"
47
+ DEFAULT_EXPORT_ENTITY = "eagle_100_total_energy_received"
48
+
49
+ #: Home Assistant's InfluxDB integration writes one row per numeric sample.
50
+ DEFAULT_TABLE = "sensor_numeric"
51
+
52
+ #: How far before the window to look for a counter reading to subtract from.
53
+ #: Sampling has been as sparse as one reading every three hours, and without a
54
+ #: baseline the first interval would silently start from zero.
55
+ BASELINE_LOOKBACK = timedelta(days=3)
56
+
57
+ #: Entity ids and the table name are interpolated into SQL, so they are
58
+ #: constrained rather than quoted: anything outside this is rejected instead of
59
+ #: escaped.
60
+ _ENTITY_RE = re.compile(r"^[A-Za-z0-9_.]+$")
61
+
62
+
63
+ @dataclass(frozen=True, slots=True)
64
+ class InfluxSettings:
65
+ """Where to reach InfluxDB 3, and which series carry grid exchange."""
66
+
67
+ host: str
68
+ database: str
69
+ token: str
70
+ import_entity: str = DEFAULT_IMPORT_ENTITY
71
+ export_entity: str = DEFAULT_EXPORT_ENTITY
72
+ table: str = DEFAULT_TABLE
73
+
74
+ @property
75
+ def query_url(self) -> str:
76
+ base = self.host if "://" in self.host else f"https://{self.host}"
77
+ return f"{base.rstrip('/')}/api/v3/query_sql"
78
+
79
+ @classmethod
80
+ def load(
81
+ cls,
82
+ config_path: str | Path | None = None,
83
+ dotenv_path: str | Path = ".env",
84
+ profile_source: MeterSource | None = None,
85
+ **overrides: str | None,
86
+ ) -> InfluxSettings:
87
+ """Resolve from config file, ``.env``, environment, profile, then args.
88
+
89
+ Same split as the Home Assistant source: entity ids and the database are
90
+ configuration and may live in the config file; a profile mapping can
91
+ replace the grid-import/grid-export series; the token is a secret and
92
+ is read only from ``.env`` or the environment.
93
+ """
94
+ import os
95
+
96
+ values: dict[str, str] = {}
97
+ path = Path(config_path) if config_path else default_config_path()
98
+ if path.is_file():
99
+ table = tomllib.loads(path.read_text(encoding="utf-8")).get("influxdb", {})
100
+ for key in ("host", "database", "import_entity", "export_entity", "table"):
101
+ if key in table:
102
+ values[key] = str(table[key])
103
+
104
+ env = {**load_dotenv(dotenv_path), **os.environ}
105
+ for key, name in (
106
+ ("host", "INFLUXDB3_HOST"),
107
+ ("database", "INFLUXDB3_DATABASE"),
108
+ ("token", "INFLUXDB3_AUTH_TOKEN"),
109
+ ("import_entity", "TARIFFKIT_INFLUX_IMPORT_ENTITY"),
110
+ ("export_entity", "TARIFFKIT_INFLUX_EXPORT_ENTITY"),
111
+ ):
112
+ if value := env.get(name):
113
+ values[key] = value
114
+ if "token" not in values and (token := get_secret("influxdb.token")):
115
+ values["token"] = token
116
+ if profile_source is not None:
117
+ if not isinstance(profile_source, MeterSource):
118
+ raise ConfigError("profile_source must be a MeterSource")
119
+ values["import_entity"] = profile_source.grid_import_entity
120
+ values["export_entity"] = profile_source.grid_export_entity
121
+ values.update({k: v for k, v in overrides.items() if v})
122
+
123
+ missing = [k for k in ("host", "database", "token") if not values.get(k)]
124
+ if missing:
125
+ raise ConfigError(
126
+ f"InfluxDB {', '.join(missing)} not set; put INFLUXDB3_HOST, "
127
+ f"INFLUXDB3_DATABASE and INFLUXDB3_AUTH_TOKEN in "
128
+ f"{Path(dotenv_path)}, the environment, or store influxdb.token with "
129
+ f"`tariffkit credentials set`"
130
+ )
131
+ return cls(
132
+ host=values["host"],
133
+ database=values["database"],
134
+ token=values["token"],
135
+ import_entity=_clean_entity(values.get("import_entity", DEFAULT_IMPORT_ENTITY)),
136
+ export_entity=_clean_entity(values.get("export_entity", DEFAULT_EXPORT_ENTITY)),
137
+ table=_sql_name(values.get("table", DEFAULT_TABLE), "table name"),
138
+ )
139
+
140
+
141
+ def _clean_entity(entity: str) -> str:
142
+ """Accept either ``sensor.foo`` or ``foo``; InfluxDB stores the latter."""
143
+ name = entity.split(".", 1)[1] if entity.startswith("sensor.") else entity
144
+ return _sql_name(name, "entity id")
145
+
146
+
147
+ def _sql_name(name: str, what: str) -> str:
148
+ """Guard an identifier that will be interpolated into SQL.
149
+
150
+ Both the entity ids and the table name reach the query as text rather than
151
+ bound parameters, and both can come from a config file. Constrain rather
152
+ than escape: a name outside this alphabet is a mistake, not something to
153
+ quote around.
154
+ """
155
+ if not _ENTITY_RE.match(name):
156
+ raise ConfigError(f"unsupported {what} {name!r}")
157
+ return name
158
+
159
+
160
+ def monotonic(samples: list[tuple[datetime, float]]) -> list[tuple[datetime, float]]:
161
+ """Drop readings that cannot be a cumulative counter moving forward.
162
+
163
+ The Eagle-100 re-establishes its meter session several times a day and
164
+ publishes exactly ``0.0`` while it does -- about one sample in ten on this
165
+ data. A reading that is zero, negative, or lower than one already seen is a
166
+ device artefact, not energy, and differencing across it would invent a huge
167
+ interval and then a compensating hole.
168
+
169
+ This is the same rule the Home Assistant template filter applies, reproduced
170
+ here so the unfiltered series -- which reaches back nine months further --
171
+ can be used directly.
172
+ """
173
+ kept: list[tuple[datetime, float]] = []
174
+ highest: float | None = None
175
+ for moment, value in samples:
176
+ if value is None or value <= 0:
177
+ continue
178
+ if highest is not None and value < highest:
179
+ continue
180
+ highest = value
181
+ kept.append((moment, value))
182
+ return kept
183
+
184
+
185
+ def _query(settings: InfluxSettings, sql: str) -> list[dict[str, Any]]:
186
+ try:
187
+ import httpx
188
+ except ImportError as exc: # pragma: no cover - exercised by packaging
189
+ raise RuntimeError(
190
+ "the InfluxDB source requires the 'influx' extra: pip install 'tariffkit[influx]'"
191
+ ) from exc
192
+
193
+ response = httpx.post(
194
+ settings.query_url,
195
+ headers={"Authorization": f"Bearer {settings.token}", "Content-Type": "application/json"},
196
+ json={"db": settings.database, "q": sql, "format": "json"},
197
+ timeout=120,
198
+ )
199
+ if response.status_code != 200:
200
+ raise DataError(
201
+ f"InfluxDB refused the query ({response.status_code}): {response.text[:200]}"
202
+ )
203
+ payload = response.json()
204
+ if not isinstance(payload, list):
205
+ raise DataError(f"unexpected InfluxDB response: {str(payload)[:200]}")
206
+ return payload
207
+
208
+
209
+ def _samples(
210
+ settings: InfluxSettings, entity: str, start: datetime, end: datetime
211
+ ) -> list[tuple[datetime, float]]:
212
+ # Guard here as well as in load(): settings can be constructed directly, and
213
+ # nothing downstream would notice a name that is not an identifier.
214
+ sql = (
215
+ f"SELECT time, value FROM {_sql_name(settings.table, 'table name')} "
216
+ f"WHERE entity_id = '{_sql_name(entity, 'entity id')}' "
217
+ f"AND time >= '{start.astimezone(UTC):%Y-%m-%dT%H:%M:%SZ}' "
218
+ f"AND time <= '{end.astimezone(UTC):%Y-%m-%dT%H:%M:%SZ}' "
219
+ f"ORDER BY time"
220
+ )
221
+ rows = _query(settings, sql)
222
+ out: list[tuple[datetime, float]] = []
223
+ for row in rows:
224
+ raw = row.get("time")
225
+ value = row.get("value")
226
+ if raw is None or value is None:
227
+ continue
228
+ try:
229
+ moment = datetime.fromisoformat(str(raw))
230
+ reading = float(value)
231
+ except (TypeError, ValueError) as exc:
232
+ # A row we cannot read is a wire-format problem, not a programming
233
+ # error; say which row rather than letting a bare ValueError out.
234
+ raise DataError(
235
+ f"could not read a {entity} sample from InfluxDB "
236
+ f"(time={raw!r}, value={value!r}): {exc}"
237
+ ) from exc
238
+ # InfluxDB stores UTC and returns it without an offset, so a naive
239
+ # timestamp is UTC rather than local.
240
+ out.append((moment if moment.tzinfo else moment.replace(tzinfo=UTC), reading))
241
+ return out
242
+
243
+
244
+ #: A sample gap wider than this makes the *shape* of what it spans a guess.
245
+ #: Spreading it evenly is still the best available estimate of the total, which
246
+ #: a cumulative counter fixes exactly at the endpoints, but the split between
247
+ #: intervals stops being measured and starts being assumed. Chosen at one hour
248
+ #: because that is the granularity a time-of-use tariff prices at: a gap inside
249
+ #: one interval cannot move energy across a rate boundary, and a gap spanning
250
+ #: several can.
251
+ SMEARED_GAP = timedelta(hours=1)
252
+
253
+
254
+ def _per_interval(
255
+ samples: list[tuple[datetime, float]], start: datetime, end: datetime, step: timedelta
256
+ ) -> tuple[dict[datetime, float], set[datetime]]:
257
+ """Counter advance per interval, spread across the span it accrued over.
258
+
259
+ A sample says only that the counter advanced by some amount *since the
260
+ previous sample*, not that the energy arrived at the instant of reading.
261
+ Crediting the whole advance to the interval holding the later sample biases
262
+ energy forward across every boundary it spans, and boundaries are where the
263
+ money is: the export delivery credit is roughly 500x larger during the 4-9pm
264
+ peak than outside it. On a real July statement that rule put 55.52 kWh of
265
+ export in peak where PG&E's own 15-minute data has 52.08; spreading pro rata
266
+ gives 52.62. Median sample spacing is five minutes, but the 90th percentile
267
+ is three quarters of an hour, so the error is not marginal.
268
+
269
+ Advance that accrued before ``start`` is clipped rather than counted, so a
270
+ baseline sample reaching back days does not dump its whole span into the
271
+ first interval.
272
+
273
+ Intervals are stepped in absolute time so the two DST transitions produce
274
+ 23 and 25 hour days rather than being assumed 24.
275
+ """
276
+ edges: list[datetime] = []
277
+ cursor = start.astimezone(UTC)
278
+ limit = end.astimezone(UTC)
279
+ while cursor < limit:
280
+ edges.append(cursor)
281
+ cursor += step
282
+ if not edges:
283
+ return {}, set()
284
+
285
+ totals = dict.fromkeys(edges, 0.0)
286
+ smeared: set[datetime] = set()
287
+ previous: tuple[datetime, float] | None = None
288
+ for moment, value in samples:
289
+ instant = moment.astimezone(UTC)
290
+ if previous is None:
291
+ previous = (instant, value)
292
+ continue
293
+ was_at, was = previous
294
+ previous = (instant, value)
295
+ advance = value - was
296
+ span = (instant - was_at).total_seconds()
297
+ if advance <= 0 or span <= 0 or instant <= edges[0]:
298
+ continue
299
+ # Walk only the intervals the advance actually touches. Indices are
300
+ # arithmetic because intervals are uniform in absolute time.
301
+ lower = max(was_at, edges[0])
302
+ while lower < instant:
303
+ index = int((lower - edges[0]) // step)
304
+ if index >= len(edges):
305
+ break
306
+ upper = min(instant, edges[index] + step)
307
+ totals[edges[index]] += advance * (upper - lower).total_seconds() / span
308
+ if instant - was_at > SMEARED_GAP:
309
+ smeared.add(edges[index])
310
+ lower = upper
311
+ return totals, smeared
312
+
313
+
314
+ def read_counters(
315
+ settings: InfluxSettings,
316
+ start: datetime,
317
+ end: datetime,
318
+ resolution: timedelta = timedelta(hours=1),
319
+ ) -> list[IntervalReading]:
320
+ """Interval readings for ``[start, end)`` from raw counter samples.
321
+
322
+ Totals over the window are exact regardless of sampling density, because a
323
+ cumulative counter only depends on its endpoints. How that energy divides
324
+ between intervals does depend on density; see the module docstring.
325
+ """
326
+ for name, moment in (("start", start), ("end", end)):
327
+ if moment.tzinfo is None:
328
+ raise ConfigError(f"{name} must be timezone-aware; got {moment.isoformat()}")
329
+ if end <= start:
330
+ raise ConfigError(f"end {end.isoformat()} is not after start {start.isoformat()}")
331
+ if resolution <= timedelta(0):
332
+ raise ConfigError(f"resolution must be positive, got {resolution}")
333
+
334
+ # Reach back before the window so the first interval has something to
335
+ # subtract from; otherwise it would silently start from zero.
336
+ lookback = start - BASELINE_LOOKBACK
337
+ import_samples = monotonic(_samples(settings, settings.import_entity, lookback, end))
338
+ export_samples = monotonic(_samples(settings, settings.export_entity, lookback, end))
339
+ # Test the samples, not the bucketed result: bucketing always yields an
340
+ # entry per interval, so an empty window is indistinguishable from a quiet
341
+ # one once it has been through _per_interval.
342
+ if not import_samples and not export_samples:
343
+ raise DataError(
344
+ f"no samples for {settings.import_entity} / {settings.export_entity} "
345
+ f"between {start.isoformat()} and {end.isoformat()}"
346
+ )
347
+ imported, smeared_in = _per_interval(import_samples, start, end, resolution)
348
+ exported, smeared_out = _per_interval(export_samples, start, end, resolution)
349
+ smeared = smeared_in | smeared_out
350
+ return [
351
+ IntervalReading(
352
+ start=to_pacific(edge),
353
+ imported=max(imported.get(edge, 0.0), 0.0),
354
+ exported=max(exported.get(edge, 0.0), 0.0),
355
+ duration=resolution,
356
+ estimated=edge in smeared,
357
+ )
358
+ for edge in sorted(set(imported) | set(exported))
359
+ ]