cloudsealed-jit 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cloudsealed_jit/__init__.py +24 -0
- cloudsealed_jit/analysis.py +423 -0
- cloudsealed_jit/api.py +165 -0
- cloudsealed_jit/cli.py +57 -0
- cloudsealed_jit/kernels.py +140 -0
- cloudsealed_jit/parsing.py +321 -0
- cloudsealed_jit-0.2.0.dist-info/METADATA +222 -0
- cloudsealed_jit-0.2.0.dist-info/RECORD +12 -0
- cloudsealed_jit-0.2.0.dist-info/WHEEL +5 -0
- cloudsealed_jit-0.2.0.dist-info/entry_points.txt +2 -0
- cloudsealed_jit-0.2.0.dist-info/licenses/LICENSE +21 -0
- cloudsealed_jit-0.2.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""
|
|
2
|
+
cloudsealed-jit — cloud billing waste analysis.
|
|
3
|
+
|
|
4
|
+
Detects structural cost waste in cloud billing exports by modelling a robust
|
|
5
|
+
day-of-week aware baseline and measuring the excess spend above it.
|
|
6
|
+
|
|
7
|
+
Public API:
|
|
8
|
+
parse_billing_csv(csv_text) -> BillingSeries
|
|
9
|
+
analyze(series, analysis_type="waste-audit") -> AnalysisResult
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from .parsing import BillingSeries, ParseError, parse_billing_csv
|
|
13
|
+
from .analysis import AnalysisResult, analyze
|
|
14
|
+
|
|
15
|
+
__version__ = "0.2.0"
|
|
16
|
+
|
|
17
|
+
__all__ = [
|
|
18
|
+
"BillingSeries",
|
|
19
|
+
"ParseError",
|
|
20
|
+
"parse_billing_csv",
|
|
21
|
+
"AnalysisResult",
|
|
22
|
+
"analyze",
|
|
23
|
+
"__version__",
|
|
24
|
+
]
|
|
@@ -0,0 +1,423 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Waste analysis over a daily billing series.
|
|
3
|
+
|
|
4
|
+
Method
|
|
5
|
+
------
|
|
6
|
+
|
|
7
|
+
**Baseline.** Expected spend for a day is the product of a level term and a
|
|
8
|
+
weekday term:
|
|
9
|
+
|
|
10
|
+
expected[i] = rolling_median(cost, 7)[i] * dow_factor[weekday(i)]
|
|
11
|
+
|
|
12
|
+
The rolling median tracks growth and step changes without being dragged by
|
|
13
|
+
spikes. The weekday factor is the median ratio of observed spend to the level
|
|
14
|
+
term for that weekday, which captures the weekday/weekend cycle that dominates
|
|
15
|
+
most cloud bills. It is only estimated when at least two full weeks are
|
|
16
|
+
available; below that the factor is 1.0 for every day.
|
|
17
|
+
|
|
18
|
+
**Anomalies.** Residuals against the baseline are scored with a modified
|
|
19
|
+
z-score (median absolute deviation, see :mod:`cloudsealed_jit.kernels`). Days
|
|
20
|
+
scoring at or above ``Z_THRESHOLD`` are reported. The threshold of 3.5 is the
|
|
21
|
+
value recommended by Iglewicz & Hoaglin for the modified z-score.
|
|
22
|
+
|
|
23
|
+
**Waste.** Only *positive* excess counts as waste: spending less than expected
|
|
24
|
+
is not an opportunity. Waste percentage is the share of total spend that sits
|
|
25
|
+
above the baseline on anomalous days, which makes it directly convertible to
|
|
26
|
+
currency rather than being a count of unusual days.
|
|
27
|
+
|
|
28
|
+
**Recommendations.** Every recommendation carries a saving derived from the
|
|
29
|
+
series itself, normalised to 30 days, and states the assumption behind it in
|
|
30
|
+
its description. Estimates that depend on facts the analyser cannot observe --
|
|
31
|
+
whether a workload is production, whether a commitment is acceptable -- are
|
|
32
|
+
labelled as conditional rather than presented as findings.
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
from __future__ import annotations
|
|
36
|
+
|
|
37
|
+
import statistics
|
|
38
|
+
from dataclasses import asdict, dataclass, field
|
|
39
|
+
from datetime import date
|
|
40
|
+
from typing import Literal
|
|
41
|
+
|
|
42
|
+
import numpy as np
|
|
43
|
+
|
|
44
|
+
from .kernels import modified_zscores, rolling_median
|
|
45
|
+
from .parsing import BillingSeries
|
|
46
|
+
|
|
47
|
+
__all__ = [
|
|
48
|
+
"Anomaly",
|
|
49
|
+
"Metrics",
|
|
50
|
+
"Recommendation",
|
|
51
|
+
"AnalysisResult",
|
|
52
|
+
"analyze",
|
|
53
|
+
"Z_THRESHOLD",
|
|
54
|
+
]
|
|
55
|
+
|
|
56
|
+
AnalysisType = Literal["waste-audit", "cost-forecast", "efficiency"]
|
|
57
|
+
|
|
58
|
+
#: Modified z-score at which a day is reported as anomalous.
|
|
59
|
+
Z_THRESHOLD = 3.5
|
|
60
|
+
|
|
61
|
+
#: Minimum days required before the weekday seasonality term is estimated.
|
|
62
|
+
MIN_DAYS_FOR_SEASONALITY = 14
|
|
63
|
+
|
|
64
|
+
#: Assumed discount for a one-year commitment against the sustained baseline.
|
|
65
|
+
#: Conservative relative to published AWS/GCP/Azure commitment discounts.
|
|
66
|
+
COMMITMENT_DISCOUNT = 0.25
|
|
67
|
+
|
|
68
|
+
#: A service whose weekend spend is at least this fraction of its weekday
|
|
69
|
+
#: spend is running through the weekend at full cost.
|
|
70
|
+
WEEKEND_RATIO_FLAG = 0.80
|
|
71
|
+
|
|
72
|
+
# Severity combines statistical strength with financial size. Either alone is
|
|
73
|
+
# misleading: a very tight baseline turns a 5x spike into a moderate z-score,
|
|
74
|
+
# while a noisy series can produce a large z-score over a trivial amount.
|
|
75
|
+
# Each tier is (min |z|, min |deviation %|, label) and matches on either.
|
|
76
|
+
_SEVERITY_TIERS = (
|
|
77
|
+
(10.0, 200.0, "CRITICAL"),
|
|
78
|
+
(7.0, 100.0, "HIGH"),
|
|
79
|
+
(5.0, 25.0, "MEDIUM"),
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
@dataclass
|
|
84
|
+
class Anomaly:
|
|
85
|
+
date: str
|
|
86
|
+
expectedCost: float
|
|
87
|
+
actualCost: float
|
|
88
|
+
deviation: float # percent against expected
|
|
89
|
+
zScore: float
|
|
90
|
+
severity: str
|
|
91
|
+
description: str
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
@dataclass
|
|
95
|
+
class Metrics:
|
|
96
|
+
averageDailyCost: float
|
|
97
|
+
stdDeviation: float
|
|
98
|
+
sharpeRatio: float # spend stability: mean / stddev of daily cost
|
|
99
|
+
wastePercentage: float
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
@dataclass
|
|
103
|
+
class Recommendation:
|
|
104
|
+
title: str
|
|
105
|
+
description: str
|
|
106
|
+
potentialSavings: float # per 30 days, in the export's currency
|
|
107
|
+
effort: str
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
@dataclass
|
|
111
|
+
class AnalysisResult:
|
|
112
|
+
anomalies: list[Anomaly] = field(default_factory=list)
|
|
113
|
+
metrics: Metrics = field(
|
|
114
|
+
default_factory=lambda: Metrics(0.0, 0.0, 0.0, 0.0)
|
|
115
|
+
)
|
|
116
|
+
recommendations: list[Recommendation] = field(default_factory=list)
|
|
117
|
+
summary: str = ""
|
|
118
|
+
|
|
119
|
+
def to_dict(self) -> dict:
|
|
120
|
+
return asdict(self)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
# --------------------------------------------------------------------------
|
|
124
|
+
# Baseline
|
|
125
|
+
# --------------------------------------------------------------------------
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _weekday_factors(costs: np.ndarray, level: np.ndarray,
|
|
129
|
+
days: list[date]) -> np.ndarray:
|
|
130
|
+
"""Median ratio of observed spend to the level term, per weekday."""
|
|
131
|
+
factors = np.ones(7, dtype=np.float64)
|
|
132
|
+
if len(days) < MIN_DAYS_FOR_SEASONALITY:
|
|
133
|
+
return factors
|
|
134
|
+
|
|
135
|
+
buckets: dict[int, list[float]] = {i: [] for i in range(7)}
|
|
136
|
+
for i, day in enumerate(days):
|
|
137
|
+
if level[i] > 0:
|
|
138
|
+
buckets[day.weekday()].append(costs[i] / level[i])
|
|
139
|
+
|
|
140
|
+
for weekday, ratios in buckets.items():
|
|
141
|
+
# Two observations is not enough to separate a pattern from noise.
|
|
142
|
+
if len(ratios) >= 2:
|
|
143
|
+
factors[weekday] = float(statistics.median(ratios))
|
|
144
|
+
|
|
145
|
+
# Renormalise so the factors do not shift the overall level.
|
|
146
|
+
mean_factor = float(np.mean(factors))
|
|
147
|
+
if mean_factor > 0:
|
|
148
|
+
factors /= mean_factor
|
|
149
|
+
return factors
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _baseline(costs: np.ndarray, days: list[date]) -> np.ndarray:
|
|
153
|
+
level = rolling_median(costs, window=7)
|
|
154
|
+
factors = _weekday_factors(costs, level, days)
|
|
155
|
+
expected = np.array(
|
|
156
|
+
[level[i] * factors[day.weekday()] for i, day in enumerate(days)],
|
|
157
|
+
dtype=np.float64,
|
|
158
|
+
)
|
|
159
|
+
return np.maximum(expected, 0.0)
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def _severity(z: float, deviation_pct: float) -> str:
|
|
163
|
+
magnitude = abs(z)
|
|
164
|
+
deviation = abs(deviation_pct)
|
|
165
|
+
for min_z, min_deviation, label in _SEVERITY_TIERS:
|
|
166
|
+
if magnitude >= min_z or deviation >= min_deviation:
|
|
167
|
+
return label
|
|
168
|
+
return "LOW"
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
# --------------------------------------------------------------------------
|
|
172
|
+
# Recommendations
|
|
173
|
+
# --------------------------------------------------------------------------
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _weekend_recommendations(series: BillingSeries,
|
|
177
|
+
currency: str) -> list[Recommendation]:
|
|
178
|
+
"""Flag services that cost the same on weekends as on weekdays."""
|
|
179
|
+
out: list[Recommendation] = []
|
|
180
|
+
days = series.days
|
|
181
|
+
if len(days) < MIN_DAYS_FOR_SEASONALITY:
|
|
182
|
+
return out
|
|
183
|
+
|
|
184
|
+
weekend_idx = [i for i, d in enumerate(days) if d.weekday() >= 5]
|
|
185
|
+
weekday_idx = [i for i, d in enumerate(days) if d.weekday() < 5]
|
|
186
|
+
if len(weekend_idx) < 4 or len(weekday_idx) < 8:
|
|
187
|
+
return out
|
|
188
|
+
|
|
189
|
+
span = len(days)
|
|
190
|
+
for name, values in series.by_service.items():
|
|
191
|
+
weekend = statistics.median(values[i] for i in weekend_idx)
|
|
192
|
+
weekday = statistics.median(values[i] for i in weekday_idx)
|
|
193
|
+
if weekday <= 0:
|
|
194
|
+
continue
|
|
195
|
+
ratio = weekend / weekday
|
|
196
|
+
if ratio < WEEKEND_RATIO_FLAG:
|
|
197
|
+
continue
|
|
198
|
+
|
|
199
|
+
service_total = sum(values)
|
|
200
|
+
if service_total <= 0:
|
|
201
|
+
continue
|
|
202
|
+
|
|
203
|
+
# Upper bound: the weekend spend itself, normalised to 30 days.
|
|
204
|
+
weekend_spend = sum(values[i] for i in weekend_idx)
|
|
205
|
+
monthly = weekend_spend / span * 30
|
|
206
|
+
if monthly < 1.0:
|
|
207
|
+
continue
|
|
208
|
+
|
|
209
|
+
out.append(
|
|
210
|
+
Recommendation(
|
|
211
|
+
title=f"Idle weekend spend in {name}",
|
|
212
|
+
description=(
|
|
213
|
+
f"{name} costs {ratio:.0%} as much on weekends as on weekdays "
|
|
214
|
+
f"(weekday median {currency} {weekday:,.2f}/day, weekend median "
|
|
215
|
+
f"{currency} {weekend:,.2f}/day). If this workload is not "
|
|
216
|
+
f"production, stopping it outside business days removes the "
|
|
217
|
+
f"weekend spend entirely. The figure is the observed weekend "
|
|
218
|
+
f"spend normalised to 30 days and assumes the workload can be "
|
|
219
|
+
f"stopped; verify before acting."
|
|
220
|
+
),
|
|
221
|
+
potentialSavings=round(monthly, 2),
|
|
222
|
+
effort="LOW",
|
|
223
|
+
)
|
|
224
|
+
)
|
|
225
|
+
return out
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def _commitment_recommendation(costs: np.ndarray,
|
|
229
|
+
currency: str) -> Recommendation | None:
|
|
230
|
+
"""Estimate the saving available on the always-on portion of spend."""
|
|
231
|
+
if costs.size < MIN_DAYS_FOR_SEASONALITY:
|
|
232
|
+
return None
|
|
233
|
+
floor = float(np.percentile(costs, 10))
|
|
234
|
+
if floor <= 0:
|
|
235
|
+
return None
|
|
236
|
+
|
|
237
|
+
monthly = floor * 30 * COMMITMENT_DISCOUNT
|
|
238
|
+
if monthly < 1.0:
|
|
239
|
+
return None
|
|
240
|
+
|
|
241
|
+
return Recommendation(
|
|
242
|
+
title="Sustained baseline eligible for commitment pricing",
|
|
243
|
+
description=(
|
|
244
|
+
f"Daily spend never fell below {currency} {floor:,.2f} (10th percentile) "
|
|
245
|
+
f"over the period, so that portion is always-on. Committed-use discounts "
|
|
246
|
+
f"or savings plans typically price this tier {COMMITMENT_DISCOUNT:.0%} "
|
|
247
|
+
f"below on-demand. The figure applies that rate to the observed floor "
|
|
248
|
+
f"over 30 days and assumes a commitment is commercially acceptable."
|
|
249
|
+
),
|
|
250
|
+
potentialSavings=round(monthly, 2),
|
|
251
|
+
effort="MEDIUM",
|
|
252
|
+
)
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def _excess_recommendation(excess_total: float, span: int,
|
|
256
|
+
currency: str, count: int) -> Recommendation | None:
|
|
257
|
+
if excess_total <= 0 or count == 0:
|
|
258
|
+
return None
|
|
259
|
+
monthly = excess_total / span * 30
|
|
260
|
+
if monthly < 1.0:
|
|
261
|
+
return None
|
|
262
|
+
return Recommendation(
|
|
263
|
+
title="Investigate spend above the modelled baseline",
|
|
264
|
+
description=(
|
|
265
|
+
f"{count} day(s) spent {currency} {excess_total:,.2f} more than the "
|
|
266
|
+
f"day-of-week baseline predicted. This is unplanned spend rather than "
|
|
267
|
+
f"growth: the baseline already tracks trend and weekly seasonality. "
|
|
268
|
+
f"The figure is that excess normalised to 30 days."
|
|
269
|
+
),
|
|
270
|
+
potentialSavings=round(monthly, 2),
|
|
271
|
+
effort="MEDIUM",
|
|
272
|
+
)
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def _concentration_recommendation(series: BillingSeries,
|
|
276
|
+
currency: str) -> Recommendation | None:
|
|
277
|
+
total = series.total_cost
|
|
278
|
+
if total <= 0 or not series.by_service:
|
|
279
|
+
return None
|
|
280
|
+
name, values = max(series.by_service.items(), key=lambda kv: sum(kv[1]))
|
|
281
|
+
share = sum(values) / total
|
|
282
|
+
if share < 0.40:
|
|
283
|
+
return None
|
|
284
|
+
return Recommendation(
|
|
285
|
+
title=f"Spend concentrated in {name}",
|
|
286
|
+
description=(
|
|
287
|
+
f"{name} accounts for {share:.0%} of total spend ({currency} "
|
|
288
|
+
f"{sum(values):,.2f} of {currency} {total:,.2f}). Optimisation effort "
|
|
289
|
+
f"applied here has the highest leverage; a 10% reduction on this "
|
|
290
|
+
f"service alone yields the figure shown."
|
|
291
|
+
),
|
|
292
|
+
potentialSavings=round(sum(values) / max(len(series.days), 1) * 30 * 0.10, 2),
|
|
293
|
+
effort="MEDIUM",
|
|
294
|
+
)
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
def _volatility_recommendation(stability: float, mean: float,
|
|
298
|
+
std: float, currency: str) -> Recommendation | None:
|
|
299
|
+
if stability >= 2.0 or mean <= 0:
|
|
300
|
+
return None
|
|
301
|
+
return Recommendation(
|
|
302
|
+
title="Daily spend is highly volatile",
|
|
303
|
+
description=(
|
|
304
|
+
f"Daily cost varies with a standard deviation of {currency} {std:,.2f} "
|
|
305
|
+
f"against a mean of {currency} {mean:,.2f} (stability ratio "
|
|
306
|
+
f"{stability:.2f}). Spend this irregular cannot be budgeted or alerted "
|
|
307
|
+
f"on reliably. Establishing budget alerts and per-environment cost "
|
|
308
|
+
f"attribution is a prerequisite for any further optimisation. No direct "
|
|
309
|
+
f"saving is claimed."
|
|
310
|
+
),
|
|
311
|
+
potentialSavings=0.0,
|
|
312
|
+
effort="LOW",
|
|
313
|
+
)
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
# --------------------------------------------------------------------------
|
|
317
|
+
# Entry point
|
|
318
|
+
# --------------------------------------------------------------------------
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
def analyze(series: BillingSeries,
|
|
322
|
+
analysis_type: AnalysisType = "waste-audit",
|
|
323
|
+
*, max_anomalies: int = 25) -> AnalysisResult:
|
|
324
|
+
"""Run waste analysis over a parsed billing series."""
|
|
325
|
+
currency = series.currency or "USD"
|
|
326
|
+
span = series.span_days
|
|
327
|
+
|
|
328
|
+
if span < 3:
|
|
329
|
+
return AnalysisResult(
|
|
330
|
+
summary=(
|
|
331
|
+
f"Only {span} day(s) of billing data. At least 3 days are needed "
|
|
332
|
+
f"to model a baseline, and 14 for weekday seasonality."
|
|
333
|
+
)
|
|
334
|
+
)
|
|
335
|
+
|
|
336
|
+
costs = np.asarray(series.costs, dtype=np.float64)
|
|
337
|
+
expected = _baseline(costs, series.days)
|
|
338
|
+
zscores = modified_zscores(costs, expected)
|
|
339
|
+
|
|
340
|
+
mean = float(np.mean(costs))
|
|
341
|
+
std = float(np.std(costs, ddof=1)) if span > 1 else 0.0
|
|
342
|
+
stability = float(mean / std) if std > 0 else 0.0
|
|
343
|
+
|
|
344
|
+
anomalies: list[Anomaly] = []
|
|
345
|
+
excess_total = 0.0
|
|
346
|
+
for i in range(span):
|
|
347
|
+
z = float(zscores[i])
|
|
348
|
+
if abs(z) < Z_THRESHOLD:
|
|
349
|
+
continue
|
|
350
|
+
exp = float(expected[i])
|
|
351
|
+
act = float(costs[i])
|
|
352
|
+
deviation = ((act - exp) / exp * 100.0) if exp > 0 else 0.0
|
|
353
|
+
if act > exp:
|
|
354
|
+
excess_total += act - exp
|
|
355
|
+
|
|
356
|
+
anomalies.append(
|
|
357
|
+
Anomaly(
|
|
358
|
+
date=series.days[i].isoformat(),
|
|
359
|
+
expectedCost=round(exp, 2),
|
|
360
|
+
actualCost=round(act, 2),
|
|
361
|
+
deviation=round(deviation, 2),
|
|
362
|
+
zScore=round(z, 2),
|
|
363
|
+
severity=_severity(z, deviation),
|
|
364
|
+
description=(
|
|
365
|
+
f"Spend {'above' if act > exp else 'below'} the day-of-week "
|
|
366
|
+
f"baseline by {currency} {abs(act - exp):,.2f} "
|
|
367
|
+
f"({abs(deviation):.1f}%)."
|
|
368
|
+
),
|
|
369
|
+
)
|
|
370
|
+
)
|
|
371
|
+
|
|
372
|
+
total = series.total_cost
|
|
373
|
+
waste_pct = (excess_total / total * 100.0) if total > 0 else 0.0
|
|
374
|
+
|
|
375
|
+
metrics = Metrics(
|
|
376
|
+
averageDailyCost=round(mean, 2),
|
|
377
|
+
stdDeviation=round(std, 2),
|
|
378
|
+
sharpeRatio=round(stability, 2),
|
|
379
|
+
wastePercentage=round(waste_pct, 2),
|
|
380
|
+
)
|
|
381
|
+
|
|
382
|
+
recommendations: list[Recommendation] = []
|
|
383
|
+
if analysis_type in ("waste-audit", "cost-forecast"):
|
|
384
|
+
over = sum(1 for a in anomalies if a.actualCost > a.expectedCost)
|
|
385
|
+
excess_rec = _excess_recommendation(excess_total, span, currency, over)
|
|
386
|
+
if excess_rec:
|
|
387
|
+
recommendations.append(excess_rec)
|
|
388
|
+
recommendations.extend(_weekend_recommendations(series, currency))
|
|
389
|
+
commitment = _commitment_recommendation(costs, currency)
|
|
390
|
+
if commitment:
|
|
391
|
+
recommendations.append(commitment)
|
|
392
|
+
concentration = _concentration_recommendation(series, currency)
|
|
393
|
+
if concentration:
|
|
394
|
+
recommendations.append(concentration)
|
|
395
|
+
|
|
396
|
+
volatility = _volatility_recommendation(stability, mean, std, currency)
|
|
397
|
+
if volatility:
|
|
398
|
+
recommendations.append(volatility)
|
|
399
|
+
|
|
400
|
+
recommendations.sort(key=lambda r: r.potentialSavings, reverse=True)
|
|
401
|
+
|
|
402
|
+
identified = sum(r.potentialSavings for r in recommendations)
|
|
403
|
+
summary = (
|
|
404
|
+
f"Analysed {span} day(s) of billing ({series.rows_parsed:,} line items) "
|
|
405
|
+
f"totalling {currency} {total:,.2f}. "
|
|
406
|
+
f"Mean daily spend {currency} {mean:,.2f}, stability ratio {stability:.2f}. "
|
|
407
|
+
f"{len(anomalies)} anomalous day(s) detected; {waste_pct:.1f}% of total "
|
|
408
|
+
f"spend sits above the modelled baseline. "
|
|
409
|
+
f"Identified up to {currency} {identified:,.2f} in addressable monthly spend."
|
|
410
|
+
)
|
|
411
|
+
if series.rows_skipped:
|
|
412
|
+
summary += f" {series.rows_skipped:,} row(s) could not be parsed."
|
|
413
|
+
if analysis_type == "cost-forecast":
|
|
414
|
+
summary += f" Projected 30-day spend at current run rate: {currency} {mean * 30:,.2f}."
|
|
415
|
+
|
|
416
|
+
anomalies.sort(key=lambda a: abs(a.zScore), reverse=True)
|
|
417
|
+
|
|
418
|
+
return AnalysisResult(
|
|
419
|
+
anomalies=anomalies[:max_anomalies],
|
|
420
|
+
metrics=metrics,
|
|
421
|
+
recommendations=recommendations,
|
|
422
|
+
summary=summary,
|
|
423
|
+
)
|
cloudsealed_jit/api.py
ADDED
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
"""
|
|
2
|
+
HTTP surface for the waste analyser.
|
|
3
|
+
|
|
4
|
+
The request and response schemas are the contract consumed by the CloudSealed
|
|
5
|
+
Framework4D assessment engine (``framework4d-jit-client.ts``). Field names and
|
|
6
|
+
types are part of that contract and must not change without a version bump.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import logging
|
|
12
|
+
import os
|
|
13
|
+
import time
|
|
14
|
+
from typing import Literal, Optional
|
|
15
|
+
|
|
16
|
+
from fastapi import Depends, FastAPI, Header, HTTPException, status
|
|
17
|
+
from pydantic import BaseModel, Field
|
|
18
|
+
|
|
19
|
+
from . import __version__
|
|
20
|
+
from .analysis import analyze
|
|
21
|
+
from .kernels import JIT_ENABLED
|
|
22
|
+
from .parsing import ParseError, parse_billing_csv
|
|
23
|
+
|
|
24
|
+
logger = logging.getLogger("cloudsealed_jit.api")
|
|
25
|
+
|
|
26
|
+
#: Reject payloads above this size before parsing. Billing exports are large,
|
|
27
|
+
#: but an unbounded body is a denial-of-service surface.
|
|
28
|
+
MAX_CSV_BYTES = int(os.getenv("JIT_MAX_CSV_BYTES", str(64 * 1024 * 1024)))
|
|
29
|
+
|
|
30
|
+
#: When set, every request must present a matching ``X-Api-Key`` header.
|
|
31
|
+
API_KEY = os.getenv("JIT_OPTIMIZATION_API_KEY", "")
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
# --------------------------------------------------------------------------
|
|
35
|
+
# Contract
|
|
36
|
+
# --------------------------------------------------------------------------
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class AnalyzeBillingRequest(BaseModel):
|
|
40
|
+
companyName: str = Field(min_length=1, max_length=200)
|
|
41
|
+
csvContent: str = Field(description="Raw contents of the billing export.")
|
|
42
|
+
analysisType: Optional[Literal["waste-audit", "cost-forecast", "efficiency"]] = (
|
|
43
|
+
"waste-audit"
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class CostAnomaly(BaseModel):
|
|
48
|
+
date: str
|
|
49
|
+
expectedCost: float
|
|
50
|
+
actualCost: float
|
|
51
|
+
deviation: float
|
|
52
|
+
zScore: float
|
|
53
|
+
severity: Literal["LOW", "MEDIUM", "HIGH", "CRITICAL"]
|
|
54
|
+
description: str
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class Metrics(BaseModel):
|
|
58
|
+
averageDailyCost: float
|
|
59
|
+
stdDeviation: float
|
|
60
|
+
sharpeRatio: float
|
|
61
|
+
wastePercentage: float
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class Recommendation(BaseModel):
|
|
65
|
+
title: str
|
|
66
|
+
description: str
|
|
67
|
+
potentialSavings: float
|
|
68
|
+
effort: Literal["LOW", "MEDIUM", "HIGH"]
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
class AnalyzeBillingResponse(BaseModel):
|
|
72
|
+
anomalies: list[CostAnomaly]
|
|
73
|
+
metrics: Metrics
|
|
74
|
+
recommendations: list[Recommendation]
|
|
75
|
+
summary: str
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
# --------------------------------------------------------------------------
|
|
79
|
+
# Application
|
|
80
|
+
# --------------------------------------------------------------------------
|
|
81
|
+
|
|
82
|
+
app = FastAPI(
|
|
83
|
+
title="CloudSealed JIT Optimization Engine",
|
|
84
|
+
version=__version__,
|
|
85
|
+
description=(
|
|
86
|
+
"Detects structural waste in cloud billing exports using a robust, "
|
|
87
|
+
"day-of-week aware baseline and modified z-scores."
|
|
88
|
+
),
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
async def require_api_key(x_api_key: str = Header(default="")) -> None:
|
|
93
|
+
"""Enforce ``X-Api-Key`` when an API key is configured."""
|
|
94
|
+
if API_KEY and x_api_key != API_KEY:
|
|
95
|
+
raise HTTPException(
|
|
96
|
+
status_code=status.HTTP_401_UNAUTHORIZED,
|
|
97
|
+
detail="Invalid or missing X-Api-Key.",
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
@app.get("/health")
|
|
102
|
+
def health() -> dict:
|
|
103
|
+
return {
|
|
104
|
+
"status": "ok",
|
|
105
|
+
"version": __version__,
|
|
106
|
+
"jitEnabled": JIT_ENABLED,
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
@app.post(
|
|
111
|
+
"/v1/analyze-billing",
|
|
112
|
+
response_model=AnalyzeBillingResponse,
|
|
113
|
+
dependencies=[Depends(require_api_key)],
|
|
114
|
+
)
|
|
115
|
+
def analyze_billing(payload: AnalyzeBillingRequest) -> AnalyzeBillingResponse:
|
|
116
|
+
"""Analyse a cloud billing export and return waste findings.
|
|
117
|
+
|
|
118
|
+
Returns 422 when the export cannot be interpreted, so callers can tell a
|
|
119
|
+
malformed file from an internal failure.
|
|
120
|
+
"""
|
|
121
|
+
size = len(payload.csvContent.encode("utf-8", errors="ignore"))
|
|
122
|
+
if size > MAX_CSV_BYTES:
|
|
123
|
+
raise HTTPException(
|
|
124
|
+
status_code=status.HTTP_413_REQUEST_ENTITY_TOO_LARGE,
|
|
125
|
+
detail=f"Billing export is {size:,} bytes; limit is {MAX_CSV_BYTES:,}.",
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
started = time.perf_counter()
|
|
129
|
+
try:
|
|
130
|
+
series = parse_billing_csv(payload.csvContent)
|
|
131
|
+
except ParseError as exc:
|
|
132
|
+
raise HTTPException(
|
|
133
|
+
status_code=status.HTTP_422_UNPROCESSABLE_ENTITY, detail=str(exc)
|
|
134
|
+
) from exc
|
|
135
|
+
|
|
136
|
+
try:
|
|
137
|
+
result = analyze(series, payload.analysisType or "waste-audit")
|
|
138
|
+
except Exception: # pragma: no cover - defensive
|
|
139
|
+
logger.exception("Analysis failed for %s", payload.companyName)
|
|
140
|
+
raise HTTPException(
|
|
141
|
+
status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
|
|
142
|
+
detail="Analysis failed.",
|
|
143
|
+
)
|
|
144
|
+
|
|
145
|
+
elapsed_ms = (time.perf_counter() - started) * 1000
|
|
146
|
+
logger.info(
|
|
147
|
+
"analyzed company=%s days=%d rows=%d anomalies=%d elapsed_ms=%.1f",
|
|
148
|
+
payload.companyName, series.span_days, series.rows_parsed,
|
|
149
|
+
len(result.anomalies), elapsed_ms,
|
|
150
|
+
)
|
|
151
|
+
return AnalyzeBillingResponse(**result.to_dict())
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def main() -> None: # pragma: no cover - entry point
|
|
155
|
+
import uvicorn
|
|
156
|
+
|
|
157
|
+
uvicorn.run(
|
|
158
|
+
app,
|
|
159
|
+
host=os.getenv("JIT_HOST", "0.0.0.0"),
|
|
160
|
+
port=int(os.getenv("JIT_PORT", "8091")),
|
|
161
|
+
)
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
if __name__ == "__main__": # pragma: no cover
|
|
165
|
+
main()
|
cloudsealed_jit/cli.py
ADDED
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"""Command-line entry point: analyse a billing export without running a server."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import json
|
|
7
|
+
import sys
|
|
8
|
+
|
|
9
|
+
from .analysis import analyze
|
|
10
|
+
from .parsing import ParseError, parse_billing_csv
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def main(argv: list[str] | None = None) -> int:
|
|
14
|
+
parser = argparse.ArgumentParser(
|
|
15
|
+
prog="cloudsealed-jit",
|
|
16
|
+
description="Detect structural waste in a cloud billing export.",
|
|
17
|
+
)
|
|
18
|
+
parser.add_argument("csv", help="path to the billing export, or - for stdin")
|
|
19
|
+
parser.add_argument(
|
|
20
|
+
"--type",
|
|
21
|
+
default="waste-audit",
|
|
22
|
+
choices=("waste-audit", "cost-forecast", "efficiency"),
|
|
23
|
+
help="analysis mode (default: waste-audit)",
|
|
24
|
+
)
|
|
25
|
+
parser.add_argument("--json", action="store_true", help="emit raw JSON")
|
|
26
|
+
args = parser.parse_args(argv)
|
|
27
|
+
|
|
28
|
+
text = sys.stdin.read() if args.csv == "-" else open(args.csv, encoding="utf-8").read()
|
|
29
|
+
|
|
30
|
+
try:
|
|
31
|
+
result = analyze(parse_billing_csv(text), args.type)
|
|
32
|
+
except ParseError as exc:
|
|
33
|
+
print(f"error: {exc}", file=sys.stderr)
|
|
34
|
+
return 2
|
|
35
|
+
|
|
36
|
+
if args.json:
|
|
37
|
+
print(json.dumps(result.to_dict(), indent=2))
|
|
38
|
+
return 0
|
|
39
|
+
|
|
40
|
+
print(result.summary)
|
|
41
|
+
if result.anomalies:
|
|
42
|
+
print("\nAnomalies")
|
|
43
|
+
for a in result.anomalies:
|
|
44
|
+
print(
|
|
45
|
+
f" {a.date} {a.severity:<8} expected {a.expectedCost:>12,.2f} "
|
|
46
|
+
f"actual {a.actualCost:>12,.2f} ({a.deviation:+.1f}%, z={a.zScore})"
|
|
47
|
+
)
|
|
48
|
+
if result.recommendations:
|
|
49
|
+
print("\nRecommendations")
|
|
50
|
+
for r in result.recommendations:
|
|
51
|
+
print(f" [{r.effort:<6}] {r.title} — {r.potentialSavings:,.2f}/30d")
|
|
52
|
+
print(f" {r.description}")
|
|
53
|
+
return 0
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
if __name__ == "__main__": # pragma: no cover
|
|
57
|
+
raise SystemExit(main())
|