cloudsealed-jit 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,24 @@
1
+ """
2
+ cloudsealed-jit — cloud billing waste analysis.
3
+
4
+ Detects structural cost waste in cloud billing exports by modelling a robust
5
+ day-of-week aware baseline and measuring the excess spend above it.
6
+
7
+ Public API:
8
+ parse_billing_csv(csv_text) -> BillingSeries
9
+ analyze(series, analysis_type="waste-audit") -> AnalysisResult
10
+ """
11
+
12
+ from .parsing import BillingSeries, ParseError, parse_billing_csv
13
+ from .analysis import AnalysisResult, analyze
14
+
15
+ __version__ = "0.2.0"
16
+
17
+ __all__ = [
18
+ "BillingSeries",
19
+ "ParseError",
20
+ "parse_billing_csv",
21
+ "AnalysisResult",
22
+ "analyze",
23
+ "__version__",
24
+ ]
@@ -0,0 +1,423 @@
1
+ """
2
+ Waste analysis over a daily billing series.
3
+
4
+ Method
5
+ ------
6
+
7
+ **Baseline.** Expected spend for a day is the product of a level term and a
8
+ weekday term:
9
+
10
+ expected[i] = rolling_median(cost, 7)[i] * dow_factor[weekday(i)]
11
+
12
+ The rolling median tracks growth and step changes without being dragged by
13
+ spikes. The weekday factor is the median ratio of observed spend to the level
14
+ term for that weekday, which captures the weekday/weekend cycle that dominates
15
+ most cloud bills. It is only estimated when at least two full weeks are
16
+ available; below that the factor is 1.0 for every day.
17
+
18
+ **Anomalies.** Residuals against the baseline are scored with a modified
19
+ z-score (median absolute deviation, see :mod:`cloudsealed_jit.kernels`). Days
20
+ scoring at or above ``Z_THRESHOLD`` are reported. The threshold of 3.5 is the
21
+ value recommended by Iglewicz & Hoaglin for the modified z-score.
22
+
23
+ **Waste.** Only *positive* excess counts as waste: spending less than expected
24
+ is not an opportunity. Waste percentage is the share of total spend that sits
25
+ above the baseline on anomalous days, which makes it directly convertible to
26
+ currency rather than being a count of unusual days.
27
+
28
+ **Recommendations.** Every recommendation carries a saving derived from the
29
+ series itself, normalised to 30 days, and states the assumption behind it in
30
+ its description. Estimates that depend on facts the analyser cannot observe --
31
+ whether a workload is production, whether a commitment is acceptable -- are
32
+ labelled as conditional rather than presented as findings.
33
+ """
34
+
35
+ from __future__ import annotations
36
+
37
+ import statistics
38
+ from dataclasses import asdict, dataclass, field
39
+ from datetime import date
40
+ from typing import Literal
41
+
42
+ import numpy as np
43
+
44
+ from .kernels import modified_zscores, rolling_median
45
+ from .parsing import BillingSeries
46
+
47
+ __all__ = [
48
+ "Anomaly",
49
+ "Metrics",
50
+ "Recommendation",
51
+ "AnalysisResult",
52
+ "analyze",
53
+ "Z_THRESHOLD",
54
+ ]
55
+
56
+ AnalysisType = Literal["waste-audit", "cost-forecast", "efficiency"]
57
+
58
+ #: Modified z-score at which a day is reported as anomalous.
59
+ Z_THRESHOLD = 3.5
60
+
61
+ #: Minimum days required before the weekday seasonality term is estimated.
62
+ MIN_DAYS_FOR_SEASONALITY = 14
63
+
64
+ #: Assumed discount for a one-year commitment against the sustained baseline.
65
+ #: Conservative relative to published AWS/GCP/Azure commitment discounts.
66
+ COMMITMENT_DISCOUNT = 0.25
67
+
68
+ #: A service whose weekend spend is at least this fraction of its weekday
69
+ #: spend is running through the weekend at full cost.
70
+ WEEKEND_RATIO_FLAG = 0.80
71
+
72
+ # Severity combines statistical strength with financial size. Either alone is
73
+ # misleading: a very tight baseline turns a 5x spike into a moderate z-score,
74
+ # while a noisy series can produce a large z-score over a trivial amount.
75
+ # Each tier is (min |z|, min |deviation %|, label) and matches on either.
76
+ _SEVERITY_TIERS = (
77
+ (10.0, 200.0, "CRITICAL"),
78
+ (7.0, 100.0, "HIGH"),
79
+ (5.0, 25.0, "MEDIUM"),
80
+ )
81
+
82
+
83
+ @dataclass
84
+ class Anomaly:
85
+ date: str
86
+ expectedCost: float
87
+ actualCost: float
88
+ deviation: float # percent against expected
89
+ zScore: float
90
+ severity: str
91
+ description: str
92
+
93
+
94
+ @dataclass
95
+ class Metrics:
96
+ averageDailyCost: float
97
+ stdDeviation: float
98
+ sharpeRatio: float # spend stability: mean / stddev of daily cost
99
+ wastePercentage: float
100
+
101
+
102
+ @dataclass
103
+ class Recommendation:
104
+ title: str
105
+ description: str
106
+ potentialSavings: float # per 30 days, in the export's currency
107
+ effort: str
108
+
109
+
110
+ @dataclass
111
+ class AnalysisResult:
112
+ anomalies: list[Anomaly] = field(default_factory=list)
113
+ metrics: Metrics = field(
114
+ default_factory=lambda: Metrics(0.0, 0.0, 0.0, 0.0)
115
+ )
116
+ recommendations: list[Recommendation] = field(default_factory=list)
117
+ summary: str = ""
118
+
119
+ def to_dict(self) -> dict:
120
+ return asdict(self)
121
+
122
+
123
+ # --------------------------------------------------------------------------
124
+ # Baseline
125
+ # --------------------------------------------------------------------------
126
+
127
+
128
+ def _weekday_factors(costs: np.ndarray, level: np.ndarray,
129
+ days: list[date]) -> np.ndarray:
130
+ """Median ratio of observed spend to the level term, per weekday."""
131
+ factors = np.ones(7, dtype=np.float64)
132
+ if len(days) < MIN_DAYS_FOR_SEASONALITY:
133
+ return factors
134
+
135
+ buckets: dict[int, list[float]] = {i: [] for i in range(7)}
136
+ for i, day in enumerate(days):
137
+ if level[i] > 0:
138
+ buckets[day.weekday()].append(costs[i] / level[i])
139
+
140
+ for weekday, ratios in buckets.items():
141
+ # Two observations is not enough to separate a pattern from noise.
142
+ if len(ratios) >= 2:
143
+ factors[weekday] = float(statistics.median(ratios))
144
+
145
+ # Renormalise so the factors do not shift the overall level.
146
+ mean_factor = float(np.mean(factors))
147
+ if mean_factor > 0:
148
+ factors /= mean_factor
149
+ return factors
150
+
151
+
152
+ def _baseline(costs: np.ndarray, days: list[date]) -> np.ndarray:
153
+ level = rolling_median(costs, window=7)
154
+ factors = _weekday_factors(costs, level, days)
155
+ expected = np.array(
156
+ [level[i] * factors[day.weekday()] for i, day in enumerate(days)],
157
+ dtype=np.float64,
158
+ )
159
+ return np.maximum(expected, 0.0)
160
+
161
+
162
+ def _severity(z: float, deviation_pct: float) -> str:
163
+ magnitude = abs(z)
164
+ deviation = abs(deviation_pct)
165
+ for min_z, min_deviation, label in _SEVERITY_TIERS:
166
+ if magnitude >= min_z or deviation >= min_deviation:
167
+ return label
168
+ return "LOW"
169
+
170
+
171
+ # --------------------------------------------------------------------------
172
+ # Recommendations
173
+ # --------------------------------------------------------------------------
174
+
175
+
176
+ def _weekend_recommendations(series: BillingSeries,
177
+ currency: str) -> list[Recommendation]:
178
+ """Flag services that cost the same on weekends as on weekdays."""
179
+ out: list[Recommendation] = []
180
+ days = series.days
181
+ if len(days) < MIN_DAYS_FOR_SEASONALITY:
182
+ return out
183
+
184
+ weekend_idx = [i for i, d in enumerate(days) if d.weekday() >= 5]
185
+ weekday_idx = [i for i, d in enumerate(days) if d.weekday() < 5]
186
+ if len(weekend_idx) < 4 or len(weekday_idx) < 8:
187
+ return out
188
+
189
+ span = len(days)
190
+ for name, values in series.by_service.items():
191
+ weekend = statistics.median(values[i] for i in weekend_idx)
192
+ weekday = statistics.median(values[i] for i in weekday_idx)
193
+ if weekday <= 0:
194
+ continue
195
+ ratio = weekend / weekday
196
+ if ratio < WEEKEND_RATIO_FLAG:
197
+ continue
198
+
199
+ service_total = sum(values)
200
+ if service_total <= 0:
201
+ continue
202
+
203
+ # Upper bound: the weekend spend itself, normalised to 30 days.
204
+ weekend_spend = sum(values[i] for i in weekend_idx)
205
+ monthly = weekend_spend / span * 30
206
+ if monthly < 1.0:
207
+ continue
208
+
209
+ out.append(
210
+ Recommendation(
211
+ title=f"Idle weekend spend in {name}",
212
+ description=(
213
+ f"{name} costs {ratio:.0%} as much on weekends as on weekdays "
214
+ f"(weekday median {currency} {weekday:,.2f}/day, weekend median "
215
+ f"{currency} {weekend:,.2f}/day). If this workload is not "
216
+ f"production, stopping it outside business days removes the "
217
+ f"weekend spend entirely. The figure is the observed weekend "
218
+ f"spend normalised to 30 days and assumes the workload can be "
219
+ f"stopped; verify before acting."
220
+ ),
221
+ potentialSavings=round(monthly, 2),
222
+ effort="LOW",
223
+ )
224
+ )
225
+ return out
226
+
227
+
228
+ def _commitment_recommendation(costs: np.ndarray,
229
+ currency: str) -> Recommendation | None:
230
+ """Estimate the saving available on the always-on portion of spend."""
231
+ if costs.size < MIN_DAYS_FOR_SEASONALITY:
232
+ return None
233
+ floor = float(np.percentile(costs, 10))
234
+ if floor <= 0:
235
+ return None
236
+
237
+ monthly = floor * 30 * COMMITMENT_DISCOUNT
238
+ if monthly < 1.0:
239
+ return None
240
+
241
+ return Recommendation(
242
+ title="Sustained baseline eligible for commitment pricing",
243
+ description=(
244
+ f"Daily spend never fell below {currency} {floor:,.2f} (10th percentile) "
245
+ f"over the period, so that portion is always-on. Committed-use discounts "
246
+ f"or savings plans typically price this tier {COMMITMENT_DISCOUNT:.0%} "
247
+ f"below on-demand. The figure applies that rate to the observed floor "
248
+ f"over 30 days and assumes a commitment is commercially acceptable."
249
+ ),
250
+ potentialSavings=round(monthly, 2),
251
+ effort="MEDIUM",
252
+ )
253
+
254
+
255
+ def _excess_recommendation(excess_total: float, span: int,
256
+ currency: str, count: int) -> Recommendation | None:
257
+ if excess_total <= 0 or count == 0:
258
+ return None
259
+ monthly = excess_total / span * 30
260
+ if monthly < 1.0:
261
+ return None
262
+ return Recommendation(
263
+ title="Investigate spend above the modelled baseline",
264
+ description=(
265
+ f"{count} day(s) spent {currency} {excess_total:,.2f} more than the "
266
+ f"day-of-week baseline predicted. This is unplanned spend rather than "
267
+ f"growth: the baseline already tracks trend and weekly seasonality. "
268
+ f"The figure is that excess normalised to 30 days."
269
+ ),
270
+ potentialSavings=round(monthly, 2),
271
+ effort="MEDIUM",
272
+ )
273
+
274
+
275
+ def _concentration_recommendation(series: BillingSeries,
276
+ currency: str) -> Recommendation | None:
277
+ total = series.total_cost
278
+ if total <= 0 or not series.by_service:
279
+ return None
280
+ name, values = max(series.by_service.items(), key=lambda kv: sum(kv[1]))
281
+ share = sum(values) / total
282
+ if share < 0.40:
283
+ return None
284
+ return Recommendation(
285
+ title=f"Spend concentrated in {name}",
286
+ description=(
287
+ f"{name} accounts for {share:.0%} of total spend ({currency} "
288
+ f"{sum(values):,.2f} of {currency} {total:,.2f}). Optimisation effort "
289
+ f"applied here has the highest leverage; a 10% reduction on this "
290
+ f"service alone yields the figure shown."
291
+ ),
292
+ potentialSavings=round(sum(values) / max(len(series.days), 1) * 30 * 0.10, 2),
293
+ effort="MEDIUM",
294
+ )
295
+
296
+
297
+ def _volatility_recommendation(stability: float, mean: float,
298
+ std: float, currency: str) -> Recommendation | None:
299
+ if stability >= 2.0 or mean <= 0:
300
+ return None
301
+ return Recommendation(
302
+ title="Daily spend is highly volatile",
303
+ description=(
304
+ f"Daily cost varies with a standard deviation of {currency} {std:,.2f} "
305
+ f"against a mean of {currency} {mean:,.2f} (stability ratio "
306
+ f"{stability:.2f}). Spend this irregular cannot be budgeted or alerted "
307
+ f"on reliably. Establishing budget alerts and per-environment cost "
308
+ f"attribution is a prerequisite for any further optimisation. No direct "
309
+ f"saving is claimed."
310
+ ),
311
+ potentialSavings=0.0,
312
+ effort="LOW",
313
+ )
314
+
315
+
316
+ # --------------------------------------------------------------------------
317
+ # Entry point
318
+ # --------------------------------------------------------------------------
319
+
320
+
321
+ def analyze(series: BillingSeries,
322
+ analysis_type: AnalysisType = "waste-audit",
323
+ *, max_anomalies: int = 25) -> AnalysisResult:
324
+ """Run waste analysis over a parsed billing series."""
325
+ currency = series.currency or "USD"
326
+ span = series.span_days
327
+
328
+ if span < 3:
329
+ return AnalysisResult(
330
+ summary=(
331
+ f"Only {span} day(s) of billing data. At least 3 days are needed "
332
+ f"to model a baseline, and 14 for weekday seasonality."
333
+ )
334
+ )
335
+
336
+ costs = np.asarray(series.costs, dtype=np.float64)
337
+ expected = _baseline(costs, series.days)
338
+ zscores = modified_zscores(costs, expected)
339
+
340
+ mean = float(np.mean(costs))
341
+ std = float(np.std(costs, ddof=1)) if span > 1 else 0.0
342
+ stability = float(mean / std) if std > 0 else 0.0
343
+
344
+ anomalies: list[Anomaly] = []
345
+ excess_total = 0.0
346
+ for i in range(span):
347
+ z = float(zscores[i])
348
+ if abs(z) < Z_THRESHOLD:
349
+ continue
350
+ exp = float(expected[i])
351
+ act = float(costs[i])
352
+ deviation = ((act - exp) / exp * 100.0) if exp > 0 else 0.0
353
+ if act > exp:
354
+ excess_total += act - exp
355
+
356
+ anomalies.append(
357
+ Anomaly(
358
+ date=series.days[i].isoformat(),
359
+ expectedCost=round(exp, 2),
360
+ actualCost=round(act, 2),
361
+ deviation=round(deviation, 2),
362
+ zScore=round(z, 2),
363
+ severity=_severity(z, deviation),
364
+ description=(
365
+ f"Spend {'above' if act > exp else 'below'} the day-of-week "
366
+ f"baseline by {currency} {abs(act - exp):,.2f} "
367
+ f"({abs(deviation):.1f}%)."
368
+ ),
369
+ )
370
+ )
371
+
372
+ total = series.total_cost
373
+ waste_pct = (excess_total / total * 100.0) if total > 0 else 0.0
374
+
375
+ metrics = Metrics(
376
+ averageDailyCost=round(mean, 2),
377
+ stdDeviation=round(std, 2),
378
+ sharpeRatio=round(stability, 2),
379
+ wastePercentage=round(waste_pct, 2),
380
+ )
381
+
382
+ recommendations: list[Recommendation] = []
383
+ if analysis_type in ("waste-audit", "cost-forecast"):
384
+ over = sum(1 for a in anomalies if a.actualCost > a.expectedCost)
385
+ excess_rec = _excess_recommendation(excess_total, span, currency, over)
386
+ if excess_rec:
387
+ recommendations.append(excess_rec)
388
+ recommendations.extend(_weekend_recommendations(series, currency))
389
+ commitment = _commitment_recommendation(costs, currency)
390
+ if commitment:
391
+ recommendations.append(commitment)
392
+ concentration = _concentration_recommendation(series, currency)
393
+ if concentration:
394
+ recommendations.append(concentration)
395
+
396
+ volatility = _volatility_recommendation(stability, mean, std, currency)
397
+ if volatility:
398
+ recommendations.append(volatility)
399
+
400
+ recommendations.sort(key=lambda r: r.potentialSavings, reverse=True)
401
+
402
+ identified = sum(r.potentialSavings for r in recommendations)
403
+ summary = (
404
+ f"Analysed {span} day(s) of billing ({series.rows_parsed:,} line items) "
405
+ f"totalling {currency} {total:,.2f}. "
406
+ f"Mean daily spend {currency} {mean:,.2f}, stability ratio {stability:.2f}. "
407
+ f"{len(anomalies)} anomalous day(s) detected; {waste_pct:.1f}% of total "
408
+ f"spend sits above the modelled baseline. "
409
+ f"Identified up to {currency} {identified:,.2f} in addressable monthly spend."
410
+ )
411
+ if series.rows_skipped:
412
+ summary += f" {series.rows_skipped:,} row(s) could not be parsed."
413
+ if analysis_type == "cost-forecast":
414
+ summary += f" Projected 30-day spend at current run rate: {currency} {mean * 30:,.2f}."
415
+
416
+ anomalies.sort(key=lambda a: abs(a.zScore), reverse=True)
417
+
418
+ return AnalysisResult(
419
+ anomalies=anomalies[:max_anomalies],
420
+ metrics=metrics,
421
+ recommendations=recommendations,
422
+ summary=summary,
423
+ )
cloudsealed_jit/api.py ADDED
@@ -0,0 +1,165 @@
1
+ """
2
+ HTTP surface for the waste analyser.
3
+
4
+ The request and response schemas are the contract consumed by the CloudSealed
5
+ Framework4D assessment engine (``framework4d-jit-client.ts``). Field names and
6
+ types are part of that contract and must not change without a version bump.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import logging
12
+ import os
13
+ import time
14
+ from typing import Literal, Optional
15
+
16
+ from fastapi import Depends, FastAPI, Header, HTTPException, status
17
+ from pydantic import BaseModel, Field
18
+
19
+ from . import __version__
20
+ from .analysis import analyze
21
+ from .kernels import JIT_ENABLED
22
+ from .parsing import ParseError, parse_billing_csv
23
+
24
+ logger = logging.getLogger("cloudsealed_jit.api")
25
+
26
+ #: Reject payloads above this size before parsing. Billing exports are large,
27
+ #: but an unbounded body is a denial-of-service surface.
28
+ MAX_CSV_BYTES = int(os.getenv("JIT_MAX_CSV_BYTES", str(64 * 1024 * 1024)))
29
+
30
+ #: When set, every request must present a matching ``X-Api-Key`` header.
31
+ API_KEY = os.getenv("JIT_OPTIMIZATION_API_KEY", "")
32
+
33
+
34
+ # --------------------------------------------------------------------------
35
+ # Contract
36
+ # --------------------------------------------------------------------------
37
+
38
+
39
+ class AnalyzeBillingRequest(BaseModel):
40
+ companyName: str = Field(min_length=1, max_length=200)
41
+ csvContent: str = Field(description="Raw contents of the billing export.")
42
+ analysisType: Optional[Literal["waste-audit", "cost-forecast", "efficiency"]] = (
43
+ "waste-audit"
44
+ )
45
+
46
+
47
+ class CostAnomaly(BaseModel):
48
+ date: str
49
+ expectedCost: float
50
+ actualCost: float
51
+ deviation: float
52
+ zScore: float
53
+ severity: Literal["LOW", "MEDIUM", "HIGH", "CRITICAL"]
54
+ description: str
55
+
56
+
57
+ class Metrics(BaseModel):
58
+ averageDailyCost: float
59
+ stdDeviation: float
60
+ sharpeRatio: float
61
+ wastePercentage: float
62
+
63
+
64
+ class Recommendation(BaseModel):
65
+ title: str
66
+ description: str
67
+ potentialSavings: float
68
+ effort: Literal["LOW", "MEDIUM", "HIGH"]
69
+
70
+
71
+ class AnalyzeBillingResponse(BaseModel):
72
+ anomalies: list[CostAnomaly]
73
+ metrics: Metrics
74
+ recommendations: list[Recommendation]
75
+ summary: str
76
+
77
+
78
+ # --------------------------------------------------------------------------
79
+ # Application
80
+ # --------------------------------------------------------------------------
81
+
82
+ app = FastAPI(
83
+ title="CloudSealed JIT Optimization Engine",
84
+ version=__version__,
85
+ description=(
86
+ "Detects structural waste in cloud billing exports using a robust, "
87
+ "day-of-week aware baseline and modified z-scores."
88
+ ),
89
+ )
90
+
91
+
92
+ async def require_api_key(x_api_key: str = Header(default="")) -> None:
93
+ """Enforce ``X-Api-Key`` when an API key is configured."""
94
+ if API_KEY and x_api_key != API_KEY:
95
+ raise HTTPException(
96
+ status_code=status.HTTP_401_UNAUTHORIZED,
97
+ detail="Invalid or missing X-Api-Key.",
98
+ )
99
+
100
+
101
+ @app.get("/health")
102
+ def health() -> dict:
103
+ return {
104
+ "status": "ok",
105
+ "version": __version__,
106
+ "jitEnabled": JIT_ENABLED,
107
+ }
108
+
109
+
110
+ @app.post(
111
+ "/v1/analyze-billing",
112
+ response_model=AnalyzeBillingResponse,
113
+ dependencies=[Depends(require_api_key)],
114
+ )
115
+ def analyze_billing(payload: AnalyzeBillingRequest) -> AnalyzeBillingResponse:
116
+ """Analyse a cloud billing export and return waste findings.
117
+
118
+ Returns 422 when the export cannot be interpreted, so callers can tell a
119
+ malformed file from an internal failure.
120
+ """
121
+ size = len(payload.csvContent.encode("utf-8", errors="ignore"))
122
+ if size > MAX_CSV_BYTES:
123
+ raise HTTPException(
124
+ status_code=status.HTTP_413_REQUEST_ENTITY_TOO_LARGE,
125
+ detail=f"Billing export is {size:,} bytes; limit is {MAX_CSV_BYTES:,}.",
126
+ )
127
+
128
+ started = time.perf_counter()
129
+ try:
130
+ series = parse_billing_csv(payload.csvContent)
131
+ except ParseError as exc:
132
+ raise HTTPException(
133
+ status_code=status.HTTP_422_UNPROCESSABLE_ENTITY, detail=str(exc)
134
+ ) from exc
135
+
136
+ try:
137
+ result = analyze(series, payload.analysisType or "waste-audit")
138
+ except Exception: # pragma: no cover - defensive
139
+ logger.exception("Analysis failed for %s", payload.companyName)
140
+ raise HTTPException(
141
+ status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
142
+ detail="Analysis failed.",
143
+ )
144
+
145
+ elapsed_ms = (time.perf_counter() - started) * 1000
146
+ logger.info(
147
+ "analyzed company=%s days=%d rows=%d anomalies=%d elapsed_ms=%.1f",
148
+ payload.companyName, series.span_days, series.rows_parsed,
149
+ len(result.anomalies), elapsed_ms,
150
+ )
151
+ return AnalyzeBillingResponse(**result.to_dict())
152
+
153
+
154
+ def main() -> None: # pragma: no cover - entry point
155
+ import uvicorn
156
+
157
+ uvicorn.run(
158
+ app,
159
+ host=os.getenv("JIT_HOST", "0.0.0.0"),
160
+ port=int(os.getenv("JIT_PORT", "8091")),
161
+ )
162
+
163
+
164
+ if __name__ == "__main__": # pragma: no cover
165
+ main()
cloudsealed_jit/cli.py ADDED
@@ -0,0 +1,57 @@
1
+ """Command-line entry point: analyse a billing export without running a server."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import json
7
+ import sys
8
+
9
+ from .analysis import analyze
10
+ from .parsing import ParseError, parse_billing_csv
11
+
12
+
13
+ def main(argv: list[str] | None = None) -> int:
14
+ parser = argparse.ArgumentParser(
15
+ prog="cloudsealed-jit",
16
+ description="Detect structural waste in a cloud billing export.",
17
+ )
18
+ parser.add_argument("csv", help="path to the billing export, or - for stdin")
19
+ parser.add_argument(
20
+ "--type",
21
+ default="waste-audit",
22
+ choices=("waste-audit", "cost-forecast", "efficiency"),
23
+ help="analysis mode (default: waste-audit)",
24
+ )
25
+ parser.add_argument("--json", action="store_true", help="emit raw JSON")
26
+ args = parser.parse_args(argv)
27
+
28
+ text = sys.stdin.read() if args.csv == "-" else open(args.csv, encoding="utf-8").read()
29
+
30
+ try:
31
+ result = analyze(parse_billing_csv(text), args.type)
32
+ except ParseError as exc:
33
+ print(f"error: {exc}", file=sys.stderr)
34
+ return 2
35
+
36
+ if args.json:
37
+ print(json.dumps(result.to_dict(), indent=2))
38
+ return 0
39
+
40
+ print(result.summary)
41
+ if result.anomalies:
42
+ print("\nAnomalies")
43
+ for a in result.anomalies:
44
+ print(
45
+ f" {a.date} {a.severity:<8} expected {a.expectedCost:>12,.2f} "
46
+ f"actual {a.actualCost:>12,.2f} ({a.deviation:+.1f}%, z={a.zScore})"
47
+ )
48
+ if result.recommendations:
49
+ print("\nRecommendations")
50
+ for r in result.recommendations:
51
+ print(f" [{r.effort:<6}] {r.title} — {r.potentialSavings:,.2f}/30d")
52
+ print(f" {r.description}")
53
+ return 0
54
+
55
+
56
+ if __name__ == "__main__": # pragma: no cover
57
+ raise SystemExit(main())