cloudsealed-jit 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cloudsealed_jit-0.2.0/LICENSE +21 -0
- cloudsealed_jit-0.2.0/PKG-INFO +222 -0
- cloudsealed_jit-0.2.0/README.md +188 -0
- cloudsealed_jit-0.2.0/cloudsealed_jit/__init__.py +24 -0
- cloudsealed_jit-0.2.0/cloudsealed_jit/analysis.py +423 -0
- cloudsealed_jit-0.2.0/cloudsealed_jit/api.py +165 -0
- cloudsealed_jit-0.2.0/cloudsealed_jit/cli.py +57 -0
- cloudsealed_jit-0.2.0/cloudsealed_jit/kernels.py +140 -0
- cloudsealed_jit-0.2.0/cloudsealed_jit/parsing.py +321 -0
- cloudsealed_jit-0.2.0/cloudsealed_jit.egg-info/PKG-INFO +222 -0
- cloudsealed_jit-0.2.0/cloudsealed_jit.egg-info/SOURCES.txt +18 -0
- cloudsealed_jit-0.2.0/cloudsealed_jit.egg-info/dependency_links.txt +1 -0
- cloudsealed_jit-0.2.0/cloudsealed_jit.egg-info/entry_points.txt +2 -0
- cloudsealed_jit-0.2.0/cloudsealed_jit.egg-info/requires.txt +13 -0
- cloudsealed_jit-0.2.0/cloudsealed_jit.egg-info/top_level.txt +1 -0
- cloudsealed_jit-0.2.0/pyproject.toml +51 -0
- cloudsealed_jit-0.2.0/setup.cfg +4 -0
- cloudsealed_jit-0.2.0/tests/test_analysis.py +121 -0
- cloudsealed_jit-0.2.0/tests/test_api.py +88 -0
- cloudsealed_jit-0.2.0/tests/test_parsing.py +82 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 cloudsealed
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,222 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: cloudsealed-jit
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Detect structural waste in cloud billing exports using robust day-of-week baselines
|
|
5
|
+
Author: Rodrigo Martinez Pinto
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/cloudsealed/JIT-Optimization-Engine
|
|
8
|
+
Project-URL: Source, https://github.com/cloudsealed/JIT-Optimization-Engine
|
|
9
|
+
Project-URL: Issues, https://github.com/cloudsealed/JIT-Optimization-Engine/issues
|
|
10
|
+
Keywords: finops,cloud-cost,aws,gcp,azure,anomaly-detection,billing
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: System Administrators
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Topic :: System :: Monitoring
|
|
19
|
+
Classifier: Topic :: Office/Business :: Financial
|
|
20
|
+
Requires-Python: >=3.10
|
|
21
|
+
Description-Content-Type: text/markdown
|
|
22
|
+
License-File: LICENSE
|
|
23
|
+
Requires-Dist: numpy>=1.24
|
|
24
|
+
Provides-Extra: jit
|
|
25
|
+
Requires-Dist: numba>=0.57; extra == "jit"
|
|
26
|
+
Provides-Extra: api
|
|
27
|
+
Requires-Dist: fastapi>=0.110; extra == "api"
|
|
28
|
+
Requires-Dist: uvicorn[standard]>=0.29; extra == "api"
|
|
29
|
+
Requires-Dist: pydantic>=2.6; extra == "api"
|
|
30
|
+
Provides-Extra: dev
|
|
31
|
+
Requires-Dist: pytest>=7.4; extra == "dev"
|
|
32
|
+
Requires-Dist: httpx>=0.27; extra == "dev"
|
|
33
|
+
Dynamic: license-file
|
|
34
|
+
|
|
35
|
+
# cloudsealed-jit
|
|
36
|
+
|
|
37
|
+
Detects structural waste in cloud billing exports.
|
|
38
|
+
|
|
39
|
+
Given a billing export from AWS, GCP or Azure, it models what each day *should*
|
|
40
|
+
have cost, reports the days that did not match, and turns the excess into a
|
|
41
|
+
monthly figure. It is a library, a CLI and an HTTP service.
|
|
42
|
+
|
|
43
|
+
[](LICENSE)
|
|
44
|
+
[](https://www.python.org/downloads/)
|
|
45
|
+
|
|
46
|
+
---
|
|
47
|
+
|
|
48
|
+
## The problem
|
|
49
|
+
|
|
50
|
+
Cloud cost anomaly detection is usually done by comparing each day against the
|
|
51
|
+
period average and flagging anything beyond two or three standard deviations.
|
|
52
|
+
On billing data that method fails in two specific ways.
|
|
53
|
+
|
|
54
|
+
**Standard deviation is inflated by the very spikes you are looking for.** A
|
|
55
|
+
handful of large anomalies raises σ enough to pull themselves back inside the
|
|
56
|
+
threshold, and to hide every smaller anomaly with them. This is the masking
|
|
57
|
+
effect, and it gets worse as the anomalies get bigger.
|
|
58
|
+
|
|
59
|
+
**A flat average ignores the weekly cycle.** Most cloud bills have a pronounced
|
|
60
|
+
weekday/weekend shape. Measured against a flat mean, ordinary Mondays look like
|
|
61
|
+
overspend and ordinary Sundays look like savings.
|
|
62
|
+
|
|
63
|
+
## The method
|
|
64
|
+
|
|
65
|
+
**Baseline.** Expected spend for a day is a level term times a weekday term:
|
|
66
|
+
|
|
67
|
+
```
|
|
68
|
+
expected[i] = rolling_median(cost, 7)[i] × dow_factor[weekday(i)]
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
The rolling median follows growth and step changes without being dragged by
|
|
72
|
+
spikes. The weekday factor is the median ratio of observed spend to the level
|
|
73
|
+
term for that weekday. It is only estimated with at least two full weeks of
|
|
74
|
+
data; below that every factor is 1.0.
|
|
75
|
+
|
|
76
|
+
**Scoring.** Residuals are scored with a modified z-score built on the median
|
|
77
|
+
absolute deviation:
|
|
78
|
+
|
|
79
|
+
```
|
|
80
|
+
z = 0.6745 × (x − baseline) / MAD
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
The 0.6745 constant makes MAD a consistent estimator of σ for normal data, so
|
|
84
|
+
the score keeps the familiar "number of deviations" reading while tolerating
|
|
85
|
+
contamination in roughly half the sample. Days at or above |z| = 3.5 are
|
|
86
|
+
reported — the threshold recommended by Iglewicz & Hoaglin (1993).
|
|
87
|
+
|
|
88
|
+
**Waste.** Only positive excess counts. Waste percentage is the share of total
|
|
89
|
+
spend sitting above the baseline on anomalous days, which converts directly to
|
|
90
|
+
currency instead of being a count of unusual days.
|
|
91
|
+
|
|
92
|
+
**Recommendations.** Each carries a figure derived from the series itself,
|
|
93
|
+
normalised to 30 days, and states its assumption in the description. Estimates
|
|
94
|
+
that depend on facts the analyser cannot observe — whether a workload is
|
|
95
|
+
production, whether a commitment is acceptable — are labelled conditional
|
|
96
|
+
rather than presented as findings.
|
|
97
|
+
|
|
98
|
+
## Does it actually work better?
|
|
99
|
+
|
|
100
|
+
Yes, and it is measured, not asserted. `benchmarks/masking_benchmark.py` builds
|
|
101
|
+
synthetic bills whose anomalies are known by construction and scores this
|
|
102
|
+
method against the textbook mean+standard-deviation approach:
|
|
103
|
+
|
|
104
|
+
| scenario | textbook F1 | this method F1 |
|
|
105
|
+
|---|---|---|
|
|
106
|
+
| masking (scale estimator) | 0.667 | **0.923** |
|
|
107
|
+
| seasonality (baseline) | 0.667 | **1.000** |
|
|
108
|
+
| end-to-end | 0.667 | **1.000** |
|
|
109
|
+
|
|
110
|
+
Full derivation and reproduction steps in [METHODOLOGY.md](METHODOLOGY.md); the
|
|
111
|
+
design of the codebase is in [architecture.md](architecture.md). The benchmark
|
|
112
|
+
runs in CI (`--check`) and fails the build if the advantage ever regresses.
|
|
113
|
+
|
|
114
|
+
## Install
|
|
115
|
+
|
|
116
|
+
```bash
|
|
117
|
+
pip install cloudsealed-jit # library + CLI
|
|
118
|
+
pip install "cloudsealed-jit[jit]" # + numba-compiled kernels
|
|
119
|
+
pip install "cloudsealed-jit[jit,api]" # + HTTP service
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
`numba` is optional. Without it the kernels run on pure NumPy and the results
|
|
123
|
+
are identical; only large inputs get slower.
|
|
124
|
+
|
|
125
|
+
## Use
|
|
126
|
+
|
|
127
|
+
### CLI
|
|
128
|
+
|
|
129
|
+
```bash
|
|
130
|
+
cloudsealed-jit billing-export.csv
|
|
131
|
+
cloudsealed-jit billing-export.csv --json > findings.json
|
|
132
|
+
cat export.csv | cloudsealed-jit - --type cost-forecast
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
### Library
|
|
136
|
+
|
|
137
|
+
```python
|
|
138
|
+
from cloudsealed_jit import parse_billing_csv, analyze
|
|
139
|
+
|
|
140
|
+
series = parse_billing_csv(open("export.csv").read())
|
|
141
|
+
result = analyze(series)
|
|
142
|
+
|
|
143
|
+
print(result.metrics.wastePercentage)
|
|
144
|
+
for r in result.recommendations:
|
|
145
|
+
print(r.title, r.potentialSavings)
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
### HTTP service
|
|
149
|
+
|
|
150
|
+
```bash
|
|
151
|
+
docker run -p 8091:8091 cloudsealed/jit-optimization-engine
|
|
152
|
+
```
|
|
153
|
+
|
|
154
|
+
```
|
|
155
|
+
GET /health
|
|
156
|
+
POST /v1/analyze-billing
|
|
157
|
+
```
|
|
158
|
+
|
|
159
|
+
```bash
|
|
160
|
+
curl -X POST localhost:8091/v1/analyze-billing \
|
|
161
|
+
-H 'Content-Type: application/json' \
|
|
162
|
+
-d '{"companyName":"Acme","csvContent":"date,cost\n2026-01-01,100\n..."}'
|
|
163
|
+
```
|
|
164
|
+
|
|
165
|
+
Set `JIT_OPTIMIZATION_API_KEY` to require an `X-Api-Key` header. Set
|
|
166
|
+
`JIT_MAX_CSV_BYTES` to change the 64 MB upload ceiling.
|
|
167
|
+
|
|
168
|
+
Response shape:
|
|
169
|
+
|
|
170
|
+
```jsonc
|
|
171
|
+
{
|
|
172
|
+
"anomalies": [
|
|
173
|
+
{ "date": "2026-01-31", "expectedCost": 99.0, "actualCost": 500.0,
|
|
174
|
+
"deviation": 405.05, "zScore": 7.82, "severity": "CRITICAL",
|
|
175
|
+
"description": "Spend above the day-of-week baseline by USD 401.00 (405.1%)." }
|
|
176
|
+
],
|
|
177
|
+
"metrics": {
|
|
178
|
+
"averageDailyCost": 106.32,
|
|
179
|
+
"stdDeviation": 51.69,
|
|
180
|
+
"sharpeRatio": 2.06, // spend stability: mean / stddev of daily cost
|
|
181
|
+
"wastePercentage": 6.29 // share of total spend above the baseline
|
|
182
|
+
},
|
|
183
|
+
"recommendations": [
|
|
184
|
+
{ "title": "...", "description": "...", "potentialSavings": 200.5, "effort": "MEDIUM" }
|
|
185
|
+
],
|
|
186
|
+
"summary": "..."
|
|
187
|
+
}
|
|
188
|
+
```
|
|
189
|
+
|
|
190
|
+
`sharpeRatio` is a **spend stability ratio** — mean daily cost divided by its
|
|
191
|
+
standard deviation, the reciprocal of the coefficient of variation. Higher
|
|
192
|
+
means more predictable spend. It is named for the field in the consuming API
|
|
193
|
+
contract; it is not a risk-adjusted return.
|
|
194
|
+
|
|
195
|
+
## Supported exports
|
|
196
|
+
|
|
197
|
+
| Provider | Date column | Cost column |
|
|
198
|
+
|---|---|---|
|
|
199
|
+
| AWS Cost and Usage Report | `lineItem/UsageStartDate` | `lineItem/UnblendedCost` |
|
|
200
|
+
| GCP billing export | `usage_start_time` | `cost` |
|
|
201
|
+
| Azure cost export | `Date`, `UsageDateTime` | `Cost`, `CostInBillingCurrency` |
|
|
202
|
+
| Generic | heuristic | heuristic |
|
|
203
|
+
|
|
204
|
+
Line items are aggregated to calendar days. Days with no line items are
|
|
205
|
+
inserted as zero-spend days rather than skipped. Rows that cannot be parsed are
|
|
206
|
+
counted and reported in the summary rather than dropped silently.
|
|
207
|
+
|
|
208
|
+
## Development
|
|
209
|
+
|
|
210
|
+
```bash
|
|
211
|
+
pip install -e ".[jit,api,dev]"
|
|
212
|
+
pytest
|
|
213
|
+
```
|
|
214
|
+
|
|
215
|
+
The test suite builds synthetic exports whose correct answer is known in
|
|
216
|
+
advance — a known spike at a known date, a known weekend-idle service, a stable
|
|
217
|
+
series that must produce no findings — so the assertions test behaviour rather
|
|
218
|
+
than the current output.
|
|
219
|
+
|
|
220
|
+
## License
|
|
221
|
+
|
|
222
|
+
MIT. See [LICENSE](LICENSE).
|
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
# cloudsealed-jit
|
|
2
|
+
|
|
3
|
+
Detects structural waste in cloud billing exports.
|
|
4
|
+
|
|
5
|
+
Given a billing export from AWS, GCP or Azure, it models what each day *should*
|
|
6
|
+
have cost, reports the days that did not match, and turns the excess into a
|
|
7
|
+
monthly figure. It is a library, a CLI and an HTTP service.
|
|
8
|
+
|
|
9
|
+
[](LICENSE)
|
|
10
|
+
[](https://www.python.org/downloads/)
|
|
11
|
+
|
|
12
|
+
---
|
|
13
|
+
|
|
14
|
+
## The problem
|
|
15
|
+
|
|
16
|
+
Cloud cost anomaly detection is usually done by comparing each day against the
|
|
17
|
+
period average and flagging anything beyond two or three standard deviations.
|
|
18
|
+
On billing data that method fails in two specific ways.
|
|
19
|
+
|
|
20
|
+
**Standard deviation is inflated by the very spikes you are looking for.** A
|
|
21
|
+
handful of large anomalies raises σ enough to pull themselves back inside the
|
|
22
|
+
threshold, and to hide every smaller anomaly with them. This is the masking
|
|
23
|
+
effect, and it gets worse as the anomalies get bigger.
|
|
24
|
+
|
|
25
|
+
**A flat average ignores the weekly cycle.** Most cloud bills have a pronounced
|
|
26
|
+
weekday/weekend shape. Measured against a flat mean, ordinary Mondays look like
|
|
27
|
+
overspend and ordinary Sundays look like savings.
|
|
28
|
+
|
|
29
|
+
## The method
|
|
30
|
+
|
|
31
|
+
**Baseline.** Expected spend for a day is a level term times a weekday term:
|
|
32
|
+
|
|
33
|
+
```
|
|
34
|
+
expected[i] = rolling_median(cost, 7)[i] × dow_factor[weekday(i)]
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
The rolling median follows growth and step changes without being dragged by
|
|
38
|
+
spikes. The weekday factor is the median ratio of observed spend to the level
|
|
39
|
+
term for that weekday. It is only estimated with at least two full weeks of
|
|
40
|
+
data; below that every factor is 1.0.
|
|
41
|
+
|
|
42
|
+
**Scoring.** Residuals are scored with a modified z-score built on the median
|
|
43
|
+
absolute deviation:
|
|
44
|
+
|
|
45
|
+
```
|
|
46
|
+
z = 0.6745 × (x − baseline) / MAD
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
The 0.6745 constant makes MAD a consistent estimator of σ for normal data, so
|
|
50
|
+
the score keeps the familiar "number of deviations" reading while tolerating
|
|
51
|
+
contamination in roughly half the sample. Days at or above |z| = 3.5 are
|
|
52
|
+
reported — the threshold recommended by Iglewicz & Hoaglin (1993).
|
|
53
|
+
|
|
54
|
+
**Waste.** Only positive excess counts. Waste percentage is the share of total
|
|
55
|
+
spend sitting above the baseline on anomalous days, which converts directly to
|
|
56
|
+
currency instead of being a count of unusual days.
|
|
57
|
+
|
|
58
|
+
**Recommendations.** Each carries a figure derived from the series itself,
|
|
59
|
+
normalised to 30 days, and states its assumption in the description. Estimates
|
|
60
|
+
that depend on facts the analyser cannot observe — whether a workload is
|
|
61
|
+
production, whether a commitment is acceptable — are labelled conditional
|
|
62
|
+
rather than presented as findings.
|
|
63
|
+
|
|
64
|
+
## Does it actually work better?
|
|
65
|
+
|
|
66
|
+
Yes, and it is measured, not asserted. `benchmarks/masking_benchmark.py` builds
|
|
67
|
+
synthetic bills whose anomalies are known by construction and scores this
|
|
68
|
+
method against the textbook mean+standard-deviation approach:
|
|
69
|
+
|
|
70
|
+
| scenario | textbook F1 | this method F1 |
|
|
71
|
+
|---|---|---|
|
|
72
|
+
| masking (scale estimator) | 0.667 | **0.923** |
|
|
73
|
+
| seasonality (baseline) | 0.667 | **1.000** |
|
|
74
|
+
| end-to-end | 0.667 | **1.000** |
|
|
75
|
+
|
|
76
|
+
Full derivation and reproduction steps in [METHODOLOGY.md](METHODOLOGY.md); the
|
|
77
|
+
design of the codebase is in [architecture.md](architecture.md). The benchmark
|
|
78
|
+
runs in CI (`--check`) and fails the build if the advantage ever regresses.
|
|
79
|
+
|
|
80
|
+
## Install
|
|
81
|
+
|
|
82
|
+
```bash
|
|
83
|
+
pip install cloudsealed-jit # library + CLI
|
|
84
|
+
pip install "cloudsealed-jit[jit]" # + numba-compiled kernels
|
|
85
|
+
pip install "cloudsealed-jit[jit,api]" # + HTTP service
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
`numba` is optional. Without it the kernels run on pure NumPy and the results
|
|
89
|
+
are identical; only large inputs get slower.
|
|
90
|
+
|
|
91
|
+
## Use
|
|
92
|
+
|
|
93
|
+
### CLI
|
|
94
|
+
|
|
95
|
+
```bash
|
|
96
|
+
cloudsealed-jit billing-export.csv
|
|
97
|
+
cloudsealed-jit billing-export.csv --json > findings.json
|
|
98
|
+
cat export.csv | cloudsealed-jit - --type cost-forecast
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
### Library
|
|
102
|
+
|
|
103
|
+
```python
|
|
104
|
+
from cloudsealed_jit import parse_billing_csv, analyze
|
|
105
|
+
|
|
106
|
+
series = parse_billing_csv(open("export.csv").read())
|
|
107
|
+
result = analyze(series)
|
|
108
|
+
|
|
109
|
+
print(result.metrics.wastePercentage)
|
|
110
|
+
for r in result.recommendations:
|
|
111
|
+
print(r.title, r.potentialSavings)
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
### HTTP service
|
|
115
|
+
|
|
116
|
+
```bash
|
|
117
|
+
docker run -p 8091:8091 cloudsealed/jit-optimization-engine
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
```
|
|
121
|
+
GET /health
|
|
122
|
+
POST /v1/analyze-billing
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
```bash
|
|
126
|
+
curl -X POST localhost:8091/v1/analyze-billing \
|
|
127
|
+
-H 'Content-Type: application/json' \
|
|
128
|
+
-d '{"companyName":"Acme","csvContent":"date,cost\n2026-01-01,100\n..."}'
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
Set `JIT_OPTIMIZATION_API_KEY` to require an `X-Api-Key` header. Set
|
|
132
|
+
`JIT_MAX_CSV_BYTES` to change the 64 MB upload ceiling.
|
|
133
|
+
|
|
134
|
+
Response shape:
|
|
135
|
+
|
|
136
|
+
```jsonc
|
|
137
|
+
{
|
|
138
|
+
"anomalies": [
|
|
139
|
+
{ "date": "2026-01-31", "expectedCost": 99.0, "actualCost": 500.0,
|
|
140
|
+
"deviation": 405.05, "zScore": 7.82, "severity": "CRITICAL",
|
|
141
|
+
"description": "Spend above the day-of-week baseline by USD 401.00 (405.1%)." }
|
|
142
|
+
],
|
|
143
|
+
"metrics": {
|
|
144
|
+
"averageDailyCost": 106.32,
|
|
145
|
+
"stdDeviation": 51.69,
|
|
146
|
+
"sharpeRatio": 2.06, // spend stability: mean / stddev of daily cost
|
|
147
|
+
"wastePercentage": 6.29 // share of total spend above the baseline
|
|
148
|
+
},
|
|
149
|
+
"recommendations": [
|
|
150
|
+
{ "title": "...", "description": "...", "potentialSavings": 200.5, "effort": "MEDIUM" }
|
|
151
|
+
],
|
|
152
|
+
"summary": "..."
|
|
153
|
+
}
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
`sharpeRatio` is a **spend stability ratio** — mean daily cost divided by its
|
|
157
|
+
standard deviation, the reciprocal of the coefficient of variation. Higher
|
|
158
|
+
means more predictable spend. It is named for the field in the consuming API
|
|
159
|
+
contract; it is not a risk-adjusted return.
|
|
160
|
+
|
|
161
|
+
## Supported exports
|
|
162
|
+
|
|
163
|
+
| Provider | Date column | Cost column |
|
|
164
|
+
|---|---|---|
|
|
165
|
+
| AWS Cost and Usage Report | `lineItem/UsageStartDate` | `lineItem/UnblendedCost` |
|
|
166
|
+
| GCP billing export | `usage_start_time` | `cost` |
|
|
167
|
+
| Azure cost export | `Date`, `UsageDateTime` | `Cost`, `CostInBillingCurrency` |
|
|
168
|
+
| Generic | heuristic | heuristic |
|
|
169
|
+
|
|
170
|
+
Line items are aggregated to calendar days. Days with no line items are
|
|
171
|
+
inserted as zero-spend days rather than skipped. Rows that cannot be parsed are
|
|
172
|
+
counted and reported in the summary rather than dropped silently.
|
|
173
|
+
|
|
174
|
+
## Development
|
|
175
|
+
|
|
176
|
+
```bash
|
|
177
|
+
pip install -e ".[jit,api,dev]"
|
|
178
|
+
pytest
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
The test suite builds synthetic exports whose correct answer is known in
|
|
182
|
+
advance — a known spike at a known date, a known weekend-idle service, a stable
|
|
183
|
+
series that must produce no findings — so the assertions test behaviour rather
|
|
184
|
+
than the current output.
|
|
185
|
+
|
|
186
|
+
## License
|
|
187
|
+
|
|
188
|
+
MIT. See [LICENSE](LICENSE).
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""
|
|
2
|
+
cloudsealed-jit — cloud billing waste analysis.
|
|
3
|
+
|
|
4
|
+
Detects structural cost waste in cloud billing exports by modelling a robust
|
|
5
|
+
day-of-week aware baseline and measuring the excess spend above it.
|
|
6
|
+
|
|
7
|
+
Public API:
|
|
8
|
+
parse_billing_csv(csv_text) -> BillingSeries
|
|
9
|
+
analyze(series, analysis_type="waste-audit") -> AnalysisResult
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from .parsing import BillingSeries, ParseError, parse_billing_csv
|
|
13
|
+
from .analysis import AnalysisResult, analyze
|
|
14
|
+
|
|
15
|
+
__version__ = "0.2.0"
|
|
16
|
+
|
|
17
|
+
__all__ = [
|
|
18
|
+
"BillingSeries",
|
|
19
|
+
"ParseError",
|
|
20
|
+
"parse_billing_csv",
|
|
21
|
+
"AnalysisResult",
|
|
22
|
+
"analyze",
|
|
23
|
+
"__version__",
|
|
24
|
+
]
|