focus-data-toolkit 0.11.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- focus_data_toolkit/__init__.py +69 -0
- focus_data_toolkit/__main__.py +6 -0
- focus_data_toolkit/_version.py +8 -0
- focus_data_toolkit/cli.py +968 -0
- focus_data_toolkit/context/__init__.py +88 -0
- focus_data_toolkit/context/billing.py +54 -0
- focus_data_toolkit/context/provider.py +90 -0
- focus_data_toolkit/convert/__init__.py +708 -0
- focus_data_toolkit/convert/billing_period.py +65 -0
- focus_data_toolkit/convert/contract_applied.py +235 -0
- focus_data_toolkit/convert/contract_commitment.py +182 -0
- focus_data_toolkit/convert/cost_and_usage.py +179 -0
- focus_data_toolkit/convert/detect.py +39 -0
- focus_data_toolkit/convert/invoice_detail.py +199 -0
- focus_data_toolkit/convert/streaming.py +1030 -0
- focus_data_toolkit/errors.py +145 -0
- focus_data_toolkit/focus_json.py +68 -0
- focus_data_toolkit/generators/__init__.py +61 -0
- focus_data_toolkit/generators/_shim.py +43 -0
- focus_data_toolkit/generators/engine/__init__.py +14 -0
- focus_data_toolkit/generators/engine/context.py +12 -0
- focus_data_toolkit/generators/engine/determinism.py +117 -0
- focus_data_toolkit/generators/engine/json_focus.py +63 -0
- focus_data_toolkit/generators/engine/ladder.py +71 -0
- focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
- focus_data_toolkit/generators/engine/serialize.py +151 -0
- focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
- focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
- focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
- focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
- focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
- focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
- focus_data_toolkit/generators/providers/__init__.py +29 -0
- focus_data_toolkit/generators/providers/aws.py +186 -0
- focus_data_toolkit/generators/providers/azure.py +191 -0
- focus_data_toolkit/generators/providers/gcp.py +194 -0
- focus_data_toolkit/generators/providers/profile.py +123 -0
- focus_data_toolkit/generators/scenarios.py +178 -0
- focus_data_toolkit/generators/versions/__init__.py +17 -0
- focus_data_toolkit/generators/versions/adapter.py +41 -0
- focus_data_toolkit/generators/versions/v1_2.py +111 -0
- focus_data_toolkit/generators/versions/v1_3.py +154 -0
- focus_data_toolkit/io/__init__.py +1 -0
- focus_data_toolkit/io/atomic_writer.py +462 -0
- focus_data_toolkit/io/csv_io.py +128 -0
- focus_data_toolkit/io/parquet_io.py +528 -0
- focus_data_toolkit/io/records.py +92 -0
- focus_data_toolkit/io/row_source.py +117 -0
- focus_data_toolkit/lifecycle.py +342 -0
- focus_data_toolkit/manifest.py +114 -0
- focus_data_toolkit/model/__init__.py +43 -0
- focus_data_toolkit/model/capabilities.py +66 -0
- focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
- focus_data_toolkit/model/focus_1_4_model.json +1913 -0
- focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
- focus_data_toolkit/model/focus_json_keys.py +112 -0
- focus_data_toolkit/model/iso_4217_currencies.json +23 -0
- focus_data_toolkit/model/json_schema_check.py +205 -0
- focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
- focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
- focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
- focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
- focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
- focus_data_toolkit/model/model_provenance.json +58 -0
- focus_data_toolkit/model/validator.py +498 -0
- focus_data_toolkit/modes.py +18 -0
- focus_data_toolkit/official_validator.py +61 -0
- focus_data_toolkit/progress.py +89 -0
- focus_data_toolkit/provenance.py +106 -0
- focus_data_toolkit/py.typed +1 -0
- focus_data_toolkit/runtime.py +243 -0
- focus_data_toolkit/schema/__init__.py +17 -0
- focus_data_toolkit/schema/detection.py +274 -0
- focus_data_toolkit/schema/registry.py +127 -0
- focus_data_toolkit/storage/__init__.py +1 -0
- focus_data_toolkit/storage/external_index.py +99 -0
- focus_data_toolkit/storage/spill.py +150 -0
- focus_data_toolkit/studio/__init__.py +19 -0
- focus_data_toolkit/studio/app.py +467 -0
- focus_data_toolkit/studio/config.py +42 -0
- focus_data_toolkit/studio/frontend/app.js +214 -0
- focus_data_toolkit/studio/frontend/index.html +101 -0
- focus_data_toolkit/studio/frontend/style.css +60 -0
- focus_data_toolkit/studio/jobs.py +142 -0
- focus_data_toolkit/studio/preview.py +32 -0
- focus_data_toolkit/studio/security.py +125 -0
- focus_data_toolkit/studio/server.py +71 -0
- focus_data_toolkit/supplement/__init__.py +50 -0
- focus_data_toolkit/supplement/adapters/__init__.py +21 -0
- focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
- focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
- focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
- focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
- focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
- focus_data_toolkit/supplement/adapters/registry.py +215 -0
- focus_data_toolkit/supplement/apply.py +318 -0
- focus_data_toolkit/supplement/gaps.py +219 -0
- focus_data_toolkit/supplement/kinds.py +118 -0
- focus_data_toolkit/supplement/loader.py +409 -0
- focus_data_toolkit/supplement/spec.py +74 -0
- focus_data_toolkit/supplement/validate.py +215 -0
- focus_data_toolkit/validate/__init__.py +15 -0
- focus_data_toolkit/validate/allocation.py +333 -0
- focus_data_toolkit/validate/bundle.py +254 -0
- focus_data_toolkit/validate/codes.py +93 -0
- focus_data_toolkit/validate/corrections.py +245 -0
- focus_data_toolkit/validate/reconciliation.py +98 -0
- focus_data_toolkit/validate/referential.py +289 -0
- focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
- focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
- focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
- focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
- focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,215 @@
|
|
|
1
|
+
"""Cross-validate loaded supplements against the source and the FOCUS model.
|
|
2
|
+
|
|
3
|
+
Diagnostics (``FDT-SUPP-0xx``): ERRORs block use of the bundle, WARNINGs don't, and the
|
|
4
|
+
coverage report (``FDT-SUPP-010``, INFO) is what drives strict gating — a blocking
|
|
5
|
+
column flips to ``ENRICHED`` only at 100 % key coverage.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from collections.abc import Mapping, Sequence
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
from decimal import Decimal, InvalidOperation
|
|
13
|
+
|
|
14
|
+
from focus_data_toolkit.convert.invoice_detail import (
|
|
15
|
+
_COST_QUANTUM,
|
|
16
|
+
GrainKey,
|
|
17
|
+
invoice_detail_grain_key,
|
|
18
|
+
)
|
|
19
|
+
from focus_data_toolkit.errors import Diagnostic, Severity
|
|
20
|
+
from focus_data_toolkit.model.validator import check_column_value
|
|
21
|
+
from focus_data_toolkit.supplement.loader import JoinKey, SupplementBundle, SupplementTable
|
|
22
|
+
|
|
23
|
+
_SAMPLE_CAP = 25
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass
|
|
27
|
+
class SourceKeySets:
|
|
28
|
+
"""The join keys the source actually derives, per supplement kind."""
|
|
29
|
+
|
|
30
|
+
billing_periods: set[JoinKey] = field(default_factory=set)
|
|
31
|
+
invoices: set[JoinKey] = field(default_factory=set)
|
|
32
|
+
invoice_grains: set[GrainKey] = field(default_factory=set)
|
|
33
|
+
contract_commitment_ids: set[JoinKey] = field(default_factory=set)
|
|
34
|
+
grain_billed: dict[GrainKey, Decimal] = field(default_factory=dict)
|
|
35
|
+
|
|
36
|
+
def keys_for(self, kind_name: str) -> set[JoinKey]:
|
|
37
|
+
return {
|
|
38
|
+
"billing_period": self.billing_periods,
|
|
39
|
+
"invoice": self.invoices,
|
|
40
|
+
"invoice_line": self.invoice_grains,
|
|
41
|
+
"contract_commitment": self.contract_commitment_ids,
|
|
42
|
+
}[kind_name]
|
|
43
|
+
|
|
44
|
+
def observe_cau_row(self, row: Mapping[str, str]) -> None:
|
|
45
|
+
"""Accumulate the keys of one Cost and Usage row (shared eager/streaming)."""
|
|
46
|
+
start = (row.get("BillingPeriodStart") or "").strip()
|
|
47
|
+
end = (row.get("BillingPeriodEnd") or "").strip()
|
|
48
|
+
issuer = (row.get("InvoiceIssuerName") or "").strip()
|
|
49
|
+
if start and end:
|
|
50
|
+
self.billing_periods.add((issuer, start, end))
|
|
51
|
+
grain = invoice_detail_grain_key(row)
|
|
52
|
+
if grain[1]: # InvoiceId present
|
|
53
|
+
self.invoice_grains.add(grain)
|
|
54
|
+
self.invoices.add((grain[0], grain[1]))
|
|
55
|
+
try:
|
|
56
|
+
cost = Decimal((row.get("BilledCost") or "0").strip() or "0")
|
|
57
|
+
except InvalidOperation:
|
|
58
|
+
cost = Decimal(0)
|
|
59
|
+
self.grain_billed[grain] = self.grain_billed.get(grain, Decimal(0)) + cost
|
|
60
|
+
|
|
61
|
+
def observe_cc_row(self, row: Mapping[str, str]) -> None:
|
|
62
|
+
cc_id = (row.get("ContractCommitmentId") or "").strip()
|
|
63
|
+
if cc_id:
|
|
64
|
+
self.contract_commitment_ids.add((cc_id,))
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def source_key_sets(
|
|
68
|
+
cau_rows: Sequence[Mapping[str, str]],
|
|
69
|
+
cc_rows: Sequence[Mapping[str, str]] | None = None,
|
|
70
|
+
) -> SourceKeySets:
|
|
71
|
+
keys = SourceKeySets()
|
|
72
|
+
for row in cau_rows:
|
|
73
|
+
keys.observe_cau_row(row)
|
|
74
|
+
for row in cc_rows or ():
|
|
75
|
+
keys.observe_cc_row(row)
|
|
76
|
+
return keys
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
@dataclass(frozen=True)
|
|
80
|
+
class ColumnCoverage:
|
|
81
|
+
"""How many source keys have a non-empty supplied value for one column."""
|
|
82
|
+
|
|
83
|
+
total_keys: int
|
|
84
|
+
covered: int
|
|
85
|
+
|
|
86
|
+
@property
|
|
87
|
+
def complete(self) -> bool:
|
|
88
|
+
return self.total_keys > 0 and self.covered == self.total_keys
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def coverage(table: SupplementTable, source_keys: set[JoinKey]) -> dict[str, ColumnCoverage]:
|
|
92
|
+
"""Per fact column: how many of the source's keys this table covers."""
|
|
93
|
+
total = len(source_keys)
|
|
94
|
+
out: dict[str, ColumnCoverage] = {}
|
|
95
|
+
for column in table.fact_columns:
|
|
96
|
+
covered = sum(1 for key in source_keys if table.value(key, column))
|
|
97
|
+
out[column] = ColumnCoverage(total_keys=total, covered=covered)
|
|
98
|
+
return out
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _sample(keys: Sequence[JoinKey]) -> str:
|
|
102
|
+
return "; ".join("|".join(k) for k in list(keys)[:_SAMPLE_CAP])
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def validate_supplements(bundle: SupplementBundle, source: SourceKeySets) -> list[Diagnostic]:
|
|
106
|
+
"""All supplement diagnostics: structural + values + joins + coverage."""
|
|
107
|
+
diagnostics = bundle.structural_diagnostics()
|
|
108
|
+
for name in sorted(bundle.tables):
|
|
109
|
+
table = bundle.tables[name]
|
|
110
|
+
source_keys = source.keys_for(name)
|
|
111
|
+
|
|
112
|
+
# FDT-SUPP-004 — supplied values must obey the model's format rules.
|
|
113
|
+
bad: dict[str, list[str]] = {}
|
|
114
|
+
for key, facts in table.rows.items():
|
|
115
|
+
for column, value in facts.items():
|
|
116
|
+
if not value:
|
|
117
|
+
continue
|
|
118
|
+
rule = check_column_value(table.kind.target_dataset, column, value)
|
|
119
|
+
if rule:
|
|
120
|
+
bad.setdefault(f"{column}:{rule}", []).append("|".join(key))
|
|
121
|
+
for column_rule, keys in sorted(bad.items()):
|
|
122
|
+
column, _, rule = column_rule.partition(":")
|
|
123
|
+
diagnostics.append(
|
|
124
|
+
Diagnostic(
|
|
125
|
+
code="FDT-SUPP-004",
|
|
126
|
+
severity=Severity.ERROR,
|
|
127
|
+
message=f"supplement value(s) for {column} violate the model rule {rule!r}",
|
|
128
|
+
datasets=(table.kind.target_dataset,),
|
|
129
|
+
file=str(table.path),
|
|
130
|
+
context={
|
|
131
|
+
"kind": name,
|
|
132
|
+
"column": column,
|
|
133
|
+
"rule": rule,
|
|
134
|
+
"row_count": str(len(keys)),
|
|
135
|
+
"sample": "; ".join(keys[:_SAMPLE_CAP]),
|
|
136
|
+
},
|
|
137
|
+
)
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
# FDT-SUPP-005 — orphan supplement rows (key never derived from the source).
|
|
141
|
+
orphans = sorted(set(table.rows) - source_keys)
|
|
142
|
+
if orphans:
|
|
143
|
+
diagnostics.append(
|
|
144
|
+
Diagnostic(
|
|
145
|
+
code="FDT-SUPP-005",
|
|
146
|
+
severity=Severity.WARNING,
|
|
147
|
+
message=f"{len(orphans)} supplement row(s) match nothing in the source "
|
|
148
|
+
"(often a wider export period; they are ignored)",
|
|
149
|
+
datasets=(table.kind.target_dataset,),
|
|
150
|
+
file=str(table.path),
|
|
151
|
+
context={"kind": name, "sample": _sample(orphans)},
|
|
152
|
+
)
|
|
153
|
+
)
|
|
154
|
+
|
|
155
|
+
# FDT-SUPP-006 — invoice_line BilledCost must reconcile with the derived sums.
|
|
156
|
+
if name == "invoice_line" and "BilledCost" in table.fact_columns and source.grain_billed:
|
|
157
|
+
conflicts: list[JoinKey] = []
|
|
158
|
+
for key in sorted(set(table.rows) & source_keys):
|
|
159
|
+
supplied = table.value(key, "BilledCost")
|
|
160
|
+
if not supplied:
|
|
161
|
+
continue
|
|
162
|
+
try:
|
|
163
|
+
supplied_cost = Decimal(supplied)
|
|
164
|
+
except InvalidOperation:
|
|
165
|
+
continue # already reported by FDT-SUPP-004
|
|
166
|
+
derived = source.grain_billed.get(key, Decimal(0))
|
|
167
|
+
if supplied_cost.quantize(_COST_QUANTUM) != derived.quantize(_COST_QUANTUM):
|
|
168
|
+
conflicts.append(key)
|
|
169
|
+
if conflicts:
|
|
170
|
+
diagnostics.append(
|
|
171
|
+
Diagnostic(
|
|
172
|
+
code="FDT-SUPP-006",
|
|
173
|
+
severity=Severity.ERROR,
|
|
174
|
+
message="supplement BilledCost conflicts with the Cost and Usage "
|
|
175
|
+
"grain sums; the supplement does not describe this source",
|
|
176
|
+
datasets=(table.kind.target_dataset,),
|
|
177
|
+
file=str(table.path),
|
|
178
|
+
context={
|
|
179
|
+
"kind": name,
|
|
180
|
+
"row_count": str(len(conflicts)),
|
|
181
|
+
"sample": _sample(conflicts),
|
|
182
|
+
},
|
|
183
|
+
)
|
|
184
|
+
)
|
|
185
|
+
|
|
186
|
+
# FDT-SUPP-010 — coverage per column (drives strict gating; INFO, never an error).
|
|
187
|
+
for column, cov in sorted(coverage(table, source_keys).items()):
|
|
188
|
+
if column == "BilledCost" and name == "invoice_line":
|
|
189
|
+
continue # reconciliation-only, never applied
|
|
190
|
+
if cov.covered < cov.total_keys:
|
|
191
|
+
missing = sorted(
|
|
192
|
+
k for k in source_keys if not table.value(k, column)
|
|
193
|
+
)
|
|
194
|
+
diagnostics.append(
|
|
195
|
+
Diagnostic(
|
|
196
|
+
code="FDT-SUPP-010",
|
|
197
|
+
severity=Severity.INFO,
|
|
198
|
+
message=f"partial coverage for {column}: "
|
|
199
|
+
f"{cov.covered}/{cov.total_keys} source key(s) supplied",
|
|
200
|
+
datasets=(table.kind.target_dataset,),
|
|
201
|
+
file=str(table.path),
|
|
202
|
+
context={
|
|
203
|
+
"kind": name,
|
|
204
|
+
"column": column,
|
|
205
|
+
"covered": str(cov.covered),
|
|
206
|
+
"total": str(cov.total_keys),
|
|
207
|
+
"missing_sample": _sample(missing),
|
|
208
|
+
},
|
|
209
|
+
)
|
|
210
|
+
)
|
|
211
|
+
return diagnostics
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def has_blocking_errors(diagnostics: Sequence[Diagnostic]) -> bool:
|
|
215
|
+
return any(d.severity is Severity.ERROR for d in diagnostics)
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""Cross-dataset validation layer (P1.4 / P1.8 / P1.9).
|
|
2
|
+
|
|
3
|
+
Distinct from the per-dataset linter in ``model/validator.py``: this layer validates a bundle
|
|
4
|
+
of datasets against each other.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from focus_data_toolkit.validate.bundle import (
|
|
10
|
+
Bundle,
|
|
11
|
+
BundleReport,
|
|
12
|
+
validate_dataset_bundle,
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
__all__ = ["Bundle", "BundleReport", "validate_dataset_bundle"]
|
|
@@ -0,0 +1,333 @@
|
|
|
1
|
+
"""Validate Split Cost Allocation groups within a Cost and Usage dataset.
|
|
2
|
+
|
|
3
|
+
An allocation *group* redistributes one origin charge across several consumers. FOCUS carries
|
|
4
|
+
the per-line ratio inside ``AllocatedMethodDetails`` (an ``Elements`` array of
|
|
5
|
+
``{AllocatedRatio, UsageUnit, UsageQuantity}``). To make a group's origin identifiable and its
|
|
6
|
+
origin cost checkable, rows are tied together by the toolkit extension columns
|
|
7
|
+
``x_SplitOriginId`` (a stable origin-charge key) and ``x_SplitOriginCost`` (the origin amount,
|
|
8
|
+
identical across the group). Rows without ``x_SplitOriginId`` are not allocation rows and are
|
|
9
|
+
ignored, so the check is a no-op on datasets that do not use split cost allocation.
|
|
10
|
+
|
|
11
|
+
Per group it checks: ratios sum to 1 (within tolerance) and each ratio is in [0, 1]; allocated
|
|
12
|
+
costs sum to the origin cost (within tolerance); a single consistent method and unit; unique
|
|
13
|
+
allocated resources; and that every row carries the information the group needs.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import json
|
|
19
|
+
from collections.abc import Callable, Iterable, Mapping, MutableMapping, Sequence
|
|
20
|
+
from dataclasses import dataclass, field
|
|
21
|
+
from decimal import Decimal, InvalidOperation
|
|
22
|
+
|
|
23
|
+
from focus_data_toolkit.errors import Diagnostic, Severity
|
|
24
|
+
|
|
25
|
+
Rows = Sequence[Mapping[str, str]]
|
|
26
|
+
#: The input is consumed in a single forward pass, so any iterable of rows works.
|
|
27
|
+
RowStream = Iterable[Mapping[str, str]]
|
|
28
|
+
#: Factory for per-key lookup state (``dict`` by default; a spillable map for streaming).
|
|
29
|
+
IndexFactory = Callable[[], MutableMapping[str, str]]
|
|
30
|
+
|
|
31
|
+
ORIGIN_ID_COLUMN = "x_SplitOriginId"
|
|
32
|
+
ORIGIN_COST_COLUMN = "x_SplitOriginCost"
|
|
33
|
+
|
|
34
|
+
DEFAULT_RATIO_TOLERANCE = Decimal("0.0001")
|
|
35
|
+
DEFAULT_COST_TOLERANCE = Decimal("0.01")
|
|
36
|
+
|
|
37
|
+
# Line numbers recorded per group for the incomplete-group diagnostic's ``rows`` context;
|
|
38
|
+
# beyond this the context reports the overflow as ``+N more`` instead of growing unboundedly.
|
|
39
|
+
_LINE_SAMPLE = 100
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _dec(value: str | None) -> Decimal | None:
|
|
43
|
+
try:
|
|
44
|
+
parsed = Decimal((value or "").strip())
|
|
45
|
+
except InvalidOperation:
|
|
46
|
+
return None
|
|
47
|
+
return parsed if parsed.is_finite() else None
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _row_ratio_and_units(row: Mapping[str, str]) -> tuple[Decimal | None, frozenset[str]]:
|
|
51
|
+
"""Extract (summed AllocatedRatio, set of UsageUnits) from a row's AllocatedMethodDetails.
|
|
52
|
+
|
|
53
|
+
Every element's ``UsageUnit`` is collected — not just the first — so a row mixing units
|
|
54
|
+
across its elements is caught by the group's single-unit check.
|
|
55
|
+
"""
|
|
56
|
+
text = (row.get("AllocatedMethodDetails") or "").strip()
|
|
57
|
+
if not text:
|
|
58
|
+
return None, frozenset()
|
|
59
|
+
try:
|
|
60
|
+
obj = json.loads(text, parse_float=Decimal, parse_int=Decimal)
|
|
61
|
+
except (ValueError, TypeError):
|
|
62
|
+
return None, frozenset()
|
|
63
|
+
elements = obj.get("Elements") if isinstance(obj, dict) else None
|
|
64
|
+
if not isinstance(elements, list) or not elements:
|
|
65
|
+
return None, frozenset()
|
|
66
|
+
ratio = Decimal(0)
|
|
67
|
+
units: set[str] = set()
|
|
68
|
+
for el in elements:
|
|
69
|
+
if not isinstance(el, dict):
|
|
70
|
+
return None, frozenset(units)
|
|
71
|
+
raw = el.get("AllocatedRatio")
|
|
72
|
+
if not isinstance(raw, Decimal) or not raw.is_finite():
|
|
73
|
+
return None, frozenset(units)
|
|
74
|
+
ratio += raw
|
|
75
|
+
unit = el.get("UsageUnit")
|
|
76
|
+
if unit:
|
|
77
|
+
units.add(unit)
|
|
78
|
+
return ratio, frozenset(units)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
@dataclass
|
|
82
|
+
class _GroupState:
|
|
83
|
+
"""Running aggregate of one allocation group — fixed-size scalars, never member rows.
|
|
84
|
+
|
|
85
|
+
Mirrors the checks of the original list-of-members implementation exactly: the first
|
|
86
|
+
incomplete condition (in row order) short-circuits further accumulation, line numbers
|
|
87
|
+
keep counting so the incomplete diagnostic still describes every member row. The state
|
|
88
|
+
is JSON round-trippable (:meth:`dump` / :meth:`load`), so it can live as a string value
|
|
89
|
+
in a disk-spilling map when the caller supplies an ``index_factory``.
|
|
90
|
+
"""
|
|
91
|
+
|
|
92
|
+
lines: list[int] = field(default_factory=list) # first _LINE_SAMPLE member lines
|
|
93
|
+
line_count: int = 0
|
|
94
|
+
incomplete: tuple[str, str | None] | None = None # (reason, column)
|
|
95
|
+
ratio_sum: Decimal = Decimal(0)
|
|
96
|
+
out_of_range: list[Decimal] = field(default_factory=list) # ratios outside [0, 1]
|
|
97
|
+
units: set[str] = field(default_factory=set)
|
|
98
|
+
methods: set[str] = field(default_factory=set)
|
|
99
|
+
resource_count: int = 0
|
|
100
|
+
duplicate_resource: bool = False
|
|
101
|
+
missing_resource: bool = False
|
|
102
|
+
total_cost: Decimal = Decimal(0)
|
|
103
|
+
origin_cost: Decimal | None = None
|
|
104
|
+
origin_disagrees: bool = False
|
|
105
|
+
|
|
106
|
+
def observe(self, line: int, row: Mapping[str, str], *, seen_resource: bool) -> None:
|
|
107
|
+
self.line_count += 1
|
|
108
|
+
if len(self.lines) < _LINE_SAMPLE:
|
|
109
|
+
self.lines.append(line)
|
|
110
|
+
if self.incomplete is not None:
|
|
111
|
+
return
|
|
112
|
+
ratio, row_units = _row_ratio_and_units(row)
|
|
113
|
+
if ratio is None:
|
|
114
|
+
self.incomplete = ("a row has no usable AllocatedRatio", "AllocatedMethodDetails")
|
|
115
|
+
return
|
|
116
|
+
self.ratio_sum += ratio
|
|
117
|
+
if ratio < 0 or ratio > 1:
|
|
118
|
+
self.out_of_range.append(ratio)
|
|
119
|
+
self.units |= row_units
|
|
120
|
+
self.methods.add((row.get("AllocatedMethodId") or "").strip())
|
|
121
|
+
resource = (row.get("AllocatedResourceId") or "").strip()
|
|
122
|
+
self.resource_count += 1
|
|
123
|
+
if seen_resource:
|
|
124
|
+
self.duplicate_resource = True
|
|
125
|
+
if not resource:
|
|
126
|
+
self.missing_resource = True
|
|
127
|
+
cost = _dec(row.get("BilledCost"))
|
|
128
|
+
if cost is None:
|
|
129
|
+
self.incomplete = ("a row has no numeric BilledCost", "BilledCost")
|
|
130
|
+
return
|
|
131
|
+
self.total_cost += cost
|
|
132
|
+
origin_cost = _dec(row.get(ORIGIN_COST_COLUMN))
|
|
133
|
+
if origin_cost is None:
|
|
134
|
+
self.incomplete = ("missing x_SplitOriginCost", ORIGIN_COST_COLUMN)
|
|
135
|
+
return
|
|
136
|
+
if self.origin_cost is None:
|
|
137
|
+
self.origin_cost = origin_cost
|
|
138
|
+
elif origin_cost != self.origin_cost:
|
|
139
|
+
self.origin_disagrees = True
|
|
140
|
+
|
|
141
|
+
def dump(self) -> str:
|
|
142
|
+
return json.dumps({
|
|
143
|
+
"ln": self.lines,
|
|
144
|
+
"lc": self.line_count,
|
|
145
|
+
"inc": list(self.incomplete) if self.incomplete else None,
|
|
146
|
+
"rs": str(self.ratio_sum),
|
|
147
|
+
"oor": [str(r) for r in self.out_of_range],
|
|
148
|
+
"un": sorted(self.units),
|
|
149
|
+
"me": sorted(self.methods),
|
|
150
|
+
"rc": self.resource_count,
|
|
151
|
+
"dup": self.duplicate_resource,
|
|
152
|
+
"mr": self.missing_resource,
|
|
153
|
+
"tc": str(self.total_cost),
|
|
154
|
+
"oc": str(self.origin_cost) if self.origin_cost is not None else None,
|
|
155
|
+
"od": self.origin_disagrees,
|
|
156
|
+
}, separators=(",", ":"))
|
|
157
|
+
|
|
158
|
+
@classmethod
|
|
159
|
+
def load(cls, text: str) -> _GroupState:
|
|
160
|
+
d = json.loads(text)
|
|
161
|
+
return cls(
|
|
162
|
+
lines=d["ln"],
|
|
163
|
+
line_count=d["lc"],
|
|
164
|
+
incomplete=tuple(d["inc"]) if d["inc"] else None,
|
|
165
|
+
ratio_sum=Decimal(d["rs"]),
|
|
166
|
+
out_of_range=[Decimal(r) for r in d["oor"]],
|
|
167
|
+
units=set(d["un"]),
|
|
168
|
+
methods=set(d["me"]),
|
|
169
|
+
resource_count=d["rc"],
|
|
170
|
+
duplicate_resource=d["dup"],
|
|
171
|
+
missing_resource=d["mr"],
|
|
172
|
+
total_cost=Decimal(d["tc"]),
|
|
173
|
+
origin_cost=Decimal(d["oc"]) if d["oc"] is not None else None,
|
|
174
|
+
origin_disagrees=d["od"],
|
|
175
|
+
)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def validate_split_allocation(
|
|
179
|
+
cost_and_usage: RowStream,
|
|
180
|
+
*,
|
|
181
|
+
ratio_tolerance: Decimal = DEFAULT_RATIO_TOLERANCE,
|
|
182
|
+
cost_tolerance: Decimal = DEFAULT_COST_TOLERANCE,
|
|
183
|
+
index_factory: IndexFactory = dict,
|
|
184
|
+
) -> list[Diagnostic]:
|
|
185
|
+
"""Validate every split-cost-allocation group in ``cost_and_usage`` (single pass).
|
|
186
|
+
|
|
187
|
+
Per-group state is a fixed-size JSON-serialised aggregate held in an ``index_factory``
|
|
188
|
+
map (as is the resource-duplicate lookup), so with a spillable factory memory stays
|
|
189
|
+
bounded even when allocation-group cardinality approaches the row count.
|
|
190
|
+
"""
|
|
191
|
+
groups = index_factory()
|
|
192
|
+
resources_seen = index_factory()
|
|
193
|
+
for i, row in enumerate(cost_and_usage, start=1):
|
|
194
|
+
origin = (row.get(ORIGIN_ID_COLUMN) or "").strip()
|
|
195
|
+
if not origin:
|
|
196
|
+
continue
|
|
197
|
+
raw = groups.get(origin)
|
|
198
|
+
state = _GroupState.load(raw) if raw is not None else _GroupState()
|
|
199
|
+
resource = (row.get("AllocatedResourceId") or "").strip()
|
|
200
|
+
resource_key = json.dumps([origin, resource], separators=(",", ":"))
|
|
201
|
+
state.observe(i, row, seen_resource=resource_key in resources_seen)
|
|
202
|
+
if state.incomplete is None:
|
|
203
|
+
resources_seen[resource_key] = ""
|
|
204
|
+
groups[origin] = state.dump()
|
|
205
|
+
|
|
206
|
+
out: list[Diagnostic] = []
|
|
207
|
+
for origin_id in sorted(groups):
|
|
208
|
+
state = _GroupState.load(groups[origin_id])
|
|
209
|
+
out.extend(_group_diagnostics(origin_id, state, ratio_tolerance, cost_tolerance))
|
|
210
|
+
return out
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def _group_diagnostics(
|
|
214
|
+
origin_id: str,
|
|
215
|
+
state: _GroupState,
|
|
216
|
+
ratio_tolerance: Decimal,
|
|
217
|
+
cost_tolerance: Decimal,
|
|
218
|
+
) -> list[Diagnostic]:
|
|
219
|
+
keys = {ORIGIN_ID_COLUMN: origin_id}
|
|
220
|
+
rows_context = ",".join(map(str, sorted(state.lines)))
|
|
221
|
+
if state.line_count > len(state.lines):
|
|
222
|
+
rows_context += f",+{state.line_count - len(state.lines)} more"
|
|
223
|
+
out: list[Diagnostic] = []
|
|
224
|
+
|
|
225
|
+
def incomplete(reason: str, column: str | None = None) -> Diagnostic:
|
|
226
|
+
return Diagnostic(
|
|
227
|
+
code="FDT-ALLOC-005",
|
|
228
|
+
severity=Severity.ERROR,
|
|
229
|
+
message=f"split allocation group {origin_id!r} is incomplete: {reason}",
|
|
230
|
+
datasets=("Cost and Usage",),
|
|
231
|
+
dataset="Cost and Usage",
|
|
232
|
+
column=column,
|
|
233
|
+
record_keys=keys,
|
|
234
|
+
context={"rows": rows_context},
|
|
235
|
+
)
|
|
236
|
+
|
|
237
|
+
if state.incomplete is not None:
|
|
238
|
+
return [incomplete(*state.incomplete)]
|
|
239
|
+
if "" in state.methods or state.missing_resource:
|
|
240
|
+
return [incomplete("a row is missing AllocatedMethodId or AllocatedResourceId")]
|
|
241
|
+
if state.origin_disagrees:
|
|
242
|
+
return [incomplete("rows disagree on x_SplitOriginCost", ORIGIN_COST_COLUMN)]
|
|
243
|
+
|
|
244
|
+
for ratio in state.out_of_range:
|
|
245
|
+
out.append(
|
|
246
|
+
Diagnostic(
|
|
247
|
+
code="FDT-ALLOC-006",
|
|
248
|
+
severity=Severity.ERROR,
|
|
249
|
+
message=f"allocation ratio {ratio} outside [0, 1] in group {origin_id!r}",
|
|
250
|
+
datasets=("Cost and Usage",),
|
|
251
|
+
dataset="Cost and Usage",
|
|
252
|
+
column="AllocatedMethodDetails",
|
|
253
|
+
actual=str(ratio),
|
|
254
|
+
record_keys=keys,
|
|
255
|
+
)
|
|
256
|
+
)
|
|
257
|
+
|
|
258
|
+
ratio_sum = state.ratio_sum
|
|
259
|
+
if abs(ratio_sum - Decimal(1)) > ratio_tolerance:
|
|
260
|
+
out.append(
|
|
261
|
+
Diagnostic(
|
|
262
|
+
code="FDT-ALLOC-001",
|
|
263
|
+
severity=Severity.ERROR,
|
|
264
|
+
message=f"allocation ratios in group {origin_id!r} sum to {ratio_sum}, not 1",
|
|
265
|
+
datasets=("Cost and Usage",),
|
|
266
|
+
dataset="Cost and Usage",
|
|
267
|
+
column="AllocatedMethodDetails",
|
|
268
|
+
expected="1",
|
|
269
|
+
actual=str(ratio_sum),
|
|
270
|
+
record_keys=keys,
|
|
271
|
+
)
|
|
272
|
+
)
|
|
273
|
+
|
|
274
|
+
origin_cost = state.origin_cost
|
|
275
|
+
assert origin_cost is not None # a complete group recorded one on every row
|
|
276
|
+
if abs(state.total_cost - origin_cost) > cost_tolerance:
|
|
277
|
+
out.append(
|
|
278
|
+
Diagnostic(
|
|
279
|
+
code="FDT-ALLOC-002",
|
|
280
|
+
severity=Severity.ERROR,
|
|
281
|
+
message=f"allocated costs in group {origin_id!r} sum to {state.total_cost}, "
|
|
282
|
+
f"not the origin cost {origin_cost}",
|
|
283
|
+
datasets=("Cost and Usage",),
|
|
284
|
+
dataset="Cost and Usage",
|
|
285
|
+
column="BilledCost",
|
|
286
|
+
expected=str(origin_cost),
|
|
287
|
+
actual=str(state.total_cost),
|
|
288
|
+
record_keys=keys,
|
|
289
|
+
)
|
|
290
|
+
)
|
|
291
|
+
|
|
292
|
+
if len({m for m in state.methods if m}) > 1:
|
|
293
|
+
out.append(
|
|
294
|
+
Diagnostic(
|
|
295
|
+
code="FDT-ALLOC-003",
|
|
296
|
+
severity=Severity.ERROR,
|
|
297
|
+
message=f"inconsistent AllocatedMethodId within group {origin_id!r}: "
|
|
298
|
+
f"{sorted(state.methods)}",
|
|
299
|
+
datasets=("Cost and Usage",),
|
|
300
|
+
dataset="Cost and Usage",
|
|
301
|
+
column="AllocatedMethodId",
|
|
302
|
+
record_keys=keys,
|
|
303
|
+
)
|
|
304
|
+
)
|
|
305
|
+
|
|
306
|
+
if len(state.units) > 1:
|
|
307
|
+
out.append(
|
|
308
|
+
Diagnostic(
|
|
309
|
+
code="FDT-ALLOC-007",
|
|
310
|
+
severity=Severity.ERROR,
|
|
311
|
+
message=f"inconsistent UsageUnit within group {origin_id!r}: "
|
|
312
|
+
f"{sorted(state.units)}",
|
|
313
|
+
datasets=("Cost and Usage",),
|
|
314
|
+
dataset="Cost and Usage",
|
|
315
|
+
column="AllocatedMethodDetails",
|
|
316
|
+
record_keys=keys,
|
|
317
|
+
)
|
|
318
|
+
)
|
|
319
|
+
|
|
320
|
+
if state.duplicate_resource:
|
|
321
|
+
out.append(
|
|
322
|
+
Diagnostic(
|
|
323
|
+
code="FDT-ALLOC-004",
|
|
324
|
+
severity=Severity.ERROR,
|
|
325
|
+
message=f"duplicate AllocatedResourceId within group {origin_id!r}",
|
|
326
|
+
datasets=("Cost and Usage",),
|
|
327
|
+
dataset="Cost and Usage",
|
|
328
|
+
column="AllocatedResourceId",
|
|
329
|
+
record_keys=keys,
|
|
330
|
+
)
|
|
331
|
+
)
|
|
332
|
+
|
|
333
|
+
return out
|