focus-data-toolkit 0.11.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- focus_data_toolkit/__init__.py +69 -0
- focus_data_toolkit/__main__.py +6 -0
- focus_data_toolkit/_version.py +8 -0
- focus_data_toolkit/cli.py +968 -0
- focus_data_toolkit/context/__init__.py +88 -0
- focus_data_toolkit/context/billing.py +54 -0
- focus_data_toolkit/context/provider.py +90 -0
- focus_data_toolkit/convert/__init__.py +708 -0
- focus_data_toolkit/convert/billing_period.py +65 -0
- focus_data_toolkit/convert/contract_applied.py +235 -0
- focus_data_toolkit/convert/contract_commitment.py +182 -0
- focus_data_toolkit/convert/cost_and_usage.py +179 -0
- focus_data_toolkit/convert/detect.py +39 -0
- focus_data_toolkit/convert/invoice_detail.py +199 -0
- focus_data_toolkit/convert/streaming.py +1030 -0
- focus_data_toolkit/errors.py +145 -0
- focus_data_toolkit/focus_json.py +68 -0
- focus_data_toolkit/generators/__init__.py +61 -0
- focus_data_toolkit/generators/_shim.py +43 -0
- focus_data_toolkit/generators/engine/__init__.py +14 -0
- focus_data_toolkit/generators/engine/context.py +12 -0
- focus_data_toolkit/generators/engine/determinism.py +117 -0
- focus_data_toolkit/generators/engine/json_focus.py +63 -0
- focus_data_toolkit/generators/engine/ladder.py +71 -0
- focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
- focus_data_toolkit/generators/engine/serialize.py +151 -0
- focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
- focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
- focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
- focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
- focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
- focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
- focus_data_toolkit/generators/providers/__init__.py +29 -0
- focus_data_toolkit/generators/providers/aws.py +186 -0
- focus_data_toolkit/generators/providers/azure.py +191 -0
- focus_data_toolkit/generators/providers/gcp.py +194 -0
- focus_data_toolkit/generators/providers/profile.py +123 -0
- focus_data_toolkit/generators/scenarios.py +178 -0
- focus_data_toolkit/generators/versions/__init__.py +17 -0
- focus_data_toolkit/generators/versions/adapter.py +41 -0
- focus_data_toolkit/generators/versions/v1_2.py +111 -0
- focus_data_toolkit/generators/versions/v1_3.py +154 -0
- focus_data_toolkit/io/__init__.py +1 -0
- focus_data_toolkit/io/atomic_writer.py +462 -0
- focus_data_toolkit/io/csv_io.py +128 -0
- focus_data_toolkit/io/parquet_io.py +528 -0
- focus_data_toolkit/io/records.py +92 -0
- focus_data_toolkit/io/row_source.py +117 -0
- focus_data_toolkit/lifecycle.py +342 -0
- focus_data_toolkit/manifest.py +114 -0
- focus_data_toolkit/model/__init__.py +43 -0
- focus_data_toolkit/model/capabilities.py +66 -0
- focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
- focus_data_toolkit/model/focus_1_4_model.json +1913 -0
- focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
- focus_data_toolkit/model/focus_json_keys.py +112 -0
- focus_data_toolkit/model/iso_4217_currencies.json +23 -0
- focus_data_toolkit/model/json_schema_check.py +205 -0
- focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
- focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
- focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
- focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
- focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
- focus_data_toolkit/model/model_provenance.json +58 -0
- focus_data_toolkit/model/validator.py +498 -0
- focus_data_toolkit/modes.py +18 -0
- focus_data_toolkit/official_validator.py +61 -0
- focus_data_toolkit/progress.py +89 -0
- focus_data_toolkit/provenance.py +106 -0
- focus_data_toolkit/py.typed +1 -0
- focus_data_toolkit/runtime.py +243 -0
- focus_data_toolkit/schema/__init__.py +17 -0
- focus_data_toolkit/schema/detection.py +274 -0
- focus_data_toolkit/schema/registry.py +127 -0
- focus_data_toolkit/storage/__init__.py +1 -0
- focus_data_toolkit/storage/external_index.py +99 -0
- focus_data_toolkit/storage/spill.py +150 -0
- focus_data_toolkit/studio/__init__.py +19 -0
- focus_data_toolkit/studio/app.py +467 -0
- focus_data_toolkit/studio/config.py +42 -0
- focus_data_toolkit/studio/frontend/app.js +214 -0
- focus_data_toolkit/studio/frontend/index.html +101 -0
- focus_data_toolkit/studio/frontend/style.css +60 -0
- focus_data_toolkit/studio/jobs.py +142 -0
- focus_data_toolkit/studio/preview.py +32 -0
- focus_data_toolkit/studio/security.py +125 -0
- focus_data_toolkit/studio/server.py +71 -0
- focus_data_toolkit/supplement/__init__.py +50 -0
- focus_data_toolkit/supplement/adapters/__init__.py +21 -0
- focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
- focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
- focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
- focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
- focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
- focus_data_toolkit/supplement/adapters/registry.py +215 -0
- focus_data_toolkit/supplement/apply.py +318 -0
- focus_data_toolkit/supplement/gaps.py +219 -0
- focus_data_toolkit/supplement/kinds.py +118 -0
- focus_data_toolkit/supplement/loader.py +409 -0
- focus_data_toolkit/supplement/spec.py +74 -0
- focus_data_toolkit/supplement/validate.py +215 -0
- focus_data_toolkit/validate/__init__.py +15 -0
- focus_data_toolkit/validate/allocation.py +333 -0
- focus_data_toolkit/validate/bundle.py +254 -0
- focus_data_toolkit/validate/codes.py +93 -0
- focus_data_toolkit/validate/corrections.py +245 -0
- focus_data_toolkit/validate/reconciliation.py +98 -0
- focus_data_toolkit/validate/referential.py +289 -0
- focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
- focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
- focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
- focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
- focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,254 @@
|
|
|
1
|
+
"""Validate a *bundle* of FOCUS datasets together (P1.4).
|
|
2
|
+
|
|
3
|
+
This layer is deliberately separate from the per-dataset linter (``model/validator.py``): it
|
|
4
|
+
asserts the cross-dataset guarantees the linter explicitly does not — referential integrity,
|
|
5
|
+
uniqueness, currency/period/issuer coherence, reconciliation, split cost allocation, and
|
|
6
|
+
commitment lifecycle. ``validate_dataset_bundle`` returns a :class:`BundleReport` whose
|
|
7
|
+
diagnostics are grouped by severity (error / warning / info / not-executable / not-applicable)
|
|
8
|
+
and which serialises to JSON.
|
|
9
|
+
|
|
10
|
+
Memory model: no dataset is ever materialised. Each check consumes its inputs in independent
|
|
11
|
+
forward passes, so every bundle value must be **re-iterable** — a list, or an object whose
|
|
12
|
+
``__iter__`` opens a fresh scan (e.g. a staged-file reader); a one-shot generator is rejected.
|
|
13
|
+
Per-key lookup state (seen ids, foreign-key targets, running sums) is created through
|
|
14
|
+
``index_factory``, which a streaming caller points at a disk-spilling map
|
|
15
|
+
(:class:`~focus_data_toolkit.storage.spill.SpillableIndexPool`) to validate bundles far
|
|
16
|
+
larger than RAM.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
from collections.abc import Callable, Iterable, Mapping, MutableMapping
|
|
22
|
+
from dataclasses import dataclass
|
|
23
|
+
from decimal import Decimal
|
|
24
|
+
|
|
25
|
+
from focus_data_toolkit.errors import Diagnostic, Severity
|
|
26
|
+
from focus_data_toolkit.validate import allocation, corrections, reconciliation, referential
|
|
27
|
+
|
|
28
|
+
Rows = Iterable[Mapping[str, str]]
|
|
29
|
+
Bundle = Mapping[str, Rows]
|
|
30
|
+
IndexFactory = Callable[[], MutableMapping[str, str]]
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass
|
|
34
|
+
class BundleReport:
|
|
35
|
+
"""Result of validating a dataset bundle."""
|
|
36
|
+
|
|
37
|
+
diagnostics: list[Diagnostic]
|
|
38
|
+
checks_run: tuple[str, ...] = ()
|
|
39
|
+
|
|
40
|
+
@property
|
|
41
|
+
def ok(self) -> bool:
|
|
42
|
+
"""No failing (ERROR) diagnostics."""
|
|
43
|
+
return not any(d.is_failure for d in self.diagnostics)
|
|
44
|
+
|
|
45
|
+
def by_severity(self, severity: Severity) -> list[Diagnostic]:
|
|
46
|
+
return [d for d in self.diagnostics if d.severity is severity]
|
|
47
|
+
|
|
48
|
+
@property
|
|
49
|
+
def errors(self) -> list[Diagnostic]:
|
|
50
|
+
return self.by_severity(Severity.ERROR)
|
|
51
|
+
|
|
52
|
+
@property
|
|
53
|
+
def warnings(self) -> list[Diagnostic]:
|
|
54
|
+
return self.by_severity(Severity.WARNING)
|
|
55
|
+
|
|
56
|
+
def counts(self) -> dict[str, int]:
|
|
57
|
+
counts: dict[str, int] = {s.value: 0 for s in Severity}
|
|
58
|
+
for diag in self.diagnostics:
|
|
59
|
+
counts[diag.severity.value] += 1
|
|
60
|
+
return counts
|
|
61
|
+
|
|
62
|
+
def as_dict(self) -> dict:
|
|
63
|
+
return {
|
|
64
|
+
"ok": self.ok,
|
|
65
|
+
"counts": self.counts(),
|
|
66
|
+
"checks_run": list(self.checks_run),
|
|
67
|
+
"diagnostics": [d.as_dict() for d in self.diagnostics],
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
def format(self) -> str:
|
|
71
|
+
counts = self.counts()
|
|
72
|
+
header = "bundle validation: " + ("OK" if self.ok else "FAILED")
|
|
73
|
+
summary = ", ".join(f"{k.lower()}={v}" for k, v in counts.items() if v)
|
|
74
|
+
blocks = [header + (f" ({summary})" if summary else "")]
|
|
75
|
+
blocks.extend(d.format() for d in self.diagnostics)
|
|
76
|
+
return "\n".join(blocks)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _note(code: str, severity: Severity, message: str, datasets: tuple[str, ...]) -> Diagnostic:
|
|
80
|
+
return Diagnostic(code=code, severity=severity, message=message, datasets=datasets)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _reiterable(name: str, rows: Rows) -> Rows:
|
|
84
|
+
"""Reject one-shot iterators: every check opens its own fresh pass over the rows."""
|
|
85
|
+
if iter(rows) is rows:
|
|
86
|
+
raise TypeError(
|
|
87
|
+
f"bundle dataset {name!r} is a one-shot iterator; validate_dataset_bundle "
|
|
88
|
+
"requires re-iterable rows (a list, or an object whose __iter__ opens a fresh scan)"
|
|
89
|
+
)
|
|
90
|
+
return rows
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _has_rows(rows: Rows) -> bool:
|
|
94
|
+
for _ in rows:
|
|
95
|
+
return True
|
|
96
|
+
return False
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def validate_dataset_bundle(
|
|
100
|
+
bundle: Bundle,
|
|
101
|
+
*,
|
|
102
|
+
invoice_detail_authoritative: bool = False,
|
|
103
|
+
rounding_tolerance: Decimal | None = None,
|
|
104
|
+
index_factory: IndexFactory | None = None,
|
|
105
|
+
) -> BundleReport:
|
|
106
|
+
"""Validate the datasets in ``bundle`` against each other.
|
|
107
|
+
|
|
108
|
+
``bundle`` maps FOCUS dataset names to their rows — each value must be re-iterable (see
|
|
109
|
+
the module docstring); nothing is materialised. ``invoice_detail_authoritative`` gates
|
|
110
|
+
the Cost-and-Usage <-> Invoice-Detail sum reconciliation: it runs only when the Invoice
|
|
111
|
+
Detail comes from a real invoice (a toolkit-derived Invoice Detail reconciles by
|
|
112
|
+
construction, so reconciling it would be circular). ``rounding_tolerance`` overrides the
|
|
113
|
+
reconciliation tolerance. ``index_factory`` supplies the per-key lookup state (``dict``
|
|
114
|
+
when omitted; pass ``SpillableIndexPool(...).make_map`` for bounded memory).
|
|
115
|
+
"""
|
|
116
|
+
factory: IndexFactory = index_factory if index_factory is not None else dict
|
|
117
|
+
cu = _reiterable("Cost and Usage", bundle.get("Cost and Usage", ()))
|
|
118
|
+
invd = _reiterable("Invoice Detail", bundle.get("Invoice Detail", ()))
|
|
119
|
+
bp = _reiterable("Billing Period", bundle.get("Billing Period", ()))
|
|
120
|
+
cc = _reiterable("Contract Commitment", bundle.get("Contract Commitment", ()))
|
|
121
|
+
has_cu, has_invd, has_bp, has_cc = (_has_rows(r) for r in (cu, invd, bp, cc))
|
|
122
|
+
|
|
123
|
+
diagnostics: list[Diagnostic] = []
|
|
124
|
+
checks: list[str] = []
|
|
125
|
+
|
|
126
|
+
def run(name: str, produced: list[Diagnostic]) -> None:
|
|
127
|
+
checks.append(name)
|
|
128
|
+
diagnostics.extend(produced)
|
|
129
|
+
|
|
130
|
+
def _refs_present(rows: Rows, column: str) -> bool:
|
|
131
|
+
return any((r.get(column) or "").strip() for r in rows)
|
|
132
|
+
|
|
133
|
+
# Referential integrity.
|
|
134
|
+
if has_invd:
|
|
135
|
+
run(
|
|
136
|
+
"unique_invoice_detail_ids",
|
|
137
|
+
referential.check_unique_invoice_detail_ids(invd, index_factory=factory),
|
|
138
|
+
)
|
|
139
|
+
if has_cc:
|
|
140
|
+
run(
|
|
141
|
+
"unique_contract_commitment_ids",
|
|
142
|
+
referential.check_unique_contract_commitment_ids(cc, index_factory=factory),
|
|
143
|
+
)
|
|
144
|
+
if has_cu and has_invd:
|
|
145
|
+
run(
|
|
146
|
+
"cost_and_usage_invoice_detail_fk",
|
|
147
|
+
referential.check_cost_and_usage_invoice_detail_fk(
|
|
148
|
+
cu, invd, index_factory=factory
|
|
149
|
+
),
|
|
150
|
+
)
|
|
151
|
+
run(
|
|
152
|
+
"cost_and_usage_invoice_detail_consistency",
|
|
153
|
+
referential.check_cost_and_usage_invoice_detail_consistency(
|
|
154
|
+
cu, invd, index_factory=factory
|
|
155
|
+
),
|
|
156
|
+
)
|
|
157
|
+
elif has_cu and _refs_present(cu, "InvoiceDetailId"):
|
|
158
|
+
# References exist but their target table is absent -> the FK check cannot resolve them.
|
|
159
|
+
diagnostics.append(
|
|
160
|
+
_note(
|
|
161
|
+
"FDT-BUNDLE-001",
|
|
162
|
+
Severity.NOT_EXECUTABLE,
|
|
163
|
+
"Cost and Usage InvoiceDetailId references cannot be checked: the Invoice Detail "
|
|
164
|
+
"dataset is absent from the bundle",
|
|
165
|
+
("Cost and Usage", "Invoice Detail"),
|
|
166
|
+
)
|
|
167
|
+
)
|
|
168
|
+
if has_cu and has_cc:
|
|
169
|
+
run(
|
|
170
|
+
"contract_applied_fk",
|
|
171
|
+
referential.check_contract_applied_fk(cu, cc, index_factory=factory),
|
|
172
|
+
)
|
|
173
|
+
elif has_cu and _refs_present(cu, "ContractApplied"):
|
|
174
|
+
diagnostics.append(
|
|
175
|
+
_note(
|
|
176
|
+
"FDT-BUNDLE-001",
|
|
177
|
+
Severity.NOT_EXECUTABLE,
|
|
178
|
+
"Cost and Usage ContractApplied references cannot be checked: the Contract "
|
|
179
|
+
"Commitment dataset is absent from the bundle",
|
|
180
|
+
("Cost and Usage", "Contract Commitment"),
|
|
181
|
+
)
|
|
182
|
+
)
|
|
183
|
+
if has_cu and has_bp:
|
|
184
|
+
run(
|
|
185
|
+
"billing_period_coverage",
|
|
186
|
+
referential.check_billing_period_coverage(cu, bp, index_factory=factory),
|
|
187
|
+
)
|
|
188
|
+
|
|
189
|
+
# Reconciliation (only for an authoritative Invoice Detail).
|
|
190
|
+
if has_cu and has_invd:
|
|
191
|
+
if invoice_detail_authoritative:
|
|
192
|
+
tolerance = (
|
|
193
|
+
rounding_tolerance
|
|
194
|
+
if rounding_tolerance is not None
|
|
195
|
+
else reconciliation.DEFAULT_TOLERANCE
|
|
196
|
+
)
|
|
197
|
+
run(
|
|
198
|
+
"reconcile_invoice_detail",
|
|
199
|
+
reconciliation.reconcile_invoice_detail(
|
|
200
|
+
cu, invd, tolerance=tolerance, index_factory=factory
|
|
201
|
+
),
|
|
202
|
+
)
|
|
203
|
+
else:
|
|
204
|
+
diagnostics.append(
|
|
205
|
+
_note(
|
|
206
|
+
"FDT-BUNDLE-002",
|
|
207
|
+
Severity.NOT_APPLICABLE,
|
|
208
|
+
"Cost and Usage <-> Invoice Detail reconciliation skipped: Invoice Detail is "
|
|
209
|
+
"not marked authoritative (a toolkit-derived Invoice Detail reconciles by "
|
|
210
|
+
"construction)",
|
|
211
|
+
("Cost and Usage", "Invoice Detail"),
|
|
212
|
+
)
|
|
213
|
+
)
|
|
214
|
+
elif invoice_detail_authoritative:
|
|
215
|
+
diagnostics.append(
|
|
216
|
+
_note(
|
|
217
|
+
"FDT-BUNDLE-001",
|
|
218
|
+
Severity.NOT_EXECUTABLE,
|
|
219
|
+
"reconciliation not executable: Cost and Usage or Invoice Detail is absent",
|
|
220
|
+
("Cost and Usage", "Invoice Detail"),
|
|
221
|
+
)
|
|
222
|
+
)
|
|
223
|
+
|
|
224
|
+
# Split cost allocation and correction integrity (self-contained within Cost and Usage).
|
|
225
|
+
if has_cu:
|
|
226
|
+
run(
|
|
227
|
+
"split_allocation",
|
|
228
|
+
allocation.validate_split_allocation(cu, index_factory=factory),
|
|
229
|
+
)
|
|
230
|
+
run(
|
|
231
|
+
"correction_references",
|
|
232
|
+
corrections.check_correction_references(cu, index_factory=factory),
|
|
233
|
+
)
|
|
234
|
+
run(
|
|
235
|
+
"correction_net_sums",
|
|
236
|
+
corrections.check_correction_net_sums(cu, index_factory=factory),
|
|
237
|
+
)
|
|
238
|
+
run(
|
|
239
|
+
"no_duplicate_charge_keys",
|
|
240
|
+
corrections.check_no_duplicate_charge_keys(cu, index_factory=factory),
|
|
241
|
+
)
|
|
242
|
+
|
|
243
|
+
# Commitment lifecycle.
|
|
244
|
+
if has_cc:
|
|
245
|
+
run("contract_commitment_periods", corrections.check_contract_commitment_periods(cc))
|
|
246
|
+
run(
|
|
247
|
+
"contract_commitment_percentages",
|
|
248
|
+
corrections.check_contract_commitment_percentages(cc),
|
|
249
|
+
)
|
|
250
|
+
|
|
251
|
+
return BundleReport(diagnostics=diagnostics, checks_run=tuple(checks))
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
__all__ = ["Bundle", "BundleReport", "validate_dataset_bundle"]
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
"""Stable catalogue of focus-data-toolkit diagnostic codes.
|
|
2
|
+
|
|
3
|
+
Codes are **stable** identifiers — once assigned they are never renumbered or reused — so
|
|
4
|
+
downstream tooling can key on them across releases. Namespaces:
|
|
5
|
+
|
|
6
|
+
* ``FDT-DET-*`` — schema / version detection
|
|
7
|
+
* ``FDT-CROSS-*`` — inter-dataset referential integrity & reconciliation
|
|
8
|
+
* ``FDT-ALLOC-*`` — split cost allocation
|
|
9
|
+
* ``FDT-CORR-*`` — corrections / credits / billing lifecycle
|
|
10
|
+
* ``FDT-IO-*`` — input / output / format
|
|
11
|
+
|
|
12
|
+
Each code has a default severity and a one-line summary; call :func:`spec` to look one up.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
|
|
19
|
+
from focus_data_toolkit.errors import Severity
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass(frozen=True)
|
|
23
|
+
class CodeSpec:
|
|
24
|
+
code: str
|
|
25
|
+
default_severity: Severity
|
|
26
|
+
summary: str
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _s(code: str, sev: Severity, summary: str) -> tuple[str, CodeSpec]:
|
|
30
|
+
return code, CodeSpec(code, sev, summary)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
CATALOG: dict[str, CodeSpec] = dict([
|
|
34
|
+
# --- detection ---------------------------------------------------------------
|
|
35
|
+
_s("FDT-DET-001", Severity.ERROR, "schema/version detection confidence too low"),
|
|
36
|
+
_s("FDT-DET-002", Severity.ERROR, "ambiguous schema (multiple candidate versions/datasets)"),
|
|
37
|
+
_s("FDT-DET-003", Severity.ERROR, "forced version incompatible with the header"),
|
|
38
|
+
_s("FDT-DET-004", Severity.WARNING, "unknown non-x_ columns present in the source"),
|
|
39
|
+
# --- multi-provider / context ------------------------------------------------
|
|
40
|
+
_s("FDT-CTX-001", Severity.WARNING, "source carries multiple provider contexts"),
|
|
41
|
+
_s("FDT-CTX-002", Severity.WARNING, "source carries multiple invoice issuers"),
|
|
42
|
+
_s("FDT-CTX-003", Severity.WARNING, "source carries multiple billing currencies"),
|
|
43
|
+
_s("FDT-CTX-004", Severity.INFO, "a representative context was chosen for enrichment"),
|
|
44
|
+
# --- cross-dataset referential integrity & reconciliation --------------------
|
|
45
|
+
_s("FDT-CROSS-001", Severity.ERROR, "duplicate identifier where uniqueness is required"),
|
|
46
|
+
_s("FDT-CROSS-002", Severity.ERROR, "identifier collides across datasets"),
|
|
47
|
+
_s("FDT-CROSS-010", Severity.ERROR, "ContractApplied references an unknown ContractCommitmentId"),
|
|
48
|
+
_s("FDT-CROSS-011", Severity.ERROR, "referenced Contract Commitment not found"),
|
|
49
|
+
_s("FDT-CROSS-014", Severity.ERROR, "Cost and Usage InvoiceDetailId not found in Invoice Detail"),
|
|
50
|
+
_s("FDT-CROSS-015", Severity.ERROR, "linked record differs on an identifying attribute (wrong invoice line)"),
|
|
51
|
+
_s("FDT-CROSS-020", Severity.ERROR, "currency mismatch between linked records"),
|
|
52
|
+
_s("FDT-CROSS-021", Severity.ERROR, "billing period mismatch between linked records"),
|
|
53
|
+
_s("FDT-CROSS-022", Severity.ERROR, "invoice issuer mismatch between linked records"),
|
|
54
|
+
_s("FDT-CROSS-023", Severity.ERROR, "billing account mismatch between linked records"),
|
|
55
|
+
_s("FDT-CROSS-030", Severity.ERROR, "sum reconciliation mismatch beyond tolerance"),
|
|
56
|
+
_s("FDT-CROSS-031", Severity.WARNING, "invoice-detail line has no matching Cost and Usage rows"),
|
|
57
|
+
_s("FDT-CROSS-040", Severity.ERROR, "no Billing Period for a (period, issuer) seen in Cost and Usage"),
|
|
58
|
+
_s("FDT-CROSS-041", Severity.ERROR, "a closed billing period was modified"),
|
|
59
|
+
_s("FDT-CROSS-050", Severity.ERROR, "contract commitment period start is not before its end"),
|
|
60
|
+
_s("FDT-CROSS-051", Severity.ERROR, "percentage value outside its allowed range"),
|
|
61
|
+
# --- split cost allocation ---------------------------------------------------
|
|
62
|
+
_s("FDT-ALLOC-001", Severity.ERROR, "allocation ratios do not sum to 1 within tolerance"),
|
|
63
|
+
_s("FDT-ALLOC-002", Severity.ERROR, "allocated costs do not sum to the origin charge"),
|
|
64
|
+
_s("FDT-ALLOC-003", Severity.ERROR, "inconsistent allocation method within a group"),
|
|
65
|
+
_s("FDT-ALLOC-004", Severity.ERROR, "duplicate allocated resource within a group"),
|
|
66
|
+
_s("FDT-ALLOC-005", Severity.ERROR, "incomplete allocation group (missing required information)"),
|
|
67
|
+
_s("FDT-ALLOC-006", Severity.ERROR, "allocation ratio outside [0, 1]"),
|
|
68
|
+
_s("FDT-ALLOC-007", Severity.ERROR, "inconsistent unit within an allocation group"),
|
|
69
|
+
# --- corrections / lifecycle -------------------------------------------------
|
|
70
|
+
_s("FDT-CORR-001", Severity.ERROR, "correction references a missing invoice/charge"),
|
|
71
|
+
_s("FDT-CORR-002", Severity.ERROR, "net sum of a correction set does not reconcile"),
|
|
72
|
+
_s("FDT-CORR-003", Severity.ERROR, "correction overwrites an original row (no audit trail)"),
|
|
73
|
+
_s("FDT-CORR-004", Severity.ERROR, "invoice status transition is not allowed"),
|
|
74
|
+
# --- bundle coverage (a check could not run / does not apply) ----------------
|
|
75
|
+
_s("FDT-BUNDLE-001", Severity.NOT_EXECUTABLE, "a cross-dataset check could not run (data absent)"),
|
|
76
|
+
_s("FDT-BUNDLE-002", Severity.NOT_APPLICABLE, "a cross-dataset check does not apply to this bundle"),
|
|
77
|
+
# --- io / format -------------------------------------------------------------
|
|
78
|
+
_s("FDT-IO-001", Severity.ERROR, "malformed input record (wrong field count)"),
|
|
79
|
+
_s("FDT-IO-002", Severity.ERROR, "decimal value exceeds the target Parquet scale/precision"),
|
|
80
|
+
_s("FDT-IO-003", Severity.ERROR, "destination already exists"),
|
|
81
|
+
_s("FDT-IO-004", Severity.WARNING, "high-cardinality Parquet partition key (many small files)"),
|
|
82
|
+
_s("FDT-IO-005", Severity.ERROR, "insufficient free space on the output filesystem"),
|
|
83
|
+
_s("FDT-IO-006", Severity.ERROR, "insufficient work-filesystem space or temp budget exceeded"),
|
|
84
|
+
])
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def spec(code: str) -> CodeSpec:
|
|
88
|
+
"""Return the :class:`CodeSpec` for ``code`` (raises ``KeyError`` if unknown)."""
|
|
89
|
+
return CATALOG[code]
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def default_severity(code: str) -> Severity:
|
|
93
|
+
return CATALOG[code].default_severity
|
|
@@ -0,0 +1,245 @@
|
|
|
1
|
+
"""Commitment lifecycle and correction-integrity checks.
|
|
2
|
+
|
|
3
|
+
Phase A covers the cheap, spec-grounded temporal and percentage-range checks on Contract
|
|
4
|
+
Commitment, plus a forward-compatible correction-reference check: a Cost and Usage correction
|
|
5
|
+
line (``ChargeClass="Correction"``) may point at the charge it corrects via the toolkit
|
|
6
|
+
extension ``x_CorrectionOf`` -> an original row's ``x_ChargeKey``; the referenced original must
|
|
7
|
+
still be present (corrections never silently overwrite history). Deeper correction scenarios
|
|
8
|
+
arrive with the Phase B generators.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from collections.abc import Callable, Iterable, Mapping, MutableMapping, Sequence
|
|
14
|
+
from datetime import datetime
|
|
15
|
+
from decimal import Decimal, InvalidOperation
|
|
16
|
+
|
|
17
|
+
from focus_data_toolkit.errors import Diagnostic, Severity
|
|
18
|
+
|
|
19
|
+
Rows = Sequence[Mapping[str, str]]
|
|
20
|
+
#: Inputs are consumed in forward passes only, so any (re-)iterable of rows works.
|
|
21
|
+
RowStream = Iterable[Mapping[str, str]]
|
|
22
|
+
#: Factory for per-key lookup state (``dict`` by default; a spillable map for streaming).
|
|
23
|
+
IndexFactory = Callable[[], MutableMapping[str, str]]
|
|
24
|
+
|
|
25
|
+
_PERCENTAGE_COLUMNS = (
|
|
26
|
+
"ContractCommitmentDiscountPercentage",
|
|
27
|
+
"ContractCommitmentPaymentUpfrontPercentage",
|
|
28
|
+
)
|
|
29
|
+
_PERIOD_PAIRS = (
|
|
30
|
+
("ContractCommitmentPeriodStart", "ContractCommitmentPeriodEnd"),
|
|
31
|
+
("ContractPeriodStart", "ContractPeriodEnd"),
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _parse_dt(value: str) -> datetime | None:
|
|
36
|
+
try:
|
|
37
|
+
parsed = datetime.fromisoformat(value.strip().replace("Z", "+00:00"))
|
|
38
|
+
except ValueError:
|
|
39
|
+
return None
|
|
40
|
+
# Require a timezone (FOCUS Date/Time is UTC 'Z'). A naive value is malformed for this
|
|
41
|
+
# lifecycle check; returning None avoids a naive-vs-aware TypeError on comparison and
|
|
42
|
+
# lets the per-dataset linter report the bad Date/Time format.
|
|
43
|
+
return parsed if parsed.tzinfo is not None else None
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def check_contract_commitment_periods(contract_commitment: RowStream) -> list[Diagnostic]:
|
|
47
|
+
"""Each commitment/contract period must start strictly before it ends."""
|
|
48
|
+
out: list[Diagnostic] = []
|
|
49
|
+
for i, row in enumerate(contract_commitment, start=1):
|
|
50
|
+
for start_col, end_col in _PERIOD_PAIRS:
|
|
51
|
+
start, end = _parse_dt(row.get(start_col, "")), _parse_dt(row.get(end_col, ""))
|
|
52
|
+
if start is None or end is None or start < end:
|
|
53
|
+
continue
|
|
54
|
+
out.append(
|
|
55
|
+
Diagnostic(
|
|
56
|
+
code="FDT-CROSS-050",
|
|
57
|
+
severity=Severity.ERROR,
|
|
58
|
+
message=f"{start_col} is not before {end_col}",
|
|
59
|
+
datasets=("Contract Commitment",),
|
|
60
|
+
dataset="Contract Commitment",
|
|
61
|
+
line_number=i,
|
|
62
|
+
column=start_col,
|
|
63
|
+
expected=f"< {row.get(end_col)}",
|
|
64
|
+
actual=row.get(start_col, ""),
|
|
65
|
+
record_keys={"ContractCommitmentId": (row.get("ContractCommitmentId") or "")},
|
|
66
|
+
)
|
|
67
|
+
)
|
|
68
|
+
return out
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def check_contract_commitment_percentages(contract_commitment: RowStream) -> list[Diagnostic]:
|
|
72
|
+
"""Percentage columns must be within [0, 1]."""
|
|
73
|
+
out: list[Diagnostic] = []
|
|
74
|
+
for i, row in enumerate(contract_commitment, start=1):
|
|
75
|
+
for col in _PERCENTAGE_COLUMNS:
|
|
76
|
+
raw = (row.get(col) or "").strip()
|
|
77
|
+
if not raw:
|
|
78
|
+
continue
|
|
79
|
+
try:
|
|
80
|
+
value = Decimal(raw)
|
|
81
|
+
except InvalidOperation:
|
|
82
|
+
continue # value-format is the per-dataset linter's job
|
|
83
|
+
if not value.is_finite():
|
|
84
|
+
continue # NaN/Infinity is a format issue for the linter, not a range error
|
|
85
|
+
if value < 0 or value > 1:
|
|
86
|
+
out.append(
|
|
87
|
+
Diagnostic(
|
|
88
|
+
code="FDT-CROSS-051",
|
|
89
|
+
severity=Severity.ERROR,
|
|
90
|
+
message=f"{col} value {raw} is outside [0, 1]",
|
|
91
|
+
datasets=("Contract Commitment",),
|
|
92
|
+
dataset="Contract Commitment",
|
|
93
|
+
line_number=i,
|
|
94
|
+
column=col,
|
|
95
|
+
actual=raw,
|
|
96
|
+
record_keys={"ContractCommitmentId": (row.get("ContractCommitmentId") or "")},
|
|
97
|
+
)
|
|
98
|
+
)
|
|
99
|
+
return out
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _is_correction(row: Mapping[str, str]) -> bool:
|
|
103
|
+
return (row.get("ChargeClass") or "").strip().casefold() == "correction"
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _dec(value: str | None) -> Decimal | None:
|
|
107
|
+
try:
|
|
108
|
+
parsed = Decimal((value or "").strip())
|
|
109
|
+
except InvalidOperation:
|
|
110
|
+
return None
|
|
111
|
+
return parsed if parsed.is_finite() else None
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def check_no_duplicate_charge_keys(
|
|
115
|
+
cost_and_usage: RowStream, *, index_factory: IndexFactory = dict
|
|
116
|
+
) -> list[Diagnostic]:
|
|
117
|
+
"""``x_ChargeKey`` must be unique: reusing one silently overwrites an auditable line.
|
|
118
|
+
|
|
119
|
+
Corrections are appended as *new* keyed rows (``x_ChargeKey`` + ``x_CorrectionOf``), so a
|
|
120
|
+
repeated key means an original (or a prior correction) was overwritten rather than amended.
|
|
121
|
+
"""
|
|
122
|
+
seen = index_factory()
|
|
123
|
+
out: list[Diagnostic] = []
|
|
124
|
+
for i, row in enumerate(cost_and_usage, start=1):
|
|
125
|
+
key = (row.get("x_ChargeKey") or "").strip()
|
|
126
|
+
if not key:
|
|
127
|
+
continue
|
|
128
|
+
if key in seen:
|
|
129
|
+
out.append(
|
|
130
|
+
Diagnostic(
|
|
131
|
+
code="FDT-CORR-003",
|
|
132
|
+
severity=Severity.ERROR,
|
|
133
|
+
message=f"x_ChargeKey {key!r} appears on rows {seen[key]} and {i} — a "
|
|
134
|
+
"correction must add a new keyed row, never overwrite an original",
|
|
135
|
+
datasets=("Cost and Usage",),
|
|
136
|
+
dataset="Cost and Usage",
|
|
137
|
+
line_number=i,
|
|
138
|
+
column="x_ChargeKey",
|
|
139
|
+
value=key,
|
|
140
|
+
record_keys={"x_ChargeKey": key},
|
|
141
|
+
)
|
|
142
|
+
)
|
|
143
|
+
else:
|
|
144
|
+
seen[key] = str(i)
|
|
145
|
+
return out
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def check_correction_net_sums(
|
|
149
|
+
cost_and_usage: RowStream,
|
|
150
|
+
*,
|
|
151
|
+
tolerance: Decimal = Decimal("0.01"),
|
|
152
|
+
index_factory: IndexFactory = dict,
|
|
153
|
+
) -> list[Diagnostic]:
|
|
154
|
+
"""The net of a correction set must equal the declared ``x_NetCharge``.
|
|
155
|
+
|
|
156
|
+
A correction set is an original charge (``x_ChargeKey`` == the corrections'
|
|
157
|
+
``x_CorrectionOf``) plus every correction pointing at it. When a correction declares the
|
|
158
|
+
post-correction net in ``x_NetCharge``, the arithmetic sum of ``BilledCost`` over the set up
|
|
159
|
+
to and including that correction must match it — so the running net stays auditable and no
|
|
160
|
+
money is silently created or lost. Sets whose original is absent are handled by
|
|
161
|
+
:func:`check_correction_references`; this check only reconciles what is present.
|
|
162
|
+
"""
|
|
163
|
+
# Decimals are stored as exact strings so the lookup state can live in a spillable map.
|
|
164
|
+
originals = index_factory()
|
|
165
|
+
for row in cost_and_usage:
|
|
166
|
+
if _is_correction(row):
|
|
167
|
+
continue
|
|
168
|
+
key = (row.get("x_ChargeKey") or "").strip()
|
|
169
|
+
cost = _dec(row.get("BilledCost"))
|
|
170
|
+
if key and cost is not None:
|
|
171
|
+
originals[key] = str(cost)
|
|
172
|
+
|
|
173
|
+
# Accumulate corrections per original in row order (running net is order-sensitive).
|
|
174
|
+
running = index_factory()
|
|
175
|
+
out: list[Diagnostic] = []
|
|
176
|
+
for i, row in enumerate(cost_and_usage, start=1):
|
|
177
|
+
if not _is_correction(row):
|
|
178
|
+
continue
|
|
179
|
+
ref = (row.get("x_CorrectionOf") or "").strip()
|
|
180
|
+
if ref not in originals:
|
|
181
|
+
continue # missing original -> reported by check_correction_references
|
|
182
|
+
delta = _dec(row.get("BilledCost"))
|
|
183
|
+
if delta is None:
|
|
184
|
+
continue
|
|
185
|
+
base = running.get(ref) or originals[ref]
|
|
186
|
+
net = Decimal(base) + delta
|
|
187
|
+
running[ref] = str(net)
|
|
188
|
+
declared = _dec(row.get("x_NetCharge"))
|
|
189
|
+
if declared is None:
|
|
190
|
+
continue
|
|
191
|
+
if abs(net - declared) > tolerance:
|
|
192
|
+
out.append(
|
|
193
|
+
Diagnostic(
|
|
194
|
+
code="FDT-CORR-002",
|
|
195
|
+
severity=Severity.ERROR,
|
|
196
|
+
message=f"correction set for {ref!r} nets to {net} but x_NetCharge "
|
|
197
|
+
f"declares {declared}",
|
|
198
|
+
datasets=("Cost and Usage",),
|
|
199
|
+
dataset="Cost and Usage",
|
|
200
|
+
line_number=i,
|
|
201
|
+
column="x_NetCharge",
|
|
202
|
+
expected=str(declared),
|
|
203
|
+
actual=str(net),
|
|
204
|
+
record_keys={"x_CorrectionOf": ref},
|
|
205
|
+
)
|
|
206
|
+
)
|
|
207
|
+
return out
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def check_correction_references(
|
|
211
|
+
cost_and_usage: RowStream, *, index_factory: IndexFactory = dict
|
|
212
|
+
) -> list[Diagnostic]:
|
|
213
|
+
"""A correction line's ``x_CorrectionOf`` must point at a still-present *original* charge.
|
|
214
|
+
|
|
215
|
+
The lookup is built only from non-correction originals, and a correction that references
|
|
216
|
+
its own key is rejected — otherwise a correction with ``x_ChargeKey == x_CorrectionOf`` and
|
|
217
|
+
no surviving original would pass, defeating the auditability guarantee.
|
|
218
|
+
"""
|
|
219
|
+
original_keys = index_factory()
|
|
220
|
+
for r in cost_and_usage:
|
|
221
|
+
key = (r.get("x_ChargeKey") or "").strip()
|
|
222
|
+
if key and not _is_correction(r):
|
|
223
|
+
original_keys[key] = ""
|
|
224
|
+
out: list[Diagnostic] = []
|
|
225
|
+
for i, row in enumerate(cost_and_usage, start=1):
|
|
226
|
+
ref = (row.get("x_CorrectionOf") or "").strip()
|
|
227
|
+
if not ref:
|
|
228
|
+
continue
|
|
229
|
+
self_key = (row.get("x_ChargeKey") or "").strip()
|
|
230
|
+
if ref == self_key or ref not in original_keys:
|
|
231
|
+
out.append(
|
|
232
|
+
Diagnostic(
|
|
233
|
+
code="FDT-CORR-001",
|
|
234
|
+
severity=Severity.ERROR,
|
|
235
|
+
message=f"correction row {i} references original charge {ref!r} that is not "
|
|
236
|
+
"present as an original (original must remain auditable)",
|
|
237
|
+
datasets=("Cost and Usage",),
|
|
238
|
+
dataset="Cost and Usage",
|
|
239
|
+
line_number=i,
|
|
240
|
+
column="x_CorrectionOf",
|
|
241
|
+
value=ref,
|
|
242
|
+
record_keys={"x_CorrectionOf": ref},
|
|
243
|
+
)
|
|
244
|
+
)
|
|
245
|
+
return out
|