focus-data-toolkit 0.11.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- focus_data_toolkit/__init__.py +69 -0
- focus_data_toolkit/__main__.py +6 -0
- focus_data_toolkit/_version.py +8 -0
- focus_data_toolkit/cli.py +968 -0
- focus_data_toolkit/context/__init__.py +88 -0
- focus_data_toolkit/context/billing.py +54 -0
- focus_data_toolkit/context/provider.py +90 -0
- focus_data_toolkit/convert/__init__.py +708 -0
- focus_data_toolkit/convert/billing_period.py +65 -0
- focus_data_toolkit/convert/contract_applied.py +235 -0
- focus_data_toolkit/convert/contract_commitment.py +182 -0
- focus_data_toolkit/convert/cost_and_usage.py +179 -0
- focus_data_toolkit/convert/detect.py +39 -0
- focus_data_toolkit/convert/invoice_detail.py +199 -0
- focus_data_toolkit/convert/streaming.py +1030 -0
- focus_data_toolkit/errors.py +145 -0
- focus_data_toolkit/focus_json.py +68 -0
- focus_data_toolkit/generators/__init__.py +61 -0
- focus_data_toolkit/generators/_shim.py +43 -0
- focus_data_toolkit/generators/engine/__init__.py +14 -0
- focus_data_toolkit/generators/engine/context.py +12 -0
- focus_data_toolkit/generators/engine/determinism.py +117 -0
- focus_data_toolkit/generators/engine/json_focus.py +63 -0
- focus_data_toolkit/generators/engine/ladder.py +71 -0
- focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
- focus_data_toolkit/generators/engine/serialize.py +151 -0
- focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
- focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
- focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
- focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
- focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
- focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
- focus_data_toolkit/generators/providers/__init__.py +29 -0
- focus_data_toolkit/generators/providers/aws.py +186 -0
- focus_data_toolkit/generators/providers/azure.py +191 -0
- focus_data_toolkit/generators/providers/gcp.py +194 -0
- focus_data_toolkit/generators/providers/profile.py +123 -0
- focus_data_toolkit/generators/scenarios.py +178 -0
- focus_data_toolkit/generators/versions/__init__.py +17 -0
- focus_data_toolkit/generators/versions/adapter.py +41 -0
- focus_data_toolkit/generators/versions/v1_2.py +111 -0
- focus_data_toolkit/generators/versions/v1_3.py +154 -0
- focus_data_toolkit/io/__init__.py +1 -0
- focus_data_toolkit/io/atomic_writer.py +462 -0
- focus_data_toolkit/io/csv_io.py +128 -0
- focus_data_toolkit/io/parquet_io.py +528 -0
- focus_data_toolkit/io/records.py +92 -0
- focus_data_toolkit/io/row_source.py +117 -0
- focus_data_toolkit/lifecycle.py +342 -0
- focus_data_toolkit/manifest.py +114 -0
- focus_data_toolkit/model/__init__.py +43 -0
- focus_data_toolkit/model/capabilities.py +66 -0
- focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
- focus_data_toolkit/model/focus_1_4_model.json +1913 -0
- focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
- focus_data_toolkit/model/focus_json_keys.py +112 -0
- focus_data_toolkit/model/iso_4217_currencies.json +23 -0
- focus_data_toolkit/model/json_schema_check.py +205 -0
- focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
- focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
- focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
- focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
- focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
- focus_data_toolkit/model/model_provenance.json +58 -0
- focus_data_toolkit/model/validator.py +498 -0
- focus_data_toolkit/modes.py +18 -0
- focus_data_toolkit/official_validator.py +61 -0
- focus_data_toolkit/progress.py +89 -0
- focus_data_toolkit/provenance.py +106 -0
- focus_data_toolkit/py.typed +1 -0
- focus_data_toolkit/runtime.py +243 -0
- focus_data_toolkit/schema/__init__.py +17 -0
- focus_data_toolkit/schema/detection.py +274 -0
- focus_data_toolkit/schema/registry.py +127 -0
- focus_data_toolkit/storage/__init__.py +1 -0
- focus_data_toolkit/storage/external_index.py +99 -0
- focus_data_toolkit/storage/spill.py +150 -0
- focus_data_toolkit/studio/__init__.py +19 -0
- focus_data_toolkit/studio/app.py +467 -0
- focus_data_toolkit/studio/config.py +42 -0
- focus_data_toolkit/studio/frontend/app.js +214 -0
- focus_data_toolkit/studio/frontend/index.html +101 -0
- focus_data_toolkit/studio/frontend/style.css +60 -0
- focus_data_toolkit/studio/jobs.py +142 -0
- focus_data_toolkit/studio/preview.py +32 -0
- focus_data_toolkit/studio/security.py +125 -0
- focus_data_toolkit/studio/server.py +71 -0
- focus_data_toolkit/supplement/__init__.py +50 -0
- focus_data_toolkit/supplement/adapters/__init__.py +21 -0
- focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
- focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
- focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
- focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
- focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
- focus_data_toolkit/supplement/adapters/registry.py +215 -0
- focus_data_toolkit/supplement/apply.py +318 -0
- focus_data_toolkit/supplement/gaps.py +219 -0
- focus_data_toolkit/supplement/kinds.py +118 -0
- focus_data_toolkit/supplement/loader.py +409 -0
- focus_data_toolkit/supplement/spec.py +74 -0
- focus_data_toolkit/supplement/validate.py +215 -0
- focus_data_toolkit/validate/__init__.py +15 -0
- focus_data_toolkit/validate/allocation.py +333 -0
- focus_data_toolkit/validate/bundle.py +254 -0
- focus_data_toolkit/validate/codes.py +93 -0
- focus_data_toolkit/validate/corrections.py +245 -0
- focus_data_toolkit/validate/reconciliation.py +98 -0
- focus_data_toolkit/validate/referential.py +289 -0
- focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
- focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
- focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
- focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
- focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
"""Reconcile Cost and Usage against an authoritative Invoice Detail dataset.
|
|
2
|
+
|
|
3
|
+
FOCUS 1.4 explicitly allows an issued invoice to differ from summed usage (it adds an
|
|
4
|
+
*Invoice Reconciliation* feature and a *Rounding Variance Tolerance* appendix). So this check
|
|
5
|
+
runs only when Invoice Detail comes from an **authoritative** source, and compares sums with an
|
|
6
|
+
explicit, documented tolerance rather than requiring exact equality.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from collections.abc import Callable, Iterable, Mapping, MutableMapping, Sequence
|
|
12
|
+
from decimal import Decimal, InvalidOperation
|
|
13
|
+
|
|
14
|
+
from focus_data_toolkit.errors import Diagnostic, Severity
|
|
15
|
+
|
|
16
|
+
Rows = Sequence[Mapping[str, str]]
|
|
17
|
+
#: Inputs are consumed in forward passes only, so any (re-)iterable of rows works.
|
|
18
|
+
RowStream = Iterable[Mapping[str, str]]
|
|
19
|
+
#: Factory for per-key lookup state (``dict`` by default; a spillable map for streaming).
|
|
20
|
+
IndexFactory = Callable[[], MutableMapping[str, str]]
|
|
21
|
+
|
|
22
|
+
# Default rounding-variance tolerance for a reconciled sum (absolute, in billing currency).
|
|
23
|
+
DEFAULT_TOLERANCE = Decimal("0.01")
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _dec(value: str | None) -> Decimal:
|
|
27
|
+
try:
|
|
28
|
+
parsed = Decimal((value or "0").strip() or "0")
|
|
29
|
+
except InvalidOperation:
|
|
30
|
+
return Decimal(0)
|
|
31
|
+
# NaN / Infinity parse successfully but would raise InvalidOperation on comparison; a
|
|
32
|
+
# malformed numeric value must not crash reconciliation (the linter flags its format).
|
|
33
|
+
return parsed if parsed.is_finite() else Decimal(0)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def reconcile_invoice_detail(
|
|
37
|
+
cost_and_usage: RowStream,
|
|
38
|
+
invoice_detail: RowStream,
|
|
39
|
+
*,
|
|
40
|
+
tolerance: Decimal = DEFAULT_TOLERANCE,
|
|
41
|
+
index_factory: IndexFactory = dict,
|
|
42
|
+
) -> list[Diagnostic]:
|
|
43
|
+
"""Compare each Invoice Detail BilledCost to the sum of its Cost and Usage lines.
|
|
44
|
+
|
|
45
|
+
Cost and Usage rows are attributed to an invoice line by their ``InvoiceDetailId``
|
|
46
|
+
back-link. A line with no matching Cost and Usage rows is a warning; a sum that differs
|
|
47
|
+
beyond ``tolerance`` is an error carrying the expected/actual amounts.
|
|
48
|
+
"""
|
|
49
|
+
# Sums are stored as exact Decimal strings so the state can live in a spillable map.
|
|
50
|
+
sums = index_factory()
|
|
51
|
+
for row in cost_and_usage:
|
|
52
|
+
ref = (row.get("InvoiceDetailId") or "").strip()
|
|
53
|
+
if ref:
|
|
54
|
+
current = sums.get(ref)
|
|
55
|
+
sums[ref] = str(
|
|
56
|
+
(Decimal(current) if current is not None else Decimal(0))
|
|
57
|
+
+ _dec(row.get("BilledCost"))
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
out: list[Diagnostic] = []
|
|
61
|
+
for i, detail in enumerate(invoice_detail, start=1):
|
|
62
|
+
detail_id = (detail.get("InvoiceDetailId") or "").strip()
|
|
63
|
+
if not detail_id:
|
|
64
|
+
continue
|
|
65
|
+
invoiced = _dec(detail.get("BilledCost"))
|
|
66
|
+
if detail_id not in sums:
|
|
67
|
+
out.append(
|
|
68
|
+
Diagnostic(
|
|
69
|
+
code="FDT-CROSS-031",
|
|
70
|
+
severity=Severity.WARNING,
|
|
71
|
+
message=f"Invoice Detail line {detail_id!r} has no matching Cost and Usage rows",
|
|
72
|
+
datasets=("Invoice Detail", "Cost and Usage"),
|
|
73
|
+
dataset="Invoice Detail",
|
|
74
|
+
line_number=i,
|
|
75
|
+
column="InvoiceDetailId",
|
|
76
|
+
value=detail_id,
|
|
77
|
+
record_keys={"InvoiceDetailId": detail_id},
|
|
78
|
+
)
|
|
79
|
+
)
|
|
80
|
+
continue
|
|
81
|
+
summed = Decimal(sums[detail_id])
|
|
82
|
+
if abs(summed - invoiced) > tolerance:
|
|
83
|
+
out.append(
|
|
84
|
+
Diagnostic(
|
|
85
|
+
code="FDT-CROSS-030",
|
|
86
|
+
severity=Severity.ERROR,
|
|
87
|
+
message=f"Invoice Detail line {detail_id!r} BilledCost does not reconcile "
|
|
88
|
+
f"with summed Cost and Usage (tolerance {tolerance})",
|
|
89
|
+
datasets=("Invoice Detail", "Cost and Usage"),
|
|
90
|
+
dataset="Invoice Detail",
|
|
91
|
+
line_number=i,
|
|
92
|
+
column="BilledCost",
|
|
93
|
+
expected=str(summed),
|
|
94
|
+
actual=str(invoiced),
|
|
95
|
+
record_keys={"InvoiceDetailId": detail_id},
|
|
96
|
+
)
|
|
97
|
+
)
|
|
98
|
+
return out
|
|
@@ -0,0 +1,289 @@
|
|
|
1
|
+
"""Referential integrity across FOCUS datasets: uniqueness, foreign keys, coherence.
|
|
2
|
+
|
|
3
|
+
These checks operate on a *bundle* of datasets (dataset name -> rows) and never live inside
|
|
4
|
+
the per-dataset linter — they are inherently cross-dataset. Every finding is a structured
|
|
5
|
+
:class:`~focus_data_toolkit.errors.Diagnostic` carrying the offending record's business key.
|
|
6
|
+
|
|
7
|
+
Every check consumes each input in a single forward pass and keeps only per-key lookup
|
|
8
|
+
state; ``index_factory`` lets the caller back that state with a disk-spilling map
|
|
9
|
+
(:class:`~focus_data_toolkit.storage.spill.SpillableMap`) so validating datasets far larger
|
|
10
|
+
than RAM stays memory-bounded.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
from collections.abc import Callable, Iterable, Mapping, MutableMapping, Sequence
|
|
17
|
+
|
|
18
|
+
from focus_data_toolkit.convert.contract_applied import ContractAppliedError, parse
|
|
19
|
+
from focus_data_toolkit.errors import Diagnostic, Severity
|
|
20
|
+
|
|
21
|
+
Rows = Sequence[Mapping[str, str]]
|
|
22
|
+
#: Every dataset side of a cross-dataset check is consumed in a single forward pass, so it
|
|
23
|
+
#: accepts any iterable of rows (e.g. a staged-file stream) — bounded memory.
|
|
24
|
+
RowStream = Iterable[Mapping[str, str]]
|
|
25
|
+
#: Factory for the per-key lookup state (``dict`` by default; a spillable map for streaming).
|
|
26
|
+
IndexFactory = Callable[[], MutableMapping[str, str]]
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _composite(*parts: str) -> str:
|
|
30
|
+
"""Unambiguous single-string key for a tuple of values (JSON array encoding)."""
|
|
31
|
+
return json.dumps(parts, separators=(",", ":"))
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _cu_keys(row: Mapping[str, str]) -> dict[str, str]:
|
|
35
|
+
"""Business key for a Cost and Usage row, for diagnostics."""
|
|
36
|
+
keys = {}
|
|
37
|
+
for col in ("InvoiceIssuerName", "InvoiceId", "BillingAccountId", "ResourceId"):
|
|
38
|
+
val = (row.get(col) or "").strip()
|
|
39
|
+
if val:
|
|
40
|
+
keys[col] = val
|
|
41
|
+
return keys
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def check_unique_invoice_detail_ids(
|
|
45
|
+
invoice_detail: RowStream, *, index_factory: IndexFactory = dict
|
|
46
|
+
) -> list[Diagnostic]:
|
|
47
|
+
"""InvoiceDetailId must be unique within Invoice Detail."""
|
|
48
|
+
seen = index_factory()
|
|
49
|
+
out: list[Diagnostic] = []
|
|
50
|
+
for i, row in enumerate(invoice_detail, start=1):
|
|
51
|
+
detail_id = (row.get("InvoiceDetailId") or "").strip()
|
|
52
|
+
if not detail_id:
|
|
53
|
+
continue
|
|
54
|
+
if detail_id in seen:
|
|
55
|
+
out.append(
|
|
56
|
+
Diagnostic(
|
|
57
|
+
code="FDT-CROSS-001",
|
|
58
|
+
severity=Severity.ERROR,
|
|
59
|
+
message=f"InvoiceDetailId {detail_id!r} is not unique in Invoice Detail "
|
|
60
|
+
f"(rows {seen[detail_id]} and {i})",
|
|
61
|
+
datasets=("Invoice Detail",),
|
|
62
|
+
dataset="Invoice Detail",
|
|
63
|
+
line_number=i,
|
|
64
|
+
column="InvoiceDetailId",
|
|
65
|
+
value=detail_id,
|
|
66
|
+
record_keys={"InvoiceDetailId": detail_id},
|
|
67
|
+
)
|
|
68
|
+
)
|
|
69
|
+
else:
|
|
70
|
+
seen[detail_id] = str(i)
|
|
71
|
+
return out
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def check_unique_contract_commitment_ids(
|
|
75
|
+
contract_commitment: RowStream, *, index_factory: IndexFactory = dict
|
|
76
|
+
) -> list[Diagnostic]:
|
|
77
|
+
"""ContractCommitmentId must be unique (ContractApplied resolves commitments by this id)."""
|
|
78
|
+
seen = index_factory()
|
|
79
|
+
out: list[Diagnostic] = []
|
|
80
|
+
for i, row in enumerate(contract_commitment, start=1):
|
|
81
|
+
commitment_id = (row.get("ContractCommitmentId") or "").strip()
|
|
82
|
+
if not commitment_id:
|
|
83
|
+
continue
|
|
84
|
+
if commitment_id in seen:
|
|
85
|
+
out.append(
|
|
86
|
+
Diagnostic(
|
|
87
|
+
code="FDT-CROSS-001",
|
|
88
|
+
severity=Severity.ERROR,
|
|
89
|
+
message=f"ContractCommitmentId {commitment_id!r} is not unique in Contract "
|
|
90
|
+
f"Commitment (rows {seen[commitment_id]} and {i})",
|
|
91
|
+
datasets=("Contract Commitment",),
|
|
92
|
+
dataset="Contract Commitment",
|
|
93
|
+
line_number=i,
|
|
94
|
+
column="ContractCommitmentId",
|
|
95
|
+
value=commitment_id,
|
|
96
|
+
record_keys={"ContractCommitmentId": commitment_id},
|
|
97
|
+
)
|
|
98
|
+
)
|
|
99
|
+
else:
|
|
100
|
+
seen[commitment_id] = str(i)
|
|
101
|
+
return out
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def check_cost_and_usage_invoice_detail_fk(
|
|
105
|
+
cost_and_usage: RowStream,
|
|
106
|
+
invoice_detail: RowStream,
|
|
107
|
+
*,
|
|
108
|
+
index_factory: IndexFactory = dict,
|
|
109
|
+
) -> list[Diagnostic]:
|
|
110
|
+
"""Every non-empty Cost and Usage InvoiceDetailId must exist in Invoice Detail."""
|
|
111
|
+
known = index_factory()
|
|
112
|
+
for r in invoice_detail:
|
|
113
|
+
detail_id = (r.get("InvoiceDetailId") or "").strip()
|
|
114
|
+
if detail_id:
|
|
115
|
+
known[detail_id] = ""
|
|
116
|
+
out: list[Diagnostic] = []
|
|
117
|
+
for i, row in enumerate(cost_and_usage, start=1):
|
|
118
|
+
ref = (row.get("InvoiceDetailId") or "").strip()
|
|
119
|
+
if not ref or ref in known:
|
|
120
|
+
continue
|
|
121
|
+
out.append(
|
|
122
|
+
Diagnostic(
|
|
123
|
+
code="FDT-CROSS-014",
|
|
124
|
+
severity=Severity.ERROR,
|
|
125
|
+
message=f"InvoiceDetailId {ref!r} referenced by Cost and Usage row {i} was not "
|
|
126
|
+
"found in Invoice Detail",
|
|
127
|
+
datasets=("Cost and Usage", "Invoice Detail"),
|
|
128
|
+
dataset="Cost and Usage",
|
|
129
|
+
line_number=i,
|
|
130
|
+
column="InvoiceDetailId",
|
|
131
|
+
value=ref,
|
|
132
|
+
record_keys=_cu_keys(row),
|
|
133
|
+
source="Cost and Usage",
|
|
134
|
+
)
|
|
135
|
+
)
|
|
136
|
+
return out
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _commitment_ids(cost_and_usage_row: Mapping[str, str]) -> list[str]:
|
|
140
|
+
"""ContractCommitmentIds referenced by a row's ContractApplied JSON (best effort)."""
|
|
141
|
+
text = (cost_and_usage_row.get("ContractApplied") or "").strip()
|
|
142
|
+
if not text:
|
|
143
|
+
return []
|
|
144
|
+
try:
|
|
145
|
+
applied = parse(text, version="1.4")
|
|
146
|
+
except ContractAppliedError:
|
|
147
|
+
return [] # structural validity is the per-dataset linter's job
|
|
148
|
+
return [e.contract_commitment_id for e in applied.elements if e.contract_commitment_id]
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def check_contract_applied_fk(
|
|
152
|
+
cost_and_usage: RowStream,
|
|
153
|
+
contract_commitment: RowStream,
|
|
154
|
+
*,
|
|
155
|
+
index_factory: IndexFactory = dict,
|
|
156
|
+
) -> list[Diagnostic]:
|
|
157
|
+
"""Every ContractCommitmentId referenced from ContractApplied must exist."""
|
|
158
|
+
known = index_factory()
|
|
159
|
+
for r in contract_commitment:
|
|
160
|
+
commitment_id = (r.get("ContractCommitmentId") or "").strip()
|
|
161
|
+
if commitment_id:
|
|
162
|
+
known[commitment_id] = ""
|
|
163
|
+
out: list[Diagnostic] = []
|
|
164
|
+
for i, row in enumerate(cost_and_usage, start=1):
|
|
165
|
+
for commitment_id in _commitment_ids(row):
|
|
166
|
+
if commitment_id in known:
|
|
167
|
+
continue
|
|
168
|
+
out.append(
|
|
169
|
+
Diagnostic(
|
|
170
|
+
code="FDT-CROSS-010",
|
|
171
|
+
severity=Severity.ERROR,
|
|
172
|
+
message=f"ContractApplied on Cost and Usage row {i} references "
|
|
173
|
+
f"ContractCommitmentId {commitment_id!r} not found in Contract Commitment",
|
|
174
|
+
datasets=("Cost and Usage", "Contract Commitment"),
|
|
175
|
+
dataset="Cost and Usage",
|
|
176
|
+
line_number=i,
|
|
177
|
+
column="ContractApplied",
|
|
178
|
+
value=commitment_id,
|
|
179
|
+
record_keys=_cu_keys(row),
|
|
180
|
+
)
|
|
181
|
+
)
|
|
182
|
+
return out
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def check_billing_period_coverage(
|
|
186
|
+
cost_and_usage: RowStream,
|
|
187
|
+
billing_period: RowStream,
|
|
188
|
+
*,
|
|
189
|
+
index_factory: IndexFactory = dict,
|
|
190
|
+
) -> list[Diagnostic]:
|
|
191
|
+
"""Every (period, issuer) seen in Cost and Usage must have a Billing Period row."""
|
|
192
|
+
known = index_factory()
|
|
193
|
+
for r in billing_period:
|
|
194
|
+
known[
|
|
195
|
+
_composite(
|
|
196
|
+
(r.get("BillingPeriodStart") or "").strip(),
|
|
197
|
+
(r.get("BillingPeriodEnd") or "").strip(),
|
|
198
|
+
(r.get("InvoiceIssuerName") or "").strip(),
|
|
199
|
+
)
|
|
200
|
+
] = ""
|
|
201
|
+
reported = index_factory()
|
|
202
|
+
out: list[Diagnostic] = []
|
|
203
|
+
for i, row in enumerate(cost_and_usage, start=1):
|
|
204
|
+
key = (
|
|
205
|
+
(row.get("BillingPeriodStart") or "").strip(),
|
|
206
|
+
(row.get("BillingPeriodEnd") or "").strip(),
|
|
207
|
+
(row.get("InvoiceIssuerName") or "").strip(),
|
|
208
|
+
)
|
|
209
|
+
composite = _composite(*key)
|
|
210
|
+
if not key[0] or composite in known or composite in reported:
|
|
211
|
+
continue
|
|
212
|
+
reported[composite] = ""
|
|
213
|
+
out.append(
|
|
214
|
+
Diagnostic(
|
|
215
|
+
code="FDT-CROSS-040",
|
|
216
|
+
severity=Severity.ERROR,
|
|
217
|
+
message="no Billing Period row for a (period, issuer) present in Cost and Usage",
|
|
218
|
+
datasets=("Cost and Usage", "Billing Period"),
|
|
219
|
+
dataset="Cost and Usage",
|
|
220
|
+
line_number=i,
|
|
221
|
+
record_keys={
|
|
222
|
+
"BillingPeriodStart": key[0],
|
|
223
|
+
"BillingPeriodEnd": key[1],
|
|
224
|
+
"InvoiceIssuerName": key[2],
|
|
225
|
+
},
|
|
226
|
+
)
|
|
227
|
+
)
|
|
228
|
+
return out
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
# Attributes that must agree between a Cost and Usage row and the Invoice Detail line it
|
|
232
|
+
# links to. InvoiceId / ChargeCategory identify *which* invoice line the row belongs to: a
|
|
233
|
+
# same-amount line under a different invoice must not be silently accepted.
|
|
234
|
+
_CONSISTENCY_COLUMNS = {
|
|
235
|
+
"InvoiceId": "FDT-CROSS-015",
|
|
236
|
+
"ChargeCategory": "FDT-CROSS-015",
|
|
237
|
+
"BillingCurrency": "FDT-CROSS-020",
|
|
238
|
+
"BillingPeriodStart": "FDT-CROSS-021",
|
|
239
|
+
"BillingPeriodEnd": "FDT-CROSS-021",
|
|
240
|
+
"InvoiceIssuerName": "FDT-CROSS-022",
|
|
241
|
+
"BillingAccountId": "FDT-CROSS-023",
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def check_cost_and_usage_invoice_detail_consistency(
|
|
246
|
+
cost_and_usage: RowStream,
|
|
247
|
+
invoice_detail: RowStream,
|
|
248
|
+
*,
|
|
249
|
+
index_factory: IndexFactory = dict,
|
|
250
|
+
) -> list[Diagnostic]:
|
|
251
|
+
"""A Cost and Usage row's invoice/category/issuer/account/currency/period must match its
|
|
252
|
+
Invoice Detail line (so a row cannot be attached to the wrong invoice line)."""
|
|
253
|
+
# Only the compared columns are indexed (id -> JSON array of their stripped values), so
|
|
254
|
+
# the lookup stays small per line and spills cleanly through a string map.
|
|
255
|
+
index = index_factory()
|
|
256
|
+
for r in invoice_detail:
|
|
257
|
+
detail_id = (r.get("InvoiceDetailId") or "").strip()
|
|
258
|
+
if detail_id:
|
|
259
|
+
index[detail_id] = _composite(
|
|
260
|
+
*((r.get(c) or "").strip() for c in _CONSISTENCY_COLUMNS)
|
|
261
|
+
)
|
|
262
|
+
out: list[Diagnostic] = []
|
|
263
|
+
for i, row in enumerate(cost_and_usage, start=1):
|
|
264
|
+
ref = (row.get("InvoiceDetailId") or "").strip()
|
|
265
|
+
packed = index.get(ref) if ref else None
|
|
266
|
+
if packed is None:
|
|
267
|
+
continue # missing FK already reported elsewhere
|
|
268
|
+
expected_values = json.loads(packed)
|
|
269
|
+
for (column, code), expected in zip(
|
|
270
|
+
_CONSISTENCY_COLUMNS.items(), expected_values, strict=True
|
|
271
|
+
):
|
|
272
|
+
actual = (row.get(column) or "").strip()
|
|
273
|
+
if expected != actual:
|
|
274
|
+
out.append(
|
|
275
|
+
Diagnostic(
|
|
276
|
+
code=code,
|
|
277
|
+
severity=Severity.ERROR,
|
|
278
|
+
message=f"{column} differs between Cost and Usage row {i} and its "
|
|
279
|
+
f"Invoice Detail line {ref!r}",
|
|
280
|
+
datasets=("Cost and Usage", "Invoice Detail"),
|
|
281
|
+
dataset="Cost and Usage",
|
|
282
|
+
line_number=i,
|
|
283
|
+
column=column,
|
|
284
|
+
expected=expected,
|
|
285
|
+
actual=actual,
|
|
286
|
+
record_keys={"InvoiceDetailId": ref, **_cu_keys(row)},
|
|
287
|
+
)
|
|
288
|
+
)
|
|
289
|
+
return out
|