focus-data-toolkit 0.11.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. focus_data_toolkit/__init__.py +69 -0
  2. focus_data_toolkit/__main__.py +6 -0
  3. focus_data_toolkit/_version.py +8 -0
  4. focus_data_toolkit/cli.py +968 -0
  5. focus_data_toolkit/context/__init__.py +88 -0
  6. focus_data_toolkit/context/billing.py +54 -0
  7. focus_data_toolkit/context/provider.py +90 -0
  8. focus_data_toolkit/convert/__init__.py +708 -0
  9. focus_data_toolkit/convert/billing_period.py +65 -0
  10. focus_data_toolkit/convert/contract_applied.py +235 -0
  11. focus_data_toolkit/convert/contract_commitment.py +182 -0
  12. focus_data_toolkit/convert/cost_and_usage.py +179 -0
  13. focus_data_toolkit/convert/detect.py +39 -0
  14. focus_data_toolkit/convert/invoice_detail.py +199 -0
  15. focus_data_toolkit/convert/streaming.py +1030 -0
  16. focus_data_toolkit/errors.py +145 -0
  17. focus_data_toolkit/focus_json.py +68 -0
  18. focus_data_toolkit/generators/__init__.py +61 -0
  19. focus_data_toolkit/generators/_shim.py +43 -0
  20. focus_data_toolkit/generators/engine/__init__.py +14 -0
  21. focus_data_toolkit/generators/engine/context.py +12 -0
  22. focus_data_toolkit/generators/engine/determinism.py +117 -0
  23. focus_data_toolkit/generators/engine/json_focus.py +63 -0
  24. focus_data_toolkit/generators/engine/ladder.py +71 -0
  25. focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
  26. focus_data_toolkit/generators/engine/serialize.py +151 -0
  27. focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
  28. focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
  29. focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
  30. focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
  31. focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
  32. focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
  33. focus_data_toolkit/generators/providers/__init__.py +29 -0
  34. focus_data_toolkit/generators/providers/aws.py +186 -0
  35. focus_data_toolkit/generators/providers/azure.py +191 -0
  36. focus_data_toolkit/generators/providers/gcp.py +194 -0
  37. focus_data_toolkit/generators/providers/profile.py +123 -0
  38. focus_data_toolkit/generators/scenarios.py +178 -0
  39. focus_data_toolkit/generators/versions/__init__.py +17 -0
  40. focus_data_toolkit/generators/versions/adapter.py +41 -0
  41. focus_data_toolkit/generators/versions/v1_2.py +111 -0
  42. focus_data_toolkit/generators/versions/v1_3.py +154 -0
  43. focus_data_toolkit/io/__init__.py +1 -0
  44. focus_data_toolkit/io/atomic_writer.py +462 -0
  45. focus_data_toolkit/io/csv_io.py +128 -0
  46. focus_data_toolkit/io/parquet_io.py +528 -0
  47. focus_data_toolkit/io/records.py +92 -0
  48. focus_data_toolkit/io/row_source.py +117 -0
  49. focus_data_toolkit/lifecycle.py +342 -0
  50. focus_data_toolkit/manifest.py +114 -0
  51. focus_data_toolkit/model/__init__.py +43 -0
  52. focus_data_toolkit/model/capabilities.py +66 -0
  53. focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
  54. focus_data_toolkit/model/focus_1_4_model.json +1913 -0
  55. focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
  56. focus_data_toolkit/model/focus_json_keys.py +112 -0
  57. focus_data_toolkit/model/iso_4217_currencies.json +23 -0
  58. focus_data_toolkit/model/json_schema_check.py +205 -0
  59. focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
  60. focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
  61. focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
  62. focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
  63. focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
  64. focus_data_toolkit/model/model_provenance.json +58 -0
  65. focus_data_toolkit/model/validator.py +498 -0
  66. focus_data_toolkit/modes.py +18 -0
  67. focus_data_toolkit/official_validator.py +61 -0
  68. focus_data_toolkit/progress.py +89 -0
  69. focus_data_toolkit/provenance.py +106 -0
  70. focus_data_toolkit/py.typed +1 -0
  71. focus_data_toolkit/runtime.py +243 -0
  72. focus_data_toolkit/schema/__init__.py +17 -0
  73. focus_data_toolkit/schema/detection.py +274 -0
  74. focus_data_toolkit/schema/registry.py +127 -0
  75. focus_data_toolkit/storage/__init__.py +1 -0
  76. focus_data_toolkit/storage/external_index.py +99 -0
  77. focus_data_toolkit/storage/spill.py +150 -0
  78. focus_data_toolkit/studio/__init__.py +19 -0
  79. focus_data_toolkit/studio/app.py +467 -0
  80. focus_data_toolkit/studio/config.py +42 -0
  81. focus_data_toolkit/studio/frontend/app.js +214 -0
  82. focus_data_toolkit/studio/frontend/index.html +101 -0
  83. focus_data_toolkit/studio/frontend/style.css +60 -0
  84. focus_data_toolkit/studio/jobs.py +142 -0
  85. focus_data_toolkit/studio/preview.py +32 -0
  86. focus_data_toolkit/studio/security.py +125 -0
  87. focus_data_toolkit/studio/server.py +71 -0
  88. focus_data_toolkit/supplement/__init__.py +50 -0
  89. focus_data_toolkit/supplement/adapters/__init__.py +21 -0
  90. focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
  91. focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
  92. focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
  93. focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
  94. focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
  95. focus_data_toolkit/supplement/adapters/registry.py +215 -0
  96. focus_data_toolkit/supplement/apply.py +318 -0
  97. focus_data_toolkit/supplement/gaps.py +219 -0
  98. focus_data_toolkit/supplement/kinds.py +118 -0
  99. focus_data_toolkit/supplement/loader.py +409 -0
  100. focus_data_toolkit/supplement/spec.py +74 -0
  101. focus_data_toolkit/supplement/validate.py +215 -0
  102. focus_data_toolkit/validate/__init__.py +15 -0
  103. focus_data_toolkit/validate/allocation.py +333 -0
  104. focus_data_toolkit/validate/bundle.py +254 -0
  105. focus_data_toolkit/validate/codes.py +93 -0
  106. focus_data_toolkit/validate/corrections.py +245 -0
  107. focus_data_toolkit/validate/reconciliation.py +98 -0
  108. focus_data_toolkit/validate/referential.py +289 -0
  109. focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
  110. focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
  111. focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
  112. focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
  113. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
  114. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
  115. focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
  116. focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,98 @@
1
+ """Reconcile Cost and Usage against an authoritative Invoice Detail dataset.
2
+
3
+ FOCUS 1.4 explicitly allows an issued invoice to differ from summed usage (it adds an
4
+ *Invoice Reconciliation* feature and a *Rounding Variance Tolerance* appendix). So this check
5
+ runs only when Invoice Detail comes from an **authoritative** source, and compares sums with an
6
+ explicit, documented tolerance rather than requiring exact equality.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from collections.abc import Callable, Iterable, Mapping, MutableMapping, Sequence
12
+ from decimal import Decimal, InvalidOperation
13
+
14
+ from focus_data_toolkit.errors import Diagnostic, Severity
15
+
16
+ Rows = Sequence[Mapping[str, str]]
17
+ #: Inputs are consumed in forward passes only, so any (re-)iterable of rows works.
18
+ RowStream = Iterable[Mapping[str, str]]
19
+ #: Factory for per-key lookup state (``dict`` by default; a spillable map for streaming).
20
+ IndexFactory = Callable[[], MutableMapping[str, str]]
21
+
22
+ # Default rounding-variance tolerance for a reconciled sum (absolute, in billing currency).
23
+ DEFAULT_TOLERANCE = Decimal("0.01")
24
+
25
+
26
+ def _dec(value: str | None) -> Decimal:
27
+ try:
28
+ parsed = Decimal((value or "0").strip() or "0")
29
+ except InvalidOperation:
30
+ return Decimal(0)
31
+ # NaN / Infinity parse successfully but would raise InvalidOperation on comparison; a
32
+ # malformed numeric value must not crash reconciliation (the linter flags its format).
33
+ return parsed if parsed.is_finite() else Decimal(0)
34
+
35
+
36
+ def reconcile_invoice_detail(
37
+ cost_and_usage: RowStream,
38
+ invoice_detail: RowStream,
39
+ *,
40
+ tolerance: Decimal = DEFAULT_TOLERANCE,
41
+ index_factory: IndexFactory = dict,
42
+ ) -> list[Diagnostic]:
43
+ """Compare each Invoice Detail BilledCost to the sum of its Cost and Usage lines.
44
+
45
+ Cost and Usage rows are attributed to an invoice line by their ``InvoiceDetailId``
46
+ back-link. A line with no matching Cost and Usage rows is a warning; a sum that differs
47
+ beyond ``tolerance`` is an error carrying the expected/actual amounts.
48
+ """
49
+ # Sums are stored as exact Decimal strings so the state can live in a spillable map.
50
+ sums = index_factory()
51
+ for row in cost_and_usage:
52
+ ref = (row.get("InvoiceDetailId") or "").strip()
53
+ if ref:
54
+ current = sums.get(ref)
55
+ sums[ref] = str(
56
+ (Decimal(current) if current is not None else Decimal(0))
57
+ + _dec(row.get("BilledCost"))
58
+ )
59
+
60
+ out: list[Diagnostic] = []
61
+ for i, detail in enumerate(invoice_detail, start=1):
62
+ detail_id = (detail.get("InvoiceDetailId") or "").strip()
63
+ if not detail_id:
64
+ continue
65
+ invoiced = _dec(detail.get("BilledCost"))
66
+ if detail_id not in sums:
67
+ out.append(
68
+ Diagnostic(
69
+ code="FDT-CROSS-031",
70
+ severity=Severity.WARNING,
71
+ message=f"Invoice Detail line {detail_id!r} has no matching Cost and Usage rows",
72
+ datasets=("Invoice Detail", "Cost and Usage"),
73
+ dataset="Invoice Detail",
74
+ line_number=i,
75
+ column="InvoiceDetailId",
76
+ value=detail_id,
77
+ record_keys={"InvoiceDetailId": detail_id},
78
+ )
79
+ )
80
+ continue
81
+ summed = Decimal(sums[detail_id])
82
+ if abs(summed - invoiced) > tolerance:
83
+ out.append(
84
+ Diagnostic(
85
+ code="FDT-CROSS-030",
86
+ severity=Severity.ERROR,
87
+ message=f"Invoice Detail line {detail_id!r} BilledCost does not reconcile "
88
+ f"with summed Cost and Usage (tolerance {tolerance})",
89
+ datasets=("Invoice Detail", "Cost and Usage"),
90
+ dataset="Invoice Detail",
91
+ line_number=i,
92
+ column="BilledCost",
93
+ expected=str(summed),
94
+ actual=str(invoiced),
95
+ record_keys={"InvoiceDetailId": detail_id},
96
+ )
97
+ )
98
+ return out
@@ -0,0 +1,289 @@
1
+ """Referential integrity across FOCUS datasets: uniqueness, foreign keys, coherence.
2
+
3
+ These checks operate on a *bundle* of datasets (dataset name -> rows) and never live inside
4
+ the per-dataset linter — they are inherently cross-dataset. Every finding is a structured
5
+ :class:`~focus_data_toolkit.errors.Diagnostic` carrying the offending record's business key.
6
+
7
+ Every check consumes each input in a single forward pass and keeps only per-key lookup
8
+ state; ``index_factory`` lets the caller back that state with a disk-spilling map
9
+ (:class:`~focus_data_toolkit.storage.spill.SpillableMap`) so validating datasets far larger
10
+ than RAM stays memory-bounded.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import json
16
+ from collections.abc import Callable, Iterable, Mapping, MutableMapping, Sequence
17
+
18
+ from focus_data_toolkit.convert.contract_applied import ContractAppliedError, parse
19
+ from focus_data_toolkit.errors import Diagnostic, Severity
20
+
21
+ Rows = Sequence[Mapping[str, str]]
22
+ #: Every dataset side of a cross-dataset check is consumed in a single forward pass, so it
23
+ #: accepts any iterable of rows (e.g. a staged-file stream) — bounded memory.
24
+ RowStream = Iterable[Mapping[str, str]]
25
+ #: Factory for the per-key lookup state (``dict`` by default; a spillable map for streaming).
26
+ IndexFactory = Callable[[], MutableMapping[str, str]]
27
+
28
+
29
+ def _composite(*parts: str) -> str:
30
+ """Unambiguous single-string key for a tuple of values (JSON array encoding)."""
31
+ return json.dumps(parts, separators=(",", ":"))
32
+
33
+
34
+ def _cu_keys(row: Mapping[str, str]) -> dict[str, str]:
35
+ """Business key for a Cost and Usage row, for diagnostics."""
36
+ keys = {}
37
+ for col in ("InvoiceIssuerName", "InvoiceId", "BillingAccountId", "ResourceId"):
38
+ val = (row.get(col) or "").strip()
39
+ if val:
40
+ keys[col] = val
41
+ return keys
42
+
43
+
44
+ def check_unique_invoice_detail_ids(
45
+ invoice_detail: RowStream, *, index_factory: IndexFactory = dict
46
+ ) -> list[Diagnostic]:
47
+ """InvoiceDetailId must be unique within Invoice Detail."""
48
+ seen = index_factory()
49
+ out: list[Diagnostic] = []
50
+ for i, row in enumerate(invoice_detail, start=1):
51
+ detail_id = (row.get("InvoiceDetailId") or "").strip()
52
+ if not detail_id:
53
+ continue
54
+ if detail_id in seen:
55
+ out.append(
56
+ Diagnostic(
57
+ code="FDT-CROSS-001",
58
+ severity=Severity.ERROR,
59
+ message=f"InvoiceDetailId {detail_id!r} is not unique in Invoice Detail "
60
+ f"(rows {seen[detail_id]} and {i})",
61
+ datasets=("Invoice Detail",),
62
+ dataset="Invoice Detail",
63
+ line_number=i,
64
+ column="InvoiceDetailId",
65
+ value=detail_id,
66
+ record_keys={"InvoiceDetailId": detail_id},
67
+ )
68
+ )
69
+ else:
70
+ seen[detail_id] = str(i)
71
+ return out
72
+
73
+
74
+ def check_unique_contract_commitment_ids(
75
+ contract_commitment: RowStream, *, index_factory: IndexFactory = dict
76
+ ) -> list[Diagnostic]:
77
+ """ContractCommitmentId must be unique (ContractApplied resolves commitments by this id)."""
78
+ seen = index_factory()
79
+ out: list[Diagnostic] = []
80
+ for i, row in enumerate(contract_commitment, start=1):
81
+ commitment_id = (row.get("ContractCommitmentId") or "").strip()
82
+ if not commitment_id:
83
+ continue
84
+ if commitment_id in seen:
85
+ out.append(
86
+ Diagnostic(
87
+ code="FDT-CROSS-001",
88
+ severity=Severity.ERROR,
89
+ message=f"ContractCommitmentId {commitment_id!r} is not unique in Contract "
90
+ f"Commitment (rows {seen[commitment_id]} and {i})",
91
+ datasets=("Contract Commitment",),
92
+ dataset="Contract Commitment",
93
+ line_number=i,
94
+ column="ContractCommitmentId",
95
+ value=commitment_id,
96
+ record_keys={"ContractCommitmentId": commitment_id},
97
+ )
98
+ )
99
+ else:
100
+ seen[commitment_id] = str(i)
101
+ return out
102
+
103
+
104
+ def check_cost_and_usage_invoice_detail_fk(
105
+ cost_and_usage: RowStream,
106
+ invoice_detail: RowStream,
107
+ *,
108
+ index_factory: IndexFactory = dict,
109
+ ) -> list[Diagnostic]:
110
+ """Every non-empty Cost and Usage InvoiceDetailId must exist in Invoice Detail."""
111
+ known = index_factory()
112
+ for r in invoice_detail:
113
+ detail_id = (r.get("InvoiceDetailId") or "").strip()
114
+ if detail_id:
115
+ known[detail_id] = ""
116
+ out: list[Diagnostic] = []
117
+ for i, row in enumerate(cost_and_usage, start=1):
118
+ ref = (row.get("InvoiceDetailId") or "").strip()
119
+ if not ref or ref in known:
120
+ continue
121
+ out.append(
122
+ Diagnostic(
123
+ code="FDT-CROSS-014",
124
+ severity=Severity.ERROR,
125
+ message=f"InvoiceDetailId {ref!r} referenced by Cost and Usage row {i} was not "
126
+ "found in Invoice Detail",
127
+ datasets=("Cost and Usage", "Invoice Detail"),
128
+ dataset="Cost and Usage",
129
+ line_number=i,
130
+ column="InvoiceDetailId",
131
+ value=ref,
132
+ record_keys=_cu_keys(row),
133
+ source="Cost and Usage",
134
+ )
135
+ )
136
+ return out
137
+
138
+
139
+ def _commitment_ids(cost_and_usage_row: Mapping[str, str]) -> list[str]:
140
+ """ContractCommitmentIds referenced by a row's ContractApplied JSON (best effort)."""
141
+ text = (cost_and_usage_row.get("ContractApplied") or "").strip()
142
+ if not text:
143
+ return []
144
+ try:
145
+ applied = parse(text, version="1.4")
146
+ except ContractAppliedError:
147
+ return [] # structural validity is the per-dataset linter's job
148
+ return [e.contract_commitment_id for e in applied.elements if e.contract_commitment_id]
149
+
150
+
151
+ def check_contract_applied_fk(
152
+ cost_and_usage: RowStream,
153
+ contract_commitment: RowStream,
154
+ *,
155
+ index_factory: IndexFactory = dict,
156
+ ) -> list[Diagnostic]:
157
+ """Every ContractCommitmentId referenced from ContractApplied must exist."""
158
+ known = index_factory()
159
+ for r in contract_commitment:
160
+ commitment_id = (r.get("ContractCommitmentId") or "").strip()
161
+ if commitment_id:
162
+ known[commitment_id] = ""
163
+ out: list[Diagnostic] = []
164
+ for i, row in enumerate(cost_and_usage, start=1):
165
+ for commitment_id in _commitment_ids(row):
166
+ if commitment_id in known:
167
+ continue
168
+ out.append(
169
+ Diagnostic(
170
+ code="FDT-CROSS-010",
171
+ severity=Severity.ERROR,
172
+ message=f"ContractApplied on Cost and Usage row {i} references "
173
+ f"ContractCommitmentId {commitment_id!r} not found in Contract Commitment",
174
+ datasets=("Cost and Usage", "Contract Commitment"),
175
+ dataset="Cost and Usage",
176
+ line_number=i,
177
+ column="ContractApplied",
178
+ value=commitment_id,
179
+ record_keys=_cu_keys(row),
180
+ )
181
+ )
182
+ return out
183
+
184
+
185
+ def check_billing_period_coverage(
186
+ cost_and_usage: RowStream,
187
+ billing_period: RowStream,
188
+ *,
189
+ index_factory: IndexFactory = dict,
190
+ ) -> list[Diagnostic]:
191
+ """Every (period, issuer) seen in Cost and Usage must have a Billing Period row."""
192
+ known = index_factory()
193
+ for r in billing_period:
194
+ known[
195
+ _composite(
196
+ (r.get("BillingPeriodStart") or "").strip(),
197
+ (r.get("BillingPeriodEnd") or "").strip(),
198
+ (r.get("InvoiceIssuerName") or "").strip(),
199
+ )
200
+ ] = ""
201
+ reported = index_factory()
202
+ out: list[Diagnostic] = []
203
+ for i, row in enumerate(cost_and_usage, start=1):
204
+ key = (
205
+ (row.get("BillingPeriodStart") or "").strip(),
206
+ (row.get("BillingPeriodEnd") or "").strip(),
207
+ (row.get("InvoiceIssuerName") or "").strip(),
208
+ )
209
+ composite = _composite(*key)
210
+ if not key[0] or composite in known or composite in reported:
211
+ continue
212
+ reported[composite] = ""
213
+ out.append(
214
+ Diagnostic(
215
+ code="FDT-CROSS-040",
216
+ severity=Severity.ERROR,
217
+ message="no Billing Period row for a (period, issuer) present in Cost and Usage",
218
+ datasets=("Cost and Usage", "Billing Period"),
219
+ dataset="Cost and Usage",
220
+ line_number=i,
221
+ record_keys={
222
+ "BillingPeriodStart": key[0],
223
+ "BillingPeriodEnd": key[1],
224
+ "InvoiceIssuerName": key[2],
225
+ },
226
+ )
227
+ )
228
+ return out
229
+
230
+
231
+ # Attributes that must agree between a Cost and Usage row and the Invoice Detail line it
232
+ # links to. InvoiceId / ChargeCategory identify *which* invoice line the row belongs to: a
233
+ # same-amount line under a different invoice must not be silently accepted.
234
+ _CONSISTENCY_COLUMNS = {
235
+ "InvoiceId": "FDT-CROSS-015",
236
+ "ChargeCategory": "FDT-CROSS-015",
237
+ "BillingCurrency": "FDT-CROSS-020",
238
+ "BillingPeriodStart": "FDT-CROSS-021",
239
+ "BillingPeriodEnd": "FDT-CROSS-021",
240
+ "InvoiceIssuerName": "FDT-CROSS-022",
241
+ "BillingAccountId": "FDT-CROSS-023",
242
+ }
243
+
244
+
245
+ def check_cost_and_usage_invoice_detail_consistency(
246
+ cost_and_usage: RowStream,
247
+ invoice_detail: RowStream,
248
+ *,
249
+ index_factory: IndexFactory = dict,
250
+ ) -> list[Diagnostic]:
251
+ """A Cost and Usage row's invoice/category/issuer/account/currency/period must match its
252
+ Invoice Detail line (so a row cannot be attached to the wrong invoice line)."""
253
+ # Only the compared columns are indexed (id -> JSON array of their stripped values), so
254
+ # the lookup stays small per line and spills cleanly through a string map.
255
+ index = index_factory()
256
+ for r in invoice_detail:
257
+ detail_id = (r.get("InvoiceDetailId") or "").strip()
258
+ if detail_id:
259
+ index[detail_id] = _composite(
260
+ *((r.get(c) or "").strip() for c in _CONSISTENCY_COLUMNS)
261
+ )
262
+ out: list[Diagnostic] = []
263
+ for i, row in enumerate(cost_and_usage, start=1):
264
+ ref = (row.get("InvoiceDetailId") or "").strip()
265
+ packed = index.get(ref) if ref else None
266
+ if packed is None:
267
+ continue # missing FK already reported elsewhere
268
+ expected_values = json.loads(packed)
269
+ for (column, code), expected in zip(
270
+ _CONSISTENCY_COLUMNS.items(), expected_values, strict=True
271
+ ):
272
+ actual = (row.get(column) or "").strip()
273
+ if expected != actual:
274
+ out.append(
275
+ Diagnostic(
276
+ code=code,
277
+ severity=Severity.ERROR,
278
+ message=f"{column} differs between Cost and Usage row {i} and its "
279
+ f"Invoice Detail line {ref!r}",
280
+ datasets=("Cost and Usage", "Invoice Detail"),
281
+ dataset="Cost and Usage",
282
+ line_number=i,
283
+ column=column,
284
+ expected=expected,
285
+ actual=actual,
286
+ record_keys={"InvoiceDetailId": ref, **_cu_keys(row)},
287
+ )
288
+ )
289
+ return out