focus-data-toolkit 0.11.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. focus_data_toolkit/__init__.py +69 -0
  2. focus_data_toolkit/__main__.py +6 -0
  3. focus_data_toolkit/_version.py +8 -0
  4. focus_data_toolkit/cli.py +968 -0
  5. focus_data_toolkit/context/__init__.py +88 -0
  6. focus_data_toolkit/context/billing.py +54 -0
  7. focus_data_toolkit/context/provider.py +90 -0
  8. focus_data_toolkit/convert/__init__.py +708 -0
  9. focus_data_toolkit/convert/billing_period.py +65 -0
  10. focus_data_toolkit/convert/contract_applied.py +235 -0
  11. focus_data_toolkit/convert/contract_commitment.py +182 -0
  12. focus_data_toolkit/convert/cost_and_usage.py +179 -0
  13. focus_data_toolkit/convert/detect.py +39 -0
  14. focus_data_toolkit/convert/invoice_detail.py +199 -0
  15. focus_data_toolkit/convert/streaming.py +1030 -0
  16. focus_data_toolkit/errors.py +145 -0
  17. focus_data_toolkit/focus_json.py +68 -0
  18. focus_data_toolkit/generators/__init__.py +61 -0
  19. focus_data_toolkit/generators/_shim.py +43 -0
  20. focus_data_toolkit/generators/engine/__init__.py +14 -0
  21. focus_data_toolkit/generators/engine/context.py +12 -0
  22. focus_data_toolkit/generators/engine/determinism.py +117 -0
  23. focus_data_toolkit/generators/engine/json_focus.py +63 -0
  24. focus_data_toolkit/generators/engine/ladder.py +71 -0
  25. focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
  26. focus_data_toolkit/generators/engine/serialize.py +151 -0
  27. focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
  28. focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
  29. focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
  30. focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
  31. focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
  32. focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
  33. focus_data_toolkit/generators/providers/__init__.py +29 -0
  34. focus_data_toolkit/generators/providers/aws.py +186 -0
  35. focus_data_toolkit/generators/providers/azure.py +191 -0
  36. focus_data_toolkit/generators/providers/gcp.py +194 -0
  37. focus_data_toolkit/generators/providers/profile.py +123 -0
  38. focus_data_toolkit/generators/scenarios.py +178 -0
  39. focus_data_toolkit/generators/versions/__init__.py +17 -0
  40. focus_data_toolkit/generators/versions/adapter.py +41 -0
  41. focus_data_toolkit/generators/versions/v1_2.py +111 -0
  42. focus_data_toolkit/generators/versions/v1_3.py +154 -0
  43. focus_data_toolkit/io/__init__.py +1 -0
  44. focus_data_toolkit/io/atomic_writer.py +462 -0
  45. focus_data_toolkit/io/csv_io.py +128 -0
  46. focus_data_toolkit/io/parquet_io.py +528 -0
  47. focus_data_toolkit/io/records.py +92 -0
  48. focus_data_toolkit/io/row_source.py +117 -0
  49. focus_data_toolkit/lifecycle.py +342 -0
  50. focus_data_toolkit/manifest.py +114 -0
  51. focus_data_toolkit/model/__init__.py +43 -0
  52. focus_data_toolkit/model/capabilities.py +66 -0
  53. focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
  54. focus_data_toolkit/model/focus_1_4_model.json +1913 -0
  55. focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
  56. focus_data_toolkit/model/focus_json_keys.py +112 -0
  57. focus_data_toolkit/model/iso_4217_currencies.json +23 -0
  58. focus_data_toolkit/model/json_schema_check.py +205 -0
  59. focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
  60. focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
  61. focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
  62. focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
  63. focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
  64. focus_data_toolkit/model/model_provenance.json +58 -0
  65. focus_data_toolkit/model/validator.py +498 -0
  66. focus_data_toolkit/modes.py +18 -0
  67. focus_data_toolkit/official_validator.py +61 -0
  68. focus_data_toolkit/progress.py +89 -0
  69. focus_data_toolkit/provenance.py +106 -0
  70. focus_data_toolkit/py.typed +1 -0
  71. focus_data_toolkit/runtime.py +243 -0
  72. focus_data_toolkit/schema/__init__.py +17 -0
  73. focus_data_toolkit/schema/detection.py +274 -0
  74. focus_data_toolkit/schema/registry.py +127 -0
  75. focus_data_toolkit/storage/__init__.py +1 -0
  76. focus_data_toolkit/storage/external_index.py +99 -0
  77. focus_data_toolkit/storage/spill.py +150 -0
  78. focus_data_toolkit/studio/__init__.py +19 -0
  79. focus_data_toolkit/studio/app.py +467 -0
  80. focus_data_toolkit/studio/config.py +42 -0
  81. focus_data_toolkit/studio/frontend/app.js +214 -0
  82. focus_data_toolkit/studio/frontend/index.html +101 -0
  83. focus_data_toolkit/studio/frontend/style.css +60 -0
  84. focus_data_toolkit/studio/jobs.py +142 -0
  85. focus_data_toolkit/studio/preview.py +32 -0
  86. focus_data_toolkit/studio/security.py +125 -0
  87. focus_data_toolkit/studio/server.py +71 -0
  88. focus_data_toolkit/supplement/__init__.py +50 -0
  89. focus_data_toolkit/supplement/adapters/__init__.py +21 -0
  90. focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
  91. focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
  92. focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
  93. focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
  94. focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
  95. focus_data_toolkit/supplement/adapters/registry.py +215 -0
  96. focus_data_toolkit/supplement/apply.py +318 -0
  97. focus_data_toolkit/supplement/gaps.py +219 -0
  98. focus_data_toolkit/supplement/kinds.py +118 -0
  99. focus_data_toolkit/supplement/loader.py +409 -0
  100. focus_data_toolkit/supplement/spec.py +74 -0
  101. focus_data_toolkit/supplement/validate.py +215 -0
  102. focus_data_toolkit/validate/__init__.py +15 -0
  103. focus_data_toolkit/validate/allocation.py +333 -0
  104. focus_data_toolkit/validate/bundle.py +254 -0
  105. focus_data_toolkit/validate/codes.py +93 -0
  106. focus_data_toolkit/validate/corrections.py +245 -0
  107. focus_data_toolkit/validate/reconciliation.py +98 -0
  108. focus_data_toolkit/validate/referential.py +289 -0
  109. focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
  110. focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
  111. focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
  112. focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
  113. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
  114. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
  115. focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
  116. focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,254 @@
1
+ """Validate a *bundle* of FOCUS datasets together (P1.4).
2
+
3
+ This layer is deliberately separate from the per-dataset linter (``model/validator.py``): it
4
+ asserts the cross-dataset guarantees the linter explicitly does not — referential integrity,
5
+ uniqueness, currency/period/issuer coherence, reconciliation, split cost allocation, and
6
+ commitment lifecycle. ``validate_dataset_bundle`` returns a :class:`BundleReport` whose
7
+ diagnostics are grouped by severity (error / warning / info / not-executable / not-applicable)
8
+ and which serialises to JSON.
9
+
10
+ Memory model: no dataset is ever materialised. Each check consumes its inputs in independent
11
+ forward passes, so every bundle value must be **re-iterable** — a list, or an object whose
12
+ ``__iter__`` opens a fresh scan (e.g. a staged-file reader); a one-shot generator is rejected.
13
+ Per-key lookup state (seen ids, foreign-key targets, running sums) is created through
14
+ ``index_factory``, which a streaming caller points at a disk-spilling map
15
+ (:class:`~focus_data_toolkit.storage.spill.SpillableIndexPool`) to validate bundles far
16
+ larger than RAM.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ from collections.abc import Callable, Iterable, Mapping, MutableMapping
22
+ from dataclasses import dataclass
23
+ from decimal import Decimal
24
+
25
+ from focus_data_toolkit.errors import Diagnostic, Severity
26
+ from focus_data_toolkit.validate import allocation, corrections, reconciliation, referential
27
+
28
+ Rows = Iterable[Mapping[str, str]]
29
+ Bundle = Mapping[str, Rows]
30
+ IndexFactory = Callable[[], MutableMapping[str, str]]
31
+
32
+
33
+ @dataclass
34
+ class BundleReport:
35
+ """Result of validating a dataset bundle."""
36
+
37
+ diagnostics: list[Diagnostic]
38
+ checks_run: tuple[str, ...] = ()
39
+
40
+ @property
41
+ def ok(self) -> bool:
42
+ """No failing (ERROR) diagnostics."""
43
+ return not any(d.is_failure for d in self.diagnostics)
44
+
45
+ def by_severity(self, severity: Severity) -> list[Diagnostic]:
46
+ return [d for d in self.diagnostics if d.severity is severity]
47
+
48
+ @property
49
+ def errors(self) -> list[Diagnostic]:
50
+ return self.by_severity(Severity.ERROR)
51
+
52
+ @property
53
+ def warnings(self) -> list[Diagnostic]:
54
+ return self.by_severity(Severity.WARNING)
55
+
56
+ def counts(self) -> dict[str, int]:
57
+ counts: dict[str, int] = {s.value: 0 for s in Severity}
58
+ for diag in self.diagnostics:
59
+ counts[diag.severity.value] += 1
60
+ return counts
61
+
62
+ def as_dict(self) -> dict:
63
+ return {
64
+ "ok": self.ok,
65
+ "counts": self.counts(),
66
+ "checks_run": list(self.checks_run),
67
+ "diagnostics": [d.as_dict() for d in self.diagnostics],
68
+ }
69
+
70
+ def format(self) -> str:
71
+ counts = self.counts()
72
+ header = "bundle validation: " + ("OK" if self.ok else "FAILED")
73
+ summary = ", ".join(f"{k.lower()}={v}" for k, v in counts.items() if v)
74
+ blocks = [header + (f" ({summary})" if summary else "")]
75
+ blocks.extend(d.format() for d in self.diagnostics)
76
+ return "\n".join(blocks)
77
+
78
+
79
+ def _note(code: str, severity: Severity, message: str, datasets: tuple[str, ...]) -> Diagnostic:
80
+ return Diagnostic(code=code, severity=severity, message=message, datasets=datasets)
81
+
82
+
83
+ def _reiterable(name: str, rows: Rows) -> Rows:
84
+ """Reject one-shot iterators: every check opens its own fresh pass over the rows."""
85
+ if iter(rows) is rows:
86
+ raise TypeError(
87
+ f"bundle dataset {name!r} is a one-shot iterator; validate_dataset_bundle "
88
+ "requires re-iterable rows (a list, or an object whose __iter__ opens a fresh scan)"
89
+ )
90
+ return rows
91
+
92
+
93
+ def _has_rows(rows: Rows) -> bool:
94
+ for _ in rows:
95
+ return True
96
+ return False
97
+
98
+
99
+ def validate_dataset_bundle(
100
+ bundle: Bundle,
101
+ *,
102
+ invoice_detail_authoritative: bool = False,
103
+ rounding_tolerance: Decimal | None = None,
104
+ index_factory: IndexFactory | None = None,
105
+ ) -> BundleReport:
106
+ """Validate the datasets in ``bundle`` against each other.
107
+
108
+ ``bundle`` maps FOCUS dataset names to their rows — each value must be re-iterable (see
109
+ the module docstring); nothing is materialised. ``invoice_detail_authoritative`` gates
110
+ the Cost-and-Usage <-> Invoice-Detail sum reconciliation: it runs only when the Invoice
111
+ Detail comes from a real invoice (a toolkit-derived Invoice Detail reconciles by
112
+ construction, so reconciling it would be circular). ``rounding_tolerance`` overrides the
113
+ reconciliation tolerance. ``index_factory`` supplies the per-key lookup state (``dict``
114
+ when omitted; pass ``SpillableIndexPool(...).make_map`` for bounded memory).
115
+ """
116
+ factory: IndexFactory = index_factory if index_factory is not None else dict
117
+ cu = _reiterable("Cost and Usage", bundle.get("Cost and Usage", ()))
118
+ invd = _reiterable("Invoice Detail", bundle.get("Invoice Detail", ()))
119
+ bp = _reiterable("Billing Period", bundle.get("Billing Period", ()))
120
+ cc = _reiterable("Contract Commitment", bundle.get("Contract Commitment", ()))
121
+ has_cu, has_invd, has_bp, has_cc = (_has_rows(r) for r in (cu, invd, bp, cc))
122
+
123
+ diagnostics: list[Diagnostic] = []
124
+ checks: list[str] = []
125
+
126
+ def run(name: str, produced: list[Diagnostic]) -> None:
127
+ checks.append(name)
128
+ diagnostics.extend(produced)
129
+
130
+ def _refs_present(rows: Rows, column: str) -> bool:
131
+ return any((r.get(column) or "").strip() for r in rows)
132
+
133
+ # Referential integrity.
134
+ if has_invd:
135
+ run(
136
+ "unique_invoice_detail_ids",
137
+ referential.check_unique_invoice_detail_ids(invd, index_factory=factory),
138
+ )
139
+ if has_cc:
140
+ run(
141
+ "unique_contract_commitment_ids",
142
+ referential.check_unique_contract_commitment_ids(cc, index_factory=factory),
143
+ )
144
+ if has_cu and has_invd:
145
+ run(
146
+ "cost_and_usage_invoice_detail_fk",
147
+ referential.check_cost_and_usage_invoice_detail_fk(
148
+ cu, invd, index_factory=factory
149
+ ),
150
+ )
151
+ run(
152
+ "cost_and_usage_invoice_detail_consistency",
153
+ referential.check_cost_and_usage_invoice_detail_consistency(
154
+ cu, invd, index_factory=factory
155
+ ),
156
+ )
157
+ elif has_cu and _refs_present(cu, "InvoiceDetailId"):
158
+ # References exist but their target table is absent -> the FK check cannot resolve them.
159
+ diagnostics.append(
160
+ _note(
161
+ "FDT-BUNDLE-001",
162
+ Severity.NOT_EXECUTABLE,
163
+ "Cost and Usage InvoiceDetailId references cannot be checked: the Invoice Detail "
164
+ "dataset is absent from the bundle",
165
+ ("Cost and Usage", "Invoice Detail"),
166
+ )
167
+ )
168
+ if has_cu and has_cc:
169
+ run(
170
+ "contract_applied_fk",
171
+ referential.check_contract_applied_fk(cu, cc, index_factory=factory),
172
+ )
173
+ elif has_cu and _refs_present(cu, "ContractApplied"):
174
+ diagnostics.append(
175
+ _note(
176
+ "FDT-BUNDLE-001",
177
+ Severity.NOT_EXECUTABLE,
178
+ "Cost and Usage ContractApplied references cannot be checked: the Contract "
179
+ "Commitment dataset is absent from the bundle",
180
+ ("Cost and Usage", "Contract Commitment"),
181
+ )
182
+ )
183
+ if has_cu and has_bp:
184
+ run(
185
+ "billing_period_coverage",
186
+ referential.check_billing_period_coverage(cu, bp, index_factory=factory),
187
+ )
188
+
189
+ # Reconciliation (only for an authoritative Invoice Detail).
190
+ if has_cu and has_invd:
191
+ if invoice_detail_authoritative:
192
+ tolerance = (
193
+ rounding_tolerance
194
+ if rounding_tolerance is not None
195
+ else reconciliation.DEFAULT_TOLERANCE
196
+ )
197
+ run(
198
+ "reconcile_invoice_detail",
199
+ reconciliation.reconcile_invoice_detail(
200
+ cu, invd, tolerance=tolerance, index_factory=factory
201
+ ),
202
+ )
203
+ else:
204
+ diagnostics.append(
205
+ _note(
206
+ "FDT-BUNDLE-002",
207
+ Severity.NOT_APPLICABLE,
208
+ "Cost and Usage <-> Invoice Detail reconciliation skipped: Invoice Detail is "
209
+ "not marked authoritative (a toolkit-derived Invoice Detail reconciles by "
210
+ "construction)",
211
+ ("Cost and Usage", "Invoice Detail"),
212
+ )
213
+ )
214
+ elif invoice_detail_authoritative:
215
+ diagnostics.append(
216
+ _note(
217
+ "FDT-BUNDLE-001",
218
+ Severity.NOT_EXECUTABLE,
219
+ "reconciliation not executable: Cost and Usage or Invoice Detail is absent",
220
+ ("Cost and Usage", "Invoice Detail"),
221
+ )
222
+ )
223
+
224
+ # Split cost allocation and correction integrity (self-contained within Cost and Usage).
225
+ if has_cu:
226
+ run(
227
+ "split_allocation",
228
+ allocation.validate_split_allocation(cu, index_factory=factory),
229
+ )
230
+ run(
231
+ "correction_references",
232
+ corrections.check_correction_references(cu, index_factory=factory),
233
+ )
234
+ run(
235
+ "correction_net_sums",
236
+ corrections.check_correction_net_sums(cu, index_factory=factory),
237
+ )
238
+ run(
239
+ "no_duplicate_charge_keys",
240
+ corrections.check_no_duplicate_charge_keys(cu, index_factory=factory),
241
+ )
242
+
243
+ # Commitment lifecycle.
244
+ if has_cc:
245
+ run("contract_commitment_periods", corrections.check_contract_commitment_periods(cc))
246
+ run(
247
+ "contract_commitment_percentages",
248
+ corrections.check_contract_commitment_percentages(cc),
249
+ )
250
+
251
+ return BundleReport(diagnostics=diagnostics, checks_run=tuple(checks))
252
+
253
+
254
+ __all__ = ["Bundle", "BundleReport", "validate_dataset_bundle"]
@@ -0,0 +1,93 @@
1
+ """Stable catalogue of focus-data-toolkit diagnostic codes.
2
+
3
+ Codes are **stable** identifiers — once assigned they are never renumbered or reused — so
4
+ downstream tooling can key on them across releases. Namespaces:
5
+
6
+ * ``FDT-DET-*`` — schema / version detection
7
+ * ``FDT-CROSS-*`` — inter-dataset referential integrity & reconciliation
8
+ * ``FDT-ALLOC-*`` — split cost allocation
9
+ * ``FDT-CORR-*`` — corrections / credits / billing lifecycle
10
+ * ``FDT-IO-*`` — input / output / format
11
+
12
+ Each code has a default severity and a one-line summary; call :func:`spec` to look one up.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ from dataclasses import dataclass
18
+
19
+ from focus_data_toolkit.errors import Severity
20
+
21
+
22
+ @dataclass(frozen=True)
23
+ class CodeSpec:
24
+ code: str
25
+ default_severity: Severity
26
+ summary: str
27
+
28
+
29
+ def _s(code: str, sev: Severity, summary: str) -> tuple[str, CodeSpec]:
30
+ return code, CodeSpec(code, sev, summary)
31
+
32
+
33
+ CATALOG: dict[str, CodeSpec] = dict([
34
+ # --- detection ---------------------------------------------------------------
35
+ _s("FDT-DET-001", Severity.ERROR, "schema/version detection confidence too low"),
36
+ _s("FDT-DET-002", Severity.ERROR, "ambiguous schema (multiple candidate versions/datasets)"),
37
+ _s("FDT-DET-003", Severity.ERROR, "forced version incompatible with the header"),
38
+ _s("FDT-DET-004", Severity.WARNING, "unknown non-x_ columns present in the source"),
39
+ # --- multi-provider / context ------------------------------------------------
40
+ _s("FDT-CTX-001", Severity.WARNING, "source carries multiple provider contexts"),
41
+ _s("FDT-CTX-002", Severity.WARNING, "source carries multiple invoice issuers"),
42
+ _s("FDT-CTX-003", Severity.WARNING, "source carries multiple billing currencies"),
43
+ _s("FDT-CTX-004", Severity.INFO, "a representative context was chosen for enrichment"),
44
+ # --- cross-dataset referential integrity & reconciliation --------------------
45
+ _s("FDT-CROSS-001", Severity.ERROR, "duplicate identifier where uniqueness is required"),
46
+ _s("FDT-CROSS-002", Severity.ERROR, "identifier collides across datasets"),
47
+ _s("FDT-CROSS-010", Severity.ERROR, "ContractApplied references an unknown ContractCommitmentId"),
48
+ _s("FDT-CROSS-011", Severity.ERROR, "referenced Contract Commitment not found"),
49
+ _s("FDT-CROSS-014", Severity.ERROR, "Cost and Usage InvoiceDetailId not found in Invoice Detail"),
50
+ _s("FDT-CROSS-015", Severity.ERROR, "linked record differs on an identifying attribute (wrong invoice line)"),
51
+ _s("FDT-CROSS-020", Severity.ERROR, "currency mismatch between linked records"),
52
+ _s("FDT-CROSS-021", Severity.ERROR, "billing period mismatch between linked records"),
53
+ _s("FDT-CROSS-022", Severity.ERROR, "invoice issuer mismatch between linked records"),
54
+ _s("FDT-CROSS-023", Severity.ERROR, "billing account mismatch between linked records"),
55
+ _s("FDT-CROSS-030", Severity.ERROR, "sum reconciliation mismatch beyond tolerance"),
56
+ _s("FDT-CROSS-031", Severity.WARNING, "invoice-detail line has no matching Cost and Usage rows"),
57
+ _s("FDT-CROSS-040", Severity.ERROR, "no Billing Period for a (period, issuer) seen in Cost and Usage"),
58
+ _s("FDT-CROSS-041", Severity.ERROR, "a closed billing period was modified"),
59
+ _s("FDT-CROSS-050", Severity.ERROR, "contract commitment period start is not before its end"),
60
+ _s("FDT-CROSS-051", Severity.ERROR, "percentage value outside its allowed range"),
61
+ # --- split cost allocation ---------------------------------------------------
62
+ _s("FDT-ALLOC-001", Severity.ERROR, "allocation ratios do not sum to 1 within tolerance"),
63
+ _s("FDT-ALLOC-002", Severity.ERROR, "allocated costs do not sum to the origin charge"),
64
+ _s("FDT-ALLOC-003", Severity.ERROR, "inconsistent allocation method within a group"),
65
+ _s("FDT-ALLOC-004", Severity.ERROR, "duplicate allocated resource within a group"),
66
+ _s("FDT-ALLOC-005", Severity.ERROR, "incomplete allocation group (missing required information)"),
67
+ _s("FDT-ALLOC-006", Severity.ERROR, "allocation ratio outside [0, 1]"),
68
+ _s("FDT-ALLOC-007", Severity.ERROR, "inconsistent unit within an allocation group"),
69
+ # --- corrections / lifecycle -------------------------------------------------
70
+ _s("FDT-CORR-001", Severity.ERROR, "correction references a missing invoice/charge"),
71
+ _s("FDT-CORR-002", Severity.ERROR, "net sum of a correction set does not reconcile"),
72
+ _s("FDT-CORR-003", Severity.ERROR, "correction overwrites an original row (no audit trail)"),
73
+ _s("FDT-CORR-004", Severity.ERROR, "invoice status transition is not allowed"),
74
+ # --- bundle coverage (a check could not run / does not apply) ----------------
75
+ _s("FDT-BUNDLE-001", Severity.NOT_EXECUTABLE, "a cross-dataset check could not run (data absent)"),
76
+ _s("FDT-BUNDLE-002", Severity.NOT_APPLICABLE, "a cross-dataset check does not apply to this bundle"),
77
+ # --- io / format -------------------------------------------------------------
78
+ _s("FDT-IO-001", Severity.ERROR, "malformed input record (wrong field count)"),
79
+ _s("FDT-IO-002", Severity.ERROR, "decimal value exceeds the target Parquet scale/precision"),
80
+ _s("FDT-IO-003", Severity.ERROR, "destination already exists"),
81
+ _s("FDT-IO-004", Severity.WARNING, "high-cardinality Parquet partition key (many small files)"),
82
+ _s("FDT-IO-005", Severity.ERROR, "insufficient free space on the output filesystem"),
83
+ _s("FDT-IO-006", Severity.ERROR, "insufficient work-filesystem space or temp budget exceeded"),
84
+ ])
85
+
86
+
87
+ def spec(code: str) -> CodeSpec:
88
+ """Return the :class:`CodeSpec` for ``code`` (raises ``KeyError`` if unknown)."""
89
+ return CATALOG[code]
90
+
91
+
92
+ def default_severity(code: str) -> Severity:
93
+ return CATALOG[code].default_severity
@@ -0,0 +1,245 @@
1
+ """Commitment lifecycle and correction-integrity checks.
2
+
3
+ Phase A covers the cheap, spec-grounded temporal and percentage-range checks on Contract
4
+ Commitment, plus a forward-compatible correction-reference check: a Cost and Usage correction
5
+ line (``ChargeClass="Correction"``) may point at the charge it corrects via the toolkit
6
+ extension ``x_CorrectionOf`` -> an original row's ``x_ChargeKey``; the referenced original must
7
+ still be present (corrections never silently overwrite history). Deeper correction scenarios
8
+ arrive with the Phase B generators.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ from collections.abc import Callable, Iterable, Mapping, MutableMapping, Sequence
14
+ from datetime import datetime
15
+ from decimal import Decimal, InvalidOperation
16
+
17
+ from focus_data_toolkit.errors import Diagnostic, Severity
18
+
19
+ Rows = Sequence[Mapping[str, str]]
20
+ #: Inputs are consumed in forward passes only, so any (re-)iterable of rows works.
21
+ RowStream = Iterable[Mapping[str, str]]
22
+ #: Factory for per-key lookup state (``dict`` by default; a spillable map for streaming).
23
+ IndexFactory = Callable[[], MutableMapping[str, str]]
24
+
25
+ _PERCENTAGE_COLUMNS = (
26
+ "ContractCommitmentDiscountPercentage",
27
+ "ContractCommitmentPaymentUpfrontPercentage",
28
+ )
29
+ _PERIOD_PAIRS = (
30
+ ("ContractCommitmentPeriodStart", "ContractCommitmentPeriodEnd"),
31
+ ("ContractPeriodStart", "ContractPeriodEnd"),
32
+ )
33
+
34
+
35
+ def _parse_dt(value: str) -> datetime | None:
36
+ try:
37
+ parsed = datetime.fromisoformat(value.strip().replace("Z", "+00:00"))
38
+ except ValueError:
39
+ return None
40
+ # Require a timezone (FOCUS Date/Time is UTC 'Z'). A naive value is malformed for this
41
+ # lifecycle check; returning None avoids a naive-vs-aware TypeError on comparison and
42
+ # lets the per-dataset linter report the bad Date/Time format.
43
+ return parsed if parsed.tzinfo is not None else None
44
+
45
+
46
+ def check_contract_commitment_periods(contract_commitment: RowStream) -> list[Diagnostic]:
47
+ """Each commitment/contract period must start strictly before it ends."""
48
+ out: list[Diagnostic] = []
49
+ for i, row in enumerate(contract_commitment, start=1):
50
+ for start_col, end_col in _PERIOD_PAIRS:
51
+ start, end = _parse_dt(row.get(start_col, "")), _parse_dt(row.get(end_col, ""))
52
+ if start is None or end is None or start < end:
53
+ continue
54
+ out.append(
55
+ Diagnostic(
56
+ code="FDT-CROSS-050",
57
+ severity=Severity.ERROR,
58
+ message=f"{start_col} is not before {end_col}",
59
+ datasets=("Contract Commitment",),
60
+ dataset="Contract Commitment",
61
+ line_number=i,
62
+ column=start_col,
63
+ expected=f"< {row.get(end_col)}",
64
+ actual=row.get(start_col, ""),
65
+ record_keys={"ContractCommitmentId": (row.get("ContractCommitmentId") or "")},
66
+ )
67
+ )
68
+ return out
69
+
70
+
71
+ def check_contract_commitment_percentages(contract_commitment: RowStream) -> list[Diagnostic]:
72
+ """Percentage columns must be within [0, 1]."""
73
+ out: list[Diagnostic] = []
74
+ for i, row in enumerate(contract_commitment, start=1):
75
+ for col in _PERCENTAGE_COLUMNS:
76
+ raw = (row.get(col) or "").strip()
77
+ if not raw:
78
+ continue
79
+ try:
80
+ value = Decimal(raw)
81
+ except InvalidOperation:
82
+ continue # value-format is the per-dataset linter's job
83
+ if not value.is_finite():
84
+ continue # NaN/Infinity is a format issue for the linter, not a range error
85
+ if value < 0 or value > 1:
86
+ out.append(
87
+ Diagnostic(
88
+ code="FDT-CROSS-051",
89
+ severity=Severity.ERROR,
90
+ message=f"{col} value {raw} is outside [0, 1]",
91
+ datasets=("Contract Commitment",),
92
+ dataset="Contract Commitment",
93
+ line_number=i,
94
+ column=col,
95
+ actual=raw,
96
+ record_keys={"ContractCommitmentId": (row.get("ContractCommitmentId") or "")},
97
+ )
98
+ )
99
+ return out
100
+
101
+
102
+ def _is_correction(row: Mapping[str, str]) -> bool:
103
+ return (row.get("ChargeClass") or "").strip().casefold() == "correction"
104
+
105
+
106
+ def _dec(value: str | None) -> Decimal | None:
107
+ try:
108
+ parsed = Decimal((value or "").strip())
109
+ except InvalidOperation:
110
+ return None
111
+ return parsed if parsed.is_finite() else None
112
+
113
+
114
+ def check_no_duplicate_charge_keys(
115
+ cost_and_usage: RowStream, *, index_factory: IndexFactory = dict
116
+ ) -> list[Diagnostic]:
117
+ """``x_ChargeKey`` must be unique: reusing one silently overwrites an auditable line.
118
+
119
+ Corrections are appended as *new* keyed rows (``x_ChargeKey`` + ``x_CorrectionOf``), so a
120
+ repeated key means an original (or a prior correction) was overwritten rather than amended.
121
+ """
122
+ seen = index_factory()
123
+ out: list[Diagnostic] = []
124
+ for i, row in enumerate(cost_and_usage, start=1):
125
+ key = (row.get("x_ChargeKey") or "").strip()
126
+ if not key:
127
+ continue
128
+ if key in seen:
129
+ out.append(
130
+ Diagnostic(
131
+ code="FDT-CORR-003",
132
+ severity=Severity.ERROR,
133
+ message=f"x_ChargeKey {key!r} appears on rows {seen[key]} and {i} — a "
134
+ "correction must add a new keyed row, never overwrite an original",
135
+ datasets=("Cost and Usage",),
136
+ dataset="Cost and Usage",
137
+ line_number=i,
138
+ column="x_ChargeKey",
139
+ value=key,
140
+ record_keys={"x_ChargeKey": key},
141
+ )
142
+ )
143
+ else:
144
+ seen[key] = str(i)
145
+ return out
146
+
147
+
148
+ def check_correction_net_sums(
149
+ cost_and_usage: RowStream,
150
+ *,
151
+ tolerance: Decimal = Decimal("0.01"),
152
+ index_factory: IndexFactory = dict,
153
+ ) -> list[Diagnostic]:
154
+ """The net of a correction set must equal the declared ``x_NetCharge``.
155
+
156
+ A correction set is an original charge (``x_ChargeKey`` == the corrections'
157
+ ``x_CorrectionOf``) plus every correction pointing at it. When a correction declares the
158
+ post-correction net in ``x_NetCharge``, the arithmetic sum of ``BilledCost`` over the set up
159
+ to and including that correction must match it — so the running net stays auditable and no
160
+ money is silently created or lost. Sets whose original is absent are handled by
161
+ :func:`check_correction_references`; this check only reconciles what is present.
162
+ """
163
+ # Decimals are stored as exact strings so the lookup state can live in a spillable map.
164
+ originals = index_factory()
165
+ for row in cost_and_usage:
166
+ if _is_correction(row):
167
+ continue
168
+ key = (row.get("x_ChargeKey") or "").strip()
169
+ cost = _dec(row.get("BilledCost"))
170
+ if key and cost is not None:
171
+ originals[key] = str(cost)
172
+
173
+ # Accumulate corrections per original in row order (running net is order-sensitive).
174
+ running = index_factory()
175
+ out: list[Diagnostic] = []
176
+ for i, row in enumerate(cost_and_usage, start=1):
177
+ if not _is_correction(row):
178
+ continue
179
+ ref = (row.get("x_CorrectionOf") or "").strip()
180
+ if ref not in originals:
181
+ continue # missing original -> reported by check_correction_references
182
+ delta = _dec(row.get("BilledCost"))
183
+ if delta is None:
184
+ continue
185
+ base = running.get(ref) or originals[ref]
186
+ net = Decimal(base) + delta
187
+ running[ref] = str(net)
188
+ declared = _dec(row.get("x_NetCharge"))
189
+ if declared is None:
190
+ continue
191
+ if abs(net - declared) > tolerance:
192
+ out.append(
193
+ Diagnostic(
194
+ code="FDT-CORR-002",
195
+ severity=Severity.ERROR,
196
+ message=f"correction set for {ref!r} nets to {net} but x_NetCharge "
197
+ f"declares {declared}",
198
+ datasets=("Cost and Usage",),
199
+ dataset="Cost and Usage",
200
+ line_number=i,
201
+ column="x_NetCharge",
202
+ expected=str(declared),
203
+ actual=str(net),
204
+ record_keys={"x_CorrectionOf": ref},
205
+ )
206
+ )
207
+ return out
208
+
209
+
210
+ def check_correction_references(
211
+ cost_and_usage: RowStream, *, index_factory: IndexFactory = dict
212
+ ) -> list[Diagnostic]:
213
+ """A correction line's ``x_CorrectionOf`` must point at a still-present *original* charge.
214
+
215
+ The lookup is built only from non-correction originals, and a correction that references
216
+ its own key is rejected — otherwise a correction with ``x_ChargeKey == x_CorrectionOf`` and
217
+ no surviving original would pass, defeating the auditability guarantee.
218
+ """
219
+ original_keys = index_factory()
220
+ for r in cost_and_usage:
221
+ key = (r.get("x_ChargeKey") or "").strip()
222
+ if key and not _is_correction(r):
223
+ original_keys[key] = ""
224
+ out: list[Diagnostic] = []
225
+ for i, row in enumerate(cost_and_usage, start=1):
226
+ ref = (row.get("x_CorrectionOf") or "").strip()
227
+ if not ref:
228
+ continue
229
+ self_key = (row.get("x_ChargeKey") or "").strip()
230
+ if ref == self_key or ref not in original_keys:
231
+ out.append(
232
+ Diagnostic(
233
+ code="FDT-CORR-001",
234
+ severity=Severity.ERROR,
235
+ message=f"correction row {i} references original charge {ref!r} that is not "
236
+ "present as an original (original must remain auditable)",
237
+ datasets=("Cost and Usage",),
238
+ dataset="Cost and Usage",
239
+ line_number=i,
240
+ column="x_CorrectionOf",
241
+ value=ref,
242
+ record_keys={"x_CorrectionOf": ref},
243
+ )
244
+ )
245
+ return out