focus-data-toolkit 0.11.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. focus_data_toolkit/__init__.py +69 -0
  2. focus_data_toolkit/__main__.py +6 -0
  3. focus_data_toolkit/_version.py +8 -0
  4. focus_data_toolkit/cli.py +968 -0
  5. focus_data_toolkit/context/__init__.py +88 -0
  6. focus_data_toolkit/context/billing.py +54 -0
  7. focus_data_toolkit/context/provider.py +90 -0
  8. focus_data_toolkit/convert/__init__.py +708 -0
  9. focus_data_toolkit/convert/billing_period.py +65 -0
  10. focus_data_toolkit/convert/contract_applied.py +235 -0
  11. focus_data_toolkit/convert/contract_commitment.py +182 -0
  12. focus_data_toolkit/convert/cost_and_usage.py +179 -0
  13. focus_data_toolkit/convert/detect.py +39 -0
  14. focus_data_toolkit/convert/invoice_detail.py +199 -0
  15. focus_data_toolkit/convert/streaming.py +1030 -0
  16. focus_data_toolkit/errors.py +145 -0
  17. focus_data_toolkit/focus_json.py +68 -0
  18. focus_data_toolkit/generators/__init__.py +61 -0
  19. focus_data_toolkit/generators/_shim.py +43 -0
  20. focus_data_toolkit/generators/engine/__init__.py +14 -0
  21. focus_data_toolkit/generators/engine/context.py +12 -0
  22. focus_data_toolkit/generators/engine/determinism.py +117 -0
  23. focus_data_toolkit/generators/engine/json_focus.py +63 -0
  24. focus_data_toolkit/generators/engine/ladder.py +71 -0
  25. focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
  26. focus_data_toolkit/generators/engine/serialize.py +151 -0
  27. focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
  28. focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
  29. focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
  30. focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
  31. focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
  32. focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
  33. focus_data_toolkit/generators/providers/__init__.py +29 -0
  34. focus_data_toolkit/generators/providers/aws.py +186 -0
  35. focus_data_toolkit/generators/providers/azure.py +191 -0
  36. focus_data_toolkit/generators/providers/gcp.py +194 -0
  37. focus_data_toolkit/generators/providers/profile.py +123 -0
  38. focus_data_toolkit/generators/scenarios.py +178 -0
  39. focus_data_toolkit/generators/versions/__init__.py +17 -0
  40. focus_data_toolkit/generators/versions/adapter.py +41 -0
  41. focus_data_toolkit/generators/versions/v1_2.py +111 -0
  42. focus_data_toolkit/generators/versions/v1_3.py +154 -0
  43. focus_data_toolkit/io/__init__.py +1 -0
  44. focus_data_toolkit/io/atomic_writer.py +462 -0
  45. focus_data_toolkit/io/csv_io.py +128 -0
  46. focus_data_toolkit/io/parquet_io.py +528 -0
  47. focus_data_toolkit/io/records.py +92 -0
  48. focus_data_toolkit/io/row_source.py +117 -0
  49. focus_data_toolkit/lifecycle.py +342 -0
  50. focus_data_toolkit/manifest.py +114 -0
  51. focus_data_toolkit/model/__init__.py +43 -0
  52. focus_data_toolkit/model/capabilities.py +66 -0
  53. focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
  54. focus_data_toolkit/model/focus_1_4_model.json +1913 -0
  55. focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
  56. focus_data_toolkit/model/focus_json_keys.py +112 -0
  57. focus_data_toolkit/model/iso_4217_currencies.json +23 -0
  58. focus_data_toolkit/model/json_schema_check.py +205 -0
  59. focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
  60. focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
  61. focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
  62. focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
  63. focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
  64. focus_data_toolkit/model/model_provenance.json +58 -0
  65. focus_data_toolkit/model/validator.py +498 -0
  66. focus_data_toolkit/modes.py +18 -0
  67. focus_data_toolkit/official_validator.py +61 -0
  68. focus_data_toolkit/progress.py +89 -0
  69. focus_data_toolkit/provenance.py +106 -0
  70. focus_data_toolkit/py.typed +1 -0
  71. focus_data_toolkit/runtime.py +243 -0
  72. focus_data_toolkit/schema/__init__.py +17 -0
  73. focus_data_toolkit/schema/detection.py +274 -0
  74. focus_data_toolkit/schema/registry.py +127 -0
  75. focus_data_toolkit/storage/__init__.py +1 -0
  76. focus_data_toolkit/storage/external_index.py +99 -0
  77. focus_data_toolkit/storage/spill.py +150 -0
  78. focus_data_toolkit/studio/__init__.py +19 -0
  79. focus_data_toolkit/studio/app.py +467 -0
  80. focus_data_toolkit/studio/config.py +42 -0
  81. focus_data_toolkit/studio/frontend/app.js +214 -0
  82. focus_data_toolkit/studio/frontend/index.html +101 -0
  83. focus_data_toolkit/studio/frontend/style.css +60 -0
  84. focus_data_toolkit/studio/jobs.py +142 -0
  85. focus_data_toolkit/studio/preview.py +32 -0
  86. focus_data_toolkit/studio/security.py +125 -0
  87. focus_data_toolkit/studio/server.py +71 -0
  88. focus_data_toolkit/supplement/__init__.py +50 -0
  89. focus_data_toolkit/supplement/adapters/__init__.py +21 -0
  90. focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
  91. focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
  92. focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
  93. focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
  94. focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
  95. focus_data_toolkit/supplement/adapters/registry.py +215 -0
  96. focus_data_toolkit/supplement/apply.py +318 -0
  97. focus_data_toolkit/supplement/gaps.py +219 -0
  98. focus_data_toolkit/supplement/kinds.py +118 -0
  99. focus_data_toolkit/supplement/loader.py +409 -0
  100. focus_data_toolkit/supplement/spec.py +74 -0
  101. focus_data_toolkit/supplement/validate.py +215 -0
  102. focus_data_toolkit/validate/__init__.py +15 -0
  103. focus_data_toolkit/validate/allocation.py +333 -0
  104. focus_data_toolkit/validate/bundle.py +254 -0
  105. focus_data_toolkit/validate/codes.py +93 -0
  106. focus_data_toolkit/validate/corrections.py +245 -0
  107. focus_data_toolkit/validate/reconciliation.py +98 -0
  108. focus_data_toolkit/validate/referential.py +289 -0
  109. focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
  110. focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
  111. focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
  112. focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
  113. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
  114. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
  115. focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
  116. focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,215 @@
1
+ """Cross-validate loaded supplements against the source and the FOCUS model.
2
+
3
+ Diagnostics (``FDT-SUPP-0xx``): ERRORs block use of the bundle, WARNINGs don't, and the
4
+ coverage report (``FDT-SUPP-010``, INFO) is what drives strict gating — a blocking
5
+ column flips to ``ENRICHED`` only at 100 % key coverage.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from collections.abc import Mapping, Sequence
11
+ from dataclasses import dataclass, field
12
+ from decimal import Decimal, InvalidOperation
13
+
14
+ from focus_data_toolkit.convert.invoice_detail import (
15
+ _COST_QUANTUM,
16
+ GrainKey,
17
+ invoice_detail_grain_key,
18
+ )
19
+ from focus_data_toolkit.errors import Diagnostic, Severity
20
+ from focus_data_toolkit.model.validator import check_column_value
21
+ from focus_data_toolkit.supplement.loader import JoinKey, SupplementBundle, SupplementTable
22
+
23
+ _SAMPLE_CAP = 25
24
+
25
+
26
+ @dataclass
27
+ class SourceKeySets:
28
+ """The join keys the source actually derives, per supplement kind."""
29
+
30
+ billing_periods: set[JoinKey] = field(default_factory=set)
31
+ invoices: set[JoinKey] = field(default_factory=set)
32
+ invoice_grains: set[GrainKey] = field(default_factory=set)
33
+ contract_commitment_ids: set[JoinKey] = field(default_factory=set)
34
+ grain_billed: dict[GrainKey, Decimal] = field(default_factory=dict)
35
+
36
+ def keys_for(self, kind_name: str) -> set[JoinKey]:
37
+ return {
38
+ "billing_period": self.billing_periods,
39
+ "invoice": self.invoices,
40
+ "invoice_line": self.invoice_grains,
41
+ "contract_commitment": self.contract_commitment_ids,
42
+ }[kind_name]
43
+
44
+ def observe_cau_row(self, row: Mapping[str, str]) -> None:
45
+ """Accumulate the keys of one Cost and Usage row (shared eager/streaming)."""
46
+ start = (row.get("BillingPeriodStart") or "").strip()
47
+ end = (row.get("BillingPeriodEnd") or "").strip()
48
+ issuer = (row.get("InvoiceIssuerName") or "").strip()
49
+ if start and end:
50
+ self.billing_periods.add((issuer, start, end))
51
+ grain = invoice_detail_grain_key(row)
52
+ if grain[1]: # InvoiceId present
53
+ self.invoice_grains.add(grain)
54
+ self.invoices.add((grain[0], grain[1]))
55
+ try:
56
+ cost = Decimal((row.get("BilledCost") or "0").strip() or "0")
57
+ except InvalidOperation:
58
+ cost = Decimal(0)
59
+ self.grain_billed[grain] = self.grain_billed.get(grain, Decimal(0)) + cost
60
+
61
+ def observe_cc_row(self, row: Mapping[str, str]) -> None:
62
+ cc_id = (row.get("ContractCommitmentId") or "").strip()
63
+ if cc_id:
64
+ self.contract_commitment_ids.add((cc_id,))
65
+
66
+
67
+ def source_key_sets(
68
+ cau_rows: Sequence[Mapping[str, str]],
69
+ cc_rows: Sequence[Mapping[str, str]] | None = None,
70
+ ) -> SourceKeySets:
71
+ keys = SourceKeySets()
72
+ for row in cau_rows:
73
+ keys.observe_cau_row(row)
74
+ for row in cc_rows or ():
75
+ keys.observe_cc_row(row)
76
+ return keys
77
+
78
+
79
+ @dataclass(frozen=True)
80
+ class ColumnCoverage:
81
+ """How many source keys have a non-empty supplied value for one column."""
82
+
83
+ total_keys: int
84
+ covered: int
85
+
86
+ @property
87
+ def complete(self) -> bool:
88
+ return self.total_keys > 0 and self.covered == self.total_keys
89
+
90
+
91
+ def coverage(table: SupplementTable, source_keys: set[JoinKey]) -> dict[str, ColumnCoverage]:
92
+ """Per fact column: how many of the source's keys this table covers."""
93
+ total = len(source_keys)
94
+ out: dict[str, ColumnCoverage] = {}
95
+ for column in table.fact_columns:
96
+ covered = sum(1 for key in source_keys if table.value(key, column))
97
+ out[column] = ColumnCoverage(total_keys=total, covered=covered)
98
+ return out
99
+
100
+
101
+ def _sample(keys: Sequence[JoinKey]) -> str:
102
+ return "; ".join("|".join(k) for k in list(keys)[:_SAMPLE_CAP])
103
+
104
+
105
+ def validate_supplements(bundle: SupplementBundle, source: SourceKeySets) -> list[Diagnostic]:
106
+ """All supplement diagnostics: structural + values + joins + coverage."""
107
+ diagnostics = bundle.structural_diagnostics()
108
+ for name in sorted(bundle.tables):
109
+ table = bundle.tables[name]
110
+ source_keys = source.keys_for(name)
111
+
112
+ # FDT-SUPP-004 — supplied values must obey the model's format rules.
113
+ bad: dict[str, list[str]] = {}
114
+ for key, facts in table.rows.items():
115
+ for column, value in facts.items():
116
+ if not value:
117
+ continue
118
+ rule = check_column_value(table.kind.target_dataset, column, value)
119
+ if rule:
120
+ bad.setdefault(f"{column}:{rule}", []).append("|".join(key))
121
+ for column_rule, keys in sorted(bad.items()):
122
+ column, _, rule = column_rule.partition(":")
123
+ diagnostics.append(
124
+ Diagnostic(
125
+ code="FDT-SUPP-004",
126
+ severity=Severity.ERROR,
127
+ message=f"supplement value(s) for {column} violate the model rule {rule!r}",
128
+ datasets=(table.kind.target_dataset,),
129
+ file=str(table.path),
130
+ context={
131
+ "kind": name,
132
+ "column": column,
133
+ "rule": rule,
134
+ "row_count": str(len(keys)),
135
+ "sample": "; ".join(keys[:_SAMPLE_CAP]),
136
+ },
137
+ )
138
+ )
139
+
140
+ # FDT-SUPP-005 — orphan supplement rows (key never derived from the source).
141
+ orphans = sorted(set(table.rows) - source_keys)
142
+ if orphans:
143
+ diagnostics.append(
144
+ Diagnostic(
145
+ code="FDT-SUPP-005",
146
+ severity=Severity.WARNING,
147
+ message=f"{len(orphans)} supplement row(s) match nothing in the source "
148
+ "(often a wider export period; they are ignored)",
149
+ datasets=(table.kind.target_dataset,),
150
+ file=str(table.path),
151
+ context={"kind": name, "sample": _sample(orphans)},
152
+ )
153
+ )
154
+
155
+ # FDT-SUPP-006 — invoice_line BilledCost must reconcile with the derived sums.
156
+ if name == "invoice_line" and "BilledCost" in table.fact_columns and source.grain_billed:
157
+ conflicts: list[JoinKey] = []
158
+ for key in sorted(set(table.rows) & source_keys):
159
+ supplied = table.value(key, "BilledCost")
160
+ if not supplied:
161
+ continue
162
+ try:
163
+ supplied_cost = Decimal(supplied)
164
+ except InvalidOperation:
165
+ continue # already reported by FDT-SUPP-004
166
+ derived = source.grain_billed.get(key, Decimal(0))
167
+ if supplied_cost.quantize(_COST_QUANTUM) != derived.quantize(_COST_QUANTUM):
168
+ conflicts.append(key)
169
+ if conflicts:
170
+ diagnostics.append(
171
+ Diagnostic(
172
+ code="FDT-SUPP-006",
173
+ severity=Severity.ERROR,
174
+ message="supplement BilledCost conflicts with the Cost and Usage "
175
+ "grain sums; the supplement does not describe this source",
176
+ datasets=(table.kind.target_dataset,),
177
+ file=str(table.path),
178
+ context={
179
+ "kind": name,
180
+ "row_count": str(len(conflicts)),
181
+ "sample": _sample(conflicts),
182
+ },
183
+ )
184
+ )
185
+
186
+ # FDT-SUPP-010 — coverage per column (drives strict gating; INFO, never an error).
187
+ for column, cov in sorted(coverage(table, source_keys).items()):
188
+ if column == "BilledCost" and name == "invoice_line":
189
+ continue # reconciliation-only, never applied
190
+ if cov.covered < cov.total_keys:
191
+ missing = sorted(
192
+ k for k in source_keys if not table.value(k, column)
193
+ )
194
+ diagnostics.append(
195
+ Diagnostic(
196
+ code="FDT-SUPP-010",
197
+ severity=Severity.INFO,
198
+ message=f"partial coverage for {column}: "
199
+ f"{cov.covered}/{cov.total_keys} source key(s) supplied",
200
+ datasets=(table.kind.target_dataset,),
201
+ file=str(table.path),
202
+ context={
203
+ "kind": name,
204
+ "column": column,
205
+ "covered": str(cov.covered),
206
+ "total": str(cov.total_keys),
207
+ "missing_sample": _sample(missing),
208
+ },
209
+ )
210
+ )
211
+ return diagnostics
212
+
213
+
214
+ def has_blocking_errors(diagnostics: Sequence[Diagnostic]) -> bool:
215
+ return any(d.severity is Severity.ERROR for d in diagnostics)
@@ -0,0 +1,15 @@
1
+ """Cross-dataset validation layer (P1.4 / P1.8 / P1.9).
2
+
3
+ Distinct from the per-dataset linter in ``model/validator.py``: this layer validates a bundle
4
+ of datasets against each other.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from focus_data_toolkit.validate.bundle import (
10
+ Bundle,
11
+ BundleReport,
12
+ validate_dataset_bundle,
13
+ )
14
+
15
+ __all__ = ["Bundle", "BundleReport", "validate_dataset_bundle"]
@@ -0,0 +1,333 @@
1
+ """Validate Split Cost Allocation groups within a Cost and Usage dataset.
2
+
3
+ An allocation *group* redistributes one origin charge across several consumers. FOCUS carries
4
+ the per-line ratio inside ``AllocatedMethodDetails`` (an ``Elements`` array of
5
+ ``{AllocatedRatio, UsageUnit, UsageQuantity}``). To make a group's origin identifiable and its
6
+ origin cost checkable, rows are tied together by the toolkit extension columns
7
+ ``x_SplitOriginId`` (a stable origin-charge key) and ``x_SplitOriginCost`` (the origin amount,
8
+ identical across the group). Rows without ``x_SplitOriginId`` are not allocation rows and are
9
+ ignored, so the check is a no-op on datasets that do not use split cost allocation.
10
+
11
+ Per group it checks: ratios sum to 1 (within tolerance) and each ratio is in [0, 1]; allocated
12
+ costs sum to the origin cost (within tolerance); a single consistent method and unit; unique
13
+ allocated resources; and that every row carries the information the group needs.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import json
19
+ from collections.abc import Callable, Iterable, Mapping, MutableMapping, Sequence
20
+ from dataclasses import dataclass, field
21
+ from decimal import Decimal, InvalidOperation
22
+
23
+ from focus_data_toolkit.errors import Diagnostic, Severity
24
+
25
+ Rows = Sequence[Mapping[str, str]]
26
+ #: The input is consumed in a single forward pass, so any iterable of rows works.
27
+ RowStream = Iterable[Mapping[str, str]]
28
+ #: Factory for per-key lookup state (``dict`` by default; a spillable map for streaming).
29
+ IndexFactory = Callable[[], MutableMapping[str, str]]
30
+
31
+ ORIGIN_ID_COLUMN = "x_SplitOriginId"
32
+ ORIGIN_COST_COLUMN = "x_SplitOriginCost"
33
+
34
+ DEFAULT_RATIO_TOLERANCE = Decimal("0.0001")
35
+ DEFAULT_COST_TOLERANCE = Decimal("0.01")
36
+
37
+ # Line numbers recorded per group for the incomplete-group diagnostic's ``rows`` context;
38
+ # beyond this the context reports the overflow as ``+N more`` instead of growing unboundedly.
39
+ _LINE_SAMPLE = 100
40
+
41
+
42
+ def _dec(value: str | None) -> Decimal | None:
43
+ try:
44
+ parsed = Decimal((value or "").strip())
45
+ except InvalidOperation:
46
+ return None
47
+ return parsed if parsed.is_finite() else None
48
+
49
+
50
+ def _row_ratio_and_units(row: Mapping[str, str]) -> tuple[Decimal | None, frozenset[str]]:
51
+ """Extract (summed AllocatedRatio, set of UsageUnits) from a row's AllocatedMethodDetails.
52
+
53
+ Every element's ``UsageUnit`` is collected — not just the first — so a row mixing units
54
+ across its elements is caught by the group's single-unit check.
55
+ """
56
+ text = (row.get("AllocatedMethodDetails") or "").strip()
57
+ if not text:
58
+ return None, frozenset()
59
+ try:
60
+ obj = json.loads(text, parse_float=Decimal, parse_int=Decimal)
61
+ except (ValueError, TypeError):
62
+ return None, frozenset()
63
+ elements = obj.get("Elements") if isinstance(obj, dict) else None
64
+ if not isinstance(elements, list) or not elements:
65
+ return None, frozenset()
66
+ ratio = Decimal(0)
67
+ units: set[str] = set()
68
+ for el in elements:
69
+ if not isinstance(el, dict):
70
+ return None, frozenset(units)
71
+ raw = el.get("AllocatedRatio")
72
+ if not isinstance(raw, Decimal) or not raw.is_finite():
73
+ return None, frozenset(units)
74
+ ratio += raw
75
+ unit = el.get("UsageUnit")
76
+ if unit:
77
+ units.add(unit)
78
+ return ratio, frozenset(units)
79
+
80
+
81
+ @dataclass
82
+ class _GroupState:
83
+ """Running aggregate of one allocation group — fixed-size scalars, never member rows.
84
+
85
+ Mirrors the checks of the original list-of-members implementation exactly: the first
86
+ incomplete condition (in row order) short-circuits further accumulation, line numbers
87
+ keep counting so the incomplete diagnostic still describes every member row. The state
88
+ is JSON round-trippable (:meth:`dump` / :meth:`load`), so it can live as a string value
89
+ in a disk-spilling map when the caller supplies an ``index_factory``.
90
+ """
91
+
92
+ lines: list[int] = field(default_factory=list) # first _LINE_SAMPLE member lines
93
+ line_count: int = 0
94
+ incomplete: tuple[str, str | None] | None = None # (reason, column)
95
+ ratio_sum: Decimal = Decimal(0)
96
+ out_of_range: list[Decimal] = field(default_factory=list) # ratios outside [0, 1]
97
+ units: set[str] = field(default_factory=set)
98
+ methods: set[str] = field(default_factory=set)
99
+ resource_count: int = 0
100
+ duplicate_resource: bool = False
101
+ missing_resource: bool = False
102
+ total_cost: Decimal = Decimal(0)
103
+ origin_cost: Decimal | None = None
104
+ origin_disagrees: bool = False
105
+
106
+ def observe(self, line: int, row: Mapping[str, str], *, seen_resource: bool) -> None:
107
+ self.line_count += 1
108
+ if len(self.lines) < _LINE_SAMPLE:
109
+ self.lines.append(line)
110
+ if self.incomplete is not None:
111
+ return
112
+ ratio, row_units = _row_ratio_and_units(row)
113
+ if ratio is None:
114
+ self.incomplete = ("a row has no usable AllocatedRatio", "AllocatedMethodDetails")
115
+ return
116
+ self.ratio_sum += ratio
117
+ if ratio < 0 or ratio > 1:
118
+ self.out_of_range.append(ratio)
119
+ self.units |= row_units
120
+ self.methods.add((row.get("AllocatedMethodId") or "").strip())
121
+ resource = (row.get("AllocatedResourceId") or "").strip()
122
+ self.resource_count += 1
123
+ if seen_resource:
124
+ self.duplicate_resource = True
125
+ if not resource:
126
+ self.missing_resource = True
127
+ cost = _dec(row.get("BilledCost"))
128
+ if cost is None:
129
+ self.incomplete = ("a row has no numeric BilledCost", "BilledCost")
130
+ return
131
+ self.total_cost += cost
132
+ origin_cost = _dec(row.get(ORIGIN_COST_COLUMN))
133
+ if origin_cost is None:
134
+ self.incomplete = ("missing x_SplitOriginCost", ORIGIN_COST_COLUMN)
135
+ return
136
+ if self.origin_cost is None:
137
+ self.origin_cost = origin_cost
138
+ elif origin_cost != self.origin_cost:
139
+ self.origin_disagrees = True
140
+
141
+ def dump(self) -> str:
142
+ return json.dumps({
143
+ "ln": self.lines,
144
+ "lc": self.line_count,
145
+ "inc": list(self.incomplete) if self.incomplete else None,
146
+ "rs": str(self.ratio_sum),
147
+ "oor": [str(r) for r in self.out_of_range],
148
+ "un": sorted(self.units),
149
+ "me": sorted(self.methods),
150
+ "rc": self.resource_count,
151
+ "dup": self.duplicate_resource,
152
+ "mr": self.missing_resource,
153
+ "tc": str(self.total_cost),
154
+ "oc": str(self.origin_cost) if self.origin_cost is not None else None,
155
+ "od": self.origin_disagrees,
156
+ }, separators=(",", ":"))
157
+
158
+ @classmethod
159
+ def load(cls, text: str) -> _GroupState:
160
+ d = json.loads(text)
161
+ return cls(
162
+ lines=d["ln"],
163
+ line_count=d["lc"],
164
+ incomplete=tuple(d["inc"]) if d["inc"] else None,
165
+ ratio_sum=Decimal(d["rs"]),
166
+ out_of_range=[Decimal(r) for r in d["oor"]],
167
+ units=set(d["un"]),
168
+ methods=set(d["me"]),
169
+ resource_count=d["rc"],
170
+ duplicate_resource=d["dup"],
171
+ missing_resource=d["mr"],
172
+ total_cost=Decimal(d["tc"]),
173
+ origin_cost=Decimal(d["oc"]) if d["oc"] is not None else None,
174
+ origin_disagrees=d["od"],
175
+ )
176
+
177
+
178
+ def validate_split_allocation(
179
+ cost_and_usage: RowStream,
180
+ *,
181
+ ratio_tolerance: Decimal = DEFAULT_RATIO_TOLERANCE,
182
+ cost_tolerance: Decimal = DEFAULT_COST_TOLERANCE,
183
+ index_factory: IndexFactory = dict,
184
+ ) -> list[Diagnostic]:
185
+ """Validate every split-cost-allocation group in ``cost_and_usage`` (single pass).
186
+
187
+ Per-group state is a fixed-size JSON-serialised aggregate held in an ``index_factory``
188
+ map (as is the resource-duplicate lookup), so with a spillable factory memory stays
189
+ bounded even when allocation-group cardinality approaches the row count.
190
+ """
191
+ groups = index_factory()
192
+ resources_seen = index_factory()
193
+ for i, row in enumerate(cost_and_usage, start=1):
194
+ origin = (row.get(ORIGIN_ID_COLUMN) or "").strip()
195
+ if not origin:
196
+ continue
197
+ raw = groups.get(origin)
198
+ state = _GroupState.load(raw) if raw is not None else _GroupState()
199
+ resource = (row.get("AllocatedResourceId") or "").strip()
200
+ resource_key = json.dumps([origin, resource], separators=(",", ":"))
201
+ state.observe(i, row, seen_resource=resource_key in resources_seen)
202
+ if state.incomplete is None:
203
+ resources_seen[resource_key] = ""
204
+ groups[origin] = state.dump()
205
+
206
+ out: list[Diagnostic] = []
207
+ for origin_id in sorted(groups):
208
+ state = _GroupState.load(groups[origin_id])
209
+ out.extend(_group_diagnostics(origin_id, state, ratio_tolerance, cost_tolerance))
210
+ return out
211
+
212
+
213
+ def _group_diagnostics(
214
+ origin_id: str,
215
+ state: _GroupState,
216
+ ratio_tolerance: Decimal,
217
+ cost_tolerance: Decimal,
218
+ ) -> list[Diagnostic]:
219
+ keys = {ORIGIN_ID_COLUMN: origin_id}
220
+ rows_context = ",".join(map(str, sorted(state.lines)))
221
+ if state.line_count > len(state.lines):
222
+ rows_context += f",+{state.line_count - len(state.lines)} more"
223
+ out: list[Diagnostic] = []
224
+
225
+ def incomplete(reason: str, column: str | None = None) -> Diagnostic:
226
+ return Diagnostic(
227
+ code="FDT-ALLOC-005",
228
+ severity=Severity.ERROR,
229
+ message=f"split allocation group {origin_id!r} is incomplete: {reason}",
230
+ datasets=("Cost and Usage",),
231
+ dataset="Cost and Usage",
232
+ column=column,
233
+ record_keys=keys,
234
+ context={"rows": rows_context},
235
+ )
236
+
237
+ if state.incomplete is not None:
238
+ return [incomplete(*state.incomplete)]
239
+ if "" in state.methods or state.missing_resource:
240
+ return [incomplete("a row is missing AllocatedMethodId or AllocatedResourceId")]
241
+ if state.origin_disagrees:
242
+ return [incomplete("rows disagree on x_SplitOriginCost", ORIGIN_COST_COLUMN)]
243
+
244
+ for ratio in state.out_of_range:
245
+ out.append(
246
+ Diagnostic(
247
+ code="FDT-ALLOC-006",
248
+ severity=Severity.ERROR,
249
+ message=f"allocation ratio {ratio} outside [0, 1] in group {origin_id!r}",
250
+ datasets=("Cost and Usage",),
251
+ dataset="Cost and Usage",
252
+ column="AllocatedMethodDetails",
253
+ actual=str(ratio),
254
+ record_keys=keys,
255
+ )
256
+ )
257
+
258
+ ratio_sum = state.ratio_sum
259
+ if abs(ratio_sum - Decimal(1)) > ratio_tolerance:
260
+ out.append(
261
+ Diagnostic(
262
+ code="FDT-ALLOC-001",
263
+ severity=Severity.ERROR,
264
+ message=f"allocation ratios in group {origin_id!r} sum to {ratio_sum}, not 1",
265
+ datasets=("Cost and Usage",),
266
+ dataset="Cost and Usage",
267
+ column="AllocatedMethodDetails",
268
+ expected="1",
269
+ actual=str(ratio_sum),
270
+ record_keys=keys,
271
+ )
272
+ )
273
+
274
+ origin_cost = state.origin_cost
275
+ assert origin_cost is not None # a complete group recorded one on every row
276
+ if abs(state.total_cost - origin_cost) > cost_tolerance:
277
+ out.append(
278
+ Diagnostic(
279
+ code="FDT-ALLOC-002",
280
+ severity=Severity.ERROR,
281
+ message=f"allocated costs in group {origin_id!r} sum to {state.total_cost}, "
282
+ f"not the origin cost {origin_cost}",
283
+ datasets=("Cost and Usage",),
284
+ dataset="Cost and Usage",
285
+ column="BilledCost",
286
+ expected=str(origin_cost),
287
+ actual=str(state.total_cost),
288
+ record_keys=keys,
289
+ )
290
+ )
291
+
292
+ if len({m for m in state.methods if m}) > 1:
293
+ out.append(
294
+ Diagnostic(
295
+ code="FDT-ALLOC-003",
296
+ severity=Severity.ERROR,
297
+ message=f"inconsistent AllocatedMethodId within group {origin_id!r}: "
298
+ f"{sorted(state.methods)}",
299
+ datasets=("Cost and Usage",),
300
+ dataset="Cost and Usage",
301
+ column="AllocatedMethodId",
302
+ record_keys=keys,
303
+ )
304
+ )
305
+
306
+ if len(state.units) > 1:
307
+ out.append(
308
+ Diagnostic(
309
+ code="FDT-ALLOC-007",
310
+ severity=Severity.ERROR,
311
+ message=f"inconsistent UsageUnit within group {origin_id!r}: "
312
+ f"{sorted(state.units)}",
313
+ datasets=("Cost and Usage",),
314
+ dataset="Cost and Usage",
315
+ column="AllocatedMethodDetails",
316
+ record_keys=keys,
317
+ )
318
+ )
319
+
320
+ if state.duplicate_resource:
321
+ out.append(
322
+ Diagnostic(
323
+ code="FDT-ALLOC-004",
324
+ severity=Severity.ERROR,
325
+ message=f"duplicate AllocatedResourceId within group {origin_id!r}",
326
+ datasets=("Cost and Usage",),
327
+ dataset="Cost and Usage",
328
+ column="AllocatedResourceId",
329
+ record_keys=keys,
330
+ )
331
+ )
332
+
333
+ return out