focus-data-toolkit 0.11.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. focus_data_toolkit/__init__.py +69 -0
  2. focus_data_toolkit/__main__.py +6 -0
  3. focus_data_toolkit/_version.py +8 -0
  4. focus_data_toolkit/cli.py +968 -0
  5. focus_data_toolkit/context/__init__.py +88 -0
  6. focus_data_toolkit/context/billing.py +54 -0
  7. focus_data_toolkit/context/provider.py +90 -0
  8. focus_data_toolkit/convert/__init__.py +708 -0
  9. focus_data_toolkit/convert/billing_period.py +65 -0
  10. focus_data_toolkit/convert/contract_applied.py +235 -0
  11. focus_data_toolkit/convert/contract_commitment.py +182 -0
  12. focus_data_toolkit/convert/cost_and_usage.py +179 -0
  13. focus_data_toolkit/convert/detect.py +39 -0
  14. focus_data_toolkit/convert/invoice_detail.py +199 -0
  15. focus_data_toolkit/convert/streaming.py +1030 -0
  16. focus_data_toolkit/errors.py +145 -0
  17. focus_data_toolkit/focus_json.py +68 -0
  18. focus_data_toolkit/generators/__init__.py +61 -0
  19. focus_data_toolkit/generators/_shim.py +43 -0
  20. focus_data_toolkit/generators/engine/__init__.py +14 -0
  21. focus_data_toolkit/generators/engine/context.py +12 -0
  22. focus_data_toolkit/generators/engine/determinism.py +117 -0
  23. focus_data_toolkit/generators/engine/json_focus.py +63 -0
  24. focus_data_toolkit/generators/engine/ladder.py +71 -0
  25. focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
  26. focus_data_toolkit/generators/engine/serialize.py +151 -0
  27. focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
  28. focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
  29. focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
  30. focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
  31. focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
  32. focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
  33. focus_data_toolkit/generators/providers/__init__.py +29 -0
  34. focus_data_toolkit/generators/providers/aws.py +186 -0
  35. focus_data_toolkit/generators/providers/azure.py +191 -0
  36. focus_data_toolkit/generators/providers/gcp.py +194 -0
  37. focus_data_toolkit/generators/providers/profile.py +123 -0
  38. focus_data_toolkit/generators/scenarios.py +178 -0
  39. focus_data_toolkit/generators/versions/__init__.py +17 -0
  40. focus_data_toolkit/generators/versions/adapter.py +41 -0
  41. focus_data_toolkit/generators/versions/v1_2.py +111 -0
  42. focus_data_toolkit/generators/versions/v1_3.py +154 -0
  43. focus_data_toolkit/io/__init__.py +1 -0
  44. focus_data_toolkit/io/atomic_writer.py +462 -0
  45. focus_data_toolkit/io/csv_io.py +128 -0
  46. focus_data_toolkit/io/parquet_io.py +528 -0
  47. focus_data_toolkit/io/records.py +92 -0
  48. focus_data_toolkit/io/row_source.py +117 -0
  49. focus_data_toolkit/lifecycle.py +342 -0
  50. focus_data_toolkit/manifest.py +114 -0
  51. focus_data_toolkit/model/__init__.py +43 -0
  52. focus_data_toolkit/model/capabilities.py +66 -0
  53. focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
  54. focus_data_toolkit/model/focus_1_4_model.json +1913 -0
  55. focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
  56. focus_data_toolkit/model/focus_json_keys.py +112 -0
  57. focus_data_toolkit/model/iso_4217_currencies.json +23 -0
  58. focus_data_toolkit/model/json_schema_check.py +205 -0
  59. focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
  60. focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
  61. focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
  62. focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
  63. focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
  64. focus_data_toolkit/model/model_provenance.json +58 -0
  65. focus_data_toolkit/model/validator.py +498 -0
  66. focus_data_toolkit/modes.py +18 -0
  67. focus_data_toolkit/official_validator.py +61 -0
  68. focus_data_toolkit/progress.py +89 -0
  69. focus_data_toolkit/provenance.py +106 -0
  70. focus_data_toolkit/py.typed +1 -0
  71. focus_data_toolkit/runtime.py +243 -0
  72. focus_data_toolkit/schema/__init__.py +17 -0
  73. focus_data_toolkit/schema/detection.py +274 -0
  74. focus_data_toolkit/schema/registry.py +127 -0
  75. focus_data_toolkit/storage/__init__.py +1 -0
  76. focus_data_toolkit/storage/external_index.py +99 -0
  77. focus_data_toolkit/storage/spill.py +150 -0
  78. focus_data_toolkit/studio/__init__.py +19 -0
  79. focus_data_toolkit/studio/app.py +467 -0
  80. focus_data_toolkit/studio/config.py +42 -0
  81. focus_data_toolkit/studio/frontend/app.js +214 -0
  82. focus_data_toolkit/studio/frontend/index.html +101 -0
  83. focus_data_toolkit/studio/frontend/style.css +60 -0
  84. focus_data_toolkit/studio/jobs.py +142 -0
  85. focus_data_toolkit/studio/preview.py +32 -0
  86. focus_data_toolkit/studio/security.py +125 -0
  87. focus_data_toolkit/studio/server.py +71 -0
  88. focus_data_toolkit/supplement/__init__.py +50 -0
  89. focus_data_toolkit/supplement/adapters/__init__.py +21 -0
  90. focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
  91. focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
  92. focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
  93. focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
  94. focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
  95. focus_data_toolkit/supplement/adapters/registry.py +215 -0
  96. focus_data_toolkit/supplement/apply.py +318 -0
  97. focus_data_toolkit/supplement/gaps.py +219 -0
  98. focus_data_toolkit/supplement/kinds.py +118 -0
  99. focus_data_toolkit/supplement/loader.py +409 -0
  100. focus_data_toolkit/supplement/spec.py +74 -0
  101. focus_data_toolkit/supplement/validate.py +215 -0
  102. focus_data_toolkit/validate/__init__.py +15 -0
  103. focus_data_toolkit/validate/allocation.py +333 -0
  104. focus_data_toolkit/validate/bundle.py +254 -0
  105. focus_data_toolkit/validate/codes.py +93 -0
  106. focus_data_toolkit/validate/corrections.py +245 -0
  107. focus_data_toolkit/validate/reconciliation.py +98 -0
  108. focus_data_toolkit/validate/referential.py +289 -0
  109. focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
  110. focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
  111. focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
  112. focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
  113. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
  114. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
  115. focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
  116. focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,498 @@
1
+ """Reference FOCUS 1.4 **structural linter** (model-driven).
2
+
3
+ This is a *linter*, not a full FOCUS 1.4 conformance validator. It checks that data
4
+ presented as a FOCUS 1.4 dataset is well-formed against the committed 1.4 data model
5
+ (``focus_1_4_model.json``) at two levels — and it can assert **only** those two:
6
+
7
+ * ``STRUCTURAL_VALID`` — required (Mandatory) columns present; unknown non-``x_``
8
+ columns flagged; nullability; and value **format** (NumericFormat incl. scientific
9
+ notation, Date/Time UTC ``…Z``, Currency ISO 4217, Allowed-Values enums, Unit, and
10
+ JSON/Key-Value well-formedness with the ``x_`` custom-key rule).
11
+ * ``SEMANTIC_VALID`` — single-row cross-field rules (Tax nulls, consumption gating,
12
+ ``LastUpdated >= Created``, ServiceSubcategory↔ServiceCategory, upfront-percentage vs
13
+ payment model, condition-aware required columns, ContractApplied deep structure), and
14
+ the official FOCUS JSON object schemas (vendored verbatim in ``model/json_schemas/``)
15
+ for ``ContractApplied``, ``AllocatedMethodDetails``,
16
+ ``CommitmentProgramEligibilityDetails`` and ``ContractCommitmentApplicability``.
17
+
18
+ It does **not** assert ``CROSS_DATASET_VALID`` (referential integrity across the four
19
+ datasets) or ``OFFICIALLY_VALIDATED`` (the FinOps ``focus_validator``, which does not yet
20
+ support 1.4). A clean ``LintReport`` therefore means *structurally and semantically
21
+ well-formed*, **not** fully FOCUS-conformant.
22
+
23
+ ``validate_focus_1_4`` is retained as a deprecated alias of
24
+ ``lint_focus_1_4_structure``.
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ import json
30
+ import re
31
+ import warnings
32
+ from collections.abc import Callable, Iterable
33
+ from dataclasses import dataclass
34
+ from datetime import datetime
35
+ from decimal import Decimal, InvalidOperation
36
+ from functools import lru_cache
37
+ from pathlib import Path
38
+ from typing import TYPE_CHECKING
39
+
40
+ if TYPE_CHECKING: # import cycle: capabilities re-uses the COND_* constants below
41
+ from focus_data_toolkit.model.capabilities import CapabilityProfile
42
+
43
+ from focus_data_toolkit.model.focus_json_keys import (
44
+ XPREFIX_ENFORCED_ELEMENTS_COLUMNS,
45
+ XPREFIX_ENFORCED_KEYVALUE_COLUMNS,
46
+ )
47
+ from focus_data_toolkit.model.json_schema_check import (
48
+ OFFICIAL_SCHEMA_COLUMNS,
49
+ check_against_official_schema,
50
+ )
51
+
52
+ _HERE = Path(__file__).resolve().parent
53
+ _MODEL_PATH = _HERE / "focus_1_4_model.json"
54
+ _ISO_4217_PATH = _HERE / "iso_4217_currencies.json"
55
+
56
+ # Validation levels this linter can assert. CROSS_DATASET_VALID / OFFICIALLY_VALIDATED
57
+ # are intentionally NOT checked here (documented in the module docstring).
58
+ LEVEL_STRUCTURAL = "STRUCTURAL_VALID"
59
+ LEVEL_SEMANTIC = "SEMANTIC_VALID"
60
+ LEVEL_CROSS_DATASET = "CROSS_DATASET_VALID"
61
+ LEVEL_OFFICIAL = "OFFICIALLY_VALIDATED"
62
+ _CHECKED_LEVELS: tuple[str, ...] = (LEVEL_STRUCTURAL, LEVEL_SEMANTIC)
63
+
64
+ # Applicability conditions (FOCUS 1.4 Applicability Criteria) that gate the
65
+ # "conditionally required" columns. Callers pass the subset they declare.
66
+ COND_MULTIPLE_PRICING_CATEGORIES = "SupportsMultiplePricingCategories"
67
+ COND_UNIT_PRICING = "SupportsUnitPricing"
68
+
69
+ _DATASET_ALIASES = {
70
+ "cost and usage": "Cost and Usage", "costandusage": "Cost and Usage", "cau": "Cost and Usage",
71
+ "billing period": "Billing Period", "billingperiod": "Billing Period", "bpd": "Billing Period",
72
+ "contract commitment": "Contract Commitment", "contractcommitment": "Contract Commitment",
73
+ "cct": "Contract Commitment",
74
+ "invoice detail": "Invoice Detail", "invoicedetail": "Invoice Detail", "ind": "Invoice Detail",
75
+ }
76
+
77
+ # NumericFormat (FOCUS attribute): integer, decimal, or scientific E-notation "mEn".
78
+ # The exponent sign is expressed ONLY when negative (no leading '+' on mantissa or
79
+ # exponent). So 35.2E-7 is valid; 35.2E+7 and +333 are not.
80
+ _NUMERIC_RE = re.compile(r"-?\d+(\.\d+)?(E-?\d+)?")
81
+ # DateTimeFormat: literal YYYY-MM-DDTHH:mm:ss[.fff]Z (UTC 'Z' only, ISO 8601).
82
+ _DATETIME_RE = re.compile(r"\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(\.\d+)?Z")
83
+
84
+
85
+ @dataclass(frozen=True)
86
+ class Violation:
87
+ dataset: str
88
+ rule: str
89
+ message: str
90
+ column: str | None = None
91
+ row_index: int | None = None
92
+ level: str = LEVEL_STRUCTURAL
93
+
94
+
95
+ @dataclass(frozen=True)
96
+ class LintReport:
97
+ dataset: str
98
+ row_count: int
99
+ violations: tuple[Violation, ...]
100
+ levels_checked: tuple[str, ...] = _CHECKED_LEVELS
101
+
102
+ @property
103
+ def ok(self) -> bool:
104
+ """No structural/semantic lint violations. NOT a full FOCUS conformance claim."""
105
+ return not self.violations
106
+
107
+ def passed(self, level: str) -> bool:
108
+ """True if ``level`` was checked and has no violations."""
109
+ if level not in self.levels_checked:
110
+ return False
111
+ return not any(v.level == level for v in self.violations)
112
+
113
+ @property
114
+ def levels_passed(self) -> tuple[str, ...]:
115
+ return tuple(level for level in self.levels_checked if self.passed(level))
116
+
117
+ def messages(self) -> list[str]:
118
+ return [
119
+ f"[{v.level}:{v.rule}] {v.column or '-'}"
120
+ + (f" row {v.row_index}" if v.row_index is not None else "")
121
+ + f": {v.message}"
122
+ for v in self.violations
123
+ ]
124
+
125
+
126
+ # Backwards-compatible alias (the class was previously ``ValidationReport``).
127
+ ValidationReport = LintReport
128
+
129
+
130
+ @lru_cache(maxsize=1)
131
+ def load_model() -> dict:
132
+ return json.loads(_MODEL_PATH.read_text(encoding="utf-8"))
133
+
134
+
135
+ @lru_cache(maxsize=1)
136
+ def _iso_4217() -> frozenset[str]:
137
+ return frozenset(json.loads(_ISO_4217_PATH.read_text(encoding="utf-8"))["codes"])
138
+
139
+
140
+ def resolve_dataset(name: str) -> str:
141
+ key = name.strip().lower()
142
+ if key not in _DATASET_ALIASES:
143
+ raise ValueError(f"unknown FOCUS 1.4 dataset {name!r}")
144
+ return _DATASET_ALIASES[key]
145
+
146
+
147
+ class _DuplicateKey(Exception):
148
+ pass
149
+
150
+
151
+ class _NonFiniteConstant(Exception):
152
+ pass
153
+
154
+
155
+ def _no_dup_pairs(pairs: list[tuple[str, object]]) -> dict:
156
+ seen: dict = {}
157
+ for key, value in pairs:
158
+ if key in seen:
159
+ raise _DuplicateKey(key)
160
+ seen[key] = value
161
+ return seen
162
+
163
+
164
+ def _reject_constant(const: str) -> float:
165
+ # NaN / Infinity / -Infinity are Python extensions, not valid JSON.
166
+ raise _NonFiniteConstant(const)
167
+
168
+
169
+ def _load_json_object(value: str) -> tuple[dict | None, str | None]:
170
+ try:
171
+ obj = json.loads(value, object_pairs_hook=_no_dup_pairs, parse_constant=_reject_constant)
172
+ except _DuplicateKey:
173
+ return None, "duplicate_json_key"
174
+ except _NonFiniteConstant:
175
+ return None, "bad_json"
176
+ except json.JSONDecodeError:
177
+ return None, "bad_json"
178
+ if not isinstance(obj, dict):
179
+ return None, "json_not_object"
180
+ return obj, None
181
+
182
+
183
+ def _decimal_or_none(value: str) -> Decimal | None:
184
+ if not _NUMERIC_RE.fullmatch(value):
185
+ return None
186
+ try:
187
+ d = Decimal(value)
188
+ except (InvalidOperation, ValueError):
189
+ return None
190
+ return d if d.is_finite() else None
191
+
192
+
193
+ def _parse_dt(value: str) -> datetime:
194
+ return datetime.fromisoformat(value.replace("Z", "+00:00"))
195
+
196
+
197
+ def _is_utc_datetime(value: str) -> bool:
198
+ if not _DATETIME_RE.fullmatch(value):
199
+ return False
200
+ try:
201
+ _parse_dt(value)
202
+ except ValueError:
203
+ return False
204
+ return True
205
+
206
+
207
+ def _keys_are_focus_or_prefixed(keys: Iterable[str], focus_keys: frozenset[str]) -> bool:
208
+ return all(k in focus_keys or k.startswith("x_") for k in keys)
209
+
210
+
211
+ def _validate_contract_applied(value: str) -> str | None:
212
+ # Lazy import avoids an import cycle (convert -> model.validator -> convert).
213
+ from focus_data_toolkit.convert.contract_applied import ContractAppliedError, parse
214
+
215
+ try:
216
+ parse(value, version="1.4")
217
+ except ContractAppliedError:
218
+ return "invalid_contract_applied"
219
+ return None
220
+
221
+
222
+ def _validate_json_column(column: str, value: str, value_format: str) -> str | None:
223
+ obj, err = _load_json_object(value)
224
+ if err:
225
+ return err
226
+ assert obj is not None # _load_json_object returns a dict whenever err is None
227
+ if value_format == "Key-Value":
228
+ if not all(v is None or isinstance(v, str | int | float | bool) for v in obj.values()):
229
+ return "key_value_value_not_scalar"
230
+ focus_keys = XPREFIX_ENFORCED_KEYVALUE_COLUMNS.get(column)
231
+ if focus_keys is not None and not _keys_are_focus_or_prefixed(obj, focus_keys):
232
+ return "custom_key_not_prefixed"
233
+ return None
234
+ # JSON Object columns.
235
+ if column == "ContractApplied":
236
+ err = _validate_contract_applied(value)
237
+ if err:
238
+ return err
239
+ entry = XPREFIX_ENFORCED_ELEMENTS_COLUMNS.get(column)
240
+ if entry is not None:
241
+ array_key, focus_keys = entry
242
+ # Top-level custom keys (alongside the array) must be x_-prefixed too.
243
+ if not all(k == array_key or k.startswith("x_") for k in obj):
244
+ return "custom_key_not_prefixed"
245
+ elements = obj.get(array_key)
246
+ if not isinstance(elements, list):
247
+ return "missing_elements_array"
248
+ for element in elements:
249
+ if not isinstance(element, dict):
250
+ return "element_not_object"
251
+ if not _keys_are_focus_or_prefixed(element, focus_keys):
252
+ return "custom_key_not_prefixed"
253
+ # Normative depth: the official FOCUS JSON Schemas (vendored verbatim, see
254
+ # model/json_schemas/) — conditional scope rules, metric exclusivity, ranges.
255
+ if column in OFFICIAL_SCHEMA_COLUMNS and check_against_official_schema(column, obj):
256
+ return "official_schema_violation"
257
+ return None
258
+
259
+
260
+ def _format_violation(spec: dict, column: str, value: str) -> str | None:
261
+ """Return a rule name if non-empty ``value`` violates the column's format."""
262
+ value_format = spec.get("value_format") or ""
263
+ data_type = spec.get("data_type") or ""
264
+
265
+ if value_format.startswith("Decimal") or data_type == "Decimal":
266
+ d = _decimal_or_none(value)
267
+ if d is None:
268
+ return "bad_numeric_format"
269
+ if "non-negative" in value_format and d < Decimal("0"):
270
+ return "negative_decimal"
271
+ rng = spec.get("numeric_range")
272
+ if rng and not (Decimal(str(rng[0])) <= d <= Decimal(str(rng[1]))):
273
+ return "decimal_out_of_range"
274
+ return None
275
+ if value_format == "Date/Time" or data_type == "Date/Time":
276
+ return None if _is_utc_datetime(value) else "bad_datetime"
277
+ if value_format == "Currency":
278
+ return None if value in _iso_4217() else "bad_currency"
279
+ if value_format == "Allowed Values":
280
+ allowed = spec.get("allowed_values")
281
+ if allowed is not None and value not in allowed:
282
+ return "not_in_allowed_values"
283
+ return None
284
+ if value_format == "Unit":
285
+ if value != value.strip() or not value or _decimal_or_none(value) is not None:
286
+ return "bad_unit"
287
+ return None
288
+ if value_format in ("JSON Object", "Key-Value") or data_type == "JSON":
289
+ return _validate_json_column(column, value, value_format)
290
+ if value_format == "Expected Format":
291
+ return None if re.search(r"\d", value) and re.search(r"[A-Za-z]", value) \
292
+ else "bad_expected_format"
293
+ return None
294
+
295
+
296
+ # --------------------------------------------------------------------------- #
297
+ # Cross-field (single-row, SEMANTIC) rules — each returns (column, rule, message) tuples.
298
+ # --------------------------------------------------------------------------- #
299
+ def _cost_and_usage(row: dict, model: dict, supported: frozenset[str]) -> list[tuple]:
300
+ def empty(col: str) -> bool:
301
+ return not (row.get(col) or "").strip()
302
+
303
+ out: list[tuple] = []
304
+ charge = (row.get("ChargeCategory") or "").strip()
305
+ charge_class = (row.get("ChargeClass") or "").strip()
306
+ commit_status = (row.get("CommitmentDiscountStatus") or "").strip()
307
+ non_correction_use = charge in ("Usage", "Purchase") and charge_class != "Correction"
308
+
309
+ if charge == "Tax" and not empty("PricingCategory"):
310
+ out.append(("PricingCategory", "must_be_null_for_tax",
311
+ "PricingCategory must be null when ChargeCategory is 'Tax'"))
312
+ for col in ("SkuId", "SkuPriceId"):
313
+ if charge == "Tax" and not empty(col):
314
+ out.append((col, "must_be_null_for_tax", f"{col} must be null for Tax charges"))
315
+
316
+ # Condition-aware required columns (only when the provider declares support).
317
+ if non_correction_use and COND_MULTIPLE_PRICING_CATEGORIES in supported and empty(
318
+ "PricingCategory"
319
+ ):
320
+ out.append(("PricingCategory", "required_for_usage_or_purchase",
321
+ "PricingCategory required for Usage/Purchase (multiple pricing categories)"))
322
+ if non_correction_use and COND_UNIT_PRICING in supported:
323
+ for col in ("SkuId", "SkuPriceId"):
324
+ if empty(col):
325
+ out.append((col, "required_for_usage_or_purchase",
326
+ f"{col} required for Usage/Purchase when unit pricing is supported"))
327
+
328
+ consumption_applies = charge == "Usage" and commit_status != "Unused"
329
+ for col in ("ConsumedQuantity", "ConsumedUnit"):
330
+ if not consumption_applies and not empty(col):
331
+ out.append((col, "consumption_not_applicable",
332
+ f"{col} is only valid for Usage with status != 'Unused'"))
333
+ if empty("ConsumedQuantity") and not empty("ConsumedUnit"):
334
+ out.append(("ConsumedUnit", "unit_without_quantity",
335
+ "ConsumedUnit must be null when ConsumedQuantity is null"))
336
+ if empty("SkuPriceId") and not empty("SkuPriceDetails"):
337
+ out.append(("SkuPriceDetails", "details_without_sku_price_id",
338
+ "SkuPriceDetails must be null when SkuPriceId is null"))
339
+ if empty("CommitmentDiscountQuantity") and not empty("CommitmentDiscountUnit"):
340
+ out.append(("CommitmentDiscountUnit", "unit_without_quantity",
341
+ "CommitmentDiscountUnit must be null when CommitmentDiscountQuantity is null"))
342
+
343
+ # ChargeFrequency must not be Usage-Based for Purchase charges.
344
+ if charge == "Purchase" and (row.get("ChargeFrequency") or "").strip() == "Usage-Based":
345
+ out.append(("ChargeFrequency", "usage_based_frequency_on_purchase",
346
+ "ChargeFrequency must not be 'Usage-Based' when ChargeCategory is 'Purchase'"))
347
+
348
+ # ServiceSubcategory must belong to its parent ServiceCategory.
349
+ sub = (row.get("ServiceSubcategory") or "").strip()
350
+ cat = (row.get("ServiceCategory") or "").strip()
351
+ parents = model.get("service_subcategory_parents", {})
352
+ if sub and sub in parents and parents[sub] != cat:
353
+ out.append(("ServiceSubcategory", "wrong_parent_category",
354
+ f"ServiceSubcategory '{sub}' belongs to '{parents[sub]}', not '{cat}'"))
355
+ return out
356
+
357
+
358
+ def _last_updated_rule(created: str, updated: str) -> Callable:
359
+ def _rule(row: dict, model: dict, supported: frozenset[str]) -> list[tuple]:
360
+ c, u = (row.get(created) or "").strip(), (row.get(updated) or "").strip()
361
+ if c and u and _DATETIME_RE.fullmatch(c) and _DATETIME_RE.fullmatch(u):
362
+ if _parse_dt(u) < _parse_dt(c):
363
+ return [(updated, "last_updated_before_created", f"{updated} is before {created}")]
364
+ return []
365
+
366
+ return _rule
367
+
368
+
369
+ def _contract_commitment_upfront(row: dict, model: dict, supported: frozenset[str]) -> list[tuple]:
370
+ pm = (row.get("ContractCommitmentPaymentModel") or "").strip()
371
+ pct = _decimal_or_none((row.get("ContractCommitmentPaymentUpfrontPercentage") or "").strip())
372
+ if not pm or pct is None:
373
+ return []
374
+ expected_ok = {
375
+ "All Upfront": pct == Decimal("1"),
376
+ "No Upfront": pct == Decimal("0"),
377
+ "Partial Upfront": Decimal("0") < pct < Decimal("1"),
378
+ }.get(pm)
379
+ if expected_ok is False:
380
+ return [(
381
+ "ContractCommitmentPaymentUpfrontPercentage", "upfront_percentage_mismatch",
382
+ f"upfront percentage {pct} is inconsistent with payment model '{pm}'",
383
+ )]
384
+ return []
385
+
386
+
387
+ _CROSS_FIELD: dict[str, list[Callable]] = {
388
+ "Cost and Usage": [_cost_and_usage],
389
+ "Billing Period": [_last_updated_rule("BillingPeriodCreated", "BillingPeriodLastUpdated")],
390
+ "Contract Commitment": [
391
+ _last_updated_rule("ContractCommitmentCreated", "ContractCommitmentLastUpdated"),
392
+ _contract_commitment_upfront,
393
+ ],
394
+ "Invoice Detail": [_last_updated_rule("InvoiceDetailCreated", "InvoiceDetailLastUpdated")],
395
+ }
396
+
397
+
398
+ def check_column_value(dataset: str, column: str, value: str) -> str | None:
399
+ """Format-check one non-empty value against the model spec of ``dataset.column``.
400
+
401
+ Returns the violated rule name (as in the lint report) or ``None``. Used by the
402
+ supplement validator so client-supplied facts obey exactly the same format rules
403
+ as converted data. An unknown column returns ``"unknown_column"``.
404
+ """
405
+ name = resolve_dataset(dataset)
406
+ spec = load_model()["datasets"][name]["columns"].get(column)
407
+ if spec is None:
408
+ return "unknown_column"
409
+ text = value.strip()
410
+ if not text:
411
+ return None
412
+ return _format_violation(spec, column, text)
413
+
414
+
415
+ def lint_focus_1_4_structure(
416
+ dataset: str,
417
+ rows: list[dict[str, str]],
418
+ *,
419
+ model: dict | None = None,
420
+ supported_conditions: Iterable[str] | None = None,
421
+ profile: CapabilityProfile | None = None,
422
+ ) -> LintReport:
423
+ """Structurally + semantically lint ``rows`` against the FOCUS 1.4 model.
424
+
425
+ This is a linter, not a full conformance validator: a clean report asserts
426
+ ``STRUCTURAL_VALID`` and ``SEMANTIC_VALID`` only (see :data:`LEVEL_CROSS_DATASET`
427
+ / :data:`LEVEL_OFFICIAL`, which are never asserted here).
428
+
429
+ ``supported_conditions`` (raw strings) and/or ``profile`` (a validated
430
+ :class:`~focus_data_toolkit.model.capabilities.CapabilityProfile`) declare the
431
+ FOCUS applicability conditions the provider supports; conditionally-required
432
+ columns are enforced only for those conditions (default: none enforced, so
433
+ sparse-but-valid rows pass — an undeclared condition is *not evaluated*).
434
+ """
435
+ name = resolve_dataset(dataset)
436
+ model = model or load_model()
437
+ supported = frozenset(supported_conditions or ())
438
+ if profile is not None:
439
+ supported |= profile.supported_conditions
440
+ columns: dict = model["datasets"][name]["columns"]
441
+ violations: list[Violation] = []
442
+
443
+ def add(rule, message, column=None, row_index=None, level=LEVEL_STRUCTURAL):
444
+ violations.append(Violation(name, rule, message, column, row_index, level))
445
+
446
+ if not rows:
447
+ add("empty_dataset", "no rows provided")
448
+ return LintReport(name, 0, tuple(violations))
449
+
450
+ present = set().union(*(set(r.keys()) for r in rows))
451
+ for key in sorted(present):
452
+ if key not in columns and not key.startswith("x_"):
453
+ add("unknown_column", f"{key} is not a FOCUS 1.4 {name} column", key)
454
+
455
+ cross_field = _CROSS_FIELD.get(name, [])
456
+ for i, row in enumerate(rows):
457
+ for col, spec in columns.items():
458
+ if col not in row:
459
+ if spec.get("feature_level") == "Mandatory":
460
+ add("missing_mandatory_column", f"required column {col} absent", col, i)
461
+ continue
462
+ value = (row.get(col) or "").strip()
463
+ if not value:
464
+ if not spec.get("allows_nulls", True):
465
+ add("null_not_allowed", "value is null/empty", col, i)
466
+ continue
467
+ rule = _format_violation(spec, col, value)
468
+ if rule:
469
+ add(rule, f"invalid value {value!r}", col, i)
470
+ for fn in cross_field:
471
+ for col, rule, msg in fn(row, model, supported):
472
+ add(rule, msg, col, i, level=LEVEL_SEMANTIC)
473
+
474
+ return LintReport(name, len(rows), tuple(violations))
475
+
476
+
477
+ def validate_focus_1_4(
478
+ dataset: str,
479
+ rows: list[dict[str, str]],
480
+ *,
481
+ model: dict | None = None,
482
+ supported_conditions: Iterable[str] | None = None,
483
+ ) -> LintReport:
484
+ """Deprecated alias of :func:`lint_focus_1_4_structure`.
485
+
486
+ This is a structural + semantic **linter**, not a full FOCUS 1.4 conformance
487
+ validator; ``report.ok`` means the lint passed, not that the data is fully
488
+ FOCUS-conformant.
489
+ """
490
+ warnings.warn(
491
+ "validate_focus_1_4 is deprecated; use lint_focus_1_4_structure. It is a "
492
+ "structural + semantic linter, not a full FOCUS 1.4 conformance validator.",
493
+ DeprecationWarning,
494
+ stacklevel=2,
495
+ )
496
+ return lint_focus_1_4_structure(
497
+ dataset, rows, model=model, supported_conditions=supported_conditions
498
+ )
@@ -0,0 +1,18 @@
1
+ """Conversion modes.
2
+
3
+ * ``STRICT`` (default) — never invents financial facts. A canonical FOCUS 1.4 dataset
4
+ is produced only when every Mandatory non-nullable column can be filled from factual
5
+ lineage (observed / renamed / derived / enriched). Datasets that would require assumed
6
+ provider-issued values are reported ``NOT_PRODUCED`` in the manifest, not fabricated.
7
+ * ``SYNTHETIC`` — for demos / tests / learning. Assumed values may be generated; the
8
+ result is explicitly labelled synthetic and is never presented as fully conformant.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ from enum import StrEnum
14
+
15
+
16
+ class Mode(StrEnum):
17
+ STRICT = "strict"
18
+ SYNTHETIC = "synthetic"
@@ -0,0 +1,61 @@
1
+ """Thin wrapper around the official FinOps FOCUS validator.
2
+
3
+ The official validator (https://github.com/finopsfoundation/focus_validator)
4
+ is an optional dependency — install it with::
5
+
6
+ pip install "focus-data-toolkit[validator]"
7
+
8
+ It requires Python >= 3.12 and validates against the FOCUS rule models
9
+ published with each FOCUS_Spec release (1.2/1.3 today; 1.4 rule-model support
10
+ is expected from the FinOps Foundation later in 2026 — until then the
11
+ toolkit's built-in model validator is the 1.4 conformance gate).
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import shutil
17
+ import subprocess
18
+ import sys
19
+ from pathlib import Path
20
+
21
+
22
+ class OfficialValidatorNotInstalled(RuntimeError):
23
+ """Raised when the official focus-validator is not available."""
24
+
25
+
26
+ def _executable() -> str:
27
+ # Prefer the interpreter's own environment (works even when this CLI was
28
+ # invoked by absolute path and the venv is not on PATH), then fall back
29
+ # to PATH lookup.
30
+ bin_dir = Path(sys.executable).parent
31
+ for name in ("focus-validator", "focus-validator.exe"):
32
+ candidate = bin_dir / name
33
+ if candidate.exists():
34
+ return str(candidate)
35
+ exe = shutil.which("focus-validator")
36
+ if exe is None:
37
+ raise OfficialValidatorNotInstalled(
38
+ "the official FOCUS validator is not installed; run "
39
+ "pip install 'focus-data-toolkit[validator]' (requires Python >= 3.12)"
40
+ )
41
+ return exe
42
+
43
+
44
+ def run_official_validator(
45
+ data_file: str | Path,
46
+ focus_version: str,
47
+ *,
48
+ extra_args: tuple[str, ...] = (),
49
+ ) -> int:
50
+ """Run the official validator on ``data_file``; return its exit code.
51
+
52
+ Output streams directly to the console. ``focus_version`` selects the rule
53
+ model (e.g. ``1.2.0.1``); pass-through flags go in ``extra_args``.
54
+ """
55
+ cmd = [
56
+ _executable(),
57
+ "--data-file", str(data_file),
58
+ "--validate-version", focus_version,
59
+ *extra_args,
60
+ ]
61
+ return subprocess.call(cmd)
@@ -0,0 +1,89 @@
1
+ """Progress reporting and cooperative cancellation for long-running conversions.
2
+
3
+ These are optional, dependency-free hooks. The streaming engine
4
+ (:func:`focus_data_toolkit.convert.convert_files`) emits throttled
5
+ :class:`ProgressEvent`\\ s and checks a :data:`CancelPredicate` between rows and
6
+ validation passes, so a caller — the CLI (``--progress`` + SIGINT/SIGTERM) or the
7
+ Studio backend (a cancel button) — can show progress and cancel cleanly without the
8
+ engine importing any UI concern. All hooks are opt-in: a conversion with neither a
9
+ callback nor a predicate behaves exactly as before.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from collections.abc import Callable
15
+ from dataclasses import dataclass
16
+ from typing import Literal
17
+
18
+ # The ordered phases a streaming conversion moves through. Progress is reported per phase
19
+ # rather than as a single 0..100% bar, because the passes measure different things (bytes
20
+ # read, rows aggregated, rows validated) and a single global percentage would mislead.
21
+ Phase = Literal[
22
+ "READING", # cheap key-collection pre-pass over the source (supplements only)
23
+ "TRANSFORMING", # the main Cost and Usage read + convert + write loop
24
+ "AGGREGATING", # Invoice Detail / Billing Period finalisation from the on-disk index
25
+ "WRITING", # writing the derived datasets
26
+ "VALIDATING", # per-dataset lint + cross-dataset bundle gate
27
+ "PUBLISHING", # checksums + manifest + the single atomic rename
28
+ ]
29
+
30
+ PHASES: tuple[Phase, ...] = (
31
+ "READING",
32
+ "TRANSFORMING",
33
+ "AGGREGATING",
34
+ "WRITING",
35
+ "VALIDATING",
36
+ "PUBLISHING",
37
+ )
38
+
39
+
40
+ @dataclass(frozen=True)
41
+ class ProgressEvent:
42
+ """A single progress observation emitted from a conversion phase.
43
+
44
+ ``total`` is ``None`` when it cannot be known without extra work (e.g. rows aggregated
45
+ on disk, whose count is not known until the pass completes). ``unit`` is ``"rows"`` or
46
+ ``"bytes"`` — the ``TRANSFORMING`` phase reports bytes for a CSV source (byte cursor /
47
+ file size) and rows for a Parquet source (footer row count). :attr:`fraction` is a
48
+ convenience 0..1 derived from ``completed``/``total`` (``None`` when indeterminate).
49
+ """
50
+
51
+ phase: Phase
52
+ completed: int
53
+ total: int | None = None
54
+ unit: str = "rows"
55
+ message: str | None = None
56
+
57
+ @property
58
+ def fraction(self) -> float | None:
59
+ """Completion in ``0..1`` when a total is known, else ``None`` (indeterminate)."""
60
+ if self.total is None or self.total <= 0:
61
+ return None
62
+ return min(1.0, self.completed / self.total)
63
+
64
+ def as_dict(self) -> dict:
65
+ """JSON-friendly view (e.g. for a Studio SSE stream)."""
66
+ return {
67
+ "phase": self.phase,
68
+ "completed": self.completed,
69
+ "total": self.total,
70
+ "unit": self.unit,
71
+ "fraction": self.fraction,
72
+ "message": self.message,
73
+ }
74
+
75
+
76
+ # A caller-supplied sink for progress events. It must be cheap and must not raise.
77
+ ProgressCallback = Callable[[ProgressEvent], None]
78
+ # Returns True when cancellation has been requested. Checked cooperatively between rows /
79
+ # validation passes; the engine raises ConversionCancelled and publishes nothing.
80
+ CancelPredicate = Callable[[], bool]
81
+
82
+
83
+ __all__ = [
84
+ "PHASES",
85
+ "CancelPredicate",
86
+ "Phase",
87
+ "ProgressCallback",
88
+ "ProgressEvent",
89
+ ]