focus-data-toolkit 0.11.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- focus_data_toolkit/__init__.py +69 -0
- focus_data_toolkit/__main__.py +6 -0
- focus_data_toolkit/_version.py +8 -0
- focus_data_toolkit/cli.py +968 -0
- focus_data_toolkit/context/__init__.py +88 -0
- focus_data_toolkit/context/billing.py +54 -0
- focus_data_toolkit/context/provider.py +90 -0
- focus_data_toolkit/convert/__init__.py +708 -0
- focus_data_toolkit/convert/billing_period.py +65 -0
- focus_data_toolkit/convert/contract_applied.py +235 -0
- focus_data_toolkit/convert/contract_commitment.py +182 -0
- focus_data_toolkit/convert/cost_and_usage.py +179 -0
- focus_data_toolkit/convert/detect.py +39 -0
- focus_data_toolkit/convert/invoice_detail.py +199 -0
- focus_data_toolkit/convert/streaming.py +1030 -0
- focus_data_toolkit/errors.py +145 -0
- focus_data_toolkit/focus_json.py +68 -0
- focus_data_toolkit/generators/__init__.py +61 -0
- focus_data_toolkit/generators/_shim.py +43 -0
- focus_data_toolkit/generators/engine/__init__.py +14 -0
- focus_data_toolkit/generators/engine/context.py +12 -0
- focus_data_toolkit/generators/engine/determinism.py +117 -0
- focus_data_toolkit/generators/engine/json_focus.py +63 -0
- focus_data_toolkit/generators/engine/ladder.py +71 -0
- focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
- focus_data_toolkit/generators/engine/serialize.py +151 -0
- focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
- focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
- focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
- focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
- focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
- focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
- focus_data_toolkit/generators/providers/__init__.py +29 -0
- focus_data_toolkit/generators/providers/aws.py +186 -0
- focus_data_toolkit/generators/providers/azure.py +191 -0
- focus_data_toolkit/generators/providers/gcp.py +194 -0
- focus_data_toolkit/generators/providers/profile.py +123 -0
- focus_data_toolkit/generators/scenarios.py +178 -0
- focus_data_toolkit/generators/versions/__init__.py +17 -0
- focus_data_toolkit/generators/versions/adapter.py +41 -0
- focus_data_toolkit/generators/versions/v1_2.py +111 -0
- focus_data_toolkit/generators/versions/v1_3.py +154 -0
- focus_data_toolkit/io/__init__.py +1 -0
- focus_data_toolkit/io/atomic_writer.py +462 -0
- focus_data_toolkit/io/csv_io.py +128 -0
- focus_data_toolkit/io/parquet_io.py +528 -0
- focus_data_toolkit/io/records.py +92 -0
- focus_data_toolkit/io/row_source.py +117 -0
- focus_data_toolkit/lifecycle.py +342 -0
- focus_data_toolkit/manifest.py +114 -0
- focus_data_toolkit/model/__init__.py +43 -0
- focus_data_toolkit/model/capabilities.py +66 -0
- focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
- focus_data_toolkit/model/focus_1_4_model.json +1913 -0
- focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
- focus_data_toolkit/model/focus_json_keys.py +112 -0
- focus_data_toolkit/model/iso_4217_currencies.json +23 -0
- focus_data_toolkit/model/json_schema_check.py +205 -0
- focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
- focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
- focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
- focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
- focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
- focus_data_toolkit/model/model_provenance.json +58 -0
- focus_data_toolkit/model/validator.py +498 -0
- focus_data_toolkit/modes.py +18 -0
- focus_data_toolkit/official_validator.py +61 -0
- focus_data_toolkit/progress.py +89 -0
- focus_data_toolkit/provenance.py +106 -0
- focus_data_toolkit/py.typed +1 -0
- focus_data_toolkit/runtime.py +243 -0
- focus_data_toolkit/schema/__init__.py +17 -0
- focus_data_toolkit/schema/detection.py +274 -0
- focus_data_toolkit/schema/registry.py +127 -0
- focus_data_toolkit/storage/__init__.py +1 -0
- focus_data_toolkit/storage/external_index.py +99 -0
- focus_data_toolkit/storage/spill.py +150 -0
- focus_data_toolkit/studio/__init__.py +19 -0
- focus_data_toolkit/studio/app.py +467 -0
- focus_data_toolkit/studio/config.py +42 -0
- focus_data_toolkit/studio/frontend/app.js +214 -0
- focus_data_toolkit/studio/frontend/index.html +101 -0
- focus_data_toolkit/studio/frontend/style.css +60 -0
- focus_data_toolkit/studio/jobs.py +142 -0
- focus_data_toolkit/studio/preview.py +32 -0
- focus_data_toolkit/studio/security.py +125 -0
- focus_data_toolkit/studio/server.py +71 -0
- focus_data_toolkit/supplement/__init__.py +50 -0
- focus_data_toolkit/supplement/adapters/__init__.py +21 -0
- focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
- focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
- focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
- focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
- focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
- focus_data_toolkit/supplement/adapters/registry.py +215 -0
- focus_data_toolkit/supplement/apply.py +318 -0
- focus_data_toolkit/supplement/gaps.py +219 -0
- focus_data_toolkit/supplement/kinds.py +118 -0
- focus_data_toolkit/supplement/loader.py +409 -0
- focus_data_toolkit/supplement/spec.py +74 -0
- focus_data_toolkit/supplement/validate.py +215 -0
- focus_data_toolkit/validate/__init__.py +15 -0
- focus_data_toolkit/validate/allocation.py +333 -0
- focus_data_toolkit/validate/bundle.py +254 -0
- focus_data_toolkit/validate/codes.py +93 -0
- focus_data_toolkit/validate/corrections.py +245 -0
- focus_data_toolkit/validate/reconciliation.py +98 -0
- focus_data_toolkit/validate/referential.py +289 -0
- focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
- focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
- focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
- focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
- focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,498 @@
|
|
|
1
|
+
"""Reference FOCUS 1.4 **structural linter** (model-driven).
|
|
2
|
+
|
|
3
|
+
This is a *linter*, not a full FOCUS 1.4 conformance validator. It checks that data
|
|
4
|
+
presented as a FOCUS 1.4 dataset is well-formed against the committed 1.4 data model
|
|
5
|
+
(``focus_1_4_model.json``) at two levels — and it can assert **only** those two:
|
|
6
|
+
|
|
7
|
+
* ``STRUCTURAL_VALID`` — required (Mandatory) columns present; unknown non-``x_``
|
|
8
|
+
columns flagged; nullability; and value **format** (NumericFormat incl. scientific
|
|
9
|
+
notation, Date/Time UTC ``…Z``, Currency ISO 4217, Allowed-Values enums, Unit, and
|
|
10
|
+
JSON/Key-Value well-formedness with the ``x_`` custom-key rule).
|
|
11
|
+
* ``SEMANTIC_VALID`` — single-row cross-field rules (Tax nulls, consumption gating,
|
|
12
|
+
``LastUpdated >= Created``, ServiceSubcategory↔ServiceCategory, upfront-percentage vs
|
|
13
|
+
payment model, condition-aware required columns, ContractApplied deep structure), and
|
|
14
|
+
the official FOCUS JSON object schemas (vendored verbatim in ``model/json_schemas/``)
|
|
15
|
+
for ``ContractApplied``, ``AllocatedMethodDetails``,
|
|
16
|
+
``CommitmentProgramEligibilityDetails`` and ``ContractCommitmentApplicability``.
|
|
17
|
+
|
|
18
|
+
It does **not** assert ``CROSS_DATASET_VALID`` (referential integrity across the four
|
|
19
|
+
datasets) or ``OFFICIALLY_VALIDATED`` (the FinOps ``focus_validator``, which does not yet
|
|
20
|
+
support 1.4). A clean ``LintReport`` therefore means *structurally and semantically
|
|
21
|
+
well-formed*, **not** fully FOCUS-conformant.
|
|
22
|
+
|
|
23
|
+
``validate_focus_1_4`` is retained as a deprecated alias of
|
|
24
|
+
``lint_focus_1_4_structure``.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import json
|
|
30
|
+
import re
|
|
31
|
+
import warnings
|
|
32
|
+
from collections.abc import Callable, Iterable
|
|
33
|
+
from dataclasses import dataclass
|
|
34
|
+
from datetime import datetime
|
|
35
|
+
from decimal import Decimal, InvalidOperation
|
|
36
|
+
from functools import lru_cache
|
|
37
|
+
from pathlib import Path
|
|
38
|
+
from typing import TYPE_CHECKING
|
|
39
|
+
|
|
40
|
+
if TYPE_CHECKING: # import cycle: capabilities re-uses the COND_* constants below
|
|
41
|
+
from focus_data_toolkit.model.capabilities import CapabilityProfile
|
|
42
|
+
|
|
43
|
+
from focus_data_toolkit.model.focus_json_keys import (
|
|
44
|
+
XPREFIX_ENFORCED_ELEMENTS_COLUMNS,
|
|
45
|
+
XPREFIX_ENFORCED_KEYVALUE_COLUMNS,
|
|
46
|
+
)
|
|
47
|
+
from focus_data_toolkit.model.json_schema_check import (
|
|
48
|
+
OFFICIAL_SCHEMA_COLUMNS,
|
|
49
|
+
check_against_official_schema,
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
_HERE = Path(__file__).resolve().parent
|
|
53
|
+
_MODEL_PATH = _HERE / "focus_1_4_model.json"
|
|
54
|
+
_ISO_4217_PATH = _HERE / "iso_4217_currencies.json"
|
|
55
|
+
|
|
56
|
+
# Validation levels this linter can assert. CROSS_DATASET_VALID / OFFICIALLY_VALIDATED
|
|
57
|
+
# are intentionally NOT checked here (documented in the module docstring).
|
|
58
|
+
LEVEL_STRUCTURAL = "STRUCTURAL_VALID"
|
|
59
|
+
LEVEL_SEMANTIC = "SEMANTIC_VALID"
|
|
60
|
+
LEVEL_CROSS_DATASET = "CROSS_DATASET_VALID"
|
|
61
|
+
LEVEL_OFFICIAL = "OFFICIALLY_VALIDATED"
|
|
62
|
+
_CHECKED_LEVELS: tuple[str, ...] = (LEVEL_STRUCTURAL, LEVEL_SEMANTIC)
|
|
63
|
+
|
|
64
|
+
# Applicability conditions (FOCUS 1.4 Applicability Criteria) that gate the
|
|
65
|
+
# "conditionally required" columns. Callers pass the subset they declare.
|
|
66
|
+
COND_MULTIPLE_PRICING_CATEGORIES = "SupportsMultiplePricingCategories"
|
|
67
|
+
COND_UNIT_PRICING = "SupportsUnitPricing"
|
|
68
|
+
|
|
69
|
+
_DATASET_ALIASES = {
|
|
70
|
+
"cost and usage": "Cost and Usage", "costandusage": "Cost and Usage", "cau": "Cost and Usage",
|
|
71
|
+
"billing period": "Billing Period", "billingperiod": "Billing Period", "bpd": "Billing Period",
|
|
72
|
+
"contract commitment": "Contract Commitment", "contractcommitment": "Contract Commitment",
|
|
73
|
+
"cct": "Contract Commitment",
|
|
74
|
+
"invoice detail": "Invoice Detail", "invoicedetail": "Invoice Detail", "ind": "Invoice Detail",
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
# NumericFormat (FOCUS attribute): integer, decimal, or scientific E-notation "mEn".
|
|
78
|
+
# The exponent sign is expressed ONLY when negative (no leading '+' on mantissa or
|
|
79
|
+
# exponent). So 35.2E-7 is valid; 35.2E+7 and +333 are not.
|
|
80
|
+
_NUMERIC_RE = re.compile(r"-?\d+(\.\d+)?(E-?\d+)?")
|
|
81
|
+
# DateTimeFormat: literal YYYY-MM-DDTHH:mm:ss[.fff]Z (UTC 'Z' only, ISO 8601).
|
|
82
|
+
_DATETIME_RE = re.compile(r"\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(\.\d+)?Z")
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
@dataclass(frozen=True)
|
|
86
|
+
class Violation:
|
|
87
|
+
dataset: str
|
|
88
|
+
rule: str
|
|
89
|
+
message: str
|
|
90
|
+
column: str | None = None
|
|
91
|
+
row_index: int | None = None
|
|
92
|
+
level: str = LEVEL_STRUCTURAL
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
@dataclass(frozen=True)
|
|
96
|
+
class LintReport:
|
|
97
|
+
dataset: str
|
|
98
|
+
row_count: int
|
|
99
|
+
violations: tuple[Violation, ...]
|
|
100
|
+
levels_checked: tuple[str, ...] = _CHECKED_LEVELS
|
|
101
|
+
|
|
102
|
+
@property
|
|
103
|
+
def ok(self) -> bool:
|
|
104
|
+
"""No structural/semantic lint violations. NOT a full FOCUS conformance claim."""
|
|
105
|
+
return not self.violations
|
|
106
|
+
|
|
107
|
+
def passed(self, level: str) -> bool:
|
|
108
|
+
"""True if ``level`` was checked and has no violations."""
|
|
109
|
+
if level not in self.levels_checked:
|
|
110
|
+
return False
|
|
111
|
+
return not any(v.level == level for v in self.violations)
|
|
112
|
+
|
|
113
|
+
@property
|
|
114
|
+
def levels_passed(self) -> tuple[str, ...]:
|
|
115
|
+
return tuple(level for level in self.levels_checked if self.passed(level))
|
|
116
|
+
|
|
117
|
+
def messages(self) -> list[str]:
|
|
118
|
+
return [
|
|
119
|
+
f"[{v.level}:{v.rule}] {v.column or '-'}"
|
|
120
|
+
+ (f" row {v.row_index}" if v.row_index is not None else "")
|
|
121
|
+
+ f": {v.message}"
|
|
122
|
+
for v in self.violations
|
|
123
|
+
]
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
# Backwards-compatible alias (the class was previously ``ValidationReport``).
|
|
127
|
+
ValidationReport = LintReport
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
@lru_cache(maxsize=1)
|
|
131
|
+
def load_model() -> dict:
|
|
132
|
+
return json.loads(_MODEL_PATH.read_text(encoding="utf-8"))
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
@lru_cache(maxsize=1)
|
|
136
|
+
def _iso_4217() -> frozenset[str]:
|
|
137
|
+
return frozenset(json.loads(_ISO_4217_PATH.read_text(encoding="utf-8"))["codes"])
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def resolve_dataset(name: str) -> str:
|
|
141
|
+
key = name.strip().lower()
|
|
142
|
+
if key not in _DATASET_ALIASES:
|
|
143
|
+
raise ValueError(f"unknown FOCUS 1.4 dataset {name!r}")
|
|
144
|
+
return _DATASET_ALIASES[key]
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
class _DuplicateKey(Exception):
|
|
148
|
+
pass
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
class _NonFiniteConstant(Exception):
|
|
152
|
+
pass
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def _no_dup_pairs(pairs: list[tuple[str, object]]) -> dict:
|
|
156
|
+
seen: dict = {}
|
|
157
|
+
for key, value in pairs:
|
|
158
|
+
if key in seen:
|
|
159
|
+
raise _DuplicateKey(key)
|
|
160
|
+
seen[key] = value
|
|
161
|
+
return seen
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def _reject_constant(const: str) -> float:
|
|
165
|
+
# NaN / Infinity / -Infinity are Python extensions, not valid JSON.
|
|
166
|
+
raise _NonFiniteConstant(const)
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def _load_json_object(value: str) -> tuple[dict | None, str | None]:
|
|
170
|
+
try:
|
|
171
|
+
obj = json.loads(value, object_pairs_hook=_no_dup_pairs, parse_constant=_reject_constant)
|
|
172
|
+
except _DuplicateKey:
|
|
173
|
+
return None, "duplicate_json_key"
|
|
174
|
+
except _NonFiniteConstant:
|
|
175
|
+
return None, "bad_json"
|
|
176
|
+
except json.JSONDecodeError:
|
|
177
|
+
return None, "bad_json"
|
|
178
|
+
if not isinstance(obj, dict):
|
|
179
|
+
return None, "json_not_object"
|
|
180
|
+
return obj, None
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def _decimal_or_none(value: str) -> Decimal | None:
|
|
184
|
+
if not _NUMERIC_RE.fullmatch(value):
|
|
185
|
+
return None
|
|
186
|
+
try:
|
|
187
|
+
d = Decimal(value)
|
|
188
|
+
except (InvalidOperation, ValueError):
|
|
189
|
+
return None
|
|
190
|
+
return d if d.is_finite() else None
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def _parse_dt(value: str) -> datetime:
|
|
194
|
+
return datetime.fromisoformat(value.replace("Z", "+00:00"))
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def _is_utc_datetime(value: str) -> bool:
|
|
198
|
+
if not _DATETIME_RE.fullmatch(value):
|
|
199
|
+
return False
|
|
200
|
+
try:
|
|
201
|
+
_parse_dt(value)
|
|
202
|
+
except ValueError:
|
|
203
|
+
return False
|
|
204
|
+
return True
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def _keys_are_focus_or_prefixed(keys: Iterable[str], focus_keys: frozenset[str]) -> bool:
|
|
208
|
+
return all(k in focus_keys or k.startswith("x_") for k in keys)
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def _validate_contract_applied(value: str) -> str | None:
|
|
212
|
+
# Lazy import avoids an import cycle (convert -> model.validator -> convert).
|
|
213
|
+
from focus_data_toolkit.convert.contract_applied import ContractAppliedError, parse
|
|
214
|
+
|
|
215
|
+
try:
|
|
216
|
+
parse(value, version="1.4")
|
|
217
|
+
except ContractAppliedError:
|
|
218
|
+
return "invalid_contract_applied"
|
|
219
|
+
return None
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _validate_json_column(column: str, value: str, value_format: str) -> str | None:
|
|
223
|
+
obj, err = _load_json_object(value)
|
|
224
|
+
if err:
|
|
225
|
+
return err
|
|
226
|
+
assert obj is not None # _load_json_object returns a dict whenever err is None
|
|
227
|
+
if value_format == "Key-Value":
|
|
228
|
+
if not all(v is None or isinstance(v, str | int | float | bool) for v in obj.values()):
|
|
229
|
+
return "key_value_value_not_scalar"
|
|
230
|
+
focus_keys = XPREFIX_ENFORCED_KEYVALUE_COLUMNS.get(column)
|
|
231
|
+
if focus_keys is not None and not _keys_are_focus_or_prefixed(obj, focus_keys):
|
|
232
|
+
return "custom_key_not_prefixed"
|
|
233
|
+
return None
|
|
234
|
+
# JSON Object columns.
|
|
235
|
+
if column == "ContractApplied":
|
|
236
|
+
err = _validate_contract_applied(value)
|
|
237
|
+
if err:
|
|
238
|
+
return err
|
|
239
|
+
entry = XPREFIX_ENFORCED_ELEMENTS_COLUMNS.get(column)
|
|
240
|
+
if entry is not None:
|
|
241
|
+
array_key, focus_keys = entry
|
|
242
|
+
# Top-level custom keys (alongside the array) must be x_-prefixed too.
|
|
243
|
+
if not all(k == array_key or k.startswith("x_") for k in obj):
|
|
244
|
+
return "custom_key_not_prefixed"
|
|
245
|
+
elements = obj.get(array_key)
|
|
246
|
+
if not isinstance(elements, list):
|
|
247
|
+
return "missing_elements_array"
|
|
248
|
+
for element in elements:
|
|
249
|
+
if not isinstance(element, dict):
|
|
250
|
+
return "element_not_object"
|
|
251
|
+
if not _keys_are_focus_or_prefixed(element, focus_keys):
|
|
252
|
+
return "custom_key_not_prefixed"
|
|
253
|
+
# Normative depth: the official FOCUS JSON Schemas (vendored verbatim, see
|
|
254
|
+
# model/json_schemas/) — conditional scope rules, metric exclusivity, ranges.
|
|
255
|
+
if column in OFFICIAL_SCHEMA_COLUMNS and check_against_official_schema(column, obj):
|
|
256
|
+
return "official_schema_violation"
|
|
257
|
+
return None
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def _format_violation(spec: dict, column: str, value: str) -> str | None:
|
|
261
|
+
"""Return a rule name if non-empty ``value`` violates the column's format."""
|
|
262
|
+
value_format = spec.get("value_format") or ""
|
|
263
|
+
data_type = spec.get("data_type") or ""
|
|
264
|
+
|
|
265
|
+
if value_format.startswith("Decimal") or data_type == "Decimal":
|
|
266
|
+
d = _decimal_or_none(value)
|
|
267
|
+
if d is None:
|
|
268
|
+
return "bad_numeric_format"
|
|
269
|
+
if "non-negative" in value_format and d < Decimal("0"):
|
|
270
|
+
return "negative_decimal"
|
|
271
|
+
rng = spec.get("numeric_range")
|
|
272
|
+
if rng and not (Decimal(str(rng[0])) <= d <= Decimal(str(rng[1]))):
|
|
273
|
+
return "decimal_out_of_range"
|
|
274
|
+
return None
|
|
275
|
+
if value_format == "Date/Time" or data_type == "Date/Time":
|
|
276
|
+
return None if _is_utc_datetime(value) else "bad_datetime"
|
|
277
|
+
if value_format == "Currency":
|
|
278
|
+
return None if value in _iso_4217() else "bad_currency"
|
|
279
|
+
if value_format == "Allowed Values":
|
|
280
|
+
allowed = spec.get("allowed_values")
|
|
281
|
+
if allowed is not None and value not in allowed:
|
|
282
|
+
return "not_in_allowed_values"
|
|
283
|
+
return None
|
|
284
|
+
if value_format == "Unit":
|
|
285
|
+
if value != value.strip() or not value or _decimal_or_none(value) is not None:
|
|
286
|
+
return "bad_unit"
|
|
287
|
+
return None
|
|
288
|
+
if value_format in ("JSON Object", "Key-Value") or data_type == "JSON":
|
|
289
|
+
return _validate_json_column(column, value, value_format)
|
|
290
|
+
if value_format == "Expected Format":
|
|
291
|
+
return None if re.search(r"\d", value) and re.search(r"[A-Za-z]", value) \
|
|
292
|
+
else "bad_expected_format"
|
|
293
|
+
return None
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
# --------------------------------------------------------------------------- #
|
|
297
|
+
# Cross-field (single-row, SEMANTIC) rules — each returns (column, rule, message) tuples.
|
|
298
|
+
# --------------------------------------------------------------------------- #
|
|
299
|
+
def _cost_and_usage(row: dict, model: dict, supported: frozenset[str]) -> list[tuple]:
|
|
300
|
+
def empty(col: str) -> bool:
|
|
301
|
+
return not (row.get(col) or "").strip()
|
|
302
|
+
|
|
303
|
+
out: list[tuple] = []
|
|
304
|
+
charge = (row.get("ChargeCategory") or "").strip()
|
|
305
|
+
charge_class = (row.get("ChargeClass") or "").strip()
|
|
306
|
+
commit_status = (row.get("CommitmentDiscountStatus") or "").strip()
|
|
307
|
+
non_correction_use = charge in ("Usage", "Purchase") and charge_class != "Correction"
|
|
308
|
+
|
|
309
|
+
if charge == "Tax" and not empty("PricingCategory"):
|
|
310
|
+
out.append(("PricingCategory", "must_be_null_for_tax",
|
|
311
|
+
"PricingCategory must be null when ChargeCategory is 'Tax'"))
|
|
312
|
+
for col in ("SkuId", "SkuPriceId"):
|
|
313
|
+
if charge == "Tax" and not empty(col):
|
|
314
|
+
out.append((col, "must_be_null_for_tax", f"{col} must be null for Tax charges"))
|
|
315
|
+
|
|
316
|
+
# Condition-aware required columns (only when the provider declares support).
|
|
317
|
+
if non_correction_use and COND_MULTIPLE_PRICING_CATEGORIES in supported and empty(
|
|
318
|
+
"PricingCategory"
|
|
319
|
+
):
|
|
320
|
+
out.append(("PricingCategory", "required_for_usage_or_purchase",
|
|
321
|
+
"PricingCategory required for Usage/Purchase (multiple pricing categories)"))
|
|
322
|
+
if non_correction_use and COND_UNIT_PRICING in supported:
|
|
323
|
+
for col in ("SkuId", "SkuPriceId"):
|
|
324
|
+
if empty(col):
|
|
325
|
+
out.append((col, "required_for_usage_or_purchase",
|
|
326
|
+
f"{col} required for Usage/Purchase when unit pricing is supported"))
|
|
327
|
+
|
|
328
|
+
consumption_applies = charge == "Usage" and commit_status != "Unused"
|
|
329
|
+
for col in ("ConsumedQuantity", "ConsumedUnit"):
|
|
330
|
+
if not consumption_applies and not empty(col):
|
|
331
|
+
out.append((col, "consumption_not_applicable",
|
|
332
|
+
f"{col} is only valid for Usage with status != 'Unused'"))
|
|
333
|
+
if empty("ConsumedQuantity") and not empty("ConsumedUnit"):
|
|
334
|
+
out.append(("ConsumedUnit", "unit_without_quantity",
|
|
335
|
+
"ConsumedUnit must be null when ConsumedQuantity is null"))
|
|
336
|
+
if empty("SkuPriceId") and not empty("SkuPriceDetails"):
|
|
337
|
+
out.append(("SkuPriceDetails", "details_without_sku_price_id",
|
|
338
|
+
"SkuPriceDetails must be null when SkuPriceId is null"))
|
|
339
|
+
if empty("CommitmentDiscountQuantity") and not empty("CommitmentDiscountUnit"):
|
|
340
|
+
out.append(("CommitmentDiscountUnit", "unit_without_quantity",
|
|
341
|
+
"CommitmentDiscountUnit must be null when CommitmentDiscountQuantity is null"))
|
|
342
|
+
|
|
343
|
+
# ChargeFrequency must not be Usage-Based for Purchase charges.
|
|
344
|
+
if charge == "Purchase" and (row.get("ChargeFrequency") or "").strip() == "Usage-Based":
|
|
345
|
+
out.append(("ChargeFrequency", "usage_based_frequency_on_purchase",
|
|
346
|
+
"ChargeFrequency must not be 'Usage-Based' when ChargeCategory is 'Purchase'"))
|
|
347
|
+
|
|
348
|
+
# ServiceSubcategory must belong to its parent ServiceCategory.
|
|
349
|
+
sub = (row.get("ServiceSubcategory") or "").strip()
|
|
350
|
+
cat = (row.get("ServiceCategory") or "").strip()
|
|
351
|
+
parents = model.get("service_subcategory_parents", {})
|
|
352
|
+
if sub and sub in parents and parents[sub] != cat:
|
|
353
|
+
out.append(("ServiceSubcategory", "wrong_parent_category",
|
|
354
|
+
f"ServiceSubcategory '{sub}' belongs to '{parents[sub]}', not '{cat}'"))
|
|
355
|
+
return out
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
def _last_updated_rule(created: str, updated: str) -> Callable:
|
|
359
|
+
def _rule(row: dict, model: dict, supported: frozenset[str]) -> list[tuple]:
|
|
360
|
+
c, u = (row.get(created) or "").strip(), (row.get(updated) or "").strip()
|
|
361
|
+
if c and u and _DATETIME_RE.fullmatch(c) and _DATETIME_RE.fullmatch(u):
|
|
362
|
+
if _parse_dt(u) < _parse_dt(c):
|
|
363
|
+
return [(updated, "last_updated_before_created", f"{updated} is before {created}")]
|
|
364
|
+
return []
|
|
365
|
+
|
|
366
|
+
return _rule
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
def _contract_commitment_upfront(row: dict, model: dict, supported: frozenset[str]) -> list[tuple]:
|
|
370
|
+
pm = (row.get("ContractCommitmentPaymentModel") or "").strip()
|
|
371
|
+
pct = _decimal_or_none((row.get("ContractCommitmentPaymentUpfrontPercentage") or "").strip())
|
|
372
|
+
if not pm or pct is None:
|
|
373
|
+
return []
|
|
374
|
+
expected_ok = {
|
|
375
|
+
"All Upfront": pct == Decimal("1"),
|
|
376
|
+
"No Upfront": pct == Decimal("0"),
|
|
377
|
+
"Partial Upfront": Decimal("0") < pct < Decimal("1"),
|
|
378
|
+
}.get(pm)
|
|
379
|
+
if expected_ok is False:
|
|
380
|
+
return [(
|
|
381
|
+
"ContractCommitmentPaymentUpfrontPercentage", "upfront_percentage_mismatch",
|
|
382
|
+
f"upfront percentage {pct} is inconsistent with payment model '{pm}'",
|
|
383
|
+
)]
|
|
384
|
+
return []
|
|
385
|
+
|
|
386
|
+
|
|
387
|
+
_CROSS_FIELD: dict[str, list[Callable]] = {
|
|
388
|
+
"Cost and Usage": [_cost_and_usage],
|
|
389
|
+
"Billing Period": [_last_updated_rule("BillingPeriodCreated", "BillingPeriodLastUpdated")],
|
|
390
|
+
"Contract Commitment": [
|
|
391
|
+
_last_updated_rule("ContractCommitmentCreated", "ContractCommitmentLastUpdated"),
|
|
392
|
+
_contract_commitment_upfront,
|
|
393
|
+
],
|
|
394
|
+
"Invoice Detail": [_last_updated_rule("InvoiceDetailCreated", "InvoiceDetailLastUpdated")],
|
|
395
|
+
}
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
def check_column_value(dataset: str, column: str, value: str) -> str | None:
|
|
399
|
+
"""Format-check one non-empty value against the model spec of ``dataset.column``.
|
|
400
|
+
|
|
401
|
+
Returns the violated rule name (as in the lint report) or ``None``. Used by the
|
|
402
|
+
supplement validator so client-supplied facts obey exactly the same format rules
|
|
403
|
+
as converted data. An unknown column returns ``"unknown_column"``.
|
|
404
|
+
"""
|
|
405
|
+
name = resolve_dataset(dataset)
|
|
406
|
+
spec = load_model()["datasets"][name]["columns"].get(column)
|
|
407
|
+
if spec is None:
|
|
408
|
+
return "unknown_column"
|
|
409
|
+
text = value.strip()
|
|
410
|
+
if not text:
|
|
411
|
+
return None
|
|
412
|
+
return _format_violation(spec, column, text)
|
|
413
|
+
|
|
414
|
+
|
|
415
|
+
def lint_focus_1_4_structure(
|
|
416
|
+
dataset: str,
|
|
417
|
+
rows: list[dict[str, str]],
|
|
418
|
+
*,
|
|
419
|
+
model: dict | None = None,
|
|
420
|
+
supported_conditions: Iterable[str] | None = None,
|
|
421
|
+
profile: CapabilityProfile | None = None,
|
|
422
|
+
) -> LintReport:
|
|
423
|
+
"""Structurally + semantically lint ``rows`` against the FOCUS 1.4 model.
|
|
424
|
+
|
|
425
|
+
This is a linter, not a full conformance validator: a clean report asserts
|
|
426
|
+
``STRUCTURAL_VALID`` and ``SEMANTIC_VALID`` only (see :data:`LEVEL_CROSS_DATASET`
|
|
427
|
+
/ :data:`LEVEL_OFFICIAL`, which are never asserted here).
|
|
428
|
+
|
|
429
|
+
``supported_conditions`` (raw strings) and/or ``profile`` (a validated
|
|
430
|
+
:class:`~focus_data_toolkit.model.capabilities.CapabilityProfile`) declare the
|
|
431
|
+
FOCUS applicability conditions the provider supports; conditionally-required
|
|
432
|
+
columns are enforced only for those conditions (default: none enforced, so
|
|
433
|
+
sparse-but-valid rows pass — an undeclared condition is *not evaluated*).
|
|
434
|
+
"""
|
|
435
|
+
name = resolve_dataset(dataset)
|
|
436
|
+
model = model or load_model()
|
|
437
|
+
supported = frozenset(supported_conditions or ())
|
|
438
|
+
if profile is not None:
|
|
439
|
+
supported |= profile.supported_conditions
|
|
440
|
+
columns: dict = model["datasets"][name]["columns"]
|
|
441
|
+
violations: list[Violation] = []
|
|
442
|
+
|
|
443
|
+
def add(rule, message, column=None, row_index=None, level=LEVEL_STRUCTURAL):
|
|
444
|
+
violations.append(Violation(name, rule, message, column, row_index, level))
|
|
445
|
+
|
|
446
|
+
if not rows:
|
|
447
|
+
add("empty_dataset", "no rows provided")
|
|
448
|
+
return LintReport(name, 0, tuple(violations))
|
|
449
|
+
|
|
450
|
+
present = set().union(*(set(r.keys()) for r in rows))
|
|
451
|
+
for key in sorted(present):
|
|
452
|
+
if key not in columns and not key.startswith("x_"):
|
|
453
|
+
add("unknown_column", f"{key} is not a FOCUS 1.4 {name} column", key)
|
|
454
|
+
|
|
455
|
+
cross_field = _CROSS_FIELD.get(name, [])
|
|
456
|
+
for i, row in enumerate(rows):
|
|
457
|
+
for col, spec in columns.items():
|
|
458
|
+
if col not in row:
|
|
459
|
+
if spec.get("feature_level") == "Mandatory":
|
|
460
|
+
add("missing_mandatory_column", f"required column {col} absent", col, i)
|
|
461
|
+
continue
|
|
462
|
+
value = (row.get(col) or "").strip()
|
|
463
|
+
if not value:
|
|
464
|
+
if not spec.get("allows_nulls", True):
|
|
465
|
+
add("null_not_allowed", "value is null/empty", col, i)
|
|
466
|
+
continue
|
|
467
|
+
rule = _format_violation(spec, col, value)
|
|
468
|
+
if rule:
|
|
469
|
+
add(rule, f"invalid value {value!r}", col, i)
|
|
470
|
+
for fn in cross_field:
|
|
471
|
+
for col, rule, msg in fn(row, model, supported):
|
|
472
|
+
add(rule, msg, col, i, level=LEVEL_SEMANTIC)
|
|
473
|
+
|
|
474
|
+
return LintReport(name, len(rows), tuple(violations))
|
|
475
|
+
|
|
476
|
+
|
|
477
|
+
def validate_focus_1_4(
|
|
478
|
+
dataset: str,
|
|
479
|
+
rows: list[dict[str, str]],
|
|
480
|
+
*,
|
|
481
|
+
model: dict | None = None,
|
|
482
|
+
supported_conditions: Iterable[str] | None = None,
|
|
483
|
+
) -> LintReport:
|
|
484
|
+
"""Deprecated alias of :func:`lint_focus_1_4_structure`.
|
|
485
|
+
|
|
486
|
+
This is a structural + semantic **linter**, not a full FOCUS 1.4 conformance
|
|
487
|
+
validator; ``report.ok`` means the lint passed, not that the data is fully
|
|
488
|
+
FOCUS-conformant.
|
|
489
|
+
"""
|
|
490
|
+
warnings.warn(
|
|
491
|
+
"validate_focus_1_4 is deprecated; use lint_focus_1_4_structure. It is a "
|
|
492
|
+
"structural + semantic linter, not a full FOCUS 1.4 conformance validator.",
|
|
493
|
+
DeprecationWarning,
|
|
494
|
+
stacklevel=2,
|
|
495
|
+
)
|
|
496
|
+
return lint_focus_1_4_structure(
|
|
497
|
+
dataset, rows, model=model, supported_conditions=supported_conditions
|
|
498
|
+
)
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""Conversion modes.
|
|
2
|
+
|
|
3
|
+
* ``STRICT`` (default) — never invents financial facts. A canonical FOCUS 1.4 dataset
|
|
4
|
+
is produced only when every Mandatory non-nullable column can be filled from factual
|
|
5
|
+
lineage (observed / renamed / derived / enriched). Datasets that would require assumed
|
|
6
|
+
provider-issued values are reported ``NOT_PRODUCED`` in the manifest, not fabricated.
|
|
7
|
+
* ``SYNTHETIC`` — for demos / tests / learning. Assumed values may be generated; the
|
|
8
|
+
result is explicitly labelled synthetic and is never presented as fully conformant.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from enum import StrEnum
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class Mode(StrEnum):
|
|
17
|
+
STRICT = "strict"
|
|
18
|
+
SYNTHETIC = "synthetic"
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""Thin wrapper around the official FinOps FOCUS validator.
|
|
2
|
+
|
|
3
|
+
The official validator (https://github.com/finopsfoundation/focus_validator)
|
|
4
|
+
is an optional dependency — install it with::
|
|
5
|
+
|
|
6
|
+
pip install "focus-data-toolkit[validator]"
|
|
7
|
+
|
|
8
|
+
It requires Python >= 3.12 and validates against the FOCUS rule models
|
|
9
|
+
published with each FOCUS_Spec release (1.2/1.3 today; 1.4 rule-model support
|
|
10
|
+
is expected from the FinOps Foundation later in 2026 — until then the
|
|
11
|
+
toolkit's built-in model validator is the 1.4 conformance gate).
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import shutil
|
|
17
|
+
import subprocess
|
|
18
|
+
import sys
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class OfficialValidatorNotInstalled(RuntimeError):
|
|
23
|
+
"""Raised when the official focus-validator is not available."""
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _executable() -> str:
|
|
27
|
+
# Prefer the interpreter's own environment (works even when this CLI was
|
|
28
|
+
# invoked by absolute path and the venv is not on PATH), then fall back
|
|
29
|
+
# to PATH lookup.
|
|
30
|
+
bin_dir = Path(sys.executable).parent
|
|
31
|
+
for name in ("focus-validator", "focus-validator.exe"):
|
|
32
|
+
candidate = bin_dir / name
|
|
33
|
+
if candidate.exists():
|
|
34
|
+
return str(candidate)
|
|
35
|
+
exe = shutil.which("focus-validator")
|
|
36
|
+
if exe is None:
|
|
37
|
+
raise OfficialValidatorNotInstalled(
|
|
38
|
+
"the official FOCUS validator is not installed; run "
|
|
39
|
+
"pip install 'focus-data-toolkit[validator]' (requires Python >= 3.12)"
|
|
40
|
+
)
|
|
41
|
+
return exe
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def run_official_validator(
|
|
45
|
+
data_file: str | Path,
|
|
46
|
+
focus_version: str,
|
|
47
|
+
*,
|
|
48
|
+
extra_args: tuple[str, ...] = (),
|
|
49
|
+
) -> int:
|
|
50
|
+
"""Run the official validator on ``data_file``; return its exit code.
|
|
51
|
+
|
|
52
|
+
Output streams directly to the console. ``focus_version`` selects the rule
|
|
53
|
+
model (e.g. ``1.2.0.1``); pass-through flags go in ``extra_args``.
|
|
54
|
+
"""
|
|
55
|
+
cmd = [
|
|
56
|
+
_executable(),
|
|
57
|
+
"--data-file", str(data_file),
|
|
58
|
+
"--validate-version", focus_version,
|
|
59
|
+
*extra_args,
|
|
60
|
+
]
|
|
61
|
+
return subprocess.call(cmd)
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
"""Progress reporting and cooperative cancellation for long-running conversions.
|
|
2
|
+
|
|
3
|
+
These are optional, dependency-free hooks. The streaming engine
|
|
4
|
+
(:func:`focus_data_toolkit.convert.convert_files`) emits throttled
|
|
5
|
+
:class:`ProgressEvent`\\ s and checks a :data:`CancelPredicate` between rows and
|
|
6
|
+
validation passes, so a caller — the CLI (``--progress`` + SIGINT/SIGTERM) or the
|
|
7
|
+
Studio backend (a cancel button) — can show progress and cancel cleanly without the
|
|
8
|
+
engine importing any UI concern. All hooks are opt-in: a conversion with neither a
|
|
9
|
+
callback nor a predicate behaves exactly as before.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from collections.abc import Callable
|
|
15
|
+
from dataclasses import dataclass
|
|
16
|
+
from typing import Literal
|
|
17
|
+
|
|
18
|
+
# The ordered phases a streaming conversion moves through. Progress is reported per phase
|
|
19
|
+
# rather than as a single 0..100% bar, because the passes measure different things (bytes
|
|
20
|
+
# read, rows aggregated, rows validated) and a single global percentage would mislead.
|
|
21
|
+
Phase = Literal[
|
|
22
|
+
"READING", # cheap key-collection pre-pass over the source (supplements only)
|
|
23
|
+
"TRANSFORMING", # the main Cost and Usage read + convert + write loop
|
|
24
|
+
"AGGREGATING", # Invoice Detail / Billing Period finalisation from the on-disk index
|
|
25
|
+
"WRITING", # writing the derived datasets
|
|
26
|
+
"VALIDATING", # per-dataset lint + cross-dataset bundle gate
|
|
27
|
+
"PUBLISHING", # checksums + manifest + the single atomic rename
|
|
28
|
+
]
|
|
29
|
+
|
|
30
|
+
PHASES: tuple[Phase, ...] = (
|
|
31
|
+
"READING",
|
|
32
|
+
"TRANSFORMING",
|
|
33
|
+
"AGGREGATING",
|
|
34
|
+
"WRITING",
|
|
35
|
+
"VALIDATING",
|
|
36
|
+
"PUBLISHING",
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@dataclass(frozen=True)
|
|
41
|
+
class ProgressEvent:
|
|
42
|
+
"""A single progress observation emitted from a conversion phase.
|
|
43
|
+
|
|
44
|
+
``total`` is ``None`` when it cannot be known without extra work (e.g. rows aggregated
|
|
45
|
+
on disk, whose count is not known until the pass completes). ``unit`` is ``"rows"`` or
|
|
46
|
+
``"bytes"`` — the ``TRANSFORMING`` phase reports bytes for a CSV source (byte cursor /
|
|
47
|
+
file size) and rows for a Parquet source (footer row count). :attr:`fraction` is a
|
|
48
|
+
convenience 0..1 derived from ``completed``/``total`` (``None`` when indeterminate).
|
|
49
|
+
"""
|
|
50
|
+
|
|
51
|
+
phase: Phase
|
|
52
|
+
completed: int
|
|
53
|
+
total: int | None = None
|
|
54
|
+
unit: str = "rows"
|
|
55
|
+
message: str | None = None
|
|
56
|
+
|
|
57
|
+
@property
|
|
58
|
+
def fraction(self) -> float | None:
|
|
59
|
+
"""Completion in ``0..1`` when a total is known, else ``None`` (indeterminate)."""
|
|
60
|
+
if self.total is None or self.total <= 0:
|
|
61
|
+
return None
|
|
62
|
+
return min(1.0, self.completed / self.total)
|
|
63
|
+
|
|
64
|
+
def as_dict(self) -> dict:
|
|
65
|
+
"""JSON-friendly view (e.g. for a Studio SSE stream)."""
|
|
66
|
+
return {
|
|
67
|
+
"phase": self.phase,
|
|
68
|
+
"completed": self.completed,
|
|
69
|
+
"total": self.total,
|
|
70
|
+
"unit": self.unit,
|
|
71
|
+
"fraction": self.fraction,
|
|
72
|
+
"message": self.message,
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
# A caller-supplied sink for progress events. It must be cheap and must not raise.
|
|
77
|
+
ProgressCallback = Callable[[ProgressEvent], None]
|
|
78
|
+
# Returns True when cancellation has been requested. Checked cooperatively between rows /
|
|
79
|
+
# validation passes; the engine raises ConversionCancelled and publishes nothing.
|
|
80
|
+
CancelPredicate = Callable[[], bool]
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
__all__ = [
|
|
84
|
+
"PHASES",
|
|
85
|
+
"CancelPredicate",
|
|
86
|
+
"Phase",
|
|
87
|
+
"ProgressCallback",
|
|
88
|
+
"ProgressEvent",
|
|
89
|
+
]
|