focus-data-toolkit 0.11.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. focus_data_toolkit/__init__.py +69 -0
  2. focus_data_toolkit/__main__.py +6 -0
  3. focus_data_toolkit/_version.py +8 -0
  4. focus_data_toolkit/cli.py +968 -0
  5. focus_data_toolkit/context/__init__.py +88 -0
  6. focus_data_toolkit/context/billing.py +54 -0
  7. focus_data_toolkit/context/provider.py +90 -0
  8. focus_data_toolkit/convert/__init__.py +708 -0
  9. focus_data_toolkit/convert/billing_period.py +65 -0
  10. focus_data_toolkit/convert/contract_applied.py +235 -0
  11. focus_data_toolkit/convert/contract_commitment.py +182 -0
  12. focus_data_toolkit/convert/cost_and_usage.py +179 -0
  13. focus_data_toolkit/convert/detect.py +39 -0
  14. focus_data_toolkit/convert/invoice_detail.py +199 -0
  15. focus_data_toolkit/convert/streaming.py +1030 -0
  16. focus_data_toolkit/errors.py +145 -0
  17. focus_data_toolkit/focus_json.py +68 -0
  18. focus_data_toolkit/generators/__init__.py +61 -0
  19. focus_data_toolkit/generators/_shim.py +43 -0
  20. focus_data_toolkit/generators/engine/__init__.py +14 -0
  21. focus_data_toolkit/generators/engine/context.py +12 -0
  22. focus_data_toolkit/generators/engine/determinism.py +117 -0
  23. focus_data_toolkit/generators/engine/json_focus.py +63 -0
  24. focus_data_toolkit/generators/engine/ladder.py +71 -0
  25. focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
  26. focus_data_toolkit/generators/engine/serialize.py +151 -0
  27. focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
  28. focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
  29. focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
  30. focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
  31. focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
  32. focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
  33. focus_data_toolkit/generators/providers/__init__.py +29 -0
  34. focus_data_toolkit/generators/providers/aws.py +186 -0
  35. focus_data_toolkit/generators/providers/azure.py +191 -0
  36. focus_data_toolkit/generators/providers/gcp.py +194 -0
  37. focus_data_toolkit/generators/providers/profile.py +123 -0
  38. focus_data_toolkit/generators/scenarios.py +178 -0
  39. focus_data_toolkit/generators/versions/__init__.py +17 -0
  40. focus_data_toolkit/generators/versions/adapter.py +41 -0
  41. focus_data_toolkit/generators/versions/v1_2.py +111 -0
  42. focus_data_toolkit/generators/versions/v1_3.py +154 -0
  43. focus_data_toolkit/io/__init__.py +1 -0
  44. focus_data_toolkit/io/atomic_writer.py +462 -0
  45. focus_data_toolkit/io/csv_io.py +128 -0
  46. focus_data_toolkit/io/parquet_io.py +528 -0
  47. focus_data_toolkit/io/records.py +92 -0
  48. focus_data_toolkit/io/row_source.py +117 -0
  49. focus_data_toolkit/lifecycle.py +342 -0
  50. focus_data_toolkit/manifest.py +114 -0
  51. focus_data_toolkit/model/__init__.py +43 -0
  52. focus_data_toolkit/model/capabilities.py +66 -0
  53. focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
  54. focus_data_toolkit/model/focus_1_4_model.json +1913 -0
  55. focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
  56. focus_data_toolkit/model/focus_json_keys.py +112 -0
  57. focus_data_toolkit/model/iso_4217_currencies.json +23 -0
  58. focus_data_toolkit/model/json_schema_check.py +205 -0
  59. focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
  60. focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
  61. focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
  62. focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
  63. focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
  64. focus_data_toolkit/model/model_provenance.json +58 -0
  65. focus_data_toolkit/model/validator.py +498 -0
  66. focus_data_toolkit/modes.py +18 -0
  67. focus_data_toolkit/official_validator.py +61 -0
  68. focus_data_toolkit/progress.py +89 -0
  69. focus_data_toolkit/provenance.py +106 -0
  70. focus_data_toolkit/py.typed +1 -0
  71. focus_data_toolkit/runtime.py +243 -0
  72. focus_data_toolkit/schema/__init__.py +17 -0
  73. focus_data_toolkit/schema/detection.py +274 -0
  74. focus_data_toolkit/schema/registry.py +127 -0
  75. focus_data_toolkit/storage/__init__.py +1 -0
  76. focus_data_toolkit/storage/external_index.py +99 -0
  77. focus_data_toolkit/storage/spill.py +150 -0
  78. focus_data_toolkit/studio/__init__.py +19 -0
  79. focus_data_toolkit/studio/app.py +467 -0
  80. focus_data_toolkit/studio/config.py +42 -0
  81. focus_data_toolkit/studio/frontend/app.js +214 -0
  82. focus_data_toolkit/studio/frontend/index.html +101 -0
  83. focus_data_toolkit/studio/frontend/style.css +60 -0
  84. focus_data_toolkit/studio/jobs.py +142 -0
  85. focus_data_toolkit/studio/preview.py +32 -0
  86. focus_data_toolkit/studio/security.py +125 -0
  87. focus_data_toolkit/studio/server.py +71 -0
  88. focus_data_toolkit/supplement/__init__.py +50 -0
  89. focus_data_toolkit/supplement/adapters/__init__.py +21 -0
  90. focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
  91. focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
  92. focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
  93. focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
  94. focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
  95. focus_data_toolkit/supplement/adapters/registry.py +215 -0
  96. focus_data_toolkit/supplement/apply.py +318 -0
  97. focus_data_toolkit/supplement/gaps.py +219 -0
  98. focus_data_toolkit/supplement/kinds.py +118 -0
  99. focus_data_toolkit/supplement/loader.py +409 -0
  100. focus_data_toolkit/supplement/spec.py +74 -0
  101. focus_data_toolkit/supplement/validate.py +215 -0
  102. focus_data_toolkit/validate/__init__.py +15 -0
  103. focus_data_toolkit/validate/allocation.py +333 -0
  104. focus_data_toolkit/validate/bundle.py +254 -0
  105. focus_data_toolkit/validate/codes.py +93 -0
  106. focus_data_toolkit/validate/corrections.py +245 -0
  107. focus_data_toolkit/validate/reconciliation.py +98 -0
  108. focus_data_toolkit/validate/referential.py +289 -0
  109. focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
  110. focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
  111. focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
  112. focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
  113. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
  114. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
  115. focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
  116. focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,65 @@
1
+ """Derive the FOCUS 1.4 Billing Period dataset (6 columns) from Cost and Usage.
2
+
3
+ FOCUS 1.2/1.3 sources have no Billing Period dataset: it is new in 1.4. Each distinct
4
+ ``(BillingPeriodStart, BillingPeriodEnd, InvoiceIssuerName)`` seen in the source becomes one
5
+ Billing Period row. The issuer is taken from each row itself (never from a global first-row
6
+ fallback); a row with no issuer keeps it empty, which the lint then flags rather than the
7
+ converter silently inventing one. Timestamps come from the period and the status of a period
8
+ present in historical billing data is ``"Closed"`` — the derivation is pure (no clock).
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ from collections.abc import Sequence
14
+
15
+ from focus_data_toolkit.model import dataset_columns
16
+ from focus_data_toolkit.provenance import ColumnRule, Lineage
17
+
18
+ DATASET = "Billing Period"
19
+
20
+ # Provenance of every Billing Period column. The status/timestamps are provider
21
+ # billing-cycle facts, not derivable from Cost and Usage -> ASSUMED (block strict).
22
+ PROVENANCE: dict[str, ColumnRule] = {
23
+ "BillingPeriodStart": ColumnRule(Lineage.OBSERVED, "CostAndUsage.BillingPeriodStart"),
24
+ "BillingPeriodEnd": ColumnRule(Lineage.OBSERVED, "CostAndUsage.BillingPeriodEnd"),
25
+ "InvoiceIssuerName": ColumnRule(Lineage.OBSERVED, "CostAndUsage.InvoiceIssuerName"),
26
+ "BillingPeriodCreated": ColumnRule(Lineage.ASSUMED, note="provider billing-cycle timestamp"),
27
+ "BillingPeriodLastUpdated": ColumnRule(
28
+ Lineage.ASSUMED, note="provider billing-cycle timestamp"
29
+ ),
30
+ "BillingPeriodStatus": ColumnRule(
31
+ Lineage.ASSUMED, note="provider Open/Closed state; assumed 'Closed'"
32
+ ),
33
+ }
34
+
35
+
36
+ def billing_period_row(
37
+ start: str, end: str, issuer: str, target: Sequence[str]
38
+ ) -> dict[str, str]:
39
+ """Build one Billing Period row from a ``(start, end, issuer)`` key (pure function)."""
40
+ values = {
41
+ "BillingPeriodStart": start,
42
+ "BillingPeriodEnd": end,
43
+ "BillingPeriodCreated": start,
44
+ "BillingPeriodLastUpdated": end,
45
+ "BillingPeriodStatus": "Closed",
46
+ "InvoiceIssuerName": issuer,
47
+ }
48
+ return {col: values.get(col, "") for col in target}
49
+
50
+
51
+ def build_billing_periods(cau_rows: list[dict[str, str]]) -> list[dict[str, str]]:
52
+ """Return one Billing Period row per distinct ``(start, end, issuer)`` in ``cau_rows``."""
53
+ target = dataset_columns(DATASET)
54
+ seen: dict[tuple[str, str, str], dict[str, str]] = {}
55
+ for row in cau_rows:
56
+ start = (row.get("BillingPeriodStart") or "").strip()
57
+ end = (row.get("BillingPeriodEnd") or "").strip()
58
+ issuer = (row.get("InvoiceIssuerName") or "").strip()
59
+ if not start or not end:
60
+ continue
61
+ key = (start, end, issuer)
62
+ if key in seen:
63
+ continue
64
+ seen[key] = billing_period_row(start, end, issuer, target)
65
+ return [seen[key] for key in sorted(seen)]
@@ -0,0 +1,235 @@
1
+ """Typed ``ContractApplied`` model, parser, validator and 1.3->1.4 migration.
2
+
3
+ ``ContractApplied`` (FOCUS Cost and Usage, JSON Object Format) links a usage row to
4
+ the Contract Commitment dataset. Its structure is a top-level ``Elements`` array of
5
+ objects. The identifier keys are **cased differently** across versions
6
+ (``contractapplied.md`` @ ``v1.3`` vs ``v1.4``):
7
+
8
+ * 1.3: ``ContractID`` / ``ContractCommitmentID`` (uppercase ``ID``)
9
+ * 1.4: ``ContractId`` / ``ContractCommitmentId``
10
+
11
+ The three metric keys — ``ContractCommitmentAppliedCost``,
12
+ ``ContractCommitmentAppliedQuantity``, ``ContractCommitmentAppliedUnit`` — are stable
13
+ across versions; the cost/quantity values are Numeric (JSON **numbers**, never quoted
14
+ strings). Custom keys (top level or inside elements) MUST be ``x_``-prefixed.
15
+
16
+ **1.4 metric exclusivity** (``ContractAppliedObjectSchema`` @ ``v1.4``, ``oneOf``):
17
+ an element carries *either* ``AppliedCost`` (quantity/unit absent-or-null) *or*
18
+ ``AppliedQuantity``+``AppliedUnit`` (cost absent-or-null) — never both. FOCUS 1.3
19
+ only requires *at least one* form, so a legal 1.3 element may carry all three
20
+ metrics; :func:`migrate_1_3_to_1_4` then keeps the **cost** branch (the reconciling
21
+ financial amount) and preserves quantity/unit losslessly as the custom keys
22
+ ``x_ContractCommitmentAppliedQuantity`` / ``x_ContractCommitmentAppliedUnit``.
23
+
24
+ The internal model is version-neutral; version affects (de)serialization casing and
25
+ the 1.4-only exclusivity rule.
26
+ """
27
+
28
+ from __future__ import annotations
29
+
30
+ import json
31
+ from dataclasses import dataclass, field
32
+
33
+ from focus_data_toolkit.focus_json import JsonNumber, dumps_object
34
+
35
+ _METRIC_COST = "ContractCommitmentAppliedCost"
36
+ _METRIC_QTY = "ContractCommitmentAppliedQuantity"
37
+ _METRIC_UNIT = "ContractCommitmentAppliedUnit"
38
+ _METRICS = (_METRIC_COST, _METRIC_QTY, _METRIC_UNIT)
39
+ _NUMERIC_METRIC_KEYS = frozenset({_METRIC_COST, _METRIC_QTY})
40
+
41
+ _ID_KEYS = {
42
+ "1.3": {"contract": "ContractID", "commitment": "ContractCommitmentID"},
43
+ "1.4": {"contract": "ContractId", "commitment": "ContractCommitmentId"},
44
+ }
45
+ # The 1.4 ContractAppliedObjectSchema restricts each element to one metric branch.
46
+ _EXCLUSIVE_METRICS_VERSIONS = frozenset({"1.4"})
47
+
48
+
49
+ class ContractAppliedError(ValueError):
50
+ """Raised when a ContractApplied JSON value is structurally invalid."""
51
+
52
+
53
+ @dataclass(frozen=True)
54
+ class ContractAppliedElement:
55
+ contract_id: str
56
+ contract_commitment_id: str
57
+ applied_cost: str | None = None
58
+ applied_quantity: str | None = None
59
+ applied_unit: str | None = None
60
+ custom: dict[str, object] = field(default_factory=dict)
61
+
62
+
63
+ @dataclass(frozen=True)
64
+ class ContractApplied:
65
+ elements: tuple[ContractAppliedElement, ...]
66
+ custom: dict[str, object] = field(default_factory=dict)
67
+
68
+
69
+ class _DuplicateKey(Exception):
70
+ def __init__(self, key: str) -> None:
71
+ self.key = key
72
+
73
+
74
+ def _no_dup(pairs: list[tuple[str, object]]) -> dict:
75
+ seen: dict = {}
76
+ for key, value in pairs:
77
+ if key in seen:
78
+ raise _DuplicateKey(key)
79
+ seen[key] = value
80
+ return seen
81
+
82
+
83
+ def _require_str(value: object, key: str) -> str:
84
+ if not isinstance(value, str) or isinstance(value, JsonNumber) or not value:
85
+ raise ContractAppliedError(f"{key} must be a non-empty string")
86
+ return value
87
+
88
+
89
+ def _numeric_text(value: object, key: str) -> str:
90
+ """Return the exact numeric text of a JSON **number** token; reject anything else."""
91
+ if isinstance(value, JsonNumber):
92
+ return str(value)
93
+ if isinstance(value, str):
94
+ raise ContractAppliedError(f"{key} must be a JSON number, not a quoted string")
95
+ raise ContractAppliedError(f"{key} must be a JSON number, got {value!r}")
96
+
97
+
98
+ def _parse_element(
99
+ obj: object, ids: dict[str, str], index: int, *, exclusive_metrics: bool
100
+ ) -> ContractAppliedElement:
101
+ if not isinstance(obj, dict):
102
+ raise ContractAppliedError(f"Elements[{index}] must be an object")
103
+ focus_keys = {ids["contract"], ids["commitment"], *_METRICS}
104
+ for key in obj:
105
+ if key not in focus_keys and not key.startswith("x_"):
106
+ raise ContractAppliedError(
107
+ f"Elements[{index}] custom key {key!r} must be prefixed with 'x_'"
108
+ )
109
+ contract_id = _require_str(obj.get(ids["contract"]), ids["contract"])
110
+ commitment_id = _require_str(obj.get(ids["commitment"]), ids["commitment"])
111
+ cost = obj.get(_METRIC_COST)
112
+ qty = obj.get(_METRIC_QTY)
113
+ unit = obj.get(_METRIC_UNIT)
114
+ if cost is None and qty is None:
115
+ raise ContractAppliedError(
116
+ f"Elements[{index}] must provide {_METRIC_COST} or {_METRIC_QTY}"
117
+ )
118
+ if qty is not None and unit is None:
119
+ raise ContractAppliedError(
120
+ f"Elements[{index}] must provide {_METRIC_UNIT} when {_METRIC_QTY} is present"
121
+ )
122
+ if exclusive_metrics and cost is not None and qty is not None:
123
+ raise ContractAppliedError(
124
+ f"Elements[{index}] must not provide both {_METRIC_COST} and {_METRIC_QTY} "
125
+ "(FOCUS 1.4 ContractAppliedObjectSchema oneOf)"
126
+ )
127
+ return ContractAppliedElement(
128
+ contract_id=contract_id,
129
+ contract_commitment_id=commitment_id,
130
+ applied_cost=_numeric_text(cost, _METRIC_COST) if cost is not None else None,
131
+ applied_quantity=_numeric_text(qty, _METRIC_QTY) if qty is not None else None,
132
+ applied_unit=_require_str(unit, _METRIC_UNIT) if unit is not None else None,
133
+ custom={k: v for k, v in obj.items() if k.startswith("x_")},
134
+ )
135
+
136
+
137
+ def parse(text: str, *, version: str = "1.4") -> ContractApplied:
138
+ """Parse and validate a ContractApplied JSON string for ``version``."""
139
+ if version not in _ID_KEYS:
140
+ raise ContractAppliedError(f"unsupported ContractApplied version {version!r}")
141
+ try:
142
+ obj = json.loads(
143
+ text, object_pairs_hook=_no_dup, parse_float=JsonNumber, parse_int=JsonNumber
144
+ )
145
+ except _DuplicateKey as exc:
146
+ raise ContractAppliedError(f"duplicate JSON key {exc.key!r}") from None
147
+ except json.JSONDecodeError as exc:
148
+ raise ContractAppliedError(f"invalid JSON: {exc.msg}") from None
149
+ if not isinstance(obj, dict):
150
+ raise ContractAppliedError("ContractApplied must be a JSON object")
151
+ for key in obj:
152
+ if key != "Elements" and not key.startswith("x_"):
153
+ raise ContractAppliedError(f"top-level custom key {key!r} must be prefixed with 'x_'")
154
+ elements = obj.get("Elements")
155
+ if not isinstance(elements, list):
156
+ raise ContractAppliedError("ContractApplied must have an 'Elements' array")
157
+ if not elements:
158
+ raise ContractAppliedError("'Elements' array must not be empty")
159
+ ids = _ID_KEYS[version]
160
+ exclusive = version in _EXCLUSIVE_METRICS_VERSIONS
161
+ return ContractApplied(
162
+ elements=tuple(
163
+ _parse_element(e, ids, i, exclusive_metrics=exclusive)
164
+ for i, e in enumerate(elements)
165
+ ),
166
+ custom={k: v for k, v in obj.items() if k.startswith("x_")},
167
+ )
168
+
169
+
170
+ def to_json(ca: ContractApplied, *, version: str = "1.4") -> str:
171
+ """Serialize ``ca`` to a compact ContractApplied JSON string for ``version``.
172
+
173
+ Numeric metric values are emitted as JSON numbers (not quoted strings). For 1.4
174
+ an element carrying both metric branches is refused (oneOf exclusivity).
175
+ """
176
+ if version not in _ID_KEYS:
177
+ raise ContractAppliedError(f"unsupported ContractApplied version {version!r}")
178
+ ids = _ID_KEYS[version]
179
+ exclusive = version in _EXCLUSIVE_METRICS_VERSIONS
180
+ elements: list[dict] = []
181
+ for i, el in enumerate(ca.elements):
182
+ if exclusive and el.applied_cost is not None and el.applied_quantity is not None:
183
+ raise ContractAppliedError(
184
+ f"Elements[{i}] carries both metric branches; FOCUS {version} allows only "
185
+ "one (ContractAppliedObjectSchema oneOf)"
186
+ )
187
+ obj: dict[str, object] = {
188
+ ids["contract"]: el.contract_id,
189
+ ids["commitment"]: el.contract_commitment_id,
190
+ }
191
+ if el.applied_cost is not None:
192
+ obj[_METRIC_COST] = el.applied_cost
193
+ if el.applied_quantity is not None:
194
+ obj[_METRIC_QTY] = el.applied_quantity
195
+ if el.applied_unit is not None:
196
+ obj[_METRIC_UNIT] = el.applied_unit
197
+ obj.update(el.custom)
198
+ elements.append(obj)
199
+ top: dict[str, object] = {"Elements": elements}
200
+ top.update(ca.custom)
201
+ return dumps_object(top, numeric_keys=_NUMERIC_METRIC_KEYS)
202
+
203
+
204
+ def _to_1_4_exclusive(el: ContractAppliedElement) -> ContractAppliedElement:
205
+ """Reduce a 1.3 element to one 1.4 metric branch (see the module docstring).
206
+
207
+ When both branches are populated, the cost branch is kept and quantity/unit are
208
+ preserved losslessly as ``x_``-prefixed custom keys.
209
+ """
210
+ if el.applied_cost is None or el.applied_quantity is None:
211
+ return el
212
+ custom = dict(el.custom)
213
+ custom[f"x_{_METRIC_QTY}"] = JsonNumber(el.applied_quantity)
214
+ if el.applied_unit is not None:
215
+ custom[f"x_{_METRIC_UNIT}"] = el.applied_unit
216
+ return ContractAppliedElement(
217
+ contract_id=el.contract_id,
218
+ contract_commitment_id=el.contract_commitment_id,
219
+ applied_cost=el.applied_cost,
220
+ custom=custom,
221
+ )
222
+
223
+
224
+ def migrate_1_3_to_1_4(text: str) -> str:
225
+ """Migrate a FOCUS 1.3 ContractApplied JSON string to the 1.4 schema.
226
+
227
+ Re-cases the identifier keys and enforces the 1.4 metric exclusivity (both-branch
228
+ 1.3 elements keep cost; quantity/unit move to ``x_`` custom keys, losslessly).
229
+ """
230
+ ca = parse(text, version="1.3")
231
+ ca = ContractApplied(
232
+ elements=tuple(_to_1_4_exclusive(el) for el in ca.elements),
233
+ custom=ca.custom,
234
+ )
235
+ return to_json(ca, version="1.4")
@@ -0,0 +1,182 @@
1
+ """Expand a FOCUS 1.3 Contract Commitment dataset (13 columns) to 1.4 (30 columns).
2
+
3
+ The 17 columns FOCUS 1.4 adds are populated as follows:
4
+
5
+ * Derived from the source or the Cost and Usage context:
6
+ ``ContractCommitmentCreated`` / ``ContractCommitmentLastUpdated`` (period
7
+ start), ``ContractCommitmentDurationType`` (from the commitment period),
8
+ ``InvoiceIssuerName`` / ``ServiceProviderName`` (provider context),
9
+ ``PricingCurrency`` (billing currency),
10
+ ``PricingCurrencyContractCommitmentCost`` (commitment cost).
11
+ * Deterministic documented defaults (the 1.3 source carries no equivalent):
12
+ ``ContractCommitmentBenefitCategory="Discount"``,
13
+ ``ContractCommitmentFulfillmentInterval="Monthly"``,
14
+ ``ContractCommitmentLifecycleStatus="Active"``,
15
+ ``ContractCommitmentModel="Continuous"``,
16
+ ``ContractCommitmentOfferCategory="Public"``,
17
+ ``ContractCommitmentPaymentInterval="Monthly"``,
18
+ ``ContractCommitmentPaymentModel="No Upfront"`` (with
19
+ ``ContractCommitmentPaymentUpfrontPercentage="0"`` for cross-field
20
+ consistency), and an explanatory ``ContractCommitmentApplicability`` JSON
21
+ object.
22
+ * Null where the model allows it: ``ContractCommitmentDiscountPercentage``.
23
+ """
24
+
25
+ from __future__ import annotations
26
+
27
+ import json
28
+ from datetime import datetime
29
+
30
+ from focus_data_toolkit.errors import Diagnostic, Severity
31
+ from focus_data_toolkit.model import dataset_columns
32
+ from focus_data_toolkit.provenance import ColumnRule, Lineage
33
+
34
+ DATASET = "Contract Commitment"
35
+
36
+ # Provenance of every 1.4 Contract Commitment column. The 13 source columns are
37
+ # OBSERVED; a few are derived/enriched; the 1.4-new commercial terms are ASSUMED
38
+ # (no source), which blocks strict production.
39
+ _OBSERVED_FROM_1_3 = (
40
+ "BillingCurrency", "ContractCommitmentCategory", "ContractCommitmentCost",
41
+ "ContractCommitmentDescription", "ContractCommitmentId", "ContractCommitmentPeriodEnd",
42
+ "ContractCommitmentPeriodStart", "ContractCommitmentQuantity", "ContractCommitmentType",
43
+ "ContractCommitmentUnit", "ContractId", "ContractPeriodEnd", "ContractPeriodStart",
44
+ )
45
+ PROVENANCE: dict[str, ColumnRule] = {
46
+ **{c: ColumnRule(Lineage.OBSERVED, f"ContractCommitment.{c}") for c in _OBSERVED_FROM_1_3},
47
+ "ContractCommitmentDurationType": ColumnRule(Lineage.DERIVED, "commitment period span"),
48
+ "InvoiceIssuerName": ColumnRule(Lineage.ENRICHED, "Cost and Usage provider context"),
49
+ "ServiceProviderName": ColumnRule(Lineage.ENRICHED, "Cost and Usage provider context"),
50
+ "PricingCurrency": ColumnRule(Lineage.DERIVED, "ContractCommitment.BillingCurrency"),
51
+ "PricingCurrencyContractCommitmentCost": ColumnRule(
52
+ Lineage.DERIVED, "ContractCommitment.ContractCommitmentCost"
53
+ ),
54
+ "ContractCommitmentCreated": ColumnRule(Lineage.ASSUMED, note="provider record timestamp"),
55
+ "ContractCommitmentLastUpdated": ColumnRule(Lineage.ASSUMED, note="provider record timestamp"),
56
+ "ContractCommitmentDiscountPercentage": ColumnRule(Lineage.UNAVAILABLE, note="emitted null"),
57
+ "ContractCommitmentApplicability": ColumnRule(Lineage.ASSUMED, note="terms absent from source"),
58
+ "ContractCommitmentBenefitCategory": ColumnRule(Lineage.ASSUMED, note="assumed default"),
59
+ "ContractCommitmentFulfillmentInterval": ColumnRule(Lineage.ASSUMED, note="assumed default"),
60
+ "ContractCommitmentLifecycleStatus": ColumnRule(Lineage.ASSUMED, note="assumed default"),
61
+ "ContractCommitmentModel": ColumnRule(Lineage.ASSUMED, note="assumed default"),
62
+ "ContractCommitmentOfferCategory": ColumnRule(Lineage.ASSUMED, note="assumed default"),
63
+ "ContractCommitmentPaymentInterval": ColumnRule(Lineage.ASSUMED, note="assumed default"),
64
+ "ContractCommitmentPaymentModel": ColumnRule(Lineage.ASSUMED, note="assumed default"),
65
+ "ContractCommitmentPaymentUpfrontPercentage": ColumnRule(Lineage.ASSUMED, note="assumed default"),
66
+ }
67
+
68
+ # The official ContractCommitmentApplicability object schema requires a scope
69
+ # representation: when neither IsGlobalScope nor IsComplexScope is true, Inclusions
70
+ # (min 1) and InclusionOperator become required. The authoritative terms are unknown
71
+ # here, so the minimal conformant synthetic object declares a complex scope; the
72
+ # value stays ASSUMED and never passes strict mode.
73
+ _APPLICABILITY = json.dumps(
74
+ {"IsComplexScope": True,
75
+ "x_Source": "Synthetic applicability derived from a FOCUS 1.3 Contract Commitment "
76
+ "dataset; authoritative applicability terms were not present in the source."},
77
+ separators=(",", ":"),
78
+ )
79
+
80
+ _DEFAULTS = {
81
+ "ContractCommitmentApplicability": _APPLICABILITY,
82
+ "ContractCommitmentBenefitCategory": "Discount",
83
+ "ContractCommitmentDiscountPercentage": "",
84
+ "ContractCommitmentFulfillmentInterval": "Monthly",
85
+ "ContractCommitmentLifecycleStatus": "Active",
86
+ "ContractCommitmentModel": "Continuous",
87
+ "ContractCommitmentOfferCategory": "Public",
88
+ "ContractCommitmentPaymentInterval": "Monthly",
89
+ "ContractCommitmentPaymentModel": "No Upfront",
90
+ "ContractCommitmentPaymentUpfrontPercentage": "0",
91
+ }
92
+
93
+
94
+ def _parse(ts: str) -> datetime | None:
95
+ try:
96
+ return datetime.fromisoformat(ts.replace("Z", "+00:00"))
97
+ except ValueError:
98
+ return None
99
+
100
+
101
+ def _duration_type(start: str, end: str) -> str:
102
+ """Return an Expected-Format duration like ``"12 Months"`` from the period.
103
+
104
+ An unparseable or inverted period yields ``""`` (never a fabricated duration):
105
+ the value cannot be derived from the source, and the mandatory-column lint will
106
+ flag the row rather than silently publish an arbitrary ``"12 Months"``.
107
+ """
108
+ a, b = _parse(start or ""), _parse(end or "")
109
+ if a is None or b is None or b <= a:
110
+ return ""
111
+ months = max(1, round((b - a).days / 30.44))
112
+ return f"{months} Months" if months > 1 else "1 Month"
113
+
114
+
115
+ # How many offending ContractCommitmentIds a diagnostic lists inline.
116
+ _ID_SAMPLE_CAP = 25
117
+
118
+
119
+ def convert_contract_commitment(
120
+ rows: list[dict[str, str]],
121
+ *,
122
+ service_provider_name: str,
123
+ invoice_issuer_name: str,
124
+ diagnostics: list[Diagnostic] | None = None,
125
+ ) -> list[dict[str, str]]:
126
+ """Return the 13-column 1.3 ``rows`` expanded to the 1.4 30-column shape.
127
+
128
+ Rows whose commitment period cannot be parsed get an empty
129
+ ``ContractCommitmentDurationType`` (the duration is not derivable) and are
130
+ reported through ``diagnostics`` as a single aggregated ``FDT-CC-001`` WARNING.
131
+ """
132
+ target = dataset_columns(DATASET)
133
+ out: list[dict[str, str]] = []
134
+ unparseable_ids: list[str] = []
135
+ for row in rows:
136
+ created = row.get("ContractCommitmentPeriodStart", "")
137
+ converted: dict[str, str] = {}
138
+ for col in target:
139
+ if col in row:
140
+ converted[col] = row[col]
141
+ elif col == "ContractCommitmentCreated":
142
+ converted[col] = created
143
+ elif col == "ContractCommitmentLastUpdated":
144
+ converted[col] = created
145
+ elif col == "ContractCommitmentDurationType":
146
+ duration = _duration_type(
147
+ row.get("ContractCommitmentPeriodStart", ""),
148
+ row.get("ContractCommitmentPeriodEnd", ""),
149
+ )
150
+ if not duration:
151
+ unparseable_ids.append(row.get("ContractCommitmentId", ""))
152
+ converted[col] = duration
153
+ elif col == "InvoiceIssuerName":
154
+ converted[col] = invoice_issuer_name
155
+ elif col == "ServiceProviderName":
156
+ converted[col] = service_provider_name
157
+ elif col == "PricingCurrency":
158
+ converted[col] = row.get("BillingCurrency", "")
159
+ elif col == "PricingCurrencyContractCommitmentCost":
160
+ converted[col] = row.get("ContractCommitmentCost", "")
161
+ elif col in _DEFAULTS:
162
+ converted[col] = _DEFAULTS[col]
163
+ else:
164
+ converted[col] = ""
165
+ out.append(converted)
166
+ if unparseable_ids and diagnostics is not None:
167
+ diagnostics.append(
168
+ Diagnostic(
169
+ code="FDT-CC-001",
170
+ severity=Severity.WARNING,
171
+ message="commitment period unparseable or inverted; "
172
+ "ContractCommitmentDurationType left empty (not derivable)",
173
+ datasets=(DATASET,),
174
+ context={
175
+ "row_count": str(len(unparseable_ids)),
176
+ "contract_commitment_ids": ", ".join(
177
+ sorted(set(unparseable_ids))[:_ID_SAMPLE_CAP]
178
+ ),
179
+ },
180
+ )
181
+ )
182
+ return out
@@ -0,0 +1,179 @@
1
+ """Convert FOCUS 1.2/1.3 Cost and Usage rows to the FOCUS 1.4 column set.
2
+
3
+ FOCUS 1.4 Cost and Usage keeps 1.3's 65-column count but:
4
+
5
+ * removes the deprecated ``ProviderName`` / ``PublisherName`` (superseded by
6
+ the 1.3 ``ServiceProviderName`` / ``HostProviderName`` split);
7
+ * adds ``CommitmentProgramEligibilityDetails`` and ``InvoiceDetailId``
8
+ (both conditional and nullable).
9
+
10
+ A 1.2 source is first lifted to the 1.3 shape: ``ServiceProviderName`` is
11
+ derived from ``ProviderName`` (its 1.3 replacement), ``HostProviderName``
12
+ takes the ``ServiceProviderName`` value — FOCUS requires the host to match
13
+ the service provider when the source does not expose the underlying host,
14
+ and a 1.2 source never exposes it. The deprecated ``PublisherName`` ("entity
15
+ that produced the service") is dropped: it does not identify the host. The
16
+ 1.3-only columns (Split Cost Allocation set, ``ContractApplied``) are null.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ from collections.abc import Iterable, Mapping
22
+
23
+ from focus_data_toolkit.convert.contract_applied import migrate_1_3_to_1_4
24
+ from focus_data_toolkit.convert.invoice_detail import GrainKey, invoice_detail_grain_key
25
+ from focus_data_toolkit.model import dataset_columns
26
+ from focus_data_toolkit.provenance import ColumnRule, Lineage, LineageCounters
27
+
28
+ DATASET = "Cost and Usage"
29
+
30
+ # 1.2 -> 1.3/1.4 participant-entity derivations. Both columns derive from
31
+ # ProviderName: FOCUS 1.3 replaced ProviderName with ServiceProviderName, and the
32
+ # HostProviderName rules require the value to match ServiceProviderName when the
33
+ # source does not expose the underlying host (a 1.2 source never does). The
34
+ # deprecated PublisherName is NOT a host equivalent and is dropped with the other
35
+ # removed 1.2 columns.
36
+ _DERIVED_FROM_1_2 = {
37
+ "ServiceProviderName": "ProviderName",
38
+ "HostProviderName": "ProviderName",
39
+ }
40
+
41
+
42
+ def cost_and_usage_provenance(
43
+ source_columns: Iterable[str], source_version: str, *, invoice_detail_linked: bool
44
+ ) -> dict[str, ColumnRule]:
45
+ """Return the per-column lineage of a converted Cost and Usage dataset.
46
+
47
+ ``invoice_detail_linked`` is True when an (synthetic) Invoice Detail dataset is being
48
+ produced, so ``InvoiceDetailId`` carries the back-link (assumed); otherwise it is null.
49
+ """
50
+ present = set(source_columns)
51
+ rules: dict[str, ColumnRule] = {}
52
+ for col in dataset_columns(DATASET):
53
+ if col == "ContractApplied":
54
+ rules[col] = (
55
+ ColumnRule(Lineage.DERIVED, "ContractApplied migrated 1.3->1.4")
56
+ if source_version == "1.3" and "ContractApplied" in present
57
+ else ColumnRule(Lineage.UNAVAILABLE, note="emitted null")
58
+ )
59
+ elif col in ("PricingCurrency", "PricingCurrencyEffectiveCost"):
60
+ # Non-nullable in 1.4; source value where present, nulls backfilled from
61
+ # billing-currency values -> derived at the column level (not plain observed).
62
+ rules[col] = ColumnRule(
63
+ Lineage.DERIVED, "source value; nulls backfilled from billing-currency values"
64
+ )
65
+ elif col in present:
66
+ rules[col] = ColumnRule(Lineage.OBSERVED, f"CostAndUsage.{col}")
67
+ elif source_version == "1.2" and col == "ServiceProviderName":
68
+ rules[col] = ColumnRule(
69
+ Lineage.DERIVED,
70
+ "ProviderName",
71
+ note="FOCUS 1.3 replaced ProviderName with ServiceProviderName",
72
+ )
73
+ elif source_version == "1.2" and col == "HostProviderName":
74
+ rules[col] = ColumnRule(
75
+ Lineage.DERIVED,
76
+ "ServiceProviderName (from ProviderName)",
77
+ note=(
78
+ "host not exposed by a 1.2 source; FOCUS requires "
79
+ "HostProviderName to match ServiceProviderName in that case"
80
+ ),
81
+ )
82
+ elif col == "InvoiceDetailId":
83
+ # A locally generated hash presented as an issuer-assigned id -> assumed
84
+ # when linked (so synthetic Cost and Usage is labelled synthetic); else null.
85
+ rules[col] = (
86
+ ColumnRule(
87
+ Lineage.ASSUMED, note="locally generated back-link to synthetic Invoice Detail"
88
+ )
89
+ if invoice_detail_linked
90
+ else ColumnRule(Lineage.UNAVAILABLE, note="emitted null (Invoice Detail not produced)")
91
+ )
92
+ else:
93
+ rules[col] = ColumnRule(Lineage.UNAVAILABLE, note="emitted null")
94
+ return rules
95
+
96
+
97
+ def _convert_contract_applied(raw: str | None, source_version: str) -> str:
98
+ """Migrate a source ``ContractApplied`` JSON to the FOCUS 1.4 schema.
99
+
100
+ 1.4 re-cases the identifier keys (``ContractID``->``ContractId``,
101
+ ``ContractCommitmentID``->``ContractCommitmentId``). A 1.2 source has no
102
+ ``ContractApplied`` column, so the value is empty there. Raises
103
+ ``ContractAppliedError`` (a ``ValueError``) on a structurally invalid source value.
104
+ """
105
+ text = (raw or "").strip()
106
+ if not text or source_version != "1.3":
107
+ return text
108
+ return migrate_1_3_to_1_4(text)
109
+
110
+
111
+ def convert_cost_and_usage_row(
112
+ row: Mapping[str, str],
113
+ source_version: str,
114
+ *,
115
+ detail_id: str = "",
116
+ target: tuple[str, ...] | None = None,
117
+ counters: LineageCounters | None = None,
118
+ ) -> dict[str, str]:
119
+ """Convert one source row to the FOCUS 1.4 Cost and Usage shape (pure function).
120
+
121
+ ``detail_id`` is the already-resolved ``InvoiceDetailId`` back-link (empty in strict mode
122
+ or for rows with no invoice). Shared by the eager and streaming pipelines so both produce
123
+ identical output. ``counters`` (optional) records the per-value lineage of columns whose
124
+ rule varies by row (the pricing-currency backfill pair).
125
+ """
126
+ columns = target if target is not None else dataset_columns(DATASET)
127
+ converted: dict[str, str] = {}
128
+ for col in columns:
129
+ if col == "ContractApplied":
130
+ converted[col] = _convert_contract_applied(row.get(col), source_version)
131
+ elif col in row:
132
+ converted[col] = row[col]
133
+ elif source_version == "1.2" and col in _DERIVED_FROM_1_2:
134
+ converted[col] = row.get(_DERIVED_FROM_1_2[col], "")
135
+ elif col == "InvoiceDetailId":
136
+ converted[col] = detail_id
137
+ else:
138
+ # New-in-1.4 or 1.3-only columns absent from the source: null.
139
+ converted[col] = ""
140
+ # FOCUS 1.4 makes the pricing-currency pair non-nullable. When a 1.x source leaves it
141
+ # null (e.g. tax or credit rows), pricing happened in the billing currency, so backfill.
142
+ for col, fallback in (
143
+ ("PricingCurrency", "BillingCurrency"),
144
+ ("PricingCurrencyEffectiveCost", "EffectiveCost"),
145
+ ):
146
+ if not converted.get(col):
147
+ converted[col] = converted.get(fallback, "")
148
+ if counters is not None:
149
+ counters.record(col, Lineage.DERIVED)
150
+ elif counters is not None:
151
+ counters.record(col, Lineage.OBSERVED)
152
+ return converted
153
+
154
+
155
+ def convert_cost_and_usage(
156
+ rows: list[dict[str, str]],
157
+ source_version: str,
158
+ *,
159
+ invoice_detail_ids: dict[GrainKey, str] | None = None,
160
+ counters: LineageCounters | None = None,
161
+ ) -> list[dict[str, str]]:
162
+ """Return ``rows`` reshaped to the FOCUS 1.4 Cost and Usage column set.
163
+
164
+ ``invoice_detail_ids`` maps each Invoice Detail business-grain key to the
165
+ ``InvoiceDetailId`` assigned by the Invoice Detail builder, so converted rows link back
166
+ to their invoice line item on exactly the same key.
167
+ """
168
+ target = dataset_columns(DATASET)
169
+ ids = invoice_detail_ids or {}
170
+ return [
171
+ convert_cost_and_usage_row(
172
+ row,
173
+ source_version,
174
+ detail_id=ids.get(invoice_detail_grain_key(row), ""),
175
+ target=target,
176
+ counters=counters,
177
+ )
178
+ for row in rows
179
+ ]