focus-data-toolkit 0.11.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- focus_data_toolkit/__init__.py +69 -0
- focus_data_toolkit/__main__.py +6 -0
- focus_data_toolkit/_version.py +8 -0
- focus_data_toolkit/cli.py +968 -0
- focus_data_toolkit/context/__init__.py +88 -0
- focus_data_toolkit/context/billing.py +54 -0
- focus_data_toolkit/context/provider.py +90 -0
- focus_data_toolkit/convert/__init__.py +708 -0
- focus_data_toolkit/convert/billing_period.py +65 -0
- focus_data_toolkit/convert/contract_applied.py +235 -0
- focus_data_toolkit/convert/contract_commitment.py +182 -0
- focus_data_toolkit/convert/cost_and_usage.py +179 -0
- focus_data_toolkit/convert/detect.py +39 -0
- focus_data_toolkit/convert/invoice_detail.py +199 -0
- focus_data_toolkit/convert/streaming.py +1030 -0
- focus_data_toolkit/errors.py +145 -0
- focus_data_toolkit/focus_json.py +68 -0
- focus_data_toolkit/generators/__init__.py +61 -0
- focus_data_toolkit/generators/_shim.py +43 -0
- focus_data_toolkit/generators/engine/__init__.py +14 -0
- focus_data_toolkit/generators/engine/context.py +12 -0
- focus_data_toolkit/generators/engine/determinism.py +117 -0
- focus_data_toolkit/generators/engine/json_focus.py +63 -0
- focus_data_toolkit/generators/engine/ladder.py +71 -0
- focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
- focus_data_toolkit/generators/engine/serialize.py +151 -0
- focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
- focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
- focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
- focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
- focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
- focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
- focus_data_toolkit/generators/providers/__init__.py +29 -0
- focus_data_toolkit/generators/providers/aws.py +186 -0
- focus_data_toolkit/generators/providers/azure.py +191 -0
- focus_data_toolkit/generators/providers/gcp.py +194 -0
- focus_data_toolkit/generators/providers/profile.py +123 -0
- focus_data_toolkit/generators/scenarios.py +178 -0
- focus_data_toolkit/generators/versions/__init__.py +17 -0
- focus_data_toolkit/generators/versions/adapter.py +41 -0
- focus_data_toolkit/generators/versions/v1_2.py +111 -0
- focus_data_toolkit/generators/versions/v1_3.py +154 -0
- focus_data_toolkit/io/__init__.py +1 -0
- focus_data_toolkit/io/atomic_writer.py +462 -0
- focus_data_toolkit/io/csv_io.py +128 -0
- focus_data_toolkit/io/parquet_io.py +528 -0
- focus_data_toolkit/io/records.py +92 -0
- focus_data_toolkit/io/row_source.py +117 -0
- focus_data_toolkit/lifecycle.py +342 -0
- focus_data_toolkit/manifest.py +114 -0
- focus_data_toolkit/model/__init__.py +43 -0
- focus_data_toolkit/model/capabilities.py +66 -0
- focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
- focus_data_toolkit/model/focus_1_4_model.json +1913 -0
- focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
- focus_data_toolkit/model/focus_json_keys.py +112 -0
- focus_data_toolkit/model/iso_4217_currencies.json +23 -0
- focus_data_toolkit/model/json_schema_check.py +205 -0
- focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
- focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
- focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
- focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
- focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
- focus_data_toolkit/model/model_provenance.json +58 -0
- focus_data_toolkit/model/validator.py +498 -0
- focus_data_toolkit/modes.py +18 -0
- focus_data_toolkit/official_validator.py +61 -0
- focus_data_toolkit/progress.py +89 -0
- focus_data_toolkit/provenance.py +106 -0
- focus_data_toolkit/py.typed +1 -0
- focus_data_toolkit/runtime.py +243 -0
- focus_data_toolkit/schema/__init__.py +17 -0
- focus_data_toolkit/schema/detection.py +274 -0
- focus_data_toolkit/schema/registry.py +127 -0
- focus_data_toolkit/storage/__init__.py +1 -0
- focus_data_toolkit/storage/external_index.py +99 -0
- focus_data_toolkit/storage/spill.py +150 -0
- focus_data_toolkit/studio/__init__.py +19 -0
- focus_data_toolkit/studio/app.py +467 -0
- focus_data_toolkit/studio/config.py +42 -0
- focus_data_toolkit/studio/frontend/app.js +214 -0
- focus_data_toolkit/studio/frontend/index.html +101 -0
- focus_data_toolkit/studio/frontend/style.css +60 -0
- focus_data_toolkit/studio/jobs.py +142 -0
- focus_data_toolkit/studio/preview.py +32 -0
- focus_data_toolkit/studio/security.py +125 -0
- focus_data_toolkit/studio/server.py +71 -0
- focus_data_toolkit/supplement/__init__.py +50 -0
- focus_data_toolkit/supplement/adapters/__init__.py +21 -0
- focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
- focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
- focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
- focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
- focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
- focus_data_toolkit/supplement/adapters/registry.py +215 -0
- focus_data_toolkit/supplement/apply.py +318 -0
- focus_data_toolkit/supplement/gaps.py +219 -0
- focus_data_toolkit/supplement/kinds.py +118 -0
- focus_data_toolkit/supplement/loader.py +409 -0
- focus_data_toolkit/supplement/spec.py +74 -0
- focus_data_toolkit/supplement/validate.py +215 -0
- focus_data_toolkit/validate/__init__.py +15 -0
- focus_data_toolkit/validate/allocation.py +333 -0
- focus_data_toolkit/validate/bundle.py +254 -0
- focus_data_toolkit/validate/codes.py +93 -0
- focus_data_toolkit/validate/corrections.py +245 -0
- focus_data_toolkit/validate/reconciliation.py +98 -0
- focus_data_toolkit/validate/referential.py +289 -0
- focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
- focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
- focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
- focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
- focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,318 @@
|
|
|
1
|
+
"""Apply validated supplements to the derived datasets (ENRICHED lineage).
|
|
2
|
+
|
|
3
|
+
Mode rules, per fact column present in a supplement table:
|
|
4
|
+
|
|
5
|
+
* **Strict** — a value is either supplied by the client or empty; synthetic defaults
|
|
6
|
+
are never emitted. A *non-nullable* column's rule flips to ``ENRICHED`` only at
|
|
7
|
+
100 % key coverage (otherwise it keeps blocking and the dataset stays
|
|
8
|
+
``NOT_PRODUCED``); a *nullable* column flips to ``ENRICHED`` as soon as the
|
|
9
|
+
supplement carries it, with per-value counters recording the supplied/null mix.
|
|
10
|
+
* **Synthetic** — a supplied value wins, the documented synthetic default fills the
|
|
11
|
+
rest. The headline rule flips to ``ENRICHED`` only at full coverage (otherwise it
|
|
12
|
+
stays ``ASSUMED`` — the weakest lineage present), with counters showing the mix.
|
|
13
|
+
|
|
14
|
+
The strict gate itself is untouched: ``ENRICHED`` is already factual, so a fully
|
|
15
|
+
covered dataset simply stops blocking.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
from dataclasses import dataclass, field
|
|
21
|
+
|
|
22
|
+
from focus_data_toolkit.convert.invoice_detail import (
|
|
23
|
+
GrainKey,
|
|
24
|
+
invoice_detail_grain_key,
|
|
25
|
+
)
|
|
26
|
+
from focus_data_toolkit.model import dataset_columns
|
|
27
|
+
from focus_data_toolkit.model.validator import load_model
|
|
28
|
+
from focus_data_toolkit.provenance import ColumnRule, Lineage, LineageCounters
|
|
29
|
+
from focus_data_toolkit.supplement.loader import SupplementBundle, SupplementTable
|
|
30
|
+
from focus_data_toolkit.supplement.validate import SourceKeySets, coverage
|
|
31
|
+
|
|
32
|
+
# Invoice Detail columns normally omitted (unfilled conditionals) that a supplement can
|
|
33
|
+
# activate; non-nullable ones require full coverage to be emitted at all.
|
|
34
|
+
_ACTIVATABLE_INVOICE_COLUMNS = (
|
|
35
|
+
"PaymentCurrency",
|
|
36
|
+
"PaymentCurrencyBilledCost",
|
|
37
|
+
"PaymentCurrencyInvoiceDetailId",
|
|
38
|
+
"PurchaseOrderNumber",
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class AppliedDataset:
|
|
44
|
+
"""Outcome of applying supplements to one dataset."""
|
|
45
|
+
|
|
46
|
+
rows: list[dict[str, str]] | None
|
|
47
|
+
provenance: dict[str, ColumnRule]
|
|
48
|
+
counters: LineageCounters = field(default_factory=LineageCounters)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _source_label(table: SupplementTable, column: str) -> str:
|
|
52
|
+
# Per-column attribution: a merged table (adapter export + hand-authored file) tags each
|
|
53
|
+
# column to its originating file ("supplement:<adapter>@<version>:<file>" or the kind).
|
|
54
|
+
return table.source_for(column)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _allows_nulls(dataset: str, column: str) -> bool:
|
|
58
|
+
spec = load_model()["datasets"][dataset]["columns"].get(column) or {}
|
|
59
|
+
return bool(spec.get("allows_nulls", True))
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _flip_rules(
|
|
63
|
+
base: dict[str, ColumnRule],
|
|
64
|
+
dataset: str,
|
|
65
|
+
tables: list[SupplementTable],
|
|
66
|
+
source: SourceKeySets,
|
|
67
|
+
) -> dict[str, ColumnRule]:
|
|
68
|
+
"""Return ``base`` with columns flipped to ENRICHED per the coverage rules."""
|
|
69
|
+
rules = dict(base)
|
|
70
|
+
for table in tables:
|
|
71
|
+
cov = coverage(table, source.keys_for(table.kind.name))
|
|
72
|
+
for column in table.fact_columns:
|
|
73
|
+
if table.kind.name == "invoice_line" and column == "BilledCost":
|
|
74
|
+
continue # reconciliation-only, never applied
|
|
75
|
+
if column not in dataset_columns(dataset):
|
|
76
|
+
continue
|
|
77
|
+
col_cov = cov[column]
|
|
78
|
+
complete = col_cov.complete
|
|
79
|
+
nullable = _allows_nulls(dataset, column)
|
|
80
|
+
if complete or (nullable and col_cov.covered > 0):
|
|
81
|
+
note = None if complete else "nulls where the client supplied no value"
|
|
82
|
+
rules[column] = ColumnRule(Lineage.ENRICHED, _source_label(table, column), note)
|
|
83
|
+
return rules
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
# Public alias: the streaming pipeline pre-computes the rule flips (rows-independent)
|
|
87
|
+
# to decide strict back-link gating before the main pass.
|
|
88
|
+
flip_enriched_rules = _flip_rules
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _apply_column(
|
|
92
|
+
row: dict[str, str],
|
|
93
|
+
column: str,
|
|
94
|
+
supplied: str,
|
|
95
|
+
*,
|
|
96
|
+
synthetic: bool,
|
|
97
|
+
counters: LineageCounters,
|
|
98
|
+
base_is_factual: bool = False,
|
|
99
|
+
) -> None:
|
|
100
|
+
"""Write one fact column: supplied value, else keep the base value / synthetic default.
|
|
101
|
+
|
|
102
|
+
``base_is_factual`` marks an *override* column that already carries a factual base value
|
|
103
|
+
(e.g. Contract Commitment ``ServiceProviderName`` from the provider context). For an
|
|
104
|
+
uncovered row such a column must keep its base value — never be blanked — so a partial
|
|
105
|
+
override does not turn otherwise-valid rows into mandatory-null failures.
|
|
106
|
+
"""
|
|
107
|
+
if supplied:
|
|
108
|
+
row[column] = supplied
|
|
109
|
+
counters.record(column, Lineage.ENRICHED)
|
|
110
|
+
elif base_is_factual and (row.get(column) or ""):
|
|
111
|
+
# Keep the factual base value for this uncovered row.
|
|
112
|
+
counters.record(column, Lineage.ENRICHED)
|
|
113
|
+
elif synthetic:
|
|
114
|
+
# Keep the builder's documented default (assumed) — or null if it emitted none.
|
|
115
|
+
counters.record(
|
|
116
|
+
column, Lineage.ASSUMED if (row.get(column) or "") else Lineage.UNAVAILABLE
|
|
117
|
+
)
|
|
118
|
+
else:
|
|
119
|
+
row[column] = ""
|
|
120
|
+
counters.record(column, Lineage.UNAVAILABLE)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _strict_suppress_uncovered_assumed(
|
|
124
|
+
applied: AppliedDataset,
|
|
125
|
+
dataset: str,
|
|
126
|
+
tables: list[SupplementTable],
|
|
127
|
+
*,
|
|
128
|
+
synthetic: bool,
|
|
129
|
+
) -> None:
|
|
130
|
+
"""In strict mode, blank uncovered nullable ASSUMED columns (never emit a default).
|
|
131
|
+
|
|
132
|
+
Uncovered non-nullable ASSUMED columns keep blocking (the dataset stays
|
|
133
|
+
``NOT_PRODUCED``); nullable ones must not leak their synthetic default into a
|
|
134
|
+
strictly-produced dataset, so they are emptied with ``UNAVAILABLE`` lineage.
|
|
135
|
+
"""
|
|
136
|
+
if synthetic:
|
|
137
|
+
return
|
|
138
|
+
covered = {c for t in tables for c in t.fact_columns}
|
|
139
|
+
for column, rule in list(applied.provenance.items()):
|
|
140
|
+
if rule.lineage is not Lineage.ASSUMED or column in covered:
|
|
141
|
+
continue
|
|
142
|
+
if not _allows_nulls(dataset, column):
|
|
143
|
+
continue
|
|
144
|
+
for row in applied.rows or []:
|
|
145
|
+
row[column] = ""
|
|
146
|
+
applied.provenance[column] = ColumnRule(
|
|
147
|
+
Lineage.UNAVAILABLE, note="synthetic default suppressed in strict mode"
|
|
148
|
+
)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def apply_billing_periods(
|
|
152
|
+
rows: list[dict[str, str]],
|
|
153
|
+
bundle: SupplementBundle,
|
|
154
|
+
source: SourceKeySets,
|
|
155
|
+
base_provenance: dict[str, ColumnRule],
|
|
156
|
+
*,
|
|
157
|
+
synthetic: bool,
|
|
158
|
+
) -> AppliedDataset:
|
|
159
|
+
table = bundle.get("billing_period")
|
|
160
|
+
if table is None:
|
|
161
|
+
return AppliedDataset(rows=rows, provenance=dict(base_provenance))
|
|
162
|
+
out = AppliedDataset(
|
|
163
|
+
rows=[dict(r) for r in rows],
|
|
164
|
+
provenance=_flip_rules(base_provenance, "Billing Period", [table], source),
|
|
165
|
+
)
|
|
166
|
+
for row in out.rows or []:
|
|
167
|
+
key = (
|
|
168
|
+
row.get("InvoiceIssuerName", ""),
|
|
169
|
+
row.get("BillingPeriodStart", ""),
|
|
170
|
+
row.get("BillingPeriodEnd", ""),
|
|
171
|
+
)
|
|
172
|
+
for column in table.fact_columns:
|
|
173
|
+
_apply_column(
|
|
174
|
+
row, column, table.value(key, column),
|
|
175
|
+
synthetic=synthetic, counters=out.counters,
|
|
176
|
+
)
|
|
177
|
+
_strict_suppress_uncovered_assumed(out, "Billing Period", [table], synthetic=synthetic)
|
|
178
|
+
return out
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def apply_invoice_details(
|
|
182
|
+
rows: list[dict[str, str]],
|
|
183
|
+
id_mapping: dict[GrainKey, str],
|
|
184
|
+
bundle: SupplementBundle,
|
|
185
|
+
source: SourceKeySets,
|
|
186
|
+
base_provenance: dict[str, ColumnRule],
|
|
187
|
+
*,
|
|
188
|
+
synthetic: bool,
|
|
189
|
+
) -> tuple[AppliedDataset, dict[GrainKey, str]]:
|
|
190
|
+
"""Apply invoice-header and invoice-line supplements; returns the updated id mapping."""
|
|
191
|
+
invoice = bundle.get("invoice")
|
|
192
|
+
line = bundle.get("invoice_line")
|
|
193
|
+
if invoice is None and line is None:
|
|
194
|
+
return AppliedDataset(rows=rows, provenance=dict(base_provenance)), id_mapping
|
|
195
|
+
tables = [t for t in (invoice, line) if t is not None]
|
|
196
|
+
out = AppliedDataset(
|
|
197
|
+
rows=[],
|
|
198
|
+
provenance=_flip_rules(base_provenance, "Invoice Detail", tables, source),
|
|
199
|
+
)
|
|
200
|
+
|
|
201
|
+
# Conditional columns activate only when the supplement genuinely enables them.
|
|
202
|
+
extra_emitted: list[str] = []
|
|
203
|
+
for column in _ACTIVATABLE_INVOICE_COLUMNS:
|
|
204
|
+
for table in tables:
|
|
205
|
+
if column not in table.fact_columns:
|
|
206
|
+
continue
|
|
207
|
+
col_cov = coverage(table, source.keys_for(table.kind.name))[column]
|
|
208
|
+
if col_cov.complete or (_allows_nulls("Invoice Detail", column) and col_cov.covered):
|
|
209
|
+
extra_emitted.append(column)
|
|
210
|
+
out.provenance[column] = ColumnRule(
|
|
211
|
+
Lineage.ENRICHED,
|
|
212
|
+
_source_label(table, column),
|
|
213
|
+
None if col_cov.complete else "nulls where the client supplied no value",
|
|
214
|
+
)
|
|
215
|
+
|
|
216
|
+
emitted = [c for c in dataset_columns("Invoice Detail") if rows and c in rows[0]]
|
|
217
|
+
all_emitted = [
|
|
218
|
+
c for c in dataset_columns("Invoice Detail") if c in set(emitted) | set(extra_emitted)
|
|
219
|
+
]
|
|
220
|
+
|
|
221
|
+
new_mapping = dict(id_mapping)
|
|
222
|
+
for row in rows:
|
|
223
|
+
merged = {c: row.get(c, "") for c in all_emitted}
|
|
224
|
+
grain = invoice_detail_grain_key(row)
|
|
225
|
+
if invoice is not None:
|
|
226
|
+
header_key = (row.get("InvoiceIssuerName", ""), row.get("InvoiceId", ""))
|
|
227
|
+
for column in invoice.fact_columns:
|
|
228
|
+
if column not in all_emitted:
|
|
229
|
+
continue
|
|
230
|
+
_apply_column(
|
|
231
|
+
merged, column, invoice.value(header_key, column),
|
|
232
|
+
synthetic=synthetic, counters=out.counters,
|
|
233
|
+
)
|
|
234
|
+
if line is not None:
|
|
235
|
+
for column in line.fact_columns:
|
|
236
|
+
if column == "BilledCost" or column not in all_emitted:
|
|
237
|
+
continue
|
|
238
|
+
_apply_column(
|
|
239
|
+
merged, column, line.value(grain, column),
|
|
240
|
+
synthetic=synthetic, counters=out.counters,
|
|
241
|
+
)
|
|
242
|
+
real_id = line.value(grain, "InvoiceDetailId")
|
|
243
|
+
if real_id:
|
|
244
|
+
new_mapping[grain] = real_id
|
|
245
|
+
assert out.rows is not None
|
|
246
|
+
out.rows.append(merged)
|
|
247
|
+
_strict_suppress_uncovered_assumed(out, "Invoice Detail", tables, synthetic=synthetic)
|
|
248
|
+
return out, new_mapping
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def apply_contract_commitments(
|
|
252
|
+
rows: list[dict[str, str]],
|
|
253
|
+
bundle: SupplementBundle,
|
|
254
|
+
source: SourceKeySets,
|
|
255
|
+
base_provenance: dict[str, ColumnRule],
|
|
256
|
+
*,
|
|
257
|
+
synthetic: bool,
|
|
258
|
+
) -> AppliedDataset:
|
|
259
|
+
table = bundle.get("contract_commitment")
|
|
260
|
+
if table is None:
|
|
261
|
+
return AppliedDataset(rows=rows, provenance=dict(base_provenance))
|
|
262
|
+
out = AppliedDataset(
|
|
263
|
+
rows=[dict(r) for r in rows],
|
|
264
|
+
provenance=_flip_rules(base_provenance, "Contract Commitment", [table], source),
|
|
265
|
+
)
|
|
266
|
+
upfront = "ContractCommitmentPaymentUpfrontPercentage"
|
|
267
|
+
model_col = "ContractCommitmentPaymentModel"
|
|
268
|
+
upfront_supplied = upfront in table.fact_columns
|
|
269
|
+
model_cov = (
|
|
270
|
+
coverage(table, source.keys_for("contract_commitment")).get(model_col)
|
|
271
|
+
if model_col in table.fact_columns
|
|
272
|
+
else None
|
|
273
|
+
)
|
|
274
|
+
# Override columns whose base row already holds a factual (provider-context) value:
|
|
275
|
+
# a partial override must not blank the uncovered rows.
|
|
276
|
+
factual_base = {c for c, r in base_provenance.items() if r.is_factual}
|
|
277
|
+
for row in out.rows or []:
|
|
278
|
+
key = (row.get("ContractCommitmentId", ""),)
|
|
279
|
+
for column in table.fact_columns:
|
|
280
|
+
if column == upfront:
|
|
281
|
+
continue # handled below (derivable from the payment model)
|
|
282
|
+
_apply_column(
|
|
283
|
+
row, column, table.value(key, column),
|
|
284
|
+
synthetic=synthetic, counters=out.counters,
|
|
285
|
+
base_is_factual=column in factual_base,
|
|
286
|
+
)
|
|
287
|
+
# Upfront percentage: supplied wins; else exactly derivable from the payment
|
|
288
|
+
# model ('No Upfront' -> 0, 'All Upfront' -> 1); 'Partial Upfront' without a
|
|
289
|
+
# supplied percentage is not derivable and stays empty in BOTH modes (never a
|
|
290
|
+
# guessed '0' paired with 'Partial Upfront'); the mandatory-column lint flags it.
|
|
291
|
+
supplied_pct = table.value(key, upfront) if upfront_supplied else ""
|
|
292
|
+
payment_model = row.get(model_col, "")
|
|
293
|
+
if supplied_pct:
|
|
294
|
+
row[upfront] = supplied_pct
|
|
295
|
+
out.counters.record(upfront, Lineage.ENRICHED)
|
|
296
|
+
elif payment_model == "No Upfront":
|
|
297
|
+
row[upfront] = "0"
|
|
298
|
+
out.counters.record(upfront, Lineage.DERIVED)
|
|
299
|
+
elif payment_model == "All Upfront":
|
|
300
|
+
row[upfront] = "1"
|
|
301
|
+
out.counters.record(upfront, Lineage.DERIVED)
|
|
302
|
+
else:
|
|
303
|
+
row[upfront] = ""
|
|
304
|
+
out.counters.record(upfront, Lineage.UNAVAILABLE)
|
|
305
|
+
if not any(
|
|
306
|
+
(r.get(upfront) or "") == "" for r in (out.rows or [])
|
|
307
|
+
) and (upfront_supplied or (model_cov is not None and model_cov.complete)):
|
|
308
|
+
source_note = (
|
|
309
|
+
_source_label(table, upfront)
|
|
310
|
+
if upfront_supplied
|
|
311
|
+
else f"ContractCommitmentPaymentModel ({_source_label(table, model_col)})"
|
|
312
|
+
)
|
|
313
|
+
lineage = Lineage.ENRICHED if upfront_supplied else Lineage.DERIVED
|
|
314
|
+
out.provenance[upfront] = ColumnRule(lineage, source_note)
|
|
315
|
+
_strict_suppress_uncovered_assumed(
|
|
316
|
+
out, "Contract Commitment", [table], synthetic=synthetic
|
|
317
|
+
)
|
|
318
|
+
return out
|
|
@@ -0,0 +1,219 @@
|
|
|
1
|
+
"""Gap analysis: exactly which facts a client must supply, per FOCUS 1.4 dataset.
|
|
2
|
+
|
|
3
|
+
The gap set is *computed*, never hardcoded: a column gap is a column that is Mandatory,
|
|
4
|
+
non-nullable and non-factual under the very provenance rules the converter would use for
|
|
5
|
+
this source — i.e. exactly ``strict_blockers()``. Each gap is annotated from the embedded
|
|
6
|
+
model (allowed values, format, condition text) and mapped to the supplement kind(s) able
|
|
7
|
+
to satisfy it. Nullable non-factual columns of the same kinds are reported as
|
|
8
|
+
*recommended* (they never block strict production, but supplying them makes the output
|
|
9
|
+
more complete). The JSON output doubles as a fill-in template: it carries a ready-to-use
|
|
10
|
+
CSV header line per supplement kind.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from collections.abc import Iterable
|
|
16
|
+
from dataclasses import dataclass, field
|
|
17
|
+
|
|
18
|
+
from focus_data_toolkit.convert.billing_period import PROVENANCE as BILLING_PERIOD_PROVENANCE
|
|
19
|
+
from focus_data_toolkit.convert.contract_commitment import (
|
|
20
|
+
PROVENANCE as CONTRACT_COMMITMENT_PROVENANCE,
|
|
21
|
+
)
|
|
22
|
+
from focus_data_toolkit.convert.cost_and_usage import cost_and_usage_provenance
|
|
23
|
+
from focus_data_toolkit.convert.invoice_detail import PROVENANCE as INVOICE_DETAIL_PROVENANCE
|
|
24
|
+
from focus_data_toolkit.model import FOCUS_1_4_DATASETS
|
|
25
|
+
from focus_data_toolkit.model.validator import load_model
|
|
26
|
+
from focus_data_toolkit.provenance import ColumnRule, Lineage, strict_blockers
|
|
27
|
+
from focus_data_toolkit.supplement.kinds import SUPPLEMENT_KINDS, kinds_for_column
|
|
28
|
+
|
|
29
|
+
GAP_REPORT_FORMAT = "1"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass(frozen=True)
|
|
33
|
+
class ColumnGap:
|
|
34
|
+
"""One column the source cannot factually populate."""
|
|
35
|
+
|
|
36
|
+
dataset: str
|
|
37
|
+
column: str
|
|
38
|
+
feature_level: str
|
|
39
|
+
allows_nulls: bool
|
|
40
|
+
current_lineage: str
|
|
41
|
+
current_note: str | None
|
|
42
|
+
blocking: bool
|
|
43
|
+
supplement_kinds: tuple[str, ...]
|
|
44
|
+
join_keys: tuple[str, ...]
|
|
45
|
+
value_format: str | None
|
|
46
|
+
allowed_values: tuple[str, ...]
|
|
47
|
+
condition: str | None
|
|
48
|
+
|
|
49
|
+
def as_dict(self) -> dict:
|
|
50
|
+
out: dict = {
|
|
51
|
+
"dataset": self.dataset,
|
|
52
|
+
"column": self.column,
|
|
53
|
+
"feature_level": self.feature_level,
|
|
54
|
+
"allows_nulls": self.allows_nulls,
|
|
55
|
+
"current_lineage": self.current_lineage,
|
|
56
|
+
"blocking": self.blocking,
|
|
57
|
+
"supplement_kinds": list(self.supplement_kinds),
|
|
58
|
+
"join_keys": list(self.join_keys),
|
|
59
|
+
}
|
|
60
|
+
if self.current_note:
|
|
61
|
+
out["current_note"] = self.current_note
|
|
62
|
+
if self.value_format:
|
|
63
|
+
out["value_format"] = self.value_format
|
|
64
|
+
if self.allowed_values:
|
|
65
|
+
out["allowed_values"] = list(self.allowed_values)
|
|
66
|
+
if self.condition:
|
|
67
|
+
out["condition"] = self.condition
|
|
68
|
+
return out
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
@dataclass(frozen=True)
|
|
72
|
+
class GapReport:
|
|
73
|
+
"""Everything a client needs to complete this source into factual 1.4 datasets."""
|
|
74
|
+
|
|
75
|
+
source_version: str
|
|
76
|
+
gaps: dict[str, tuple[ColumnGap, ...]] = field(default_factory=dict)
|
|
77
|
+
dataset_level_gaps: dict[str, str] = field(default_factory=dict)
|
|
78
|
+
|
|
79
|
+
def blocking(self, dataset: str) -> tuple[ColumnGap, ...]:
|
|
80
|
+
return tuple(g for g in self.gaps.get(dataset, ()) if g.blocking)
|
|
81
|
+
|
|
82
|
+
def as_dict(self) -> dict:
|
|
83
|
+
"""Deterministic JSON payload; doubles as a fill-in template."""
|
|
84
|
+
kinds_used = sorted(
|
|
85
|
+
{k for gaps in self.gaps.values() for g in gaps for k in g.supplement_kinds}
|
|
86
|
+
)
|
|
87
|
+
return {
|
|
88
|
+
"gap_report_format": GAP_REPORT_FORMAT,
|
|
89
|
+
"source_version": self.source_version,
|
|
90
|
+
"datasets": {
|
|
91
|
+
name: {
|
|
92
|
+
"column_gaps": [g.as_dict() for g in self.gaps.get(name, ())],
|
|
93
|
+
"strictly_producible_as_is": not self.blocking(name)
|
|
94
|
+
and name not in self.dataset_level_gaps,
|
|
95
|
+
**(
|
|
96
|
+
{"dataset_gap": self.dataset_level_gaps[name]}
|
|
97
|
+
if name in self.dataset_level_gaps
|
|
98
|
+
else {}
|
|
99
|
+
),
|
|
100
|
+
}
|
|
101
|
+
for name in FOCUS_1_4_DATASETS
|
|
102
|
+
},
|
|
103
|
+
"supplement_templates": {
|
|
104
|
+
name: {
|
|
105
|
+
"target_dataset": SUPPLEMENT_KINDS[name].target_dataset,
|
|
106
|
+
"join_keys": list(SUPPLEMENT_KINDS[name].join_keys),
|
|
107
|
+
"csv_header": ",".join(SUPPLEMENT_KINDS[name].header_template),
|
|
108
|
+
}
|
|
109
|
+
for name in kinds_used
|
|
110
|
+
},
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
def render_text(self) -> str:
|
|
114
|
+
lines = [f"FOCUS {self.source_version} source -> 1.4 gap report", ""]
|
|
115
|
+
for name in FOCUS_1_4_DATASETS:
|
|
116
|
+
gaps = self.gaps.get(name, ())
|
|
117
|
+
if name in self.dataset_level_gaps:
|
|
118
|
+
lines.append(f"[{name}] NOT PRODUCIBLE: {self.dataset_level_gaps[name]}")
|
|
119
|
+
lines.append("")
|
|
120
|
+
continue
|
|
121
|
+
blocking = [g for g in gaps if g.blocking]
|
|
122
|
+
recommended = [g for g in gaps if not g.blocking]
|
|
123
|
+
if not blocking:
|
|
124
|
+
lines.append(f"[{name}] strictly producible from this source as-is")
|
|
125
|
+
else:
|
|
126
|
+
lines.append(f"[{name}] blocked by {len(blocking)} column(s):")
|
|
127
|
+
for g in blocking:
|
|
128
|
+
extra = f" (allowed: {', '.join(g.allowed_values)})" if g.allowed_values else ""
|
|
129
|
+
kinds = ", ".join(g.supplement_kinds) or "-"
|
|
130
|
+
lines.append(f" - {g.column}{extra} <- supplement kind: {kinds}")
|
|
131
|
+
for g in recommended:
|
|
132
|
+
lines.append(f" ~ {g.column} (recommended, nullable)")
|
|
133
|
+
lines.append("")
|
|
134
|
+
kinds_used = sorted(
|
|
135
|
+
{k for gaps in self.gaps.values() for g in gaps for k in g.supplement_kinds}
|
|
136
|
+
)
|
|
137
|
+
if kinds_used:
|
|
138
|
+
lines.append("Supplement templates (CSV headers, ready to fill):")
|
|
139
|
+
for name in kinds_used:
|
|
140
|
+
lines.append(f" {name}: {','.join(SUPPLEMENT_KINDS[name].header_template)}")
|
|
141
|
+
return "\n".join(lines) + "\n"
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def _gap(dataset: str, column: str, spec: dict, rule: ColumnRule | None, blocking: bool) -> ColumnGap:
|
|
145
|
+
kinds = kinds_for_column(dataset, column)
|
|
146
|
+
return ColumnGap(
|
|
147
|
+
dataset=dataset,
|
|
148
|
+
column=column,
|
|
149
|
+
feature_level=spec.get("feature_level", ""),
|
|
150
|
+
allows_nulls=bool(spec.get("allows_nulls", True)),
|
|
151
|
+
current_lineage=rule.lineage.value if rule else "UNAVAILABLE",
|
|
152
|
+
current_note=(rule.note if rule else None),
|
|
153
|
+
blocking=blocking,
|
|
154
|
+
supplement_kinds=tuple(k.name for k in kinds),
|
|
155
|
+
join_keys=kinds[0].join_keys if kinds else (),
|
|
156
|
+
value_format=spec.get("value_format") or spec.get("data_type"),
|
|
157
|
+
allowed_values=tuple(spec.get("allowed_values") or ()),
|
|
158
|
+
condition=spec.get("condition"),
|
|
159
|
+
)
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def compute_gaps(
|
|
163
|
+
source_columns: Iterable[str],
|
|
164
|
+
source_version: str,
|
|
165
|
+
*,
|
|
166
|
+
cc_columns: Iterable[str] | None = None,
|
|
167
|
+
) -> GapReport:
|
|
168
|
+
"""Compute the gap report for a Cost and Usage header (and optional 1.3 CC header).
|
|
169
|
+
|
|
170
|
+
Uses the same provenance rules as the converter, so a reported blocking gap is by
|
|
171
|
+
construction exactly what would block strict production of that dataset.
|
|
172
|
+
"""
|
|
173
|
+
model = load_model()
|
|
174
|
+
# When a Contract Commitment source header is given, reflect it: a column the base
|
|
175
|
+
# provenance treats as OBSERVED-from-1.3 but that is absent from the actual header is a
|
|
176
|
+
# source-completeness gap (no supplement can fabricate it), not an observed value.
|
|
177
|
+
cc_prov = dict(CONTRACT_COMMITMENT_PROVENANCE)
|
|
178
|
+
if cc_columns is not None:
|
|
179
|
+
present_cc = set(cc_columns)
|
|
180
|
+
for col, base_rule in CONTRACT_COMMITMENT_PROVENANCE.items():
|
|
181
|
+
if base_rule.lineage is Lineage.OBSERVED and col not in present_cc:
|
|
182
|
+
cc_prov[col] = ColumnRule(
|
|
183
|
+
Lineage.UNAVAILABLE, note="absent from the Contract Commitment source"
|
|
184
|
+
)
|
|
185
|
+
provenance: dict[str, dict[str, ColumnRule]] = {
|
|
186
|
+
"Cost and Usage": cost_and_usage_provenance(
|
|
187
|
+
source_columns, source_version, invoice_detail_linked=False
|
|
188
|
+
),
|
|
189
|
+
"Billing Period": BILLING_PERIOD_PROVENANCE,
|
|
190
|
+
"Invoice Detail": INVOICE_DETAIL_PROVENANCE,
|
|
191
|
+
"Contract Commitment": cc_prov,
|
|
192
|
+
}
|
|
193
|
+
gaps: dict[str, tuple[ColumnGap, ...]] = {}
|
|
194
|
+
dataset_level: dict[str, str] = {}
|
|
195
|
+
for name in FOCUS_1_4_DATASETS:
|
|
196
|
+
columns: dict = model["datasets"][name]["columns"]
|
|
197
|
+
prov = provenance[name]
|
|
198
|
+
blockers = set(strict_blockers(prov, columns))
|
|
199
|
+
out: list[ColumnGap] = []
|
|
200
|
+
for col in sorted(columns):
|
|
201
|
+
spec = columns[col]
|
|
202
|
+
rule = prov.get(col)
|
|
203
|
+
if col in blockers:
|
|
204
|
+
out.append(_gap(name, col, spec, rule, blocking=True))
|
|
205
|
+
elif (
|
|
206
|
+
rule is not None
|
|
207
|
+
and not rule.is_factual
|
|
208
|
+
and spec.get("allows_nulls", True)
|
|
209
|
+
and kinds_for_column(name, col)
|
|
210
|
+
):
|
|
211
|
+
# Nullable, non-factual, and a supplement kind can supply it: recommended.
|
|
212
|
+
out.append(_gap(name, col, spec, rule, blocking=False))
|
|
213
|
+
gaps[name] = tuple(out)
|
|
214
|
+
if cc_columns is None:
|
|
215
|
+
dataset_level["Contract Commitment"] = (
|
|
216
|
+
"no FOCUS 1.3 Contract Commitment source provided; supply the 1.3 dataset "
|
|
217
|
+
"(13 columns) plus a 'contract_commitment' supplement for the 1.4-new terms"
|
|
218
|
+
)
|
|
219
|
+
return GapReport(source_version=source_version, gaps=gaps, dataset_level_gaps=dataset_level)
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
"""Registry of supplement kinds — what a client may supply, and how it joins.
|
|
2
|
+
|
|
3
|
+
Each kind targets one FOCUS 1.4 dataset, joins to the conversion on that dataset's
|
|
4
|
+
natural key, and may supply a fixed set of fact columns (FOCUS column names). Anything
|
|
5
|
+
else must be ``x_``-prefixed. Join-key values are compared after ``.strip()`` (the same
|
|
6
|
+
normalization the converters use); there is no fuzzy matching.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
|
|
13
|
+
from focus_data_toolkit.convert.invoice_detail import GRAIN_FIELDS
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@dataclass(frozen=True)
|
|
17
|
+
class SupplementKind:
|
|
18
|
+
"""One supplement file format the toolkit knows how to join and apply."""
|
|
19
|
+
|
|
20
|
+
name: str
|
|
21
|
+
target_dataset: str
|
|
22
|
+
join_keys: tuple[str, ...]
|
|
23
|
+
columns: frozenset[str]
|
|
24
|
+
|
|
25
|
+
@property
|
|
26
|
+
def header_template(self) -> tuple[str, ...]:
|
|
27
|
+
"""A ready-to-fill CSV header: join keys first, then the fact columns."""
|
|
28
|
+
return self.join_keys + tuple(sorted(self.columns))
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
# The billing-period cycle facts a Cost and Usage source can never carry.
|
|
32
|
+
BILLING_PERIOD_KIND = SupplementKind(
|
|
33
|
+
name="billing_period",
|
|
34
|
+
target_dataset="Billing Period",
|
|
35
|
+
join_keys=("InvoiceIssuerName", "BillingPeriodStart", "BillingPeriodEnd"),
|
|
36
|
+
columns=frozenset(
|
|
37
|
+
{"BillingPeriodCreated", "BillingPeriodLastUpdated", "BillingPeriodStatus"}
|
|
38
|
+
),
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
# Invoice-header facts: one row per issued invoice.
|
|
42
|
+
INVOICE_KIND = SupplementKind(
|
|
43
|
+
name="invoice",
|
|
44
|
+
target_dataset="Invoice Detail",
|
|
45
|
+
join_keys=("InvoiceIssuerName", "InvoiceId"),
|
|
46
|
+
columns=frozenset(
|
|
47
|
+
{
|
|
48
|
+
"InvoiceIssueDate",
|
|
49
|
+
"InvoiceIssueStatus",
|
|
50
|
+
"PaymentTerms",
|
|
51
|
+
"PaymentDueDate",
|
|
52
|
+
"ReferenceInvoiceId",
|
|
53
|
+
"PurchaseOrderNumber",
|
|
54
|
+
"PaymentCurrency",
|
|
55
|
+
}
|
|
56
|
+
),
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
# Invoice-line facts: one row per invoice line, joined on the full business grain
|
|
60
|
+
# (exactly the grain the derived Invoice Detail dataset aggregates on). ``BilledCost``
|
|
61
|
+
# is accepted only as a reconciliation check against the derived grain sum — it never
|
|
62
|
+
# replaces the derived value.
|
|
63
|
+
INVOICE_LINE_KIND = SupplementKind(
|
|
64
|
+
name="invoice_line",
|
|
65
|
+
target_dataset="Invoice Detail",
|
|
66
|
+
join_keys=GRAIN_FIELDS,
|
|
67
|
+
columns=frozenset(
|
|
68
|
+
{
|
|
69
|
+
"InvoiceDetailId",
|
|
70
|
+
"InvoiceDetailCreated",
|
|
71
|
+
"InvoiceDetailLastUpdated",
|
|
72
|
+
"InvoiceDetailDescription",
|
|
73
|
+
"InvoiceDetailGrain",
|
|
74
|
+
"PaymentCurrencyBilledCost",
|
|
75
|
+
"PaymentCurrencyInvoiceDetailId",
|
|
76
|
+
"BilledCost",
|
|
77
|
+
}
|
|
78
|
+
),
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
# The 1.4-new contract-commitment commercial terms a 1.3 source does not carry.
|
|
82
|
+
CONTRACT_COMMITMENT_KIND = SupplementKind(
|
|
83
|
+
name="contract_commitment",
|
|
84
|
+
target_dataset="Contract Commitment",
|
|
85
|
+
join_keys=("ContractCommitmentId",),
|
|
86
|
+
columns=frozenset(
|
|
87
|
+
{
|
|
88
|
+
"ContractCommitmentApplicability",
|
|
89
|
+
"ContractCommitmentBenefitCategory",
|
|
90
|
+
"ContractCommitmentCreated",
|
|
91
|
+
"ContractCommitmentDiscountPercentage",
|
|
92
|
+
"ContractCommitmentFulfillmentInterval",
|
|
93
|
+
"ContractCommitmentLastUpdated",
|
|
94
|
+
"ContractCommitmentLifecycleStatus",
|
|
95
|
+
"ContractCommitmentModel",
|
|
96
|
+
"ContractCommitmentOfferCategory",
|
|
97
|
+
"ContractCommitmentPaymentInterval",
|
|
98
|
+
"ContractCommitmentPaymentModel",
|
|
99
|
+
"ContractCommitmentPaymentUpfrontPercentage",
|
|
100
|
+
"InvoiceIssuerName",
|
|
101
|
+
"ServiceProviderName",
|
|
102
|
+
}
|
|
103
|
+
),
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
SUPPLEMENT_KINDS: dict[str, SupplementKind] = {
|
|
107
|
+
k.name: k
|
|
108
|
+
for k in (BILLING_PERIOD_KIND, INVOICE_KIND, INVOICE_LINE_KIND, CONTRACT_COMMITMENT_KIND)
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def kinds_for_column(dataset: str, column: str) -> tuple[SupplementKind, ...]:
|
|
113
|
+
"""The kinds able to supply ``column`` of ``dataset`` (deterministic order)."""
|
|
114
|
+
return tuple(
|
|
115
|
+
kind
|
|
116
|
+
for kind in SUPPLEMENT_KINDS.values()
|
|
117
|
+
if kind.target_dataset == dataset and column in kind.columns
|
|
118
|
+
)
|