focus-data-toolkit 0.11.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. focus_data_toolkit/__init__.py +69 -0
  2. focus_data_toolkit/__main__.py +6 -0
  3. focus_data_toolkit/_version.py +8 -0
  4. focus_data_toolkit/cli.py +968 -0
  5. focus_data_toolkit/context/__init__.py +88 -0
  6. focus_data_toolkit/context/billing.py +54 -0
  7. focus_data_toolkit/context/provider.py +90 -0
  8. focus_data_toolkit/convert/__init__.py +708 -0
  9. focus_data_toolkit/convert/billing_period.py +65 -0
  10. focus_data_toolkit/convert/contract_applied.py +235 -0
  11. focus_data_toolkit/convert/contract_commitment.py +182 -0
  12. focus_data_toolkit/convert/cost_and_usage.py +179 -0
  13. focus_data_toolkit/convert/detect.py +39 -0
  14. focus_data_toolkit/convert/invoice_detail.py +199 -0
  15. focus_data_toolkit/convert/streaming.py +1030 -0
  16. focus_data_toolkit/errors.py +145 -0
  17. focus_data_toolkit/focus_json.py +68 -0
  18. focus_data_toolkit/generators/__init__.py +61 -0
  19. focus_data_toolkit/generators/_shim.py +43 -0
  20. focus_data_toolkit/generators/engine/__init__.py +14 -0
  21. focus_data_toolkit/generators/engine/context.py +12 -0
  22. focus_data_toolkit/generators/engine/determinism.py +117 -0
  23. focus_data_toolkit/generators/engine/json_focus.py +63 -0
  24. focus_data_toolkit/generators/engine/ladder.py +71 -0
  25. focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
  26. focus_data_toolkit/generators/engine/serialize.py +151 -0
  27. focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
  28. focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
  29. focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
  30. focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
  31. focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
  32. focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
  33. focus_data_toolkit/generators/providers/__init__.py +29 -0
  34. focus_data_toolkit/generators/providers/aws.py +186 -0
  35. focus_data_toolkit/generators/providers/azure.py +191 -0
  36. focus_data_toolkit/generators/providers/gcp.py +194 -0
  37. focus_data_toolkit/generators/providers/profile.py +123 -0
  38. focus_data_toolkit/generators/scenarios.py +178 -0
  39. focus_data_toolkit/generators/versions/__init__.py +17 -0
  40. focus_data_toolkit/generators/versions/adapter.py +41 -0
  41. focus_data_toolkit/generators/versions/v1_2.py +111 -0
  42. focus_data_toolkit/generators/versions/v1_3.py +154 -0
  43. focus_data_toolkit/io/__init__.py +1 -0
  44. focus_data_toolkit/io/atomic_writer.py +462 -0
  45. focus_data_toolkit/io/csv_io.py +128 -0
  46. focus_data_toolkit/io/parquet_io.py +528 -0
  47. focus_data_toolkit/io/records.py +92 -0
  48. focus_data_toolkit/io/row_source.py +117 -0
  49. focus_data_toolkit/lifecycle.py +342 -0
  50. focus_data_toolkit/manifest.py +114 -0
  51. focus_data_toolkit/model/__init__.py +43 -0
  52. focus_data_toolkit/model/capabilities.py +66 -0
  53. focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
  54. focus_data_toolkit/model/focus_1_4_model.json +1913 -0
  55. focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
  56. focus_data_toolkit/model/focus_json_keys.py +112 -0
  57. focus_data_toolkit/model/iso_4217_currencies.json +23 -0
  58. focus_data_toolkit/model/json_schema_check.py +205 -0
  59. focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
  60. focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
  61. focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
  62. focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
  63. focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
  64. focus_data_toolkit/model/model_provenance.json +58 -0
  65. focus_data_toolkit/model/validator.py +498 -0
  66. focus_data_toolkit/modes.py +18 -0
  67. focus_data_toolkit/official_validator.py +61 -0
  68. focus_data_toolkit/progress.py +89 -0
  69. focus_data_toolkit/provenance.py +106 -0
  70. focus_data_toolkit/py.typed +1 -0
  71. focus_data_toolkit/runtime.py +243 -0
  72. focus_data_toolkit/schema/__init__.py +17 -0
  73. focus_data_toolkit/schema/detection.py +274 -0
  74. focus_data_toolkit/schema/registry.py +127 -0
  75. focus_data_toolkit/storage/__init__.py +1 -0
  76. focus_data_toolkit/storage/external_index.py +99 -0
  77. focus_data_toolkit/storage/spill.py +150 -0
  78. focus_data_toolkit/studio/__init__.py +19 -0
  79. focus_data_toolkit/studio/app.py +467 -0
  80. focus_data_toolkit/studio/config.py +42 -0
  81. focus_data_toolkit/studio/frontend/app.js +214 -0
  82. focus_data_toolkit/studio/frontend/index.html +101 -0
  83. focus_data_toolkit/studio/frontend/style.css +60 -0
  84. focus_data_toolkit/studio/jobs.py +142 -0
  85. focus_data_toolkit/studio/preview.py +32 -0
  86. focus_data_toolkit/studio/security.py +125 -0
  87. focus_data_toolkit/studio/server.py +71 -0
  88. focus_data_toolkit/supplement/__init__.py +50 -0
  89. focus_data_toolkit/supplement/adapters/__init__.py +21 -0
  90. focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
  91. focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
  92. focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
  93. focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
  94. focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
  95. focus_data_toolkit/supplement/adapters/registry.py +215 -0
  96. focus_data_toolkit/supplement/apply.py +318 -0
  97. focus_data_toolkit/supplement/gaps.py +219 -0
  98. focus_data_toolkit/supplement/kinds.py +118 -0
  99. focus_data_toolkit/supplement/loader.py +409 -0
  100. focus_data_toolkit/supplement/spec.py +74 -0
  101. focus_data_toolkit/supplement/validate.py +215 -0
  102. focus_data_toolkit/validate/__init__.py +15 -0
  103. focus_data_toolkit/validate/allocation.py +333 -0
  104. focus_data_toolkit/validate/bundle.py +254 -0
  105. focus_data_toolkit/validate/codes.py +93 -0
  106. focus_data_toolkit/validate/corrections.py +245 -0
  107. focus_data_toolkit/validate/reconciliation.py +98 -0
  108. focus_data_toolkit/validate/referential.py +289 -0
  109. focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
  110. focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
  111. focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
  112. focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
  113. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
  114. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
  115. focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
  116. focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,318 @@
1
+ """Apply validated supplements to the derived datasets (ENRICHED lineage).
2
+
3
+ Mode rules, per fact column present in a supplement table:
4
+
5
+ * **Strict** — a value is either supplied by the client or empty; synthetic defaults
6
+ are never emitted. A *non-nullable* column's rule flips to ``ENRICHED`` only at
7
+ 100 % key coverage (otherwise it keeps blocking and the dataset stays
8
+ ``NOT_PRODUCED``); a *nullable* column flips to ``ENRICHED`` as soon as the
9
+ supplement carries it, with per-value counters recording the supplied/null mix.
10
+ * **Synthetic** — a supplied value wins, the documented synthetic default fills the
11
+ rest. The headline rule flips to ``ENRICHED`` only at full coverage (otherwise it
12
+ stays ``ASSUMED`` — the weakest lineage present), with counters showing the mix.
13
+
14
+ The strict gate itself is untouched: ``ENRICHED`` is already factual, so a fully
15
+ covered dataset simply stops blocking.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ from dataclasses import dataclass, field
21
+
22
+ from focus_data_toolkit.convert.invoice_detail import (
23
+ GrainKey,
24
+ invoice_detail_grain_key,
25
+ )
26
+ from focus_data_toolkit.model import dataset_columns
27
+ from focus_data_toolkit.model.validator import load_model
28
+ from focus_data_toolkit.provenance import ColumnRule, Lineage, LineageCounters
29
+ from focus_data_toolkit.supplement.loader import SupplementBundle, SupplementTable
30
+ from focus_data_toolkit.supplement.validate import SourceKeySets, coverage
31
+
32
+ # Invoice Detail columns normally omitted (unfilled conditionals) that a supplement can
33
+ # activate; non-nullable ones require full coverage to be emitted at all.
34
+ _ACTIVATABLE_INVOICE_COLUMNS = (
35
+ "PaymentCurrency",
36
+ "PaymentCurrencyBilledCost",
37
+ "PaymentCurrencyInvoiceDetailId",
38
+ "PurchaseOrderNumber",
39
+ )
40
+
41
+
42
+ @dataclass
43
+ class AppliedDataset:
44
+ """Outcome of applying supplements to one dataset."""
45
+
46
+ rows: list[dict[str, str]] | None
47
+ provenance: dict[str, ColumnRule]
48
+ counters: LineageCounters = field(default_factory=LineageCounters)
49
+
50
+
51
+ def _source_label(table: SupplementTable, column: str) -> str:
52
+ # Per-column attribution: a merged table (adapter export + hand-authored file) tags each
53
+ # column to its originating file ("supplement:<adapter>@<version>:<file>" or the kind).
54
+ return table.source_for(column)
55
+
56
+
57
+ def _allows_nulls(dataset: str, column: str) -> bool:
58
+ spec = load_model()["datasets"][dataset]["columns"].get(column) or {}
59
+ return bool(spec.get("allows_nulls", True))
60
+
61
+
62
+ def _flip_rules(
63
+ base: dict[str, ColumnRule],
64
+ dataset: str,
65
+ tables: list[SupplementTable],
66
+ source: SourceKeySets,
67
+ ) -> dict[str, ColumnRule]:
68
+ """Return ``base`` with columns flipped to ENRICHED per the coverage rules."""
69
+ rules = dict(base)
70
+ for table in tables:
71
+ cov = coverage(table, source.keys_for(table.kind.name))
72
+ for column in table.fact_columns:
73
+ if table.kind.name == "invoice_line" and column == "BilledCost":
74
+ continue # reconciliation-only, never applied
75
+ if column not in dataset_columns(dataset):
76
+ continue
77
+ col_cov = cov[column]
78
+ complete = col_cov.complete
79
+ nullable = _allows_nulls(dataset, column)
80
+ if complete or (nullable and col_cov.covered > 0):
81
+ note = None if complete else "nulls where the client supplied no value"
82
+ rules[column] = ColumnRule(Lineage.ENRICHED, _source_label(table, column), note)
83
+ return rules
84
+
85
+
86
+ # Public alias: the streaming pipeline pre-computes the rule flips (rows-independent)
87
+ # to decide strict back-link gating before the main pass.
88
+ flip_enriched_rules = _flip_rules
89
+
90
+
91
+ def _apply_column(
92
+ row: dict[str, str],
93
+ column: str,
94
+ supplied: str,
95
+ *,
96
+ synthetic: bool,
97
+ counters: LineageCounters,
98
+ base_is_factual: bool = False,
99
+ ) -> None:
100
+ """Write one fact column: supplied value, else keep the base value / synthetic default.
101
+
102
+ ``base_is_factual`` marks an *override* column that already carries a factual base value
103
+ (e.g. Contract Commitment ``ServiceProviderName`` from the provider context). For an
104
+ uncovered row such a column must keep its base value — never be blanked — so a partial
105
+ override does not turn otherwise-valid rows into mandatory-null failures.
106
+ """
107
+ if supplied:
108
+ row[column] = supplied
109
+ counters.record(column, Lineage.ENRICHED)
110
+ elif base_is_factual and (row.get(column) or ""):
111
+ # Keep the factual base value for this uncovered row.
112
+ counters.record(column, Lineage.ENRICHED)
113
+ elif synthetic:
114
+ # Keep the builder's documented default (assumed) — or null if it emitted none.
115
+ counters.record(
116
+ column, Lineage.ASSUMED if (row.get(column) or "") else Lineage.UNAVAILABLE
117
+ )
118
+ else:
119
+ row[column] = ""
120
+ counters.record(column, Lineage.UNAVAILABLE)
121
+
122
+
123
+ def _strict_suppress_uncovered_assumed(
124
+ applied: AppliedDataset,
125
+ dataset: str,
126
+ tables: list[SupplementTable],
127
+ *,
128
+ synthetic: bool,
129
+ ) -> None:
130
+ """In strict mode, blank uncovered nullable ASSUMED columns (never emit a default).
131
+
132
+ Uncovered non-nullable ASSUMED columns keep blocking (the dataset stays
133
+ ``NOT_PRODUCED``); nullable ones must not leak their synthetic default into a
134
+ strictly-produced dataset, so they are emptied with ``UNAVAILABLE`` lineage.
135
+ """
136
+ if synthetic:
137
+ return
138
+ covered = {c for t in tables for c in t.fact_columns}
139
+ for column, rule in list(applied.provenance.items()):
140
+ if rule.lineage is not Lineage.ASSUMED or column in covered:
141
+ continue
142
+ if not _allows_nulls(dataset, column):
143
+ continue
144
+ for row in applied.rows or []:
145
+ row[column] = ""
146
+ applied.provenance[column] = ColumnRule(
147
+ Lineage.UNAVAILABLE, note="synthetic default suppressed in strict mode"
148
+ )
149
+
150
+
151
+ def apply_billing_periods(
152
+ rows: list[dict[str, str]],
153
+ bundle: SupplementBundle,
154
+ source: SourceKeySets,
155
+ base_provenance: dict[str, ColumnRule],
156
+ *,
157
+ synthetic: bool,
158
+ ) -> AppliedDataset:
159
+ table = bundle.get("billing_period")
160
+ if table is None:
161
+ return AppliedDataset(rows=rows, provenance=dict(base_provenance))
162
+ out = AppliedDataset(
163
+ rows=[dict(r) for r in rows],
164
+ provenance=_flip_rules(base_provenance, "Billing Period", [table], source),
165
+ )
166
+ for row in out.rows or []:
167
+ key = (
168
+ row.get("InvoiceIssuerName", ""),
169
+ row.get("BillingPeriodStart", ""),
170
+ row.get("BillingPeriodEnd", ""),
171
+ )
172
+ for column in table.fact_columns:
173
+ _apply_column(
174
+ row, column, table.value(key, column),
175
+ synthetic=synthetic, counters=out.counters,
176
+ )
177
+ _strict_suppress_uncovered_assumed(out, "Billing Period", [table], synthetic=synthetic)
178
+ return out
179
+
180
+
181
+ def apply_invoice_details(
182
+ rows: list[dict[str, str]],
183
+ id_mapping: dict[GrainKey, str],
184
+ bundle: SupplementBundle,
185
+ source: SourceKeySets,
186
+ base_provenance: dict[str, ColumnRule],
187
+ *,
188
+ synthetic: bool,
189
+ ) -> tuple[AppliedDataset, dict[GrainKey, str]]:
190
+ """Apply invoice-header and invoice-line supplements; returns the updated id mapping."""
191
+ invoice = bundle.get("invoice")
192
+ line = bundle.get("invoice_line")
193
+ if invoice is None and line is None:
194
+ return AppliedDataset(rows=rows, provenance=dict(base_provenance)), id_mapping
195
+ tables = [t for t in (invoice, line) if t is not None]
196
+ out = AppliedDataset(
197
+ rows=[],
198
+ provenance=_flip_rules(base_provenance, "Invoice Detail", tables, source),
199
+ )
200
+
201
+ # Conditional columns activate only when the supplement genuinely enables them.
202
+ extra_emitted: list[str] = []
203
+ for column in _ACTIVATABLE_INVOICE_COLUMNS:
204
+ for table in tables:
205
+ if column not in table.fact_columns:
206
+ continue
207
+ col_cov = coverage(table, source.keys_for(table.kind.name))[column]
208
+ if col_cov.complete or (_allows_nulls("Invoice Detail", column) and col_cov.covered):
209
+ extra_emitted.append(column)
210
+ out.provenance[column] = ColumnRule(
211
+ Lineage.ENRICHED,
212
+ _source_label(table, column),
213
+ None if col_cov.complete else "nulls where the client supplied no value",
214
+ )
215
+
216
+ emitted = [c for c in dataset_columns("Invoice Detail") if rows and c in rows[0]]
217
+ all_emitted = [
218
+ c for c in dataset_columns("Invoice Detail") if c in set(emitted) | set(extra_emitted)
219
+ ]
220
+
221
+ new_mapping = dict(id_mapping)
222
+ for row in rows:
223
+ merged = {c: row.get(c, "") for c in all_emitted}
224
+ grain = invoice_detail_grain_key(row)
225
+ if invoice is not None:
226
+ header_key = (row.get("InvoiceIssuerName", ""), row.get("InvoiceId", ""))
227
+ for column in invoice.fact_columns:
228
+ if column not in all_emitted:
229
+ continue
230
+ _apply_column(
231
+ merged, column, invoice.value(header_key, column),
232
+ synthetic=synthetic, counters=out.counters,
233
+ )
234
+ if line is not None:
235
+ for column in line.fact_columns:
236
+ if column == "BilledCost" or column not in all_emitted:
237
+ continue
238
+ _apply_column(
239
+ merged, column, line.value(grain, column),
240
+ synthetic=synthetic, counters=out.counters,
241
+ )
242
+ real_id = line.value(grain, "InvoiceDetailId")
243
+ if real_id:
244
+ new_mapping[grain] = real_id
245
+ assert out.rows is not None
246
+ out.rows.append(merged)
247
+ _strict_suppress_uncovered_assumed(out, "Invoice Detail", tables, synthetic=synthetic)
248
+ return out, new_mapping
249
+
250
+
251
+ def apply_contract_commitments(
252
+ rows: list[dict[str, str]],
253
+ bundle: SupplementBundle,
254
+ source: SourceKeySets,
255
+ base_provenance: dict[str, ColumnRule],
256
+ *,
257
+ synthetic: bool,
258
+ ) -> AppliedDataset:
259
+ table = bundle.get("contract_commitment")
260
+ if table is None:
261
+ return AppliedDataset(rows=rows, provenance=dict(base_provenance))
262
+ out = AppliedDataset(
263
+ rows=[dict(r) for r in rows],
264
+ provenance=_flip_rules(base_provenance, "Contract Commitment", [table], source),
265
+ )
266
+ upfront = "ContractCommitmentPaymentUpfrontPercentage"
267
+ model_col = "ContractCommitmentPaymentModel"
268
+ upfront_supplied = upfront in table.fact_columns
269
+ model_cov = (
270
+ coverage(table, source.keys_for("contract_commitment")).get(model_col)
271
+ if model_col in table.fact_columns
272
+ else None
273
+ )
274
+ # Override columns whose base row already holds a factual (provider-context) value:
275
+ # a partial override must not blank the uncovered rows.
276
+ factual_base = {c for c, r in base_provenance.items() if r.is_factual}
277
+ for row in out.rows or []:
278
+ key = (row.get("ContractCommitmentId", ""),)
279
+ for column in table.fact_columns:
280
+ if column == upfront:
281
+ continue # handled below (derivable from the payment model)
282
+ _apply_column(
283
+ row, column, table.value(key, column),
284
+ synthetic=synthetic, counters=out.counters,
285
+ base_is_factual=column in factual_base,
286
+ )
287
+ # Upfront percentage: supplied wins; else exactly derivable from the payment
288
+ # model ('No Upfront' -> 0, 'All Upfront' -> 1); 'Partial Upfront' without a
289
+ # supplied percentage is not derivable and stays empty in BOTH modes (never a
290
+ # guessed '0' paired with 'Partial Upfront'); the mandatory-column lint flags it.
291
+ supplied_pct = table.value(key, upfront) if upfront_supplied else ""
292
+ payment_model = row.get(model_col, "")
293
+ if supplied_pct:
294
+ row[upfront] = supplied_pct
295
+ out.counters.record(upfront, Lineage.ENRICHED)
296
+ elif payment_model == "No Upfront":
297
+ row[upfront] = "0"
298
+ out.counters.record(upfront, Lineage.DERIVED)
299
+ elif payment_model == "All Upfront":
300
+ row[upfront] = "1"
301
+ out.counters.record(upfront, Lineage.DERIVED)
302
+ else:
303
+ row[upfront] = ""
304
+ out.counters.record(upfront, Lineage.UNAVAILABLE)
305
+ if not any(
306
+ (r.get(upfront) or "") == "" for r in (out.rows or [])
307
+ ) and (upfront_supplied or (model_cov is not None and model_cov.complete)):
308
+ source_note = (
309
+ _source_label(table, upfront)
310
+ if upfront_supplied
311
+ else f"ContractCommitmentPaymentModel ({_source_label(table, model_col)})"
312
+ )
313
+ lineage = Lineage.ENRICHED if upfront_supplied else Lineage.DERIVED
314
+ out.provenance[upfront] = ColumnRule(lineage, source_note)
315
+ _strict_suppress_uncovered_assumed(
316
+ out, "Contract Commitment", [table], synthetic=synthetic
317
+ )
318
+ return out
@@ -0,0 +1,219 @@
1
+ """Gap analysis: exactly which facts a client must supply, per FOCUS 1.4 dataset.
2
+
3
+ The gap set is *computed*, never hardcoded: a column gap is a column that is Mandatory,
4
+ non-nullable and non-factual under the very provenance rules the converter would use for
5
+ this source — i.e. exactly ``strict_blockers()``. Each gap is annotated from the embedded
6
+ model (allowed values, format, condition text) and mapped to the supplement kind(s) able
7
+ to satisfy it. Nullable non-factual columns of the same kinds are reported as
8
+ *recommended* (they never block strict production, but supplying them makes the output
9
+ more complete). The JSON output doubles as a fill-in template: it carries a ready-to-use
10
+ CSV header line per supplement kind.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ from collections.abc import Iterable
16
+ from dataclasses import dataclass, field
17
+
18
+ from focus_data_toolkit.convert.billing_period import PROVENANCE as BILLING_PERIOD_PROVENANCE
19
+ from focus_data_toolkit.convert.contract_commitment import (
20
+ PROVENANCE as CONTRACT_COMMITMENT_PROVENANCE,
21
+ )
22
+ from focus_data_toolkit.convert.cost_and_usage import cost_and_usage_provenance
23
+ from focus_data_toolkit.convert.invoice_detail import PROVENANCE as INVOICE_DETAIL_PROVENANCE
24
+ from focus_data_toolkit.model import FOCUS_1_4_DATASETS
25
+ from focus_data_toolkit.model.validator import load_model
26
+ from focus_data_toolkit.provenance import ColumnRule, Lineage, strict_blockers
27
+ from focus_data_toolkit.supplement.kinds import SUPPLEMENT_KINDS, kinds_for_column
28
+
29
+ GAP_REPORT_FORMAT = "1"
30
+
31
+
32
+ @dataclass(frozen=True)
33
+ class ColumnGap:
34
+ """One column the source cannot factually populate."""
35
+
36
+ dataset: str
37
+ column: str
38
+ feature_level: str
39
+ allows_nulls: bool
40
+ current_lineage: str
41
+ current_note: str | None
42
+ blocking: bool
43
+ supplement_kinds: tuple[str, ...]
44
+ join_keys: tuple[str, ...]
45
+ value_format: str | None
46
+ allowed_values: tuple[str, ...]
47
+ condition: str | None
48
+
49
+ def as_dict(self) -> dict:
50
+ out: dict = {
51
+ "dataset": self.dataset,
52
+ "column": self.column,
53
+ "feature_level": self.feature_level,
54
+ "allows_nulls": self.allows_nulls,
55
+ "current_lineage": self.current_lineage,
56
+ "blocking": self.blocking,
57
+ "supplement_kinds": list(self.supplement_kinds),
58
+ "join_keys": list(self.join_keys),
59
+ }
60
+ if self.current_note:
61
+ out["current_note"] = self.current_note
62
+ if self.value_format:
63
+ out["value_format"] = self.value_format
64
+ if self.allowed_values:
65
+ out["allowed_values"] = list(self.allowed_values)
66
+ if self.condition:
67
+ out["condition"] = self.condition
68
+ return out
69
+
70
+
71
+ @dataclass(frozen=True)
72
+ class GapReport:
73
+ """Everything a client needs to complete this source into factual 1.4 datasets."""
74
+
75
+ source_version: str
76
+ gaps: dict[str, tuple[ColumnGap, ...]] = field(default_factory=dict)
77
+ dataset_level_gaps: dict[str, str] = field(default_factory=dict)
78
+
79
+ def blocking(self, dataset: str) -> tuple[ColumnGap, ...]:
80
+ return tuple(g for g in self.gaps.get(dataset, ()) if g.blocking)
81
+
82
+ def as_dict(self) -> dict:
83
+ """Deterministic JSON payload; doubles as a fill-in template."""
84
+ kinds_used = sorted(
85
+ {k for gaps in self.gaps.values() for g in gaps for k in g.supplement_kinds}
86
+ )
87
+ return {
88
+ "gap_report_format": GAP_REPORT_FORMAT,
89
+ "source_version": self.source_version,
90
+ "datasets": {
91
+ name: {
92
+ "column_gaps": [g.as_dict() for g in self.gaps.get(name, ())],
93
+ "strictly_producible_as_is": not self.blocking(name)
94
+ and name not in self.dataset_level_gaps,
95
+ **(
96
+ {"dataset_gap": self.dataset_level_gaps[name]}
97
+ if name in self.dataset_level_gaps
98
+ else {}
99
+ ),
100
+ }
101
+ for name in FOCUS_1_4_DATASETS
102
+ },
103
+ "supplement_templates": {
104
+ name: {
105
+ "target_dataset": SUPPLEMENT_KINDS[name].target_dataset,
106
+ "join_keys": list(SUPPLEMENT_KINDS[name].join_keys),
107
+ "csv_header": ",".join(SUPPLEMENT_KINDS[name].header_template),
108
+ }
109
+ for name in kinds_used
110
+ },
111
+ }
112
+
113
+ def render_text(self) -> str:
114
+ lines = [f"FOCUS {self.source_version} source -> 1.4 gap report", ""]
115
+ for name in FOCUS_1_4_DATASETS:
116
+ gaps = self.gaps.get(name, ())
117
+ if name in self.dataset_level_gaps:
118
+ lines.append(f"[{name}] NOT PRODUCIBLE: {self.dataset_level_gaps[name]}")
119
+ lines.append("")
120
+ continue
121
+ blocking = [g for g in gaps if g.blocking]
122
+ recommended = [g for g in gaps if not g.blocking]
123
+ if not blocking:
124
+ lines.append(f"[{name}] strictly producible from this source as-is")
125
+ else:
126
+ lines.append(f"[{name}] blocked by {len(blocking)} column(s):")
127
+ for g in blocking:
128
+ extra = f" (allowed: {', '.join(g.allowed_values)})" if g.allowed_values else ""
129
+ kinds = ", ".join(g.supplement_kinds) or "-"
130
+ lines.append(f" - {g.column}{extra} <- supplement kind: {kinds}")
131
+ for g in recommended:
132
+ lines.append(f" ~ {g.column} (recommended, nullable)")
133
+ lines.append("")
134
+ kinds_used = sorted(
135
+ {k for gaps in self.gaps.values() for g in gaps for k in g.supplement_kinds}
136
+ )
137
+ if kinds_used:
138
+ lines.append("Supplement templates (CSV headers, ready to fill):")
139
+ for name in kinds_used:
140
+ lines.append(f" {name}: {','.join(SUPPLEMENT_KINDS[name].header_template)}")
141
+ return "\n".join(lines) + "\n"
142
+
143
+
144
+ def _gap(dataset: str, column: str, spec: dict, rule: ColumnRule | None, blocking: bool) -> ColumnGap:
145
+ kinds = kinds_for_column(dataset, column)
146
+ return ColumnGap(
147
+ dataset=dataset,
148
+ column=column,
149
+ feature_level=spec.get("feature_level", ""),
150
+ allows_nulls=bool(spec.get("allows_nulls", True)),
151
+ current_lineage=rule.lineage.value if rule else "UNAVAILABLE",
152
+ current_note=(rule.note if rule else None),
153
+ blocking=blocking,
154
+ supplement_kinds=tuple(k.name for k in kinds),
155
+ join_keys=kinds[0].join_keys if kinds else (),
156
+ value_format=spec.get("value_format") or spec.get("data_type"),
157
+ allowed_values=tuple(spec.get("allowed_values") or ()),
158
+ condition=spec.get("condition"),
159
+ )
160
+
161
+
162
+ def compute_gaps(
163
+ source_columns: Iterable[str],
164
+ source_version: str,
165
+ *,
166
+ cc_columns: Iterable[str] | None = None,
167
+ ) -> GapReport:
168
+ """Compute the gap report for a Cost and Usage header (and optional 1.3 CC header).
169
+
170
+ Uses the same provenance rules as the converter, so a reported blocking gap is by
171
+ construction exactly what would block strict production of that dataset.
172
+ """
173
+ model = load_model()
174
+ # When a Contract Commitment source header is given, reflect it: a column the base
175
+ # provenance treats as OBSERVED-from-1.3 but that is absent from the actual header is a
176
+ # source-completeness gap (no supplement can fabricate it), not an observed value.
177
+ cc_prov = dict(CONTRACT_COMMITMENT_PROVENANCE)
178
+ if cc_columns is not None:
179
+ present_cc = set(cc_columns)
180
+ for col, base_rule in CONTRACT_COMMITMENT_PROVENANCE.items():
181
+ if base_rule.lineage is Lineage.OBSERVED and col not in present_cc:
182
+ cc_prov[col] = ColumnRule(
183
+ Lineage.UNAVAILABLE, note="absent from the Contract Commitment source"
184
+ )
185
+ provenance: dict[str, dict[str, ColumnRule]] = {
186
+ "Cost and Usage": cost_and_usage_provenance(
187
+ source_columns, source_version, invoice_detail_linked=False
188
+ ),
189
+ "Billing Period": BILLING_PERIOD_PROVENANCE,
190
+ "Invoice Detail": INVOICE_DETAIL_PROVENANCE,
191
+ "Contract Commitment": cc_prov,
192
+ }
193
+ gaps: dict[str, tuple[ColumnGap, ...]] = {}
194
+ dataset_level: dict[str, str] = {}
195
+ for name in FOCUS_1_4_DATASETS:
196
+ columns: dict = model["datasets"][name]["columns"]
197
+ prov = provenance[name]
198
+ blockers = set(strict_blockers(prov, columns))
199
+ out: list[ColumnGap] = []
200
+ for col in sorted(columns):
201
+ spec = columns[col]
202
+ rule = prov.get(col)
203
+ if col in blockers:
204
+ out.append(_gap(name, col, spec, rule, blocking=True))
205
+ elif (
206
+ rule is not None
207
+ and not rule.is_factual
208
+ and spec.get("allows_nulls", True)
209
+ and kinds_for_column(name, col)
210
+ ):
211
+ # Nullable, non-factual, and a supplement kind can supply it: recommended.
212
+ out.append(_gap(name, col, spec, rule, blocking=False))
213
+ gaps[name] = tuple(out)
214
+ if cc_columns is None:
215
+ dataset_level["Contract Commitment"] = (
216
+ "no FOCUS 1.3 Contract Commitment source provided; supply the 1.3 dataset "
217
+ "(13 columns) plus a 'contract_commitment' supplement for the 1.4-new terms"
218
+ )
219
+ return GapReport(source_version=source_version, gaps=gaps, dataset_level_gaps=dataset_level)
@@ -0,0 +1,118 @@
1
+ """Registry of supplement kinds — what a client may supply, and how it joins.
2
+
3
+ Each kind targets one FOCUS 1.4 dataset, joins to the conversion on that dataset's
4
+ natural key, and may supply a fixed set of fact columns (FOCUS column names). Anything
5
+ else must be ``x_``-prefixed. Join-key values are compared after ``.strip()`` (the same
6
+ normalization the converters use); there is no fuzzy matching.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from dataclasses import dataclass
12
+
13
+ from focus_data_toolkit.convert.invoice_detail import GRAIN_FIELDS
14
+
15
+
16
+ @dataclass(frozen=True)
17
+ class SupplementKind:
18
+ """One supplement file format the toolkit knows how to join and apply."""
19
+
20
+ name: str
21
+ target_dataset: str
22
+ join_keys: tuple[str, ...]
23
+ columns: frozenset[str]
24
+
25
+ @property
26
+ def header_template(self) -> tuple[str, ...]:
27
+ """A ready-to-fill CSV header: join keys first, then the fact columns."""
28
+ return self.join_keys + tuple(sorted(self.columns))
29
+
30
+
31
+ # The billing-period cycle facts a Cost and Usage source can never carry.
32
+ BILLING_PERIOD_KIND = SupplementKind(
33
+ name="billing_period",
34
+ target_dataset="Billing Period",
35
+ join_keys=("InvoiceIssuerName", "BillingPeriodStart", "BillingPeriodEnd"),
36
+ columns=frozenset(
37
+ {"BillingPeriodCreated", "BillingPeriodLastUpdated", "BillingPeriodStatus"}
38
+ ),
39
+ )
40
+
41
+ # Invoice-header facts: one row per issued invoice.
42
+ INVOICE_KIND = SupplementKind(
43
+ name="invoice",
44
+ target_dataset="Invoice Detail",
45
+ join_keys=("InvoiceIssuerName", "InvoiceId"),
46
+ columns=frozenset(
47
+ {
48
+ "InvoiceIssueDate",
49
+ "InvoiceIssueStatus",
50
+ "PaymentTerms",
51
+ "PaymentDueDate",
52
+ "ReferenceInvoiceId",
53
+ "PurchaseOrderNumber",
54
+ "PaymentCurrency",
55
+ }
56
+ ),
57
+ )
58
+
59
+ # Invoice-line facts: one row per invoice line, joined on the full business grain
60
+ # (exactly the grain the derived Invoice Detail dataset aggregates on). ``BilledCost``
61
+ # is accepted only as a reconciliation check against the derived grain sum — it never
62
+ # replaces the derived value.
63
+ INVOICE_LINE_KIND = SupplementKind(
64
+ name="invoice_line",
65
+ target_dataset="Invoice Detail",
66
+ join_keys=GRAIN_FIELDS,
67
+ columns=frozenset(
68
+ {
69
+ "InvoiceDetailId",
70
+ "InvoiceDetailCreated",
71
+ "InvoiceDetailLastUpdated",
72
+ "InvoiceDetailDescription",
73
+ "InvoiceDetailGrain",
74
+ "PaymentCurrencyBilledCost",
75
+ "PaymentCurrencyInvoiceDetailId",
76
+ "BilledCost",
77
+ }
78
+ ),
79
+ )
80
+
81
+ # The 1.4-new contract-commitment commercial terms a 1.3 source does not carry.
82
+ CONTRACT_COMMITMENT_KIND = SupplementKind(
83
+ name="contract_commitment",
84
+ target_dataset="Contract Commitment",
85
+ join_keys=("ContractCommitmentId",),
86
+ columns=frozenset(
87
+ {
88
+ "ContractCommitmentApplicability",
89
+ "ContractCommitmentBenefitCategory",
90
+ "ContractCommitmentCreated",
91
+ "ContractCommitmentDiscountPercentage",
92
+ "ContractCommitmentFulfillmentInterval",
93
+ "ContractCommitmentLastUpdated",
94
+ "ContractCommitmentLifecycleStatus",
95
+ "ContractCommitmentModel",
96
+ "ContractCommitmentOfferCategory",
97
+ "ContractCommitmentPaymentInterval",
98
+ "ContractCommitmentPaymentModel",
99
+ "ContractCommitmentPaymentUpfrontPercentage",
100
+ "InvoiceIssuerName",
101
+ "ServiceProviderName",
102
+ }
103
+ ),
104
+ )
105
+
106
+ SUPPLEMENT_KINDS: dict[str, SupplementKind] = {
107
+ k.name: k
108
+ for k in (BILLING_PERIOD_KIND, INVOICE_KIND, INVOICE_LINE_KIND, CONTRACT_COMMITMENT_KIND)
109
+ }
110
+
111
+
112
+ def kinds_for_column(dataset: str, column: str) -> tuple[SupplementKind, ...]:
113
+ """The kinds able to supply ``column`` of ``dataset`` (deterministic order)."""
114
+ return tuple(
115
+ kind
116
+ for kind in SUPPLEMENT_KINDS.values()
117
+ if kind.target_dataset == dataset and column in kind.columns
118
+ )