focus-data-toolkit 0.11.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. focus_data_toolkit/__init__.py +69 -0
  2. focus_data_toolkit/__main__.py +6 -0
  3. focus_data_toolkit/_version.py +8 -0
  4. focus_data_toolkit/cli.py +968 -0
  5. focus_data_toolkit/context/__init__.py +88 -0
  6. focus_data_toolkit/context/billing.py +54 -0
  7. focus_data_toolkit/context/provider.py +90 -0
  8. focus_data_toolkit/convert/__init__.py +708 -0
  9. focus_data_toolkit/convert/billing_period.py +65 -0
  10. focus_data_toolkit/convert/contract_applied.py +235 -0
  11. focus_data_toolkit/convert/contract_commitment.py +182 -0
  12. focus_data_toolkit/convert/cost_and_usage.py +179 -0
  13. focus_data_toolkit/convert/detect.py +39 -0
  14. focus_data_toolkit/convert/invoice_detail.py +199 -0
  15. focus_data_toolkit/convert/streaming.py +1030 -0
  16. focus_data_toolkit/errors.py +145 -0
  17. focus_data_toolkit/focus_json.py +68 -0
  18. focus_data_toolkit/generators/__init__.py +61 -0
  19. focus_data_toolkit/generators/_shim.py +43 -0
  20. focus_data_toolkit/generators/engine/__init__.py +14 -0
  21. focus_data_toolkit/generators/engine/context.py +12 -0
  22. focus_data_toolkit/generators/engine/determinism.py +117 -0
  23. focus_data_toolkit/generators/engine/json_focus.py +63 -0
  24. focus_data_toolkit/generators/engine/ladder.py +71 -0
  25. focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
  26. focus_data_toolkit/generators/engine/serialize.py +151 -0
  27. focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
  28. focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
  29. focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
  30. focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
  31. focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
  32. focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
  33. focus_data_toolkit/generators/providers/__init__.py +29 -0
  34. focus_data_toolkit/generators/providers/aws.py +186 -0
  35. focus_data_toolkit/generators/providers/azure.py +191 -0
  36. focus_data_toolkit/generators/providers/gcp.py +194 -0
  37. focus_data_toolkit/generators/providers/profile.py +123 -0
  38. focus_data_toolkit/generators/scenarios.py +178 -0
  39. focus_data_toolkit/generators/versions/__init__.py +17 -0
  40. focus_data_toolkit/generators/versions/adapter.py +41 -0
  41. focus_data_toolkit/generators/versions/v1_2.py +111 -0
  42. focus_data_toolkit/generators/versions/v1_3.py +154 -0
  43. focus_data_toolkit/io/__init__.py +1 -0
  44. focus_data_toolkit/io/atomic_writer.py +462 -0
  45. focus_data_toolkit/io/csv_io.py +128 -0
  46. focus_data_toolkit/io/parquet_io.py +528 -0
  47. focus_data_toolkit/io/records.py +92 -0
  48. focus_data_toolkit/io/row_source.py +117 -0
  49. focus_data_toolkit/lifecycle.py +342 -0
  50. focus_data_toolkit/manifest.py +114 -0
  51. focus_data_toolkit/model/__init__.py +43 -0
  52. focus_data_toolkit/model/capabilities.py +66 -0
  53. focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
  54. focus_data_toolkit/model/focus_1_4_model.json +1913 -0
  55. focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
  56. focus_data_toolkit/model/focus_json_keys.py +112 -0
  57. focus_data_toolkit/model/iso_4217_currencies.json +23 -0
  58. focus_data_toolkit/model/json_schema_check.py +205 -0
  59. focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
  60. focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
  61. focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
  62. focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
  63. focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
  64. focus_data_toolkit/model/model_provenance.json +58 -0
  65. focus_data_toolkit/model/validator.py +498 -0
  66. focus_data_toolkit/modes.py +18 -0
  67. focus_data_toolkit/official_validator.py +61 -0
  68. focus_data_toolkit/progress.py +89 -0
  69. focus_data_toolkit/provenance.py +106 -0
  70. focus_data_toolkit/py.typed +1 -0
  71. focus_data_toolkit/runtime.py +243 -0
  72. focus_data_toolkit/schema/__init__.py +17 -0
  73. focus_data_toolkit/schema/detection.py +274 -0
  74. focus_data_toolkit/schema/registry.py +127 -0
  75. focus_data_toolkit/storage/__init__.py +1 -0
  76. focus_data_toolkit/storage/external_index.py +99 -0
  77. focus_data_toolkit/storage/spill.py +150 -0
  78. focus_data_toolkit/studio/__init__.py +19 -0
  79. focus_data_toolkit/studio/app.py +467 -0
  80. focus_data_toolkit/studio/config.py +42 -0
  81. focus_data_toolkit/studio/frontend/app.js +214 -0
  82. focus_data_toolkit/studio/frontend/index.html +101 -0
  83. focus_data_toolkit/studio/frontend/style.css +60 -0
  84. focus_data_toolkit/studio/jobs.py +142 -0
  85. focus_data_toolkit/studio/preview.py +32 -0
  86. focus_data_toolkit/studio/security.py +125 -0
  87. focus_data_toolkit/studio/server.py +71 -0
  88. focus_data_toolkit/supplement/__init__.py +50 -0
  89. focus_data_toolkit/supplement/adapters/__init__.py +21 -0
  90. focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
  91. focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
  92. focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
  93. focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
  94. focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
  95. focus_data_toolkit/supplement/adapters/registry.py +215 -0
  96. focus_data_toolkit/supplement/apply.py +318 -0
  97. focus_data_toolkit/supplement/gaps.py +219 -0
  98. focus_data_toolkit/supplement/kinds.py +118 -0
  99. focus_data_toolkit/supplement/loader.py +409 -0
  100. focus_data_toolkit/supplement/spec.py +74 -0
  101. focus_data_toolkit/supplement/validate.py +215 -0
  102. focus_data_toolkit/validate/__init__.py +15 -0
  103. focus_data_toolkit/validate/allocation.py +333 -0
  104. focus_data_toolkit/validate/bundle.py +254 -0
  105. focus_data_toolkit/validate/codes.py +93 -0
  106. focus_data_toolkit/validate/corrections.py +245 -0
  107. focus_data_toolkit/validate/reconciliation.py +98 -0
  108. focus_data_toolkit/validate/referential.py +289 -0
  109. focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
  110. focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
  111. focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
  112. focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
  113. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
  114. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
  115. focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
  116. focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,708 @@
1
+ """Convert FOCUS 1.2/1.3 source data into the four FOCUS 1.4 datasets.
2
+
3
+ Two modes (see :mod:`focus_data_toolkit.modes`):
4
+
5
+ * ``STRICT`` (default) — a canonical FOCUS 1.4 dataset is produced only when every
6
+ Mandatory non-nullable column has a factual lineage (observed / renamed / derived /
7
+ enriched). Datasets that would require assumed provider-issued values are reported
8
+ ``NOT_PRODUCED`` in the manifest, never fabricated. In practice only Cost and Usage is
9
+ produced from a Cost-and-Usage source; Billing Period, Invoice Detail and the expanded
10
+ 1.4 Contract Commitment require provider billing facts absent from the source.
11
+ * ``SYNTHETIC`` — for demos / tests / learning: assumed values are generated, the affected
12
+ datasets are labelled synthetic in the manifest (and filenames), and the result is never
13
+ presented as fully conformant.
14
+
15
+ Every conversion emits a deterministic manifest recording, per column, how the value was
16
+ obtained.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import csv
22
+ import hashlib
23
+ import io
24
+ import json
25
+ from collections.abc import Iterable
26
+ from dataclasses import dataclass, field
27
+ from datetime import UTC, datetime
28
+ from pathlib import Path
29
+ from typing import TYPE_CHECKING
30
+
31
+ if TYPE_CHECKING: # imported lazily at runtime (validate.bundle imports from this package)
32
+ from focus_data_toolkit.validate.bundle import BundleReport
33
+
34
+ from focus_data_toolkit import manifest as manifest_mod
35
+ from focus_data_toolkit.context import describe_source_contexts, representative_provider
36
+ from focus_data_toolkit.convert.billing_period import PROVENANCE as BILLING_PERIOD_PROVENANCE
37
+ from focus_data_toolkit.convert.billing_period import build_billing_periods
38
+ from focus_data_toolkit.convert.contract_commitment import (
39
+ PROVENANCE as CONTRACT_COMMITMENT_PROVENANCE,
40
+ )
41
+ from focus_data_toolkit.convert.contract_commitment import convert_contract_commitment
42
+ from focus_data_toolkit.convert.cost_and_usage import (
43
+ convert_cost_and_usage,
44
+ cost_and_usage_provenance,
45
+ )
46
+ from focus_data_toolkit.convert.detect import detect_focus_version
47
+ from focus_data_toolkit.convert.invoice_detail import PROVENANCE as INVOICE_DETAIL_PROVENANCE
48
+ from focus_data_toolkit.convert.invoice_detail import build_invoice_details
49
+ from focus_data_toolkit.errors import Diagnostic, Severity
50
+ from focus_data_toolkit.io.atomic_writer import (
51
+ AtomicOutputDir,
52
+ AtomicWriteError,
53
+ DestinationExistsError,
54
+ OnExists,
55
+ sha256sums_text,
56
+ )
57
+ from focus_data_toolkit.model import FOCUS_1_4_DATASETS, load_model
58
+ from focus_data_toolkit.model.capabilities import CapabilityProfile
59
+ from focus_data_toolkit.model.validator import LintReport, lint_focus_1_4_structure
60
+ from focus_data_toolkit.modes import Mode
61
+ from focus_data_toolkit.provenance import (
62
+ ColumnRule,
63
+ Lineage,
64
+ LineageCounters,
65
+ has_assumptions,
66
+ strict_blockers,
67
+ )
68
+ from focus_data_toolkit.schema import registry
69
+ from focus_data_toolkit.schema.detection import SchemaDetectionResult, detect_focus_schema
70
+ from focus_data_toolkit.supplement.apply import (
71
+ apply_billing_periods,
72
+ apply_contract_commitments,
73
+ apply_invoice_details,
74
+ )
75
+ from focus_data_toolkit.supplement.loader import SupplementBundle
76
+ from focus_data_toolkit.supplement.validate import (
77
+ SourceKeySets,
78
+ coverage,
79
+ source_key_sets,
80
+ validate_supplements,
81
+ )
82
+
83
+ # Base output file name per dataset (stable, snake_case). Synthetic datasets are written
84
+ # with a ``synthetic_`` prefix so they are unmistakable on disk.
85
+ DATASET_FILENAMES = {
86
+ "Cost and Usage": "focus_1_4_cost_and_usage.csv",
87
+ "Contract Commitment": "focus_1_4_contract_commitment.csv",
88
+ "Billing Period": "focus_1_4_billing_period.csv",
89
+ "Invoice Detail": "focus_1_4_invoice_detail.csv",
90
+ }
91
+
92
+ # Output formats the pipeline can write; the CSV path stays byte-exact, Parquet is value-exact.
93
+ OUTPUT_FORMATS = ("csv", "parquet")
94
+
95
+
96
+ def output_filename_for(
97
+ dataset: str, *, synthetic_prefix: bool, output_format: str = "csv", partitioned: bool = False
98
+ ) -> str:
99
+ """The output name of ``dataset`` for a format.
100
+
101
+ CSV/single-file Parquet get a ``.csv``/``.parquet`` file; a partitioned Parquet dataset is a
102
+ *directory* (no extension) holding the Hive partition tree.
103
+ """
104
+ base = DATASET_FILENAMES[dataset]
105
+ if output_format == "parquet" and base.endswith(".csv"):
106
+ base = base[:-4] + ("" if partitioned else ".parquet")
107
+ return f"synthetic_{base}" if synthetic_prefix else base
108
+
109
+
110
+ class ConversionError(ValueError):
111
+ """Raised when the source cannot be converted."""
112
+
113
+
114
+ class ConversionCancelled(ConversionError):
115
+ """Raised cooperatively when a cancel predicate returns True mid-conversion.
116
+
117
+ Subclasses :class:`ConversionError` so existing ``except ConversionError`` handlers
118
+ still clean up (the atomic staging directory is removed on the way out, so nothing is
119
+ published); the CLI catches it first to report a distinct cancelled exit code.
120
+ """
121
+
122
+
123
+ @dataclass
124
+ class ConversionResult:
125
+ """Outcome of a 1.x -> 1.4 conversion."""
126
+
127
+ source_version: str
128
+ mode: Mode
129
+ datasets: dict[str, list[dict[str, str]]]
130
+ provenance: dict[str, dict[str, ColumnRule]]
131
+ manifest: dict
132
+ reports: dict[str, LintReport] = field(default_factory=dict)
133
+ detection: SchemaDetectionResult | None = None
134
+ contexts: dict = field(default_factory=dict)
135
+ diagnostics: list[Diagnostic] = field(default_factory=list)
136
+ bundle_report: BundleReport | None = None
137
+
138
+ @property
139
+ def ok(self) -> bool:
140
+ """All produced datasets passed the structural + semantic lint."""
141
+ return all(r.ok for r in self.reports.values())
142
+
143
+ @property
144
+ def coverage(self) -> tuple[str, ...]:
145
+ """FOCUS 1.4 datasets actually produced (in canonical order)."""
146
+ return tuple(name for name in FOCUS_1_4_DATASETS if name in self.datasets)
147
+
148
+ @property
149
+ def not_produced(self) -> tuple[str, ...]:
150
+ return tuple(name for name in FOCUS_1_4_DATASETS if name not in self.datasets)
151
+
152
+ @property
153
+ def assumptions_present(self) -> bool:
154
+ return bool(self.manifest["assumptions_present"])
155
+
156
+ def output_filename(self, dataset: str) -> str:
157
+ return self.manifest["datasets"][dataset]["output_file"]
158
+
159
+
160
+ def _resolve_source_version(
161
+ headers: Iterable[str],
162
+ *,
163
+ source_version: str | None,
164
+ source_dataset: str | None,
165
+ mode: Mode,
166
+ ) -> tuple[str, SchemaDetectionResult]:
167
+ """Determine the convertible source version and record the detection decision.
168
+
169
+ ``headers`` is the source column set. ``source_version`` / ``source_dataset`` force the
170
+ corresponding dimension. In strict mode an ambiguous or low-confidence detection (that is
171
+ not forced) is refused with a clear error; a forced version incompatible with the header is
172
+ always refused.
173
+ """
174
+ # A bad --source-version/--source-dataset value raises ValueError from normalisation;
175
+ # surface it as a ConversionError so the CLI exits with the invalid-argument code, not a
176
+ # traceback.
177
+ try:
178
+ detection = detect_focus_schema(headers, dataset=source_dataset, version=source_version)
179
+ except ValueError as exc:
180
+ raise ConversionError(f"invalid --source-version/--source-dataset: {exc}") from exc
181
+
182
+ forced = source_version is not None or source_dataset is not None
183
+ if forced and detection.confidence == "LOW":
184
+ raise ConversionError(
185
+ "forced source schema is incompatible with the header (detected "
186
+ f"{detection.dataset} {detection.detected_version}, confidence LOW): "
187
+ + "; ".join(detection.notes)
188
+ )
189
+
190
+ if source_version is not None:
191
+ try:
192
+ version = registry.normalize_version(source_version)
193
+ except ValueError as exc:
194
+ raise ConversionError(f"invalid --source-version {source_version!r}: {exc}") from exc
195
+ else:
196
+ if mode is Mode.STRICT and not forced and detection.confidence != "HIGH":
197
+ raise ConversionError(
198
+ "strict mode refuses an ambiguous or low-confidence source schema (detected "
199
+ f"{detection.dataset} {detection.detected_version}, confidence "
200
+ f"{detection.confidence}); force it with --source-version / --source-dataset"
201
+ )
202
+ # detect_focus_version raises a clear ValueError for non-CAU / 1.4 / non-FOCUS headers.
203
+ try:
204
+ version = detect_focus_version(headers)
205
+ except ValueError as exc:
206
+ raise ConversionError(str(exc)) from exc
207
+
208
+ # This converter only accepts a Cost and Usage source. Forcing a version does not bypass
209
+ # this: a Contract Commitment header passed as --cost-and-usage --source-version 1.3 must be
210
+ # rejected, not converted into an empty manifest.
211
+ if detection.dataset != "Cost and Usage":
212
+ raise ConversionError(
213
+ "this converter requires a FOCUS Cost and Usage source; detected "
214
+ f"{detection.dataset or 'no FOCUS dataset'} (confidence {detection.confidence})"
215
+ )
216
+ if version not in ("1.2", "1.3"):
217
+ raise ConversionError(
218
+ f"unsupported source version {version!r}; this tool converts FOCUS 1.2/1.3 -> 1.4"
219
+ )
220
+ return version, detection
221
+
222
+
223
+ def assemble_manifest(
224
+ *,
225
+ version: str,
226
+ mode: Mode,
227
+ synthetic: bool,
228
+ detection: SchemaDetectionResult,
229
+ contexts: dict,
230
+ diagnostics: list[Diagnostic],
231
+ provenance: dict[str, dict[str, ColumnRule]],
232
+ source_available: dict[str, bool],
233
+ row_counts: dict[str, int],
234
+ output_format: str = "csv",
235
+ partitioned_by: dict[str, list[str]] | None = None,
236
+ lineage_counts: dict[str, LineageCounters] | None = None,
237
+ capabilities: CapabilityProfile | None = None,
238
+ supplements: list[dict] | None = None,
239
+ ) -> tuple[dict, dict, dict[str, str]]:
240
+ """Build the manifest entries + manifest from per-dataset provenance and row counts.
241
+
242
+ Shared by the eager (:func:`convert_to_focus_1_4`) and streaming (``convert_files``) paths
243
+ so both emit an identical manifest for the same input. ``output_format`` selects the output
244
+ filename extension (``csv`` default, or ``parquet``); ``partitioned_by`` maps a dataset to
245
+ the Parquet partition columns, making its output a directory. ``lineage_counts`` maps a
246
+ dataset to its per-value :class:`LineageCounters` (surfaced as ``lineage_summary``).
247
+ Returns ``(entries, manifest, produced_output_files)`` where the last maps each produced
248
+ dataset to its output filename.
249
+ """
250
+ partitioned_by = partitioned_by or {}
251
+ lineage_counts = lineage_counts or {}
252
+ from focus_data_toolkit import __version__
253
+
254
+ model = load_model()
255
+ entries: dict[str, dict] = {}
256
+ produced_output_files: dict[str, str] = {}
257
+ for name in FOCUS_1_4_DATASETS:
258
+ prov = provenance[name]
259
+ cols = model["datasets"][name]["columns"]
260
+
261
+ if not source_available[name]:
262
+ entries[name] = manifest_mod.dataset_entry(
263
+ status=manifest_mod.NOT_PRODUCED,
264
+ conformance=manifest_mod.CONF_INCOMPLETE,
265
+ provenance=prov,
266
+ reason="no source dataset available for this FOCUS 1.4 dataset",
267
+ )
268
+ continue
269
+
270
+ blockers = strict_blockers(prov, cols)
271
+ if blockers and not synthetic:
272
+ entries[name] = manifest_mod.dataset_entry(
273
+ status=manifest_mod.NOT_PRODUCED,
274
+ conformance=manifest_mod.CONF_INCOMPLETE,
275
+ provenance=prov,
276
+ reason="Mandatory provider-issued fields unavailable from Cost and Usage",
277
+ blocking_columns=blockers,
278
+ )
279
+ continue
280
+
281
+ count = row_counts.get(name) or 0
282
+ if not count:
283
+ entries[name] = manifest_mod.dataset_entry(
284
+ status=manifest_mod.NOT_PRODUCED,
285
+ conformance=manifest_mod.CONF_INCOMPLETE,
286
+ provenance=prov,
287
+ reason="source rows yield no derivable rows for this dataset",
288
+ )
289
+ continue
290
+
291
+ assumed = has_assumptions(prov) if synthetic else bool(blockers)
292
+ status = manifest_mod.PRODUCED_SYNTHETIC if assumed else manifest_mod.PRODUCED
293
+ conformance = manifest_mod.CONF_SYNTHETIC if assumed else manifest_mod.CONF_NOT_VALIDATED
294
+ parts = partitioned_by.get(name)
295
+ output_file = output_filename_for(
296
+ name, synthetic_prefix=assumed, output_format=output_format, partitioned=bool(parts)
297
+ )
298
+ produced_output_files[name] = output_file
299
+ counters = lineage_counts.get(name)
300
+ entries[name] = manifest_mod.dataset_entry(
301
+ status=status,
302
+ conformance=conformance,
303
+ provenance=prov,
304
+ row_count=count,
305
+ output_file=output_file,
306
+ partitioned_by=parts,
307
+ lineage_summary=counters.summary() if counters else None,
308
+ )
309
+
310
+ manifest = manifest_mod.build_manifest(
311
+ tool_version=__version__,
312
+ source_version=version,
313
+ mode=mode.value,
314
+ datasets=entries,
315
+ detection=detection.as_dict(),
316
+ contexts=contexts,
317
+ diagnostics=[d.as_dict() for d in diagnostics],
318
+ capability_profile=(capabilities or CapabilityProfile.none()).as_dict(),
319
+ supplements=supplements,
320
+ )
321
+ return entries, manifest, produced_output_files
322
+
323
+
324
+ def convert_to_focus_1_4(
325
+ cau_rows: list[dict[str, str]],
326
+ cc_rows: list[dict[str, str]] | None = None,
327
+ *,
328
+ source_version: str | None = None,
329
+ source_dataset: str | None = None,
330
+ mode: Mode | str = Mode.STRICT,
331
+ validate: bool = True,
332
+ capabilities: CapabilityProfile | None = None,
333
+ supplements: SupplementBundle | None = None,
334
+ ) -> ConversionResult:
335
+ """Convert FOCUS 1.2/1.3 rows into the FOCUS 1.4 datasets for the given ``mode``.
336
+
337
+ ``cau_rows`` is a FOCUS 1.2 or 1.3 Cost and Usage table; ``cc_rows`` is the optional
338
+ FOCUS 1.3 Contract Commitment table. ``source_version`` / ``source_dataset`` force schema
339
+ detection. ``supplements`` is an optional loaded supplement bundle: supplied facts are
340
+ validated against the source, applied with ``ENRICHED`` lineage, and — at full coverage —
341
+ let **strict** mode produce the derived datasets factually. Returns a
342
+ :class:`ConversionResult` carrying the produced datasets, per-column provenance, the
343
+ detected schema, a per-row context summary, diagnostics, a manifest and (when
344
+ ``validate``) lint reports.
345
+ """
346
+ if not cau_rows:
347
+ raise ConversionError("no Cost and Usage rows to convert")
348
+ mode = Mode(mode)
349
+ version, detection = _resolve_source_version(
350
+ cau_rows[0].keys(), source_version=source_version, source_dataset=source_dataset, mode=mode
351
+ )
352
+ synthetic = mode is Mode.SYNTHETIC
353
+ source_cols = set(cau_rows[0].keys())
354
+
355
+ # Provider/issuer context is derived from the whole source, never the first row. A single
356
+ # representative is needed only to enrich synthetic Contract Commitment (whose 1.3 source
357
+ # carries no provider); ambiguity is surfaced as a diagnostic, never resolved silently.
358
+ contexts = describe_source_contexts(cau_rows, version)
359
+ provider_ctx, provider_ambiguous = representative_provider(cau_rows, version)
360
+ issuers = sorted(
361
+ {(r.get("InvoiceIssuerName") or "").strip() for r in cau_rows}
362
+ - {""}
363
+ )
364
+ issuer = issuers[0] if issuers else provider_ctx.service_provider_name
365
+ diagnostics: list[Diagnostic] = []
366
+
367
+ # Supplements: validate against this exact source before any use. ERRORs block the
368
+ # conversion outright — a supplement that does not describe this source is never
369
+ # partially applied.
370
+ supp_keys: SourceKeySets | None = None
371
+ if supplements:
372
+ supp_keys = source_key_sets(cau_rows, cc_rows)
373
+ supp_diags = validate_supplements(supplements, supp_keys)
374
+ diagnostics.extend(supp_diags)
375
+ errors = [d for d in supp_diags if d.severity is Severity.ERROR]
376
+ if errors:
377
+ raise ConversionError(
378
+ f"{len(errors)} supplement validation error(s); first: "
379
+ f"[{errors[0].code}] {errors[0].message}"
380
+ )
381
+
382
+ # Derived-dataset builders. Historically synthetic-only (Billing Period / Invoice
383
+ # Detail / Contract Commitment are never strictly producible from a Cost-and-Usage
384
+ # source alone); with supplements they also run in strict mode — whether each
385
+ # dataset is then actually produced is decided by its (supplemented) provenance.
386
+ if synthetic or supplements:
387
+ invoice_rows, id_mapping = build_invoice_details(cau_rows)
388
+ billing_rows = build_billing_periods(cau_rows)
389
+ if cc_rows:
390
+ commitment_rows = convert_contract_commitment(
391
+ cc_rows,
392
+ service_provider_name=provider_ctx.service_provider_name,
393
+ invoice_issuer_name=issuer,
394
+ diagnostics=diagnostics,
395
+ )
396
+ if provider_ambiguous:
397
+ diagnostics.append(
398
+ Diagnostic(
399
+ code="FDT-CTX-001",
400
+ severity=Severity.WARNING,
401
+ message="source carries multiple provider contexts; a representative "
402
+ "was chosen to enrich synthetic Contract Commitment",
403
+ datasets=("Contract Commitment",),
404
+ context={"chosen_service_provider": provider_ctx.service_provider_name},
405
+ )
406
+ )
407
+ if len(issuers) > 1:
408
+ diagnostics.append(
409
+ Diagnostic(
410
+ code="FDT-CTX-002",
411
+ severity=Severity.WARNING,
412
+ message="source carries multiple invoice issuers; a representative was "
413
+ "chosen to enrich synthetic Contract Commitment",
414
+ datasets=("Contract Commitment",),
415
+ context={"chosen_invoice_issuer": issuer},
416
+ )
417
+ )
418
+ else:
419
+ commitment_rows = None
420
+ else:
421
+ invoice_rows, id_mapping, billing_rows, commitment_rows = None, {}, None, None
422
+
423
+ # Apply supplements to the derived datasets (ENRICHED lineage, per-value counters).
424
+ bp_prov: dict[str, ColumnRule] = BILLING_PERIOD_PROVENANCE
425
+ invd_prov: dict[str, ColumnRule] = INVOICE_DETAIL_PROVENANCE
426
+ cc_prov: dict[str, ColumnRule] = CONTRACT_COMMITMENT_PROVENANCE
427
+ lineage_counts: dict[str, LineageCounters] = {}
428
+ if supplements and supp_keys is not None:
429
+ if billing_rows is not None:
430
+ applied = apply_billing_periods(
431
+ billing_rows, supplements, supp_keys, bp_prov, synthetic=synthetic
432
+ )
433
+ billing_rows, bp_prov = applied.rows, applied.provenance
434
+ lineage_counts["Billing Period"] = applied.counters
435
+ if invoice_rows is not None:
436
+ applied, id_mapping = apply_invoice_details(
437
+ invoice_rows, id_mapping, supplements, supp_keys, invd_prov,
438
+ synthetic=synthetic,
439
+ )
440
+ invoice_rows, invd_prov = applied.rows, applied.provenance
441
+ lineage_counts["Invoice Detail"] = applied.counters
442
+ if commitment_rows is not None:
443
+ applied = apply_contract_commitments(
444
+ commitment_rows, supplements, supp_keys, cc_prov, synthetic=synthetic
445
+ )
446
+ commitment_rows, cc_prov = applied.rows, applied.provenance
447
+ lineage_counts["Contract Commitment"] = applied.counters
448
+ # In strict mode the Cost and Usage back-link only exists when Invoice Detail is
449
+ # actually produced (its supplemented provenance clears every blocker).
450
+ if not synthetic and strict_blockers(
451
+ invd_prov, load_model()["datasets"]["Invoice Detail"]["columns"]
452
+ ):
453
+ id_mapping = {}
454
+
455
+ linked = bool(id_mapping)
456
+ cu_counters = LineageCounters()
457
+ cu_rows = convert_cost_and_usage(
458
+ cau_rows, version, invoice_detail_ids=id_mapping, counters=cu_counters
459
+ )
460
+ lineage_counts["Cost and Usage"] = cu_counters
461
+ cu_prov = cost_and_usage_provenance(source_cols, version, invoice_detail_linked=linked)
462
+ if supplements and supp_keys is not None and linked:
463
+ # Real, fully-covering issuer-assigned ids from an invoice_line supplement make
464
+ # the Cost and Usage back-link factual instead of a locally generated id.
465
+ line_table = supplements.get("invoice_line")
466
+ if line_table is not None and "InvoiceDetailId" in line_table.fact_columns:
467
+ id_cov = coverage(line_table, supp_keys.invoice_grains)["InvoiceDetailId"]
468
+ if id_cov.complete:
469
+ cu_prov["InvoiceDetailId"] = ColumnRule(
470
+ Lineage.ENRICHED,
471
+ line_table.source_for("InvoiceDetailId"),
472
+ note="issuer-assigned back-link to Invoice Detail",
473
+ )
474
+
475
+ provenance: dict[str, dict[str, ColumnRule]] = {
476
+ "Cost and Usage": cu_prov,
477
+ "Contract Commitment": cc_prov,
478
+ "Billing Period": bp_prov,
479
+ "Invoice Detail": invd_prov,
480
+ }
481
+ built_rows: dict[str, list[dict[str, str]] | None] = {
482
+ "Cost and Usage": cu_rows,
483
+ "Contract Commitment": commitment_rows,
484
+ "Billing Period": billing_rows,
485
+ "Invoice Detail": invoice_rows,
486
+ }
487
+ source_available = {
488
+ "Cost and Usage": True,
489
+ "Contract Commitment": bool(cc_rows), # None or empty -> no source dataset
490
+ "Billing Period": True,
491
+ "Invoice Detail": True,
492
+ }
493
+ row_counts = {name: len(built_rows[name] or []) for name in FOCUS_1_4_DATASETS}
494
+
495
+ _entries, manifest, produced_output_files = assemble_manifest(
496
+ version=version,
497
+ mode=mode,
498
+ synthetic=synthetic,
499
+ detection=detection,
500
+ contexts=contexts,
501
+ diagnostics=diagnostics,
502
+ provenance=provenance,
503
+ source_available=source_available,
504
+ row_counts=row_counts,
505
+ lineage_counts=lineage_counts,
506
+ capabilities=capabilities,
507
+ supplements=supplements.manifest_entries() if supplements else None,
508
+ )
509
+ produced: dict[str, list[dict[str, str]]] = {
510
+ name: built_rows[name] or [] for name in produced_output_files
511
+ }
512
+
513
+ result = ConversionResult(
514
+ source_version=version,
515
+ mode=mode,
516
+ datasets=produced,
517
+ provenance=provenance,
518
+ manifest=manifest,
519
+ detection=detection,
520
+ contexts=contexts,
521
+ diagnostics=diagnostics,
522
+ )
523
+ if validate:
524
+ for name, rows in produced.items():
525
+ report = lint_focus_1_4_structure(name, rows, profile=capabilities)
526
+ result.reports[name] = report
527
+ entry = result.manifest["datasets"][name]
528
+ # Only a factual dataset advertises a lint conclusion; set it now that the
529
+ # lint has actually run (synthetic entries keep their SYNTHETIC label).
530
+ if entry["conformance"] == manifest_mod.CONF_NOT_VALIDATED:
531
+ entry["conformance"] = (
532
+ manifest_mod.CONF_STRUCTURAL_LINT if report.ok
533
+ else manifest_mod.CONF_LINT_FAILED
534
+ )
535
+ # Cross-dataset bundle validation: recorded in the manifest exactly as the
536
+ # streaming path records it, so both paths render identical manifests. The
537
+ # publication *gate* itself is enforced by write_result.
538
+ from focus_data_toolkit.validate.bundle import validate_dataset_bundle
539
+
540
+ result.bundle_report = validate_dataset_bundle(produced)
541
+ result.manifest["bundle_validation"] = result.bundle_report.as_dict()
542
+ return result
543
+
544
+
545
+ def read_csv_rows(path: str | Path) -> list[dict[str, str]]:
546
+ """Read a CSV file into a list of dict rows (all values as strings)."""
547
+ with open(path, newline="", encoding="utf-8") as fh:
548
+ return list(csv.DictReader(fh))
549
+
550
+
551
+ def rows_to_csv_bytes(rows: list[dict[str, str]]) -> bytes:
552
+ """Serialize dict rows to CSV bytes (column order taken from the first row)."""
553
+ if not rows:
554
+ return b""
555
+ buf = io.StringIO()
556
+ writer = csv.DictWriter(buf, fieldnames=list(rows[0].keys()))
557
+ writer.writeheader()
558
+ writer.writerows(rows)
559
+ return buf.getvalue().encode("utf-8")
560
+
561
+
562
+ RUN_SIDECAR_FILENAME = "_run.json"
563
+ SHA256SUMS_FILENAME = "SHA256SUMS"
564
+
565
+
566
+ def _run_metadata(
567
+ result: ConversionResult,
568
+ checksums: dict[str, str],
569
+ sizes: dict[str, int],
570
+ run_id: str,
571
+ tool_version: str,
572
+ generated_at: str,
573
+ ) -> dict:
574
+ """Operational metadata sidecar — kept OUT of the deterministic business manifest.
575
+
576
+ Carries the run id, wall-clock timestamp and per-file checksums/sizes/row-counts, so the
577
+ business datasets and manifest stay byte-reproducible while operational facts are recorded.
578
+ """
579
+ file_to_dataset = {result.output_filename(name): name for name in result.datasets}
580
+ files = []
581
+ for filename in sorted(checksums):
582
+ dataset = file_to_dataset.get(filename)
583
+ entry = result.manifest["datasets"].get(dataset, {}) if dataset else {}
584
+ files.append(
585
+ {
586
+ "name": filename,
587
+ "dataset": dataset,
588
+ "format": "csv",
589
+ "row_count": len(result.datasets.get(dataset, [])) if dataset else None,
590
+ "size_bytes": sizes.get(filename),
591
+ "sha256": checksums[filename],
592
+ "status": entry.get("status"),
593
+ "conformance": entry.get("conformance"),
594
+ }
595
+ )
596
+ return {
597
+ "run_id": run_id,
598
+ "generated_at": generated_at,
599
+ "toolkit_version": tool_version,
600
+ "mode": result.mode.value,
601
+ "source_version": result.source_version,
602
+ "manifest": manifest_mod.MANIFEST_FILENAME,
603
+ "files": files,
604
+ }
605
+
606
+
607
+ def write_result(
608
+ result: ConversionResult,
609
+ out_dir: str | Path,
610
+ *,
611
+ on_exists: OnExists | str = OnExists.REFUSE,
612
+ keep_temp: bool = False,
613
+ require_valid: bool = True,
614
+ validate_bundle: bool = True,
615
+ ) -> list[Path]:
616
+ """Write every produced dataset plus the manifest to ``out_dir`` **atomically**.
617
+
618
+ Files are staged in a temporary directory on the same filesystem; only after mandatory
619
+ validation passes and the manifest, checksums and operational sidecar are written is the
620
+ directory published with a single atomic rename. On any error the staging directory is
621
+ removed and ``out_dir`` is left untouched (existing results are never partially clobbered).
622
+
623
+ ``on_exists`` chooses the policy when ``out_dir`` already exists (refuse / replace /
624
+ version). ``require_valid`` refuses to publish when the built-in lint failed.
625
+ ``validate_bundle`` runs the cross-dataset bundle validation
626
+ (:func:`~focus_data_toolkit.validate.bundle.validate_dataset_bundle`) as a publication
627
+ gate — its result is recorded in the manifest and, under ``require_valid``, an ERROR
628
+ blocks publication. Returns the published dataset + manifest paths.
629
+ """
630
+ from focus_data_toolkit import __version__
631
+ from focus_data_toolkit.validate.bundle import validate_dataset_bundle
632
+
633
+ generated_at = datetime.now(UTC).isoformat()
634
+ data_files = [
635
+ (result.output_filename(name), rows_to_csv_bytes(rows))
636
+ for name, rows in result.datasets.items()
637
+ ]
638
+
639
+ with AtomicOutputDir(out_dir, on_exists=on_exists, keep_temp=keep_temp) as out:
640
+ for name, data in data_files:
641
+ out.write_bytes(name, data)
642
+
643
+ # Mandatory validation gate: never publish a lint-failing result.
644
+ if require_valid and result.reports and not result.ok:
645
+ failed = sorted(n for n, r in result.reports.items() if not r.ok)
646
+ raise AtomicWriteError(
647
+ f"lint failed for {failed}; final output not written to {out_dir}"
648
+ )
649
+
650
+ # Cross-dataset publication gate: the produced datasets must agree with each other
651
+ # (referential integrity, billing-period coverage, corrections, allocation) before
652
+ # anything is published. The outcome — or the explicit skip — lands in the manifest.
653
+ if validate_bundle:
654
+ # Always validated fresh at publication time (never a report cached from
655
+ # convert time), so datasets mutated in between cannot slip past the gate.
656
+ bundle_report = validate_dataset_bundle(result.datasets)
657
+ result.bundle_report = bundle_report
658
+ result.manifest["bundle_validation"] = bundle_report.as_dict()
659
+ if require_valid and not bundle_report.ok:
660
+ first = bundle_report.errors[0]
661
+ raise AtomicWriteError(
662
+ f"bundle validation failed ({len(bundle_report.errors)} error(s); "
663
+ f"first: [{first.code}] {first.message}); final output not written "
664
+ f"to {out_dir}"
665
+ )
666
+ else:
667
+ result.manifest["bundle_validation"] = {"skipped": True}
668
+
669
+ checksums = out.checksums()
670
+ manifest_bytes = manifest_mod.render(result.manifest).encode("utf-8")
671
+ sidecar = _run_metadata(
672
+ result, checksums, out.sizes(), out.run_id, __version__, generated_at
673
+ )
674
+ all_sums = dict(checksums)
675
+ all_sums[manifest_mod.MANIFEST_FILENAME] = hashlib.sha256(manifest_bytes).hexdigest()
676
+ final_files = {
677
+ manifest_mod.MANIFEST_FILENAME: manifest_bytes,
678
+ RUN_SIDECAR_FILENAME: (json.dumps(sidecar, indent=2, sort_keys=True) + "\n").encode(),
679
+ SHA256SUMS_FILENAME: sha256sums_text(all_sums).encode("utf-8"),
680
+ }
681
+ target = out.commit(final_files=final_files)
682
+
683
+ written = [target / name for name, _ in data_files]
684
+ written.append(target / manifest_mod.MANIFEST_FILENAME)
685
+ return written
686
+
687
+
688
+ __all__ = [
689
+ "DATASET_FILENAMES",
690
+ "OUTPUT_FORMATS",
691
+ "AtomicWriteError",
692
+ "ConversionCancelled",
693
+ "ConversionError",
694
+ "ConversionResult",
695
+ "DestinationExistsError",
696
+ "OnExists",
697
+ "assemble_manifest",
698
+ "convert_files",
699
+ "convert_to_focus_1_4",
700
+ "detect_focus_version",
701
+ "output_filename_for",
702
+ "read_csv_rows",
703
+ "rows_to_csv_bytes",
704
+ "write_result",
705
+ ]
706
+
707
+ # Imported last (streaming imports names from this module, which are now all defined).
708
+ from focus_data_toolkit.convert.streaming import convert_files # noqa: E402