focus-data-toolkit 0.11.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. focus_data_toolkit/__init__.py +69 -0
  2. focus_data_toolkit/__main__.py +6 -0
  3. focus_data_toolkit/_version.py +8 -0
  4. focus_data_toolkit/cli.py +968 -0
  5. focus_data_toolkit/context/__init__.py +88 -0
  6. focus_data_toolkit/context/billing.py +54 -0
  7. focus_data_toolkit/context/provider.py +90 -0
  8. focus_data_toolkit/convert/__init__.py +708 -0
  9. focus_data_toolkit/convert/billing_period.py +65 -0
  10. focus_data_toolkit/convert/contract_applied.py +235 -0
  11. focus_data_toolkit/convert/contract_commitment.py +182 -0
  12. focus_data_toolkit/convert/cost_and_usage.py +179 -0
  13. focus_data_toolkit/convert/detect.py +39 -0
  14. focus_data_toolkit/convert/invoice_detail.py +199 -0
  15. focus_data_toolkit/convert/streaming.py +1030 -0
  16. focus_data_toolkit/errors.py +145 -0
  17. focus_data_toolkit/focus_json.py +68 -0
  18. focus_data_toolkit/generators/__init__.py +61 -0
  19. focus_data_toolkit/generators/_shim.py +43 -0
  20. focus_data_toolkit/generators/engine/__init__.py +14 -0
  21. focus_data_toolkit/generators/engine/context.py +12 -0
  22. focus_data_toolkit/generators/engine/determinism.py +117 -0
  23. focus_data_toolkit/generators/engine/json_focus.py +63 -0
  24. focus_data_toolkit/generators/engine/ladder.py +71 -0
  25. focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
  26. focus_data_toolkit/generators/engine/serialize.py +151 -0
  27. focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
  28. focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
  29. focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
  30. focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
  31. focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
  32. focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
  33. focus_data_toolkit/generators/providers/__init__.py +29 -0
  34. focus_data_toolkit/generators/providers/aws.py +186 -0
  35. focus_data_toolkit/generators/providers/azure.py +191 -0
  36. focus_data_toolkit/generators/providers/gcp.py +194 -0
  37. focus_data_toolkit/generators/providers/profile.py +123 -0
  38. focus_data_toolkit/generators/scenarios.py +178 -0
  39. focus_data_toolkit/generators/versions/__init__.py +17 -0
  40. focus_data_toolkit/generators/versions/adapter.py +41 -0
  41. focus_data_toolkit/generators/versions/v1_2.py +111 -0
  42. focus_data_toolkit/generators/versions/v1_3.py +154 -0
  43. focus_data_toolkit/io/__init__.py +1 -0
  44. focus_data_toolkit/io/atomic_writer.py +462 -0
  45. focus_data_toolkit/io/csv_io.py +128 -0
  46. focus_data_toolkit/io/parquet_io.py +528 -0
  47. focus_data_toolkit/io/records.py +92 -0
  48. focus_data_toolkit/io/row_source.py +117 -0
  49. focus_data_toolkit/lifecycle.py +342 -0
  50. focus_data_toolkit/manifest.py +114 -0
  51. focus_data_toolkit/model/__init__.py +43 -0
  52. focus_data_toolkit/model/capabilities.py +66 -0
  53. focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
  54. focus_data_toolkit/model/focus_1_4_model.json +1913 -0
  55. focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
  56. focus_data_toolkit/model/focus_json_keys.py +112 -0
  57. focus_data_toolkit/model/iso_4217_currencies.json +23 -0
  58. focus_data_toolkit/model/json_schema_check.py +205 -0
  59. focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
  60. focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
  61. focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
  62. focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
  63. focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
  64. focus_data_toolkit/model/model_provenance.json +58 -0
  65. focus_data_toolkit/model/validator.py +498 -0
  66. focus_data_toolkit/modes.py +18 -0
  67. focus_data_toolkit/official_validator.py +61 -0
  68. focus_data_toolkit/progress.py +89 -0
  69. focus_data_toolkit/provenance.py +106 -0
  70. focus_data_toolkit/py.typed +1 -0
  71. focus_data_toolkit/runtime.py +243 -0
  72. focus_data_toolkit/schema/__init__.py +17 -0
  73. focus_data_toolkit/schema/detection.py +274 -0
  74. focus_data_toolkit/schema/registry.py +127 -0
  75. focus_data_toolkit/storage/__init__.py +1 -0
  76. focus_data_toolkit/storage/external_index.py +99 -0
  77. focus_data_toolkit/storage/spill.py +150 -0
  78. focus_data_toolkit/studio/__init__.py +19 -0
  79. focus_data_toolkit/studio/app.py +467 -0
  80. focus_data_toolkit/studio/config.py +42 -0
  81. focus_data_toolkit/studio/frontend/app.js +214 -0
  82. focus_data_toolkit/studio/frontend/index.html +101 -0
  83. focus_data_toolkit/studio/frontend/style.css +60 -0
  84. focus_data_toolkit/studio/jobs.py +142 -0
  85. focus_data_toolkit/studio/preview.py +32 -0
  86. focus_data_toolkit/studio/security.py +125 -0
  87. focus_data_toolkit/studio/server.py +71 -0
  88. focus_data_toolkit/supplement/__init__.py +50 -0
  89. focus_data_toolkit/supplement/adapters/__init__.py +21 -0
  90. focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
  91. focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
  92. focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
  93. focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
  94. focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
  95. focus_data_toolkit/supplement/adapters/registry.py +215 -0
  96. focus_data_toolkit/supplement/apply.py +318 -0
  97. focus_data_toolkit/supplement/gaps.py +219 -0
  98. focus_data_toolkit/supplement/kinds.py +118 -0
  99. focus_data_toolkit/supplement/loader.py +409 -0
  100. focus_data_toolkit/supplement/spec.py +74 -0
  101. focus_data_toolkit/supplement/validate.py +215 -0
  102. focus_data_toolkit/validate/__init__.py +15 -0
  103. focus_data_toolkit/validate/allocation.py +333 -0
  104. focus_data_toolkit/validate/bundle.py +254 -0
  105. focus_data_toolkit/validate/codes.py +93 -0
  106. focus_data_toolkit/validate/corrections.py +245 -0
  107. focus_data_toolkit/validate/reconciliation.py +98 -0
  108. focus_data_toolkit/validate/referential.py +289 -0
  109. focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
  110. focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
  111. focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
  112. focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
  113. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
  114. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
  115. focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
  116. focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1030 @@
1
+ """Streaming, bounded-memory conversion of large Cost and Usage files.
2
+
3
+ ``convert_files`` reads the Cost and Usage CSV once, writing the converted Cost and Usage
4
+ output incrementally and staging the Invoice Detail aggregation / Billing Period dedup in a
5
+ throwaway SQLite database (:mod:`focus_data_toolkit.storage.external_index`). Memory stays
6
+ bounded — one row plus one running group accumulator plus SQLite's capped page cache — so
7
+ files far larger than RAM convert successfully.
8
+
9
+ Equivalence with the eager :func:`focus_data_toolkit.convert.convert_to_focus_1_4` path is by
10
+ construction: both call the same pure per-row / per-group functions
11
+ (``convert_cost_and_usage_row``, ``invoice_detail_row``, ``billing_period_row``) and the same
12
+ manifest assembler (``assemble_manifest``). The output is byte-identical.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import hashlib
18
+ import json
19
+ import os
20
+ import shutil
21
+ import time
22
+ from collections.abc import Sequence
23
+ from contextlib import ExitStack
24
+ from dataclasses import astuple, replace
25
+ from datetime import UTC, datetime
26
+ from pathlib import Path
27
+
28
+ from focus_data_toolkit import manifest as manifest_mod
29
+ from focus_data_toolkit import runtime
30
+ from focus_data_toolkit.context import (
31
+ billing_context_of_row,
32
+ provider_context_of_row,
33
+ representative_from_contexts,
34
+ summarize_contexts,
35
+ )
36
+ from focus_data_toolkit.context.billing import BillingContext
37
+ from focus_data_toolkit.context.provider import ProviderContext
38
+ from focus_data_toolkit.convert import (
39
+ OUTPUT_FORMATS,
40
+ RUN_SIDECAR_FILENAME,
41
+ SHA256SUMS_FILENAME,
42
+ ConversionCancelled,
43
+ ConversionError,
44
+ assemble_manifest,
45
+ output_filename_for,
46
+ )
47
+ from focus_data_toolkit.convert.billing_period import PROVENANCE as BILLING_PERIOD_PROVENANCE
48
+ from focus_data_toolkit.convert.billing_period import billing_period_row
49
+ from focus_data_toolkit.convert.contract_commitment import (
50
+ PROVENANCE as CONTRACT_COMMITMENT_PROVENANCE,
51
+ )
52
+ from focus_data_toolkit.convert.contract_commitment import convert_contract_commitment
53
+ from focus_data_toolkit.convert.cost_and_usage import (
54
+ convert_cost_and_usage_row,
55
+ cost_and_usage_provenance,
56
+ )
57
+ from focus_data_toolkit.convert.invoice_detail import PROVENANCE as INVOICE_DETAIL_PROVENANCE
58
+ from focus_data_toolkit.convert.invoice_detail import (
59
+ emitted_invoice_detail_columns,
60
+ invoice_detail_grain_key,
61
+ invoice_detail_id,
62
+ invoice_detail_row,
63
+ )
64
+ from focus_data_toolkit.errors import Diagnostic, Severity
65
+ from focus_data_toolkit.io.atomic_writer import AtomicOutputDir, AtomicWriteError, OnExists
66
+ from focus_data_toolkit.io.csv_io import CsvRowReader, open_csv_writer
67
+ from focus_data_toolkit.io.records import DatasetSchema
68
+ from focus_data_toolkit.io.row_source import open_row_source, read_source_rows
69
+ from focus_data_toolkit.model import dataset_columns, load_model
70
+ from focus_data_toolkit.model.capabilities import CapabilityProfile
71
+ from focus_data_toolkit.modes import Mode
72
+ from focus_data_toolkit.progress import CancelPredicate, ProgressCallback, ProgressEvent
73
+ from focus_data_toolkit.provenance import (
74
+ ColumnRule,
75
+ Lineage,
76
+ LineageCounters,
77
+ has_assumptions,
78
+ strict_blockers,
79
+ )
80
+ from focus_data_toolkit.supplement.apply import (
81
+ apply_billing_periods,
82
+ apply_contract_commitments,
83
+ apply_invoice_details,
84
+ flip_enriched_rules,
85
+ )
86
+ from focus_data_toolkit.supplement.loader import SupplementBundle
87
+ from focus_data_toolkit.supplement.validate import (
88
+ SourceKeySets,
89
+ coverage,
90
+ validate_supplements,
91
+ )
92
+
93
+ # Rows per chunk when linting a produced file (bounded memory; the linter has no cross-row
94
+ # state, so chunked linting equals whole-file linting for the fixed model column set).
95
+ _LINT_CHUNK = 5000
96
+
97
+ _INDEX_DB = "_index.sqlite"
98
+ _BUNDLE_DB = "_bundle_index.sqlite"
99
+
100
+ # Cancel is checked, and a ProgressEvent considered, on this row cadence; the emitted events
101
+ # are further time-throttled (below) so a fast conversion cannot flood the callback. Capped so
102
+ # a large ``progress_interval`` never makes cancellation unresponsive.
103
+ _PROGRESS_STEP_MAX = 5000
104
+ _PROGRESS_MIN_SECONDS = 0.5
105
+
106
+
107
+ def _unlink_db(path: Path) -> None:
108
+ """Remove a scratch SQLite DB and any sidecar (-wal/-shm/-journal), ignoring absence."""
109
+ for suffix in ("", "-wal", "-shm", "-journal"):
110
+ try:
111
+ Path(str(path) + suffix).unlink()
112
+ except OSError:
113
+ pass # best-effort cleanup: a missing/locked sidecar is not an error
114
+
115
+
116
+ def _progress_totals(reader, progress) -> tuple[str, int | None]:
117
+ """(unit, total) for a source reader, computed **once** — bytes for CSV, rows for Parquet.
118
+
119
+ Returns ``("rows", None)`` when no progress callback is set, so a partitioned-Parquet source
120
+ is never row-counted (``count_rows()``) purely for metrics that no one consumes.
121
+ """
122
+ if progress is None:
123
+ return "rows", None
124
+ bytes_total = getattr(reader, "bytes_total", None)
125
+ if bytes_total:
126
+ return "bytes", bytes_total
127
+ return "rows", getattr(reader, "expected_rows", None)
128
+
129
+
130
+ def _source_completed(reader, unit: str, count: int) -> int:
131
+ """Completed amount for a source reader given the chosen unit (live byte cursor or row count)."""
132
+ if unit == "bytes":
133
+ return getattr(reader, "bytes_read", None) or count
134
+ return count
135
+
136
+
137
+ def _discard_scratch(out, config, name: str, path: Path) -> None:
138
+ """Drop a scratch DB: from the staging dir by name (default) or by path when relocated.
139
+
140
+ A relocated scratch (under ``FOCUS_TOOLKIT_WORK_DIR``) is not inside the atomic staging dir,
141
+ so ``AtomicOutputDir`` never publishes or cleans it — we unlink it explicitly here (and again,
142
+ defensively, on any error via the ExitStack callback).
143
+ """
144
+ if config.work_dir is None:
145
+ out.discard(name) # scratch DB inside staging must never be published
146
+ else:
147
+ _unlink_db(path)
148
+
149
+
150
+ class _StagedRows:
151
+ """Re-iterable row stream over a staged output file (or partition tree).
152
+
153
+ Each ``__iter__`` opens a fresh reader, so the bundle validator can run several
154
+ independent forward passes without ever materialising the dataset.
155
+ """
156
+
157
+ def __init__(
158
+ self, path: Path, output_format: str, dataset: str, partition_by=None,
159
+ *, check=None, guard=None,
160
+ ) -> None:
161
+ self._path = path
162
+ self._output_format = output_format
163
+ self._dataset = dataset
164
+ self._partition_by = partition_by
165
+ self._check = check
166
+ self._guard = guard
167
+
168
+ def __iter__(self):
169
+ reader = _open_reader(
170
+ self._path, self._output_format,
171
+ dataset=self._dataset, partition_by=self._partition_by,
172
+ )
173
+ try:
174
+ n = 0
175
+ for record in reader:
176
+ n += 1
177
+ if n % _PROGRESS_STEP_MAX == 0:
178
+ if self._check is not None:
179
+ self._check()
180
+ if self._guard is not None:
181
+ self._guard() # enforce disk budgets as the bundle spill DB grows
182
+ yield record.values
183
+ finally:
184
+ reader.close()
185
+
186
+
187
+ def _dataset_is_assumed(name: str, provenance: dict, synthetic: bool) -> bool:
188
+ prov = provenance[name]
189
+ cols = load_model()["datasets"][name]["columns"]
190
+ return has_assumptions(prov) if synthetic else bool(strict_blockers(prov, cols))
191
+
192
+
193
+ def _output_filename(
194
+ name: str, provenance: dict, synthetic: bool, output_format: str, partitioned: bool = False
195
+ ) -> str:
196
+ assumed = _dataset_is_assumed(name, provenance, synthetic)
197
+ return output_filename_for(
198
+ name, synthetic_prefix=assumed, output_format=output_format, partitioned=partitioned
199
+ )
200
+
201
+
202
+ def _open_writer(
203
+ path: Path,
204
+ schema: DatasetSchema,
205
+ output_format: str,
206
+ metadata=None,
207
+ *,
208
+ compression: str = "snappy",
209
+ partition_by: tuple[str, ...] | None = None,
210
+ target_file_size: int | None = None,
211
+ ):
212
+ """Open a format-appropriate row writer, returning ``(handle, writer)``.
213
+
214
+ ``partition_by`` (Parquet only) writes a Hive-partitioned dataset directory instead of a
215
+ single file; ``handle`` then equals the writer (it owns its own files).
216
+ """
217
+ if output_format == "parquet":
218
+ from focus_data_toolkit.io.parquet_io import PartitionedParquetWriter, open_parquet_writer
219
+
220
+ if partition_by:
221
+ w = PartitionedParquetWriter(
222
+ path,
223
+ schema,
224
+ partition_by,
225
+ metadata=metadata,
226
+ compression=compression,
227
+ target_file_size=target_file_size,
228
+ )
229
+ return w, w
230
+ return open_parquet_writer(path, schema, metadata=metadata, compression=compression)
231
+ return open_csv_writer(path, schema)
232
+
233
+
234
+ def _open_reader(
235
+ path: Path, output_format: str, *, dataset: str | None = None, partition_by=None
236
+ ):
237
+ """Open a format-appropriate row reader."""
238
+ if output_format == "parquet":
239
+ from focus_data_toolkit.io.parquet_io import ParquetRowReader, PartitionedParquetReader
240
+
241
+ if partition_by:
242
+ return PartitionedParquetReader(path, dataset or "Cost and Usage", partition_by)
243
+ return ParquetRowReader(path, dataset=dataset)
244
+ return CsvRowReader(path)
245
+
246
+
247
+ def _lint_file(
248
+ dataset: str,
249
+ path: Path,
250
+ output_format: str = "csv",
251
+ *,
252
+ partition_by=None,
253
+ capabilities: CapabilityProfile | None = None,
254
+ check=None,
255
+ on_rows=None,
256
+ ):
257
+ """Lint a produced file (or partition tree) in bounded chunks, returning a merged LintReport.
258
+
259
+ ``check`` (a no-arg callable) is invoked per chunk to honour cancellation; ``on_rows`` (an
260
+ ``int -> None`` callable) receives the running row count for progress reporting.
261
+ """
262
+ from focus_data_toolkit.model.validator import (
263
+ _CHECKED_LEVELS,
264
+ LintReport,
265
+ lint_focus_1_4_structure,
266
+ )
267
+
268
+ reader = _open_reader(path, output_format, dataset=dataset, partition_by=partition_by)
269
+ violations: list = []
270
+ total = 0
271
+ levels = _CHECKED_LEVELS
272
+ chunk: list[dict[str, str]] = []
273
+
274
+ def flush(rows: list[dict[str, str]]) -> None:
275
+ nonlocal total, levels
276
+ report = lint_focus_1_4_structure(dataset, rows, profile=capabilities)
277
+ levels = report.levels_checked
278
+ for v in report.violations:
279
+ if v.row_index is None:
280
+ violations.append(v)
281
+ else:
282
+ violations.append(replace(v, row_index=v.row_index + total))
283
+ total += len(rows)
284
+
285
+ try:
286
+ for record in reader:
287
+ chunk.append(record.values)
288
+ if len(chunk) >= _LINT_CHUNK:
289
+ flush(chunk)
290
+ chunk = []
291
+ if check is not None:
292
+ check()
293
+ if on_rows is not None:
294
+ on_rows(total)
295
+ if chunk:
296
+ flush(chunk)
297
+ finally:
298
+ reader.close()
299
+ return LintReport(
300
+ dataset=dataset, row_count=total, violations=tuple(violations), levels_checked=levels
301
+ )
302
+
303
+
304
+ def _validate_parquet_options(
305
+ output_format: str,
306
+ partition_by: tuple[str, ...],
307
+ compression: str,
308
+ target_file_size: int | None,
309
+ ) -> None:
310
+ """Reject partitioning/compression options that don't apply or aren't valid FOCUS keys."""
311
+ if output_format != "parquet":
312
+ if partition_by:
313
+ raise ConversionError("--partition-by requires --output-format parquet")
314
+ return
315
+ from focus_data_toolkit.io.parquet_io import COMPRESSIONS, partitionable_columns
316
+
317
+ if compression not in COMPRESSIONS:
318
+ raise ConversionError(
319
+ f"unsupported compression {compression!r}; choose one of {', '.join(COMPRESSIONS)}"
320
+ )
321
+ if target_file_size is not None and target_file_size <= 0:
322
+ raise ConversionError(
323
+ f"--target-file-size must be positive, got {target_file_size} bytes"
324
+ )
325
+ if partition_by:
326
+ bad = partitionable_columns("Cost and Usage", partition_by)
327
+ if bad:
328
+ raise ConversionError(
329
+ "cannot partition on "
330
+ + ", ".join(bad)
331
+ + ": partition columns must be Cost and Usage String or Date/Time columns "
332
+ "(not measures, JSON, or unknown columns)"
333
+ )
334
+
335
+
336
+ def convert_files(
337
+ cost_and_usage: str | os.PathLike[str],
338
+ out_dir: str | os.PathLike[str],
339
+ *,
340
+ contract_commitment: str | os.PathLike[str] | None = None,
341
+ source_version: str | None = None,
342
+ source_dataset: str | None = None,
343
+ mode: Mode | str = Mode.STRICT,
344
+ validate: bool = True,
345
+ on_exists: OnExists | str = OnExists.REFUSE,
346
+ keep_temp: bool = False,
347
+ output_format: str = "csv",
348
+ partition_by: Sequence[str] | None = None,
349
+ compression: str = "snappy",
350
+ target_file_size: int | None = None,
351
+ capabilities: CapabilityProfile | None = None,
352
+ supplements: SupplementBundle | None = None,
353
+ progress: ProgressCallback | None = None,
354
+ cancel: CancelPredicate | None = None,
355
+ progress_interval: int = 5000,
356
+ ) -> Path:
357
+ """Stream-convert a Cost and Usage file to the FOCUS 1.4 datasets in ``out_dir``.
358
+
359
+ Inputs (``cost_and_usage``, ``contract_commitment``) may each be CSV (gzip ok) or
360
+ Parquet — the format is sniffed per file, so they can be mixed freely.
361
+ The Cost and Usage file is read once (twice with ``supplements``: a cheap key-collection
362
+ pre-pass validates the bundle before anything is staged); Invoice Detail / Billing Period
363
+ aggregation happens on disk (SQLite) so memory stays bounded by the *supplement-scale*
364
+ cardinalities (periods, invoices, invoice lines, commitments), never by the Cost and
365
+ Usage row count. Output is published atomically (nothing appears until validation passes
366
+ and checksums + manifest are written). ``output_format`` is ``csv`` (byte-exact) or
367
+ ``parquet`` (value-exact decimal128; requires the ``[parquet]`` extra).
368
+
369
+ Parquet only: ``partition_by`` writes the Cost and Usage dataset as a Hive-partitioned tree
370
+ on the given low-cardinality String/Date-Time columns; ``compression`` selects the codec; and
371
+ ``target_file_size`` (approximate uncompressed bytes) rolls each partition to a new part file.
372
+ Returns the published path. Supplement handling is shared with the eager path
373
+ (same ``apply_*`` functions), so both produce identical bytes.
374
+
375
+ ``progress`` (an optional callback) receives throttled :class:`~focus_data_toolkit.progress.ProgressEvent`\\ s
376
+ per phase; ``cancel`` (an optional predicate) is checked cooperatively between rows and
377
+ validation passes — when it returns True the conversion raises
378
+ :class:`~focus_data_toolkit.convert.ConversionCancelled` and the atomic staging directory is
379
+ removed, so **nothing partial is ever published**. Both default to ``None`` (unchanged
380
+ behaviour). ``progress_interval`` is the row cadence (capped at 5000) at which cancel is
381
+ checked and progress considered.
382
+ """
383
+ from focus_data_toolkit import __version__
384
+ from focus_data_toolkit.convert import _resolve_source_version
385
+
386
+ if output_format not in OUTPUT_FORMATS:
387
+ raise ConversionError(
388
+ f"unsupported output format {output_format!r}; choose one of {', '.join(OUTPUT_FORMATS)}"
389
+ )
390
+ partition_by = tuple(partition_by or ())
391
+ _validate_parquet_options(output_format, partition_by, compression, target_file_size)
392
+ # Partitioning applies to the (large) Cost and Usage dataset only; the small derived datasets
393
+ # stay single files.
394
+ partition_map: dict[str, tuple[str, ...]] = (
395
+ {"Cost and Usage": partition_by} if partition_by else {}
396
+ )
397
+ mode = Mode(mode)
398
+ synthetic = mode is Mode.SYNTHETIC
399
+ generated_at = datetime.now(UTC).isoformat()
400
+
401
+ # --- runtime disk budgets (two filesystems: work scratch vs output staging) ----------
402
+ config = runtime.RuntimeConfig.from_env()
403
+ config.apply_logging()
404
+ out_parent = Path(out_dir).parent
405
+ work_dir_eff = config.work_dir or out_parent
406
+ scratch_paths: list[Path] = [] # scratch DB files, tracked for MAX_WORK_BYTES accounting
407
+ # Best-effort pre-flight before any staging: fail fast with a structured FDT-IO-005/006
408
+ # diagnostic (exit 5) rather than a raw OSError mid-run.
409
+ runtime.preflight(config, out_parent, [cost_and_usage, contract_commitment])
410
+
411
+ def _resource_guard() -> None:
412
+ scratch_bytes = 0
413
+ for p in scratch_paths:
414
+ try:
415
+ scratch_bytes += p.stat().st_size
416
+ except OSError:
417
+ pass # scratch DB may not exist yet / already discarded — not an error
418
+ runtime.enforce_limits(config, out_parent, work_dir_eff, scratch_bytes)
419
+
420
+ # --- progress + cooperative cancellation (opt-in; no-ops when unset) -----------------
421
+ step = max(1, min(int(progress_interval or _PROGRESS_STEP_MAX), _PROGRESS_STEP_MAX))
422
+ _last_emit_t = 0.0
423
+ _last_emit_completed = 0
424
+ _last_emit_phase: str | None = None
425
+
426
+ def _check() -> None:
427
+ if cancel is not None and cancel():
428
+ raise ConversionCancelled("conversion cancelled")
429
+
430
+ def _emit(
431
+ phase: str,
432
+ completed: int,
433
+ total: int | None = None,
434
+ unit: str = "rows",
435
+ message: str | None = None,
436
+ *,
437
+ force: bool = False,
438
+ ) -> None:
439
+ nonlocal _last_emit_t, _last_emit_completed, _last_emit_phase
440
+ if progress is None:
441
+ return
442
+ now = time.monotonic()
443
+ # Fire on a phase boundary, once the completed count advances by the interval, or at
444
+ # least every _PROGRESS_MIN_SECONDS — so fast runs still show movement and slow phases
445
+ # still tick, without a callback per row.
446
+ due = (
447
+ force
448
+ or phase != _last_emit_phase
449
+ or completed - _last_emit_completed >= progress_interval
450
+ or now - _last_emit_t >= _PROGRESS_MIN_SECONDS
451
+ )
452
+ if not due:
453
+ return
454
+ _last_emit_t = now
455
+ _last_emit_completed = completed
456
+ _last_emit_phase = phase
457
+ try:
458
+ progress(
459
+ ProgressEvent(
460
+ phase=phase, # type: ignore[arg-type]
461
+ completed=completed,
462
+ total=total,
463
+ unit=unit,
464
+ message=message,
465
+ )
466
+ )
467
+ except Exception: # a misbehaving progress sink must never break a conversion
468
+ pass
469
+
470
+ def _meta(dataset: str) -> dict | None:
471
+ if output_format != "parquet":
472
+ return None
473
+ from focus_data_toolkit.io.parquet_io import dataset_metadata
474
+
475
+ # The output is FOCUS 1.4; `version` is the *source* version (1.2/1.3) and belongs in
476
+ # source_version, not target_version, so Parquet-metadata readers are not misled.
477
+ return dataset_metadata(
478
+ dataset,
479
+ target_version="1.4",
480
+ source_version=version,
481
+ mode=mode.value,
482
+ conformance="see manifest",
483
+ tool_version=__version__,
484
+ )
485
+
486
+ reader = open_row_source(cost_and_usage)
487
+ try:
488
+ version, detection = _resolve_source_version(
489
+ reader.source_columns,
490
+ source_version=source_version,
491
+ source_dataset=source_dataset,
492
+ mode=mode,
493
+ )
494
+ except ConversionError:
495
+ reader.close()
496
+ raise
497
+
498
+ cc_rows = read_source_rows(contract_commitment) if contract_commitment else None
499
+ source_cols = set(reader.source_columns)
500
+ diagnostics: list[Diagnostic] = []
501
+
502
+ # Supplements: a cheap pre-pass collects the source's join keys (memory bounded by
503
+ # their distinct counts, i.e. supplement scale) so the bundle is fully validated
504
+ # before anything is staged, and the strict back-link/provenance decisions are made
505
+ # exactly as in the eager path.
506
+ supp_keys: SourceKeySets | None = None
507
+ linked = synthetic
508
+ line_table = supplements.get("invoice_line") if supplements else None
509
+ if supplements:
510
+ supp_keys = SourceKeySets()
511
+ pre = open_row_source(cost_and_usage)
512
+ try:
513
+ pre_unit, pre_total = _progress_totals(pre, progress)
514
+ _emit("READING", 0, pre_total, pre_unit, "collecting source keys", force=True)
515
+ pre_count = 0
516
+ for record in pre:
517
+ supp_keys.observe_cau_row(record.values)
518
+ pre_count += 1
519
+ if pre_count % step == 0:
520
+ _check()
521
+ _resource_guard()
522
+ if progress is not None:
523
+ _emit("READING", _source_completed(pre, pre_unit, pre_count),
524
+ pre_total, pre_unit, "collecting source keys")
525
+ finally:
526
+ pre.close()
527
+ for cc_row in cc_rows or ():
528
+ supp_keys.observe_cc_row(cc_row)
529
+ supp_diags = validate_supplements(supplements, supp_keys)
530
+ diagnostics.extend(supp_diags)
531
+ errors = [d for d in supp_diags if d.severity is Severity.ERROR]
532
+ if errors:
533
+ reader.close()
534
+ raise ConversionError(
535
+ f"{len(errors)} supplement validation error(s); first: "
536
+ f"[{errors[0].code}] {errors[0].message}"
537
+ )
538
+ # Will Invoice Detail be produced? (Rules only — the same _flip_rules the apply
539
+ # step uses, so this matches the post-pass provenance exactly.)
540
+ invd_flipped = flip_enriched_rules(
541
+ INVOICE_DETAIL_PROVENANCE, "Invoice Detail",
542
+ [t for t in (supplements.get("invoice"), line_table) if t is not None],
543
+ supp_keys,
544
+ )
545
+ invd_blocked = bool(
546
+ strict_blockers(invd_flipped, load_model()["datasets"]["Invoice Detail"]["columns"])
547
+ )
548
+ linked = bool(supp_keys.invoice_grains) and (synthetic or not invd_blocked)
549
+
550
+ cu_prov = cost_and_usage_provenance(source_cols, version, invoice_detail_linked=linked)
551
+ if supplements and supp_keys is not None and linked and line_table is not None:
552
+ if "InvoiceDetailId" in line_table.fact_columns:
553
+ id_cov = coverage(line_table, supp_keys.invoice_grains)["InvoiceDetailId"]
554
+ if id_cov.complete:
555
+ cu_prov["InvoiceDetailId"] = ColumnRule(
556
+ Lineage.ENRICHED,
557
+ line_table.source_for("InvoiceDetailId"),
558
+ note="issuer-assigned back-link to Invoice Detail",
559
+ )
560
+ provenance = {
561
+ "Cost and Usage": cu_prov,
562
+ "Contract Commitment": CONTRACT_COMMITMENT_PROVENANCE,
563
+ "Billing Period": BILLING_PERIOD_PROVENANCE,
564
+ "Invoice Detail": INVOICE_DETAIL_PROVENANCE,
565
+ }
566
+ row_counts = dict.fromkeys(load_model()["datasets"], 0)
567
+
568
+ with ExitStack() as stack:
569
+ out = stack.enter_context(
570
+ AtomicOutputDir(out_dir, on_exists=on_exists, keep_temp=keep_temp)
571
+ )
572
+ # Relocated scratch (FOCUS_TOOLKIT_WORK_DIR) lives in a per-run subdirectory so concurrent
573
+ # runs sharing one WORK_DIR never collide; it is outside the atomic staging dir, so we
574
+ # remove it ourselves on every exit path.
575
+ work_run = runtime.work_run_dir(config, out.run_id)
576
+
577
+ def _cleanup_relocated_scratch() -> None:
578
+ if work_run is not None:
579
+ shutil.rmtree(work_run, ignore_errors=True)
580
+
581
+ stack.callback(_cleanup_relocated_scratch)
582
+
583
+ def _scratch_path(name: str) -> Path:
584
+ """Scratch DB location: the per-run WORK_DIR subdir if set, else inside staging."""
585
+ return (work_run / name) if work_run is not None else out.path_for(name)
586
+
587
+ cu_columns = dataset_columns("Cost and Usage")
588
+ cu_partition = partition_map.get("Cost and Usage")
589
+ cu_file = _output_filename(
590
+ "Cost and Usage", provenance, synthetic, output_format, partitioned=bool(cu_partition)
591
+ )
592
+ cu_handle, cu_writer = _open_writer(
593
+ out.path_for(cu_file),
594
+ DatasetSchema("Cost and Usage", cu_columns),
595
+ output_format,
596
+ metadata=_meta("Cost and Usage"),
597
+ compression=compression,
598
+ partition_by=cu_partition,
599
+ target_file_size=target_file_size,
600
+ )
601
+ index_db_path = _scratch_path(_INDEX_DB)
602
+ scratch_paths.append(index_db_path) # tracked for the work budget in both configs
603
+ index = (
604
+ ExternalIndexOpener(index_db_path)
605
+ if (synthetic or supplements)
606
+ else None
607
+ )
608
+ if index is not None:
609
+ # Guarantee the SQLite handle is closed before staging / work-dir cleanup on EVERY
610
+ # exit path (a mid-run cancel or budget abort would otherwise leave it open, and
611
+ # Windows cannot remove an open file). close() is idempotent, so the explicit close
612
+ # on the success path is harmless.
613
+ stack.callback(index.close)
614
+
615
+ provider_seen: dict[tuple[str, str], ProviderContext] = {}
616
+ billing_seen: dict[tuple, BillingContext] = {}
617
+ cu_counters = LineageCounters()
618
+ cu_count = 0
619
+
620
+ tr_unit, tr_total = _progress_totals(reader, progress)
621
+ _emit("TRANSFORMING", 0, tr_total, tr_unit, "converting Cost and Usage", force=True)
622
+ try:
623
+ for record in reader:
624
+ row = record.values
625
+ pctx = provider_context_of_row(row, version)
626
+ provider_seen[(pctx.service_provider_name, pctx.host_provider_name)] = pctx
627
+ bctx = billing_context_of_row(row)
628
+ billing_seen[astuple(bctx)] = bctx
629
+
630
+ grain = invoice_detail_grain_key(row)
631
+ detail_id = ""
632
+ if grain[1] and linked:
633
+ real_id = (
634
+ line_table.value(grain, "InvoiceDetailId")
635
+ if line_table is not None
636
+ else ""
637
+ )
638
+ if real_id:
639
+ detail_id = real_id
640
+ elif synthetic:
641
+ detail_id = invoice_detail_id(grain)
642
+ cu_writer.write(
643
+ convert_cost_and_usage_row(
644
+ row, version, detail_id=detail_id, target=cu_columns,
645
+ counters=cu_counters,
646
+ )
647
+ )
648
+ cu_count += 1
649
+
650
+ if index is not None:
651
+ if grain[1]:
652
+ index.stage_invoice_line(grain, (row.get("BilledCost") or "0"))
653
+ start = (row.get("BillingPeriodStart") or "").strip()
654
+ end = (row.get("BillingPeriodEnd") or "").strip()
655
+ issuer = (row.get("InvoiceIssuerName") or "").strip()
656
+ if start and end:
657
+ index.stage_billing_period(start, end, issuer)
658
+
659
+ if cu_count % step == 0:
660
+ _check()
661
+ _resource_guard()
662
+ if progress is not None:
663
+ _emit("TRANSFORMING", _source_completed(reader, tr_unit, cu_count),
664
+ tr_total, tr_unit, "converting Cost and Usage")
665
+ finally:
666
+ cu_handle.close()
667
+ reader.close()
668
+ if not cu_count:
669
+ # Header-only input: match the eager path rather than publishing a manifest-only
670
+ # directory. Raising inside the context removes the staging dir (nothing published).
671
+ raise ConversionError("no Cost and Usage rows to convert")
672
+ row_counts["Cost and Usage"] = cu_count
673
+ # Enforce budgets once the aggregation scratch is fully written — covers inputs shorter
674
+ # than `step` and any growth after the last in-loop check.
675
+ _resource_guard()
676
+ if cu_partition is not None:
677
+ diagnostics.extend(_partition_diagnostics(cu_partition, cu_writer.partition_count()))
678
+
679
+ staged = {"Cost and Usage": cu_file}
680
+ lineage_counts: dict[str, LineageCounters] = {"Cost and Usage": cu_counters}
681
+
682
+ if index is not None:
683
+ # Invoice Detail: rows come out of the SQLite aggregation (supplement-scale
684
+ # cardinality). With supplements they are materialized and pushed through the
685
+ # same apply function as the eager path, so both stay byte-identical.
686
+ emitted = emitted_invoice_detail_columns()
687
+ _emit("AGGREGATING", 0, None, "rows", "aggregating Invoice Detail", force=True)
688
+ invd_rows = []
689
+ for j, (grain, grp_total) in enumerate(index.finalize_invoice_groups(), 1):
690
+ invd_rows.append(
691
+ invoice_detail_row(grain, grp_total, invoice_detail_id(grain), emitted)
692
+ )
693
+ if j % step == 0:
694
+ _check()
695
+ _emit("AGGREGATING", j, None, "rows", "aggregating Invoice Detail")
696
+ if supplements and supp_keys is not None and invd_rows:
697
+ applied, _mapping = apply_invoice_details(
698
+ invd_rows, {}, supplements, supp_keys, INVOICE_DETAIL_PROVENANCE,
699
+ synthetic=synthetic,
700
+ )
701
+ invd_rows = applied.rows or []
702
+ provenance["Invoice Detail"] = applied.provenance
703
+ lineage_counts["Invoice Detail"] = applied.counters
704
+ id_columns = tuple(invd_rows[0].keys()) if invd_rows else tuple(emitted)
705
+ id_file = _output_filename("Invoice Detail", provenance, synthetic, output_format)
706
+ id_handle, id_writer = _open_writer(
707
+ out.path_for(id_file),
708
+ DatasetSchema("Invoice Detail", id_columns),
709
+ output_format,
710
+ metadata=_meta("Invoice Detail"),
711
+ compression=compression,
712
+ )
713
+ _emit("WRITING", 0, len(invd_rows), "rows", "writing Invoice Detail", force=True)
714
+ for j, invd_row in enumerate(invd_rows, 1):
715
+ id_writer.write(invd_row)
716
+ if j % step == 0:
717
+ _check()
718
+ _emit("WRITING", j, len(invd_rows), "rows", "writing Invoice Detail")
719
+ id_handle.close()
720
+ staged["Invoice Detail"] = id_file
721
+ row_counts["Invoice Detail"] = len(invd_rows)
722
+
723
+ bp_columns = dataset_columns("Billing Period")
724
+ _emit("AGGREGATING", 0, None, "rows", "aggregating Billing Period", force=True)
725
+ bp_rows = []
726
+ for j, (bp_start, bp_end, bp_issuer) in enumerate(index.finalize_billing_periods(), 1):
727
+ bp_rows.append(billing_period_row(bp_start, bp_end, bp_issuer, bp_columns))
728
+ if j % step == 0:
729
+ _check()
730
+ _emit("AGGREGATING", j, None, "rows", "aggregating Billing Period")
731
+ if supplements and supp_keys is not None and bp_rows:
732
+ bp_applied = apply_billing_periods(
733
+ bp_rows, supplements, supp_keys, BILLING_PERIOD_PROVENANCE,
734
+ synthetic=synthetic,
735
+ )
736
+ bp_rows = bp_applied.rows or []
737
+ provenance["Billing Period"] = bp_applied.provenance
738
+ lineage_counts["Billing Period"] = bp_applied.counters
739
+ bp_file = _output_filename("Billing Period", provenance, synthetic, output_format)
740
+ bp_handle, bp_writer = _open_writer(
741
+ out.path_for(bp_file),
742
+ DatasetSchema("Billing Period", bp_columns),
743
+ output_format,
744
+ metadata=_meta("Billing Period"),
745
+ compression=compression,
746
+ )
747
+ _emit("WRITING", 0, len(bp_rows), "rows", "writing Billing Period", force=True)
748
+ for j, bp_row in enumerate(bp_rows, 1):
749
+ bp_writer.write(bp_row)
750
+ if j % step == 0:
751
+ _check()
752
+ _emit("WRITING", j, len(bp_rows), "rows", "writing Billing Period")
753
+ bp_handle.close()
754
+ staged["Billing Period"] = bp_file
755
+ row_counts["Billing Period"] = len(bp_rows)
756
+
757
+ if cc_rows:
758
+ providers = [provider_seen[k] for k in sorted(provider_seen)]
759
+ provider_ctx, provider_ambiguous = representative_from_contexts(providers)
760
+ issuers = sorted(
761
+ {b.invoice_issuer_name for b in billing_seen.values() if b.invoice_issuer_name}
762
+ )
763
+ issuer = issuers[0] if issuers else provider_ctx.service_provider_name
764
+ cc_out = convert_contract_commitment(
765
+ cc_rows,
766
+ service_provider_name=provider_ctx.service_provider_name,
767
+ invoice_issuer_name=issuer,
768
+ diagnostics=diagnostics,
769
+ )
770
+ if supplements and supp_keys is not None and cc_out:
771
+ cc_applied = apply_contract_commitments(
772
+ cc_out, supplements, supp_keys, CONTRACT_COMMITMENT_PROVENANCE,
773
+ synthetic=synthetic,
774
+ )
775
+ cc_out = cc_applied.rows or []
776
+ provenance["Contract Commitment"] = cc_applied.provenance
777
+ lineage_counts["Contract Commitment"] = cc_applied.counters
778
+ cc_columns = dataset_columns("Contract Commitment")
779
+ cc_file = _output_filename(
780
+ "Contract Commitment", provenance, synthetic, output_format
781
+ )
782
+ cc_handle, cc_writer = _open_writer(
783
+ out.path_for(cc_file),
784
+ DatasetSchema("Contract Commitment", cc_columns),
785
+ output_format,
786
+ metadata=_meta("Contract Commitment"),
787
+ compression=compression,
788
+ )
789
+ _emit("WRITING", 0, len(cc_out), "rows", "writing Contract Commitment", force=True)
790
+ for j, r in enumerate(cc_out, 1):
791
+ cc_writer.write(r)
792
+ if j % step == 0:
793
+ _check()
794
+ _emit("WRITING", j, len(cc_out), "rows", "writing Contract Commitment")
795
+ cc_handle.close()
796
+ staged["Contract Commitment"] = cc_file
797
+ row_counts["Contract Commitment"] = len(cc_out)
798
+ diagnostics.extend(_context_diagnostics(provider_ambiguous, provider_ctx, issuers))
799
+
800
+ if index is not None:
801
+ index.close()
802
+ _discard_scratch(out, config, _INDEX_DB, index_db_path)
803
+
804
+ providers = [provider_seen[k] for k in sorted(provider_seen)]
805
+ billing = [billing_seen[k] for k in sorted(billing_seen)]
806
+ contexts = summarize_contexts(providers, billing)
807
+
808
+ source_available = {
809
+ "Cost and Usage": True,
810
+ "Contract Commitment": bool(cc_rows),
811
+ "Billing Period": True,
812
+ "Invoice Detail": True,
813
+ }
814
+ entries, manifest, produced_output_files = assemble_manifest(
815
+ version=version,
816
+ mode=mode,
817
+ synthetic=synthetic,
818
+ detection=detection,
819
+ contexts=contexts,
820
+ diagnostics=diagnostics,
821
+ provenance=provenance,
822
+ source_available=source_available,
823
+ row_counts=row_counts,
824
+ output_format=output_format,
825
+ partitioned_by={k: list(v) for k, v in partition_map.items()},
826
+ lineage_counts=lineage_counts,
827
+ capabilities=capabilities,
828
+ supplements=supplements.manifest_entries() if supplements else None,
829
+ )
830
+
831
+ # Remove any staged file whose dataset turned out NOT produced (e.g. zero derivable rows).
832
+ for name, fname in staged.items():
833
+ if name not in produced_output_files:
834
+ target = out.path_for(fname)
835
+ if name in partition_map and target.is_dir():
836
+ shutil.rmtree(target, ignore_errors=True)
837
+ else:
838
+ target.unlink(missing_ok=True)
839
+
840
+ # Mandatory validation gate (chunked, bounded memory): never publish a lint failure.
841
+ if validate:
842
+ for name, fname in produced_output_files.items():
843
+ _check()
844
+ dataset_total = row_counts.get(name) or 0
845
+ _emit("VALIDATING", 0, dataset_total, "rows", f"linting {name}", force=True)
846
+
847
+ def _lint_rows(done: int, _name: str = name, _total: int = dataset_total) -> None:
848
+ _emit("VALIDATING", done, _total, "rows", f"linting {_name}")
849
+
850
+ report = _lint_file(
851
+ name, out.path_for(fname), output_format,
852
+ partition_by=partition_map.get(name), capabilities=capabilities,
853
+ check=_check, on_rows=_lint_rows,
854
+ )
855
+ entry = manifest["datasets"][name]
856
+ if entry["conformance"] == manifest_mod.CONF_NOT_VALIDATED:
857
+ entry["conformance"] = (
858
+ manifest_mod.CONF_STRUCTURAL_LINT if report.ok
859
+ else manifest_mod.CONF_LINT_FAILED
860
+ )
861
+ if not report.ok:
862
+ raise AtomicWriteError(
863
+ f"lint failed for {name}; final output not written to {out_dir}"
864
+ )
865
+
866
+ # Cross-dataset publication gate: re-read the staged files (bounded memory — each
867
+ # check is an independent forward pass, per-key state spills to a scratch SQLite DB
868
+ # past a threshold) and refuse to publish on any ERROR. The outcome — or the
869
+ # explicit skip — lands in the manifest, exactly as in the eager path.
870
+ if validate:
871
+ from focus_data_toolkit.storage.spill import SpillableIndexPool
872
+ from focus_data_toolkit.validate.bundle import validate_dataset_bundle
873
+
874
+ _check()
875
+ _emit("VALIDATING", 0, None, "rows", "cross-dataset bundle validation", force=True)
876
+ bundle_rows = {
877
+ name: _StagedRows(
878
+ out.path_for(fname), output_format, name,
879
+ partition_by=partition_map.get(name), check=_check, guard=_resource_guard,
880
+ )
881
+ for name, fname in produced_output_files.items()
882
+ }
883
+ spill_db_path = _scratch_path(_BUNDLE_DB)
884
+ scratch_paths.append(spill_db_path) # tracked for the work budget in both configs
885
+ spill = SpillableIndexPool(spill_db_path)
886
+ stack.callback(spill.close) # backstop: close before cleanup on any exit (Windows)
887
+ try:
888
+ bundle_report = validate_dataset_bundle(
889
+ bundle_rows, index_factory=spill.make_map
890
+ )
891
+ finally:
892
+ spill.close()
893
+ # Catch a spill DB that grew past the budget before it is discarded.
894
+ _resource_guard()
895
+ _discard_scratch(out, config, _BUNDLE_DB, spill_db_path)
896
+ manifest["bundle_validation"] = bundle_report.as_dict()
897
+ if not bundle_report.ok:
898
+ first = bundle_report.errors[0]
899
+ raise AtomicWriteError(
900
+ f"bundle validation failed ({len(bundle_report.errors)} error(s); "
901
+ f"first: [{first.code}] {first.message}); final output not written "
902
+ f"to {out_dir}"
903
+ )
904
+ else:
905
+ manifest["bundle_validation"] = {"skipped": True}
906
+
907
+ # Enroll produced files for fsync + checksums: a partitioned dataset is a tree of parts.
908
+ for name, fname in produced_output_files.items():
909
+ if name in partition_map:
910
+ out.add_data_tree(fname)
911
+ else:
912
+ out.add_data_file(fname)
913
+
914
+ # Last cancel + budget check before the atomic publish; the rename itself is fast and is
915
+ # intentionally not interrupted (interrupting it would defeat atomicity).
916
+ _check()
917
+ _resource_guard()
918
+ _emit("PUBLISHING", 0, None, "rows", "writing checksums + manifest", force=True)
919
+ checksums = out.checksums()
920
+ manifest_bytes = manifest_mod.render(manifest).encode("utf-8")
921
+ sidecar = _run_metadata(
922
+ produced_output_files,
923
+ row_counts,
924
+ manifest,
925
+ out.run_id,
926
+ __version__,
927
+ generated_at,
928
+ mode,
929
+ output_format,
930
+ )
931
+ all_sums = dict(checksums)
932
+ all_sums[manifest_mod.MANIFEST_FILENAME] = hashlib.sha256(manifest_bytes).hexdigest()
933
+ from focus_data_toolkit.io.atomic_writer import sha256sums_text
934
+
935
+ final_files = {
936
+ manifest_mod.MANIFEST_FILENAME: manifest_bytes,
937
+ RUN_SIDECAR_FILENAME: (json.dumps(sidecar, indent=2, sort_keys=True) + "\n").encode(),
938
+ SHA256SUMS_FILENAME: sha256sums_text(all_sums).encode("utf-8"),
939
+ }
940
+ return out.commit(final_files=final_files)
941
+
942
+
943
+ def _partition_diagnostics(partition_by: Sequence[str], count: int) -> list[Diagnostic]:
944
+ from focus_data_toolkit.io.parquet_io import PARTITION_WARN_THRESHOLD
945
+
946
+ if count <= PARTITION_WARN_THRESHOLD:
947
+ return []
948
+ return [
949
+ Diagnostic(
950
+ code="FDT-IO-004",
951
+ severity=Severity.WARNING,
952
+ message=f"--partition-by {list(partition_by)} produced {count} partitions; a "
953
+ "high-cardinality key creates many small files — prefer a lower-cardinality key",
954
+ datasets=("Cost and Usage",),
955
+ context={"partitions": str(count)},
956
+ )
957
+ ]
958
+
959
+
960
+ def _context_diagnostics(
961
+ provider_ambiguous: bool, provider_ctx: ProviderContext, issuers: list[str]
962
+ ) -> list[Diagnostic]:
963
+ out: list[Diagnostic] = []
964
+ if provider_ambiguous:
965
+ out.append(
966
+ Diagnostic(
967
+ code="FDT-CTX-001",
968
+ severity=Severity.WARNING,
969
+ message="source carries multiple provider contexts; a representative was chosen "
970
+ "to enrich synthetic Contract Commitment",
971
+ datasets=("Contract Commitment",),
972
+ context={"chosen_service_provider": provider_ctx.service_provider_name},
973
+ )
974
+ )
975
+ if len(issuers) > 1:
976
+ out.append(
977
+ Diagnostic(
978
+ code="FDT-CTX-002",
979
+ severity=Severity.WARNING,
980
+ message="source carries multiple invoice issuers; a representative was chosen to "
981
+ "enrich synthetic Contract Commitment",
982
+ datasets=("Contract Commitment",),
983
+ context={"chosen_invoice_issuer": issuers[0]},
984
+ )
985
+ )
986
+ return out
987
+
988
+
989
+ def _run_metadata(
990
+ produced_output_files: dict[str, str],
991
+ row_counts: dict[str, int],
992
+ manifest: dict,
993
+ run_id: str,
994
+ tool_version: str,
995
+ generated_at: str,
996
+ mode: Mode,
997
+ output_format: str = "csv",
998
+ ) -> dict:
999
+ files = []
1000
+ for name, filename in sorted(produced_output_files.items(), key=lambda kv: kv[1]):
1001
+ entry = manifest["datasets"].get(name, {})
1002
+ files.append(
1003
+ {
1004
+ "name": filename,
1005
+ "dataset": name,
1006
+ "format": output_format,
1007
+ "row_count": row_counts.get(name),
1008
+ "status": entry.get("status"),
1009
+ "conformance": entry.get("conformance"),
1010
+ }
1011
+ )
1012
+ return {
1013
+ "run_id": run_id,
1014
+ "generated_at": generated_at,
1015
+ "toolkit_version": tool_version,
1016
+ "mode": mode.value,
1017
+ "source_version": manifest.get("source_version"),
1018
+ "manifest": manifest_mod.MANIFEST_FILENAME,
1019
+ "files": files,
1020
+ }
1021
+
1022
+
1023
+ def ExternalIndexOpener(db_path: Path): # noqa: N802 - factory reads clearly at call site
1024
+ """Open an :class:`ExternalIndex` (imported lazily to keep the import graph shallow)."""
1025
+ from focus_data_toolkit.storage.external_index import ExternalIndex
1026
+
1027
+ return ExternalIndex(db_path)
1028
+
1029
+
1030
+ __all__ = ["convert_files"]