focus-data-toolkit 0.11.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- focus_data_toolkit/__init__.py +69 -0
- focus_data_toolkit/__main__.py +6 -0
- focus_data_toolkit/_version.py +8 -0
- focus_data_toolkit/cli.py +968 -0
- focus_data_toolkit/context/__init__.py +88 -0
- focus_data_toolkit/context/billing.py +54 -0
- focus_data_toolkit/context/provider.py +90 -0
- focus_data_toolkit/convert/__init__.py +708 -0
- focus_data_toolkit/convert/billing_period.py +65 -0
- focus_data_toolkit/convert/contract_applied.py +235 -0
- focus_data_toolkit/convert/contract_commitment.py +182 -0
- focus_data_toolkit/convert/cost_and_usage.py +179 -0
- focus_data_toolkit/convert/detect.py +39 -0
- focus_data_toolkit/convert/invoice_detail.py +199 -0
- focus_data_toolkit/convert/streaming.py +1030 -0
- focus_data_toolkit/errors.py +145 -0
- focus_data_toolkit/focus_json.py +68 -0
- focus_data_toolkit/generators/__init__.py +61 -0
- focus_data_toolkit/generators/_shim.py +43 -0
- focus_data_toolkit/generators/engine/__init__.py +14 -0
- focus_data_toolkit/generators/engine/context.py +12 -0
- focus_data_toolkit/generators/engine/determinism.py +117 -0
- focus_data_toolkit/generators/engine/json_focus.py +63 -0
- focus_data_toolkit/generators/engine/ladder.py +71 -0
- focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
- focus_data_toolkit/generators/engine/serialize.py +151 -0
- focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
- focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
- focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
- focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
- focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
- focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
- focus_data_toolkit/generators/providers/__init__.py +29 -0
- focus_data_toolkit/generators/providers/aws.py +186 -0
- focus_data_toolkit/generators/providers/azure.py +191 -0
- focus_data_toolkit/generators/providers/gcp.py +194 -0
- focus_data_toolkit/generators/providers/profile.py +123 -0
- focus_data_toolkit/generators/scenarios.py +178 -0
- focus_data_toolkit/generators/versions/__init__.py +17 -0
- focus_data_toolkit/generators/versions/adapter.py +41 -0
- focus_data_toolkit/generators/versions/v1_2.py +111 -0
- focus_data_toolkit/generators/versions/v1_3.py +154 -0
- focus_data_toolkit/io/__init__.py +1 -0
- focus_data_toolkit/io/atomic_writer.py +462 -0
- focus_data_toolkit/io/csv_io.py +128 -0
- focus_data_toolkit/io/parquet_io.py +528 -0
- focus_data_toolkit/io/records.py +92 -0
- focus_data_toolkit/io/row_source.py +117 -0
- focus_data_toolkit/lifecycle.py +342 -0
- focus_data_toolkit/manifest.py +114 -0
- focus_data_toolkit/model/__init__.py +43 -0
- focus_data_toolkit/model/capabilities.py +66 -0
- focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
- focus_data_toolkit/model/focus_1_4_model.json +1913 -0
- focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
- focus_data_toolkit/model/focus_json_keys.py +112 -0
- focus_data_toolkit/model/iso_4217_currencies.json +23 -0
- focus_data_toolkit/model/json_schema_check.py +205 -0
- focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
- focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
- focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
- focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
- focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
- focus_data_toolkit/model/model_provenance.json +58 -0
- focus_data_toolkit/model/validator.py +498 -0
- focus_data_toolkit/modes.py +18 -0
- focus_data_toolkit/official_validator.py +61 -0
- focus_data_toolkit/progress.py +89 -0
- focus_data_toolkit/provenance.py +106 -0
- focus_data_toolkit/py.typed +1 -0
- focus_data_toolkit/runtime.py +243 -0
- focus_data_toolkit/schema/__init__.py +17 -0
- focus_data_toolkit/schema/detection.py +274 -0
- focus_data_toolkit/schema/registry.py +127 -0
- focus_data_toolkit/storage/__init__.py +1 -0
- focus_data_toolkit/storage/external_index.py +99 -0
- focus_data_toolkit/storage/spill.py +150 -0
- focus_data_toolkit/studio/__init__.py +19 -0
- focus_data_toolkit/studio/app.py +467 -0
- focus_data_toolkit/studio/config.py +42 -0
- focus_data_toolkit/studio/frontend/app.js +214 -0
- focus_data_toolkit/studio/frontend/index.html +101 -0
- focus_data_toolkit/studio/frontend/style.css +60 -0
- focus_data_toolkit/studio/jobs.py +142 -0
- focus_data_toolkit/studio/preview.py +32 -0
- focus_data_toolkit/studio/security.py +125 -0
- focus_data_toolkit/studio/server.py +71 -0
- focus_data_toolkit/supplement/__init__.py +50 -0
- focus_data_toolkit/supplement/adapters/__init__.py +21 -0
- focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
- focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
- focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
- focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
- focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
- focus_data_toolkit/supplement/adapters/registry.py +215 -0
- focus_data_toolkit/supplement/apply.py +318 -0
- focus_data_toolkit/supplement/gaps.py +219 -0
- focus_data_toolkit/supplement/kinds.py +118 -0
- focus_data_toolkit/supplement/loader.py +409 -0
- focus_data_toolkit/supplement/spec.py +74 -0
- focus_data_toolkit/supplement/validate.py +215 -0
- focus_data_toolkit/validate/__init__.py +15 -0
- focus_data_toolkit/validate/allocation.py +333 -0
- focus_data_toolkit/validate/bundle.py +254 -0
- focus_data_toolkit/validate/codes.py +93 -0
- focus_data_toolkit/validate/corrections.py +245 -0
- focus_data_toolkit/validate/reconciliation.py +98 -0
- focus_data_toolkit/validate/referential.py +289 -0
- focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
- focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
- focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
- focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
- focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,708 @@
|
|
|
1
|
+
"""Convert FOCUS 1.2/1.3 source data into the four FOCUS 1.4 datasets.
|
|
2
|
+
|
|
3
|
+
Two modes (see :mod:`focus_data_toolkit.modes`):
|
|
4
|
+
|
|
5
|
+
* ``STRICT`` (default) — a canonical FOCUS 1.4 dataset is produced only when every
|
|
6
|
+
Mandatory non-nullable column has a factual lineage (observed / renamed / derived /
|
|
7
|
+
enriched). Datasets that would require assumed provider-issued values are reported
|
|
8
|
+
``NOT_PRODUCED`` in the manifest, never fabricated. In practice only Cost and Usage is
|
|
9
|
+
produced from a Cost-and-Usage source; Billing Period, Invoice Detail and the expanded
|
|
10
|
+
1.4 Contract Commitment require provider billing facts absent from the source.
|
|
11
|
+
* ``SYNTHETIC`` — for demos / tests / learning: assumed values are generated, the affected
|
|
12
|
+
datasets are labelled synthetic in the manifest (and filenames), and the result is never
|
|
13
|
+
presented as fully conformant.
|
|
14
|
+
|
|
15
|
+
Every conversion emits a deterministic manifest recording, per column, how the value was
|
|
16
|
+
obtained.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import csv
|
|
22
|
+
import hashlib
|
|
23
|
+
import io
|
|
24
|
+
import json
|
|
25
|
+
from collections.abc import Iterable
|
|
26
|
+
from dataclasses import dataclass, field
|
|
27
|
+
from datetime import UTC, datetime
|
|
28
|
+
from pathlib import Path
|
|
29
|
+
from typing import TYPE_CHECKING
|
|
30
|
+
|
|
31
|
+
if TYPE_CHECKING: # imported lazily at runtime (validate.bundle imports from this package)
|
|
32
|
+
from focus_data_toolkit.validate.bundle import BundleReport
|
|
33
|
+
|
|
34
|
+
from focus_data_toolkit import manifest as manifest_mod
|
|
35
|
+
from focus_data_toolkit.context import describe_source_contexts, representative_provider
|
|
36
|
+
from focus_data_toolkit.convert.billing_period import PROVENANCE as BILLING_PERIOD_PROVENANCE
|
|
37
|
+
from focus_data_toolkit.convert.billing_period import build_billing_periods
|
|
38
|
+
from focus_data_toolkit.convert.contract_commitment import (
|
|
39
|
+
PROVENANCE as CONTRACT_COMMITMENT_PROVENANCE,
|
|
40
|
+
)
|
|
41
|
+
from focus_data_toolkit.convert.contract_commitment import convert_contract_commitment
|
|
42
|
+
from focus_data_toolkit.convert.cost_and_usage import (
|
|
43
|
+
convert_cost_and_usage,
|
|
44
|
+
cost_and_usage_provenance,
|
|
45
|
+
)
|
|
46
|
+
from focus_data_toolkit.convert.detect import detect_focus_version
|
|
47
|
+
from focus_data_toolkit.convert.invoice_detail import PROVENANCE as INVOICE_DETAIL_PROVENANCE
|
|
48
|
+
from focus_data_toolkit.convert.invoice_detail import build_invoice_details
|
|
49
|
+
from focus_data_toolkit.errors import Diagnostic, Severity
|
|
50
|
+
from focus_data_toolkit.io.atomic_writer import (
|
|
51
|
+
AtomicOutputDir,
|
|
52
|
+
AtomicWriteError,
|
|
53
|
+
DestinationExistsError,
|
|
54
|
+
OnExists,
|
|
55
|
+
sha256sums_text,
|
|
56
|
+
)
|
|
57
|
+
from focus_data_toolkit.model import FOCUS_1_4_DATASETS, load_model
|
|
58
|
+
from focus_data_toolkit.model.capabilities import CapabilityProfile
|
|
59
|
+
from focus_data_toolkit.model.validator import LintReport, lint_focus_1_4_structure
|
|
60
|
+
from focus_data_toolkit.modes import Mode
|
|
61
|
+
from focus_data_toolkit.provenance import (
|
|
62
|
+
ColumnRule,
|
|
63
|
+
Lineage,
|
|
64
|
+
LineageCounters,
|
|
65
|
+
has_assumptions,
|
|
66
|
+
strict_blockers,
|
|
67
|
+
)
|
|
68
|
+
from focus_data_toolkit.schema import registry
|
|
69
|
+
from focus_data_toolkit.schema.detection import SchemaDetectionResult, detect_focus_schema
|
|
70
|
+
from focus_data_toolkit.supplement.apply import (
|
|
71
|
+
apply_billing_periods,
|
|
72
|
+
apply_contract_commitments,
|
|
73
|
+
apply_invoice_details,
|
|
74
|
+
)
|
|
75
|
+
from focus_data_toolkit.supplement.loader import SupplementBundle
|
|
76
|
+
from focus_data_toolkit.supplement.validate import (
|
|
77
|
+
SourceKeySets,
|
|
78
|
+
coverage,
|
|
79
|
+
source_key_sets,
|
|
80
|
+
validate_supplements,
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
# Base output file name per dataset (stable, snake_case). Synthetic datasets are written
|
|
84
|
+
# with a ``synthetic_`` prefix so they are unmistakable on disk.
|
|
85
|
+
DATASET_FILENAMES = {
|
|
86
|
+
"Cost and Usage": "focus_1_4_cost_and_usage.csv",
|
|
87
|
+
"Contract Commitment": "focus_1_4_contract_commitment.csv",
|
|
88
|
+
"Billing Period": "focus_1_4_billing_period.csv",
|
|
89
|
+
"Invoice Detail": "focus_1_4_invoice_detail.csv",
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
# Output formats the pipeline can write; the CSV path stays byte-exact, Parquet is value-exact.
|
|
93
|
+
OUTPUT_FORMATS = ("csv", "parquet")
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def output_filename_for(
|
|
97
|
+
dataset: str, *, synthetic_prefix: bool, output_format: str = "csv", partitioned: bool = False
|
|
98
|
+
) -> str:
|
|
99
|
+
"""The output name of ``dataset`` for a format.
|
|
100
|
+
|
|
101
|
+
CSV/single-file Parquet get a ``.csv``/``.parquet`` file; a partitioned Parquet dataset is a
|
|
102
|
+
*directory* (no extension) holding the Hive partition tree.
|
|
103
|
+
"""
|
|
104
|
+
base = DATASET_FILENAMES[dataset]
|
|
105
|
+
if output_format == "parquet" and base.endswith(".csv"):
|
|
106
|
+
base = base[:-4] + ("" if partitioned else ".parquet")
|
|
107
|
+
return f"synthetic_{base}" if synthetic_prefix else base
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
class ConversionError(ValueError):
|
|
111
|
+
"""Raised when the source cannot be converted."""
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
class ConversionCancelled(ConversionError):
|
|
115
|
+
"""Raised cooperatively when a cancel predicate returns True mid-conversion.
|
|
116
|
+
|
|
117
|
+
Subclasses :class:`ConversionError` so existing ``except ConversionError`` handlers
|
|
118
|
+
still clean up (the atomic staging directory is removed on the way out, so nothing is
|
|
119
|
+
published); the CLI catches it first to report a distinct cancelled exit code.
|
|
120
|
+
"""
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
@dataclass
|
|
124
|
+
class ConversionResult:
|
|
125
|
+
"""Outcome of a 1.x -> 1.4 conversion."""
|
|
126
|
+
|
|
127
|
+
source_version: str
|
|
128
|
+
mode: Mode
|
|
129
|
+
datasets: dict[str, list[dict[str, str]]]
|
|
130
|
+
provenance: dict[str, dict[str, ColumnRule]]
|
|
131
|
+
manifest: dict
|
|
132
|
+
reports: dict[str, LintReport] = field(default_factory=dict)
|
|
133
|
+
detection: SchemaDetectionResult | None = None
|
|
134
|
+
contexts: dict = field(default_factory=dict)
|
|
135
|
+
diagnostics: list[Diagnostic] = field(default_factory=list)
|
|
136
|
+
bundle_report: BundleReport | None = None
|
|
137
|
+
|
|
138
|
+
@property
|
|
139
|
+
def ok(self) -> bool:
|
|
140
|
+
"""All produced datasets passed the structural + semantic lint."""
|
|
141
|
+
return all(r.ok for r in self.reports.values())
|
|
142
|
+
|
|
143
|
+
@property
|
|
144
|
+
def coverage(self) -> tuple[str, ...]:
|
|
145
|
+
"""FOCUS 1.4 datasets actually produced (in canonical order)."""
|
|
146
|
+
return tuple(name for name in FOCUS_1_4_DATASETS if name in self.datasets)
|
|
147
|
+
|
|
148
|
+
@property
|
|
149
|
+
def not_produced(self) -> tuple[str, ...]:
|
|
150
|
+
return tuple(name for name in FOCUS_1_4_DATASETS if name not in self.datasets)
|
|
151
|
+
|
|
152
|
+
@property
|
|
153
|
+
def assumptions_present(self) -> bool:
|
|
154
|
+
return bool(self.manifest["assumptions_present"])
|
|
155
|
+
|
|
156
|
+
def output_filename(self, dataset: str) -> str:
|
|
157
|
+
return self.manifest["datasets"][dataset]["output_file"]
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _resolve_source_version(
|
|
161
|
+
headers: Iterable[str],
|
|
162
|
+
*,
|
|
163
|
+
source_version: str | None,
|
|
164
|
+
source_dataset: str | None,
|
|
165
|
+
mode: Mode,
|
|
166
|
+
) -> tuple[str, SchemaDetectionResult]:
|
|
167
|
+
"""Determine the convertible source version and record the detection decision.
|
|
168
|
+
|
|
169
|
+
``headers`` is the source column set. ``source_version`` / ``source_dataset`` force the
|
|
170
|
+
corresponding dimension. In strict mode an ambiguous or low-confidence detection (that is
|
|
171
|
+
not forced) is refused with a clear error; a forced version incompatible with the header is
|
|
172
|
+
always refused.
|
|
173
|
+
"""
|
|
174
|
+
# A bad --source-version/--source-dataset value raises ValueError from normalisation;
|
|
175
|
+
# surface it as a ConversionError so the CLI exits with the invalid-argument code, not a
|
|
176
|
+
# traceback.
|
|
177
|
+
try:
|
|
178
|
+
detection = detect_focus_schema(headers, dataset=source_dataset, version=source_version)
|
|
179
|
+
except ValueError as exc:
|
|
180
|
+
raise ConversionError(f"invalid --source-version/--source-dataset: {exc}") from exc
|
|
181
|
+
|
|
182
|
+
forced = source_version is not None or source_dataset is not None
|
|
183
|
+
if forced and detection.confidence == "LOW":
|
|
184
|
+
raise ConversionError(
|
|
185
|
+
"forced source schema is incompatible with the header (detected "
|
|
186
|
+
f"{detection.dataset} {detection.detected_version}, confidence LOW): "
|
|
187
|
+
+ "; ".join(detection.notes)
|
|
188
|
+
)
|
|
189
|
+
|
|
190
|
+
if source_version is not None:
|
|
191
|
+
try:
|
|
192
|
+
version = registry.normalize_version(source_version)
|
|
193
|
+
except ValueError as exc:
|
|
194
|
+
raise ConversionError(f"invalid --source-version {source_version!r}: {exc}") from exc
|
|
195
|
+
else:
|
|
196
|
+
if mode is Mode.STRICT and not forced and detection.confidence != "HIGH":
|
|
197
|
+
raise ConversionError(
|
|
198
|
+
"strict mode refuses an ambiguous or low-confidence source schema (detected "
|
|
199
|
+
f"{detection.dataset} {detection.detected_version}, confidence "
|
|
200
|
+
f"{detection.confidence}); force it with --source-version / --source-dataset"
|
|
201
|
+
)
|
|
202
|
+
# detect_focus_version raises a clear ValueError for non-CAU / 1.4 / non-FOCUS headers.
|
|
203
|
+
try:
|
|
204
|
+
version = detect_focus_version(headers)
|
|
205
|
+
except ValueError as exc:
|
|
206
|
+
raise ConversionError(str(exc)) from exc
|
|
207
|
+
|
|
208
|
+
# This converter only accepts a Cost and Usage source. Forcing a version does not bypass
|
|
209
|
+
# this: a Contract Commitment header passed as --cost-and-usage --source-version 1.3 must be
|
|
210
|
+
# rejected, not converted into an empty manifest.
|
|
211
|
+
if detection.dataset != "Cost and Usage":
|
|
212
|
+
raise ConversionError(
|
|
213
|
+
"this converter requires a FOCUS Cost and Usage source; detected "
|
|
214
|
+
f"{detection.dataset or 'no FOCUS dataset'} (confidence {detection.confidence})"
|
|
215
|
+
)
|
|
216
|
+
if version not in ("1.2", "1.3"):
|
|
217
|
+
raise ConversionError(
|
|
218
|
+
f"unsupported source version {version!r}; this tool converts FOCUS 1.2/1.3 -> 1.4"
|
|
219
|
+
)
|
|
220
|
+
return version, detection
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def assemble_manifest(
|
|
224
|
+
*,
|
|
225
|
+
version: str,
|
|
226
|
+
mode: Mode,
|
|
227
|
+
synthetic: bool,
|
|
228
|
+
detection: SchemaDetectionResult,
|
|
229
|
+
contexts: dict,
|
|
230
|
+
diagnostics: list[Diagnostic],
|
|
231
|
+
provenance: dict[str, dict[str, ColumnRule]],
|
|
232
|
+
source_available: dict[str, bool],
|
|
233
|
+
row_counts: dict[str, int],
|
|
234
|
+
output_format: str = "csv",
|
|
235
|
+
partitioned_by: dict[str, list[str]] | None = None,
|
|
236
|
+
lineage_counts: dict[str, LineageCounters] | None = None,
|
|
237
|
+
capabilities: CapabilityProfile | None = None,
|
|
238
|
+
supplements: list[dict] | None = None,
|
|
239
|
+
) -> tuple[dict, dict, dict[str, str]]:
|
|
240
|
+
"""Build the manifest entries + manifest from per-dataset provenance and row counts.
|
|
241
|
+
|
|
242
|
+
Shared by the eager (:func:`convert_to_focus_1_4`) and streaming (``convert_files``) paths
|
|
243
|
+
so both emit an identical manifest for the same input. ``output_format`` selects the output
|
|
244
|
+
filename extension (``csv`` default, or ``parquet``); ``partitioned_by`` maps a dataset to
|
|
245
|
+
the Parquet partition columns, making its output a directory. ``lineage_counts`` maps a
|
|
246
|
+
dataset to its per-value :class:`LineageCounters` (surfaced as ``lineage_summary``).
|
|
247
|
+
Returns ``(entries, manifest, produced_output_files)`` where the last maps each produced
|
|
248
|
+
dataset to its output filename.
|
|
249
|
+
"""
|
|
250
|
+
partitioned_by = partitioned_by or {}
|
|
251
|
+
lineage_counts = lineage_counts or {}
|
|
252
|
+
from focus_data_toolkit import __version__
|
|
253
|
+
|
|
254
|
+
model = load_model()
|
|
255
|
+
entries: dict[str, dict] = {}
|
|
256
|
+
produced_output_files: dict[str, str] = {}
|
|
257
|
+
for name in FOCUS_1_4_DATASETS:
|
|
258
|
+
prov = provenance[name]
|
|
259
|
+
cols = model["datasets"][name]["columns"]
|
|
260
|
+
|
|
261
|
+
if not source_available[name]:
|
|
262
|
+
entries[name] = manifest_mod.dataset_entry(
|
|
263
|
+
status=manifest_mod.NOT_PRODUCED,
|
|
264
|
+
conformance=manifest_mod.CONF_INCOMPLETE,
|
|
265
|
+
provenance=prov,
|
|
266
|
+
reason="no source dataset available for this FOCUS 1.4 dataset",
|
|
267
|
+
)
|
|
268
|
+
continue
|
|
269
|
+
|
|
270
|
+
blockers = strict_blockers(prov, cols)
|
|
271
|
+
if blockers and not synthetic:
|
|
272
|
+
entries[name] = manifest_mod.dataset_entry(
|
|
273
|
+
status=manifest_mod.NOT_PRODUCED,
|
|
274
|
+
conformance=manifest_mod.CONF_INCOMPLETE,
|
|
275
|
+
provenance=prov,
|
|
276
|
+
reason="Mandatory provider-issued fields unavailable from Cost and Usage",
|
|
277
|
+
blocking_columns=blockers,
|
|
278
|
+
)
|
|
279
|
+
continue
|
|
280
|
+
|
|
281
|
+
count = row_counts.get(name) or 0
|
|
282
|
+
if not count:
|
|
283
|
+
entries[name] = manifest_mod.dataset_entry(
|
|
284
|
+
status=manifest_mod.NOT_PRODUCED,
|
|
285
|
+
conformance=manifest_mod.CONF_INCOMPLETE,
|
|
286
|
+
provenance=prov,
|
|
287
|
+
reason="source rows yield no derivable rows for this dataset",
|
|
288
|
+
)
|
|
289
|
+
continue
|
|
290
|
+
|
|
291
|
+
assumed = has_assumptions(prov) if synthetic else bool(blockers)
|
|
292
|
+
status = manifest_mod.PRODUCED_SYNTHETIC if assumed else manifest_mod.PRODUCED
|
|
293
|
+
conformance = manifest_mod.CONF_SYNTHETIC if assumed else manifest_mod.CONF_NOT_VALIDATED
|
|
294
|
+
parts = partitioned_by.get(name)
|
|
295
|
+
output_file = output_filename_for(
|
|
296
|
+
name, synthetic_prefix=assumed, output_format=output_format, partitioned=bool(parts)
|
|
297
|
+
)
|
|
298
|
+
produced_output_files[name] = output_file
|
|
299
|
+
counters = lineage_counts.get(name)
|
|
300
|
+
entries[name] = manifest_mod.dataset_entry(
|
|
301
|
+
status=status,
|
|
302
|
+
conformance=conformance,
|
|
303
|
+
provenance=prov,
|
|
304
|
+
row_count=count,
|
|
305
|
+
output_file=output_file,
|
|
306
|
+
partitioned_by=parts,
|
|
307
|
+
lineage_summary=counters.summary() if counters else None,
|
|
308
|
+
)
|
|
309
|
+
|
|
310
|
+
manifest = manifest_mod.build_manifest(
|
|
311
|
+
tool_version=__version__,
|
|
312
|
+
source_version=version,
|
|
313
|
+
mode=mode.value,
|
|
314
|
+
datasets=entries,
|
|
315
|
+
detection=detection.as_dict(),
|
|
316
|
+
contexts=contexts,
|
|
317
|
+
diagnostics=[d.as_dict() for d in diagnostics],
|
|
318
|
+
capability_profile=(capabilities or CapabilityProfile.none()).as_dict(),
|
|
319
|
+
supplements=supplements,
|
|
320
|
+
)
|
|
321
|
+
return entries, manifest, produced_output_files
|
|
322
|
+
|
|
323
|
+
|
|
324
|
+
def convert_to_focus_1_4(
|
|
325
|
+
cau_rows: list[dict[str, str]],
|
|
326
|
+
cc_rows: list[dict[str, str]] | None = None,
|
|
327
|
+
*,
|
|
328
|
+
source_version: str | None = None,
|
|
329
|
+
source_dataset: str | None = None,
|
|
330
|
+
mode: Mode | str = Mode.STRICT,
|
|
331
|
+
validate: bool = True,
|
|
332
|
+
capabilities: CapabilityProfile | None = None,
|
|
333
|
+
supplements: SupplementBundle | None = None,
|
|
334
|
+
) -> ConversionResult:
|
|
335
|
+
"""Convert FOCUS 1.2/1.3 rows into the FOCUS 1.4 datasets for the given ``mode``.
|
|
336
|
+
|
|
337
|
+
``cau_rows`` is a FOCUS 1.2 or 1.3 Cost and Usage table; ``cc_rows`` is the optional
|
|
338
|
+
FOCUS 1.3 Contract Commitment table. ``source_version`` / ``source_dataset`` force schema
|
|
339
|
+
detection. ``supplements`` is an optional loaded supplement bundle: supplied facts are
|
|
340
|
+
validated against the source, applied with ``ENRICHED`` lineage, and — at full coverage —
|
|
341
|
+
let **strict** mode produce the derived datasets factually. Returns a
|
|
342
|
+
:class:`ConversionResult` carrying the produced datasets, per-column provenance, the
|
|
343
|
+
detected schema, a per-row context summary, diagnostics, a manifest and (when
|
|
344
|
+
``validate``) lint reports.
|
|
345
|
+
"""
|
|
346
|
+
if not cau_rows:
|
|
347
|
+
raise ConversionError("no Cost and Usage rows to convert")
|
|
348
|
+
mode = Mode(mode)
|
|
349
|
+
version, detection = _resolve_source_version(
|
|
350
|
+
cau_rows[0].keys(), source_version=source_version, source_dataset=source_dataset, mode=mode
|
|
351
|
+
)
|
|
352
|
+
synthetic = mode is Mode.SYNTHETIC
|
|
353
|
+
source_cols = set(cau_rows[0].keys())
|
|
354
|
+
|
|
355
|
+
# Provider/issuer context is derived from the whole source, never the first row. A single
|
|
356
|
+
# representative is needed only to enrich synthetic Contract Commitment (whose 1.3 source
|
|
357
|
+
# carries no provider); ambiguity is surfaced as a diagnostic, never resolved silently.
|
|
358
|
+
contexts = describe_source_contexts(cau_rows, version)
|
|
359
|
+
provider_ctx, provider_ambiguous = representative_provider(cau_rows, version)
|
|
360
|
+
issuers = sorted(
|
|
361
|
+
{(r.get("InvoiceIssuerName") or "").strip() for r in cau_rows}
|
|
362
|
+
- {""}
|
|
363
|
+
)
|
|
364
|
+
issuer = issuers[0] if issuers else provider_ctx.service_provider_name
|
|
365
|
+
diagnostics: list[Diagnostic] = []
|
|
366
|
+
|
|
367
|
+
# Supplements: validate against this exact source before any use. ERRORs block the
|
|
368
|
+
# conversion outright — a supplement that does not describe this source is never
|
|
369
|
+
# partially applied.
|
|
370
|
+
supp_keys: SourceKeySets | None = None
|
|
371
|
+
if supplements:
|
|
372
|
+
supp_keys = source_key_sets(cau_rows, cc_rows)
|
|
373
|
+
supp_diags = validate_supplements(supplements, supp_keys)
|
|
374
|
+
diagnostics.extend(supp_diags)
|
|
375
|
+
errors = [d for d in supp_diags if d.severity is Severity.ERROR]
|
|
376
|
+
if errors:
|
|
377
|
+
raise ConversionError(
|
|
378
|
+
f"{len(errors)} supplement validation error(s); first: "
|
|
379
|
+
f"[{errors[0].code}] {errors[0].message}"
|
|
380
|
+
)
|
|
381
|
+
|
|
382
|
+
# Derived-dataset builders. Historically synthetic-only (Billing Period / Invoice
|
|
383
|
+
# Detail / Contract Commitment are never strictly producible from a Cost-and-Usage
|
|
384
|
+
# source alone); with supplements they also run in strict mode — whether each
|
|
385
|
+
# dataset is then actually produced is decided by its (supplemented) provenance.
|
|
386
|
+
if synthetic or supplements:
|
|
387
|
+
invoice_rows, id_mapping = build_invoice_details(cau_rows)
|
|
388
|
+
billing_rows = build_billing_periods(cau_rows)
|
|
389
|
+
if cc_rows:
|
|
390
|
+
commitment_rows = convert_contract_commitment(
|
|
391
|
+
cc_rows,
|
|
392
|
+
service_provider_name=provider_ctx.service_provider_name,
|
|
393
|
+
invoice_issuer_name=issuer,
|
|
394
|
+
diagnostics=diagnostics,
|
|
395
|
+
)
|
|
396
|
+
if provider_ambiguous:
|
|
397
|
+
diagnostics.append(
|
|
398
|
+
Diagnostic(
|
|
399
|
+
code="FDT-CTX-001",
|
|
400
|
+
severity=Severity.WARNING,
|
|
401
|
+
message="source carries multiple provider contexts; a representative "
|
|
402
|
+
"was chosen to enrich synthetic Contract Commitment",
|
|
403
|
+
datasets=("Contract Commitment",),
|
|
404
|
+
context={"chosen_service_provider": provider_ctx.service_provider_name},
|
|
405
|
+
)
|
|
406
|
+
)
|
|
407
|
+
if len(issuers) > 1:
|
|
408
|
+
diagnostics.append(
|
|
409
|
+
Diagnostic(
|
|
410
|
+
code="FDT-CTX-002",
|
|
411
|
+
severity=Severity.WARNING,
|
|
412
|
+
message="source carries multiple invoice issuers; a representative was "
|
|
413
|
+
"chosen to enrich synthetic Contract Commitment",
|
|
414
|
+
datasets=("Contract Commitment",),
|
|
415
|
+
context={"chosen_invoice_issuer": issuer},
|
|
416
|
+
)
|
|
417
|
+
)
|
|
418
|
+
else:
|
|
419
|
+
commitment_rows = None
|
|
420
|
+
else:
|
|
421
|
+
invoice_rows, id_mapping, billing_rows, commitment_rows = None, {}, None, None
|
|
422
|
+
|
|
423
|
+
# Apply supplements to the derived datasets (ENRICHED lineage, per-value counters).
|
|
424
|
+
bp_prov: dict[str, ColumnRule] = BILLING_PERIOD_PROVENANCE
|
|
425
|
+
invd_prov: dict[str, ColumnRule] = INVOICE_DETAIL_PROVENANCE
|
|
426
|
+
cc_prov: dict[str, ColumnRule] = CONTRACT_COMMITMENT_PROVENANCE
|
|
427
|
+
lineage_counts: dict[str, LineageCounters] = {}
|
|
428
|
+
if supplements and supp_keys is not None:
|
|
429
|
+
if billing_rows is not None:
|
|
430
|
+
applied = apply_billing_periods(
|
|
431
|
+
billing_rows, supplements, supp_keys, bp_prov, synthetic=synthetic
|
|
432
|
+
)
|
|
433
|
+
billing_rows, bp_prov = applied.rows, applied.provenance
|
|
434
|
+
lineage_counts["Billing Period"] = applied.counters
|
|
435
|
+
if invoice_rows is not None:
|
|
436
|
+
applied, id_mapping = apply_invoice_details(
|
|
437
|
+
invoice_rows, id_mapping, supplements, supp_keys, invd_prov,
|
|
438
|
+
synthetic=synthetic,
|
|
439
|
+
)
|
|
440
|
+
invoice_rows, invd_prov = applied.rows, applied.provenance
|
|
441
|
+
lineage_counts["Invoice Detail"] = applied.counters
|
|
442
|
+
if commitment_rows is not None:
|
|
443
|
+
applied = apply_contract_commitments(
|
|
444
|
+
commitment_rows, supplements, supp_keys, cc_prov, synthetic=synthetic
|
|
445
|
+
)
|
|
446
|
+
commitment_rows, cc_prov = applied.rows, applied.provenance
|
|
447
|
+
lineage_counts["Contract Commitment"] = applied.counters
|
|
448
|
+
# In strict mode the Cost and Usage back-link only exists when Invoice Detail is
|
|
449
|
+
# actually produced (its supplemented provenance clears every blocker).
|
|
450
|
+
if not synthetic and strict_blockers(
|
|
451
|
+
invd_prov, load_model()["datasets"]["Invoice Detail"]["columns"]
|
|
452
|
+
):
|
|
453
|
+
id_mapping = {}
|
|
454
|
+
|
|
455
|
+
linked = bool(id_mapping)
|
|
456
|
+
cu_counters = LineageCounters()
|
|
457
|
+
cu_rows = convert_cost_and_usage(
|
|
458
|
+
cau_rows, version, invoice_detail_ids=id_mapping, counters=cu_counters
|
|
459
|
+
)
|
|
460
|
+
lineage_counts["Cost and Usage"] = cu_counters
|
|
461
|
+
cu_prov = cost_and_usage_provenance(source_cols, version, invoice_detail_linked=linked)
|
|
462
|
+
if supplements and supp_keys is not None and linked:
|
|
463
|
+
# Real, fully-covering issuer-assigned ids from an invoice_line supplement make
|
|
464
|
+
# the Cost and Usage back-link factual instead of a locally generated id.
|
|
465
|
+
line_table = supplements.get("invoice_line")
|
|
466
|
+
if line_table is not None and "InvoiceDetailId" in line_table.fact_columns:
|
|
467
|
+
id_cov = coverage(line_table, supp_keys.invoice_grains)["InvoiceDetailId"]
|
|
468
|
+
if id_cov.complete:
|
|
469
|
+
cu_prov["InvoiceDetailId"] = ColumnRule(
|
|
470
|
+
Lineage.ENRICHED,
|
|
471
|
+
line_table.source_for("InvoiceDetailId"),
|
|
472
|
+
note="issuer-assigned back-link to Invoice Detail",
|
|
473
|
+
)
|
|
474
|
+
|
|
475
|
+
provenance: dict[str, dict[str, ColumnRule]] = {
|
|
476
|
+
"Cost and Usage": cu_prov,
|
|
477
|
+
"Contract Commitment": cc_prov,
|
|
478
|
+
"Billing Period": bp_prov,
|
|
479
|
+
"Invoice Detail": invd_prov,
|
|
480
|
+
}
|
|
481
|
+
built_rows: dict[str, list[dict[str, str]] | None] = {
|
|
482
|
+
"Cost and Usage": cu_rows,
|
|
483
|
+
"Contract Commitment": commitment_rows,
|
|
484
|
+
"Billing Period": billing_rows,
|
|
485
|
+
"Invoice Detail": invoice_rows,
|
|
486
|
+
}
|
|
487
|
+
source_available = {
|
|
488
|
+
"Cost and Usage": True,
|
|
489
|
+
"Contract Commitment": bool(cc_rows), # None or empty -> no source dataset
|
|
490
|
+
"Billing Period": True,
|
|
491
|
+
"Invoice Detail": True,
|
|
492
|
+
}
|
|
493
|
+
row_counts = {name: len(built_rows[name] or []) for name in FOCUS_1_4_DATASETS}
|
|
494
|
+
|
|
495
|
+
_entries, manifest, produced_output_files = assemble_manifest(
|
|
496
|
+
version=version,
|
|
497
|
+
mode=mode,
|
|
498
|
+
synthetic=synthetic,
|
|
499
|
+
detection=detection,
|
|
500
|
+
contexts=contexts,
|
|
501
|
+
diagnostics=diagnostics,
|
|
502
|
+
provenance=provenance,
|
|
503
|
+
source_available=source_available,
|
|
504
|
+
row_counts=row_counts,
|
|
505
|
+
lineage_counts=lineage_counts,
|
|
506
|
+
capabilities=capabilities,
|
|
507
|
+
supplements=supplements.manifest_entries() if supplements else None,
|
|
508
|
+
)
|
|
509
|
+
produced: dict[str, list[dict[str, str]]] = {
|
|
510
|
+
name: built_rows[name] or [] for name in produced_output_files
|
|
511
|
+
}
|
|
512
|
+
|
|
513
|
+
result = ConversionResult(
|
|
514
|
+
source_version=version,
|
|
515
|
+
mode=mode,
|
|
516
|
+
datasets=produced,
|
|
517
|
+
provenance=provenance,
|
|
518
|
+
manifest=manifest,
|
|
519
|
+
detection=detection,
|
|
520
|
+
contexts=contexts,
|
|
521
|
+
diagnostics=diagnostics,
|
|
522
|
+
)
|
|
523
|
+
if validate:
|
|
524
|
+
for name, rows in produced.items():
|
|
525
|
+
report = lint_focus_1_4_structure(name, rows, profile=capabilities)
|
|
526
|
+
result.reports[name] = report
|
|
527
|
+
entry = result.manifest["datasets"][name]
|
|
528
|
+
# Only a factual dataset advertises a lint conclusion; set it now that the
|
|
529
|
+
# lint has actually run (synthetic entries keep their SYNTHETIC label).
|
|
530
|
+
if entry["conformance"] == manifest_mod.CONF_NOT_VALIDATED:
|
|
531
|
+
entry["conformance"] = (
|
|
532
|
+
manifest_mod.CONF_STRUCTURAL_LINT if report.ok
|
|
533
|
+
else manifest_mod.CONF_LINT_FAILED
|
|
534
|
+
)
|
|
535
|
+
# Cross-dataset bundle validation: recorded in the manifest exactly as the
|
|
536
|
+
# streaming path records it, so both paths render identical manifests. The
|
|
537
|
+
# publication *gate* itself is enforced by write_result.
|
|
538
|
+
from focus_data_toolkit.validate.bundle import validate_dataset_bundle
|
|
539
|
+
|
|
540
|
+
result.bundle_report = validate_dataset_bundle(produced)
|
|
541
|
+
result.manifest["bundle_validation"] = result.bundle_report.as_dict()
|
|
542
|
+
return result
|
|
543
|
+
|
|
544
|
+
|
|
545
|
+
def read_csv_rows(path: str | Path) -> list[dict[str, str]]:
|
|
546
|
+
"""Read a CSV file into a list of dict rows (all values as strings)."""
|
|
547
|
+
with open(path, newline="", encoding="utf-8") as fh:
|
|
548
|
+
return list(csv.DictReader(fh))
|
|
549
|
+
|
|
550
|
+
|
|
551
|
+
def rows_to_csv_bytes(rows: list[dict[str, str]]) -> bytes:
|
|
552
|
+
"""Serialize dict rows to CSV bytes (column order taken from the first row)."""
|
|
553
|
+
if not rows:
|
|
554
|
+
return b""
|
|
555
|
+
buf = io.StringIO()
|
|
556
|
+
writer = csv.DictWriter(buf, fieldnames=list(rows[0].keys()))
|
|
557
|
+
writer.writeheader()
|
|
558
|
+
writer.writerows(rows)
|
|
559
|
+
return buf.getvalue().encode("utf-8")
|
|
560
|
+
|
|
561
|
+
|
|
562
|
+
RUN_SIDECAR_FILENAME = "_run.json"
|
|
563
|
+
SHA256SUMS_FILENAME = "SHA256SUMS"
|
|
564
|
+
|
|
565
|
+
|
|
566
|
+
def _run_metadata(
|
|
567
|
+
result: ConversionResult,
|
|
568
|
+
checksums: dict[str, str],
|
|
569
|
+
sizes: dict[str, int],
|
|
570
|
+
run_id: str,
|
|
571
|
+
tool_version: str,
|
|
572
|
+
generated_at: str,
|
|
573
|
+
) -> dict:
|
|
574
|
+
"""Operational metadata sidecar — kept OUT of the deterministic business manifest.
|
|
575
|
+
|
|
576
|
+
Carries the run id, wall-clock timestamp and per-file checksums/sizes/row-counts, so the
|
|
577
|
+
business datasets and manifest stay byte-reproducible while operational facts are recorded.
|
|
578
|
+
"""
|
|
579
|
+
file_to_dataset = {result.output_filename(name): name for name in result.datasets}
|
|
580
|
+
files = []
|
|
581
|
+
for filename in sorted(checksums):
|
|
582
|
+
dataset = file_to_dataset.get(filename)
|
|
583
|
+
entry = result.manifest["datasets"].get(dataset, {}) if dataset else {}
|
|
584
|
+
files.append(
|
|
585
|
+
{
|
|
586
|
+
"name": filename,
|
|
587
|
+
"dataset": dataset,
|
|
588
|
+
"format": "csv",
|
|
589
|
+
"row_count": len(result.datasets.get(dataset, [])) if dataset else None,
|
|
590
|
+
"size_bytes": sizes.get(filename),
|
|
591
|
+
"sha256": checksums[filename],
|
|
592
|
+
"status": entry.get("status"),
|
|
593
|
+
"conformance": entry.get("conformance"),
|
|
594
|
+
}
|
|
595
|
+
)
|
|
596
|
+
return {
|
|
597
|
+
"run_id": run_id,
|
|
598
|
+
"generated_at": generated_at,
|
|
599
|
+
"toolkit_version": tool_version,
|
|
600
|
+
"mode": result.mode.value,
|
|
601
|
+
"source_version": result.source_version,
|
|
602
|
+
"manifest": manifest_mod.MANIFEST_FILENAME,
|
|
603
|
+
"files": files,
|
|
604
|
+
}
|
|
605
|
+
|
|
606
|
+
|
|
607
|
+
def write_result(
|
|
608
|
+
result: ConversionResult,
|
|
609
|
+
out_dir: str | Path,
|
|
610
|
+
*,
|
|
611
|
+
on_exists: OnExists | str = OnExists.REFUSE,
|
|
612
|
+
keep_temp: bool = False,
|
|
613
|
+
require_valid: bool = True,
|
|
614
|
+
validate_bundle: bool = True,
|
|
615
|
+
) -> list[Path]:
|
|
616
|
+
"""Write every produced dataset plus the manifest to ``out_dir`` **atomically**.
|
|
617
|
+
|
|
618
|
+
Files are staged in a temporary directory on the same filesystem; only after mandatory
|
|
619
|
+
validation passes and the manifest, checksums and operational sidecar are written is the
|
|
620
|
+
directory published with a single atomic rename. On any error the staging directory is
|
|
621
|
+
removed and ``out_dir`` is left untouched (existing results are never partially clobbered).
|
|
622
|
+
|
|
623
|
+
``on_exists`` chooses the policy when ``out_dir`` already exists (refuse / replace /
|
|
624
|
+
version). ``require_valid`` refuses to publish when the built-in lint failed.
|
|
625
|
+
``validate_bundle`` runs the cross-dataset bundle validation
|
|
626
|
+
(:func:`~focus_data_toolkit.validate.bundle.validate_dataset_bundle`) as a publication
|
|
627
|
+
gate — its result is recorded in the manifest and, under ``require_valid``, an ERROR
|
|
628
|
+
blocks publication. Returns the published dataset + manifest paths.
|
|
629
|
+
"""
|
|
630
|
+
from focus_data_toolkit import __version__
|
|
631
|
+
from focus_data_toolkit.validate.bundle import validate_dataset_bundle
|
|
632
|
+
|
|
633
|
+
generated_at = datetime.now(UTC).isoformat()
|
|
634
|
+
data_files = [
|
|
635
|
+
(result.output_filename(name), rows_to_csv_bytes(rows))
|
|
636
|
+
for name, rows in result.datasets.items()
|
|
637
|
+
]
|
|
638
|
+
|
|
639
|
+
with AtomicOutputDir(out_dir, on_exists=on_exists, keep_temp=keep_temp) as out:
|
|
640
|
+
for name, data in data_files:
|
|
641
|
+
out.write_bytes(name, data)
|
|
642
|
+
|
|
643
|
+
# Mandatory validation gate: never publish a lint-failing result.
|
|
644
|
+
if require_valid and result.reports and not result.ok:
|
|
645
|
+
failed = sorted(n for n, r in result.reports.items() if not r.ok)
|
|
646
|
+
raise AtomicWriteError(
|
|
647
|
+
f"lint failed for {failed}; final output not written to {out_dir}"
|
|
648
|
+
)
|
|
649
|
+
|
|
650
|
+
# Cross-dataset publication gate: the produced datasets must agree with each other
|
|
651
|
+
# (referential integrity, billing-period coverage, corrections, allocation) before
|
|
652
|
+
# anything is published. The outcome — or the explicit skip — lands in the manifest.
|
|
653
|
+
if validate_bundle:
|
|
654
|
+
# Always validated fresh at publication time (never a report cached from
|
|
655
|
+
# convert time), so datasets mutated in between cannot slip past the gate.
|
|
656
|
+
bundle_report = validate_dataset_bundle(result.datasets)
|
|
657
|
+
result.bundle_report = bundle_report
|
|
658
|
+
result.manifest["bundle_validation"] = bundle_report.as_dict()
|
|
659
|
+
if require_valid and not bundle_report.ok:
|
|
660
|
+
first = bundle_report.errors[0]
|
|
661
|
+
raise AtomicWriteError(
|
|
662
|
+
f"bundle validation failed ({len(bundle_report.errors)} error(s); "
|
|
663
|
+
f"first: [{first.code}] {first.message}); final output not written "
|
|
664
|
+
f"to {out_dir}"
|
|
665
|
+
)
|
|
666
|
+
else:
|
|
667
|
+
result.manifest["bundle_validation"] = {"skipped": True}
|
|
668
|
+
|
|
669
|
+
checksums = out.checksums()
|
|
670
|
+
manifest_bytes = manifest_mod.render(result.manifest).encode("utf-8")
|
|
671
|
+
sidecar = _run_metadata(
|
|
672
|
+
result, checksums, out.sizes(), out.run_id, __version__, generated_at
|
|
673
|
+
)
|
|
674
|
+
all_sums = dict(checksums)
|
|
675
|
+
all_sums[manifest_mod.MANIFEST_FILENAME] = hashlib.sha256(manifest_bytes).hexdigest()
|
|
676
|
+
final_files = {
|
|
677
|
+
manifest_mod.MANIFEST_FILENAME: manifest_bytes,
|
|
678
|
+
RUN_SIDECAR_FILENAME: (json.dumps(sidecar, indent=2, sort_keys=True) + "\n").encode(),
|
|
679
|
+
SHA256SUMS_FILENAME: sha256sums_text(all_sums).encode("utf-8"),
|
|
680
|
+
}
|
|
681
|
+
target = out.commit(final_files=final_files)
|
|
682
|
+
|
|
683
|
+
written = [target / name for name, _ in data_files]
|
|
684
|
+
written.append(target / manifest_mod.MANIFEST_FILENAME)
|
|
685
|
+
return written
|
|
686
|
+
|
|
687
|
+
|
|
688
|
+
__all__ = [
|
|
689
|
+
"DATASET_FILENAMES",
|
|
690
|
+
"OUTPUT_FORMATS",
|
|
691
|
+
"AtomicWriteError",
|
|
692
|
+
"ConversionCancelled",
|
|
693
|
+
"ConversionError",
|
|
694
|
+
"ConversionResult",
|
|
695
|
+
"DestinationExistsError",
|
|
696
|
+
"OnExists",
|
|
697
|
+
"assemble_manifest",
|
|
698
|
+
"convert_files",
|
|
699
|
+
"convert_to_focus_1_4",
|
|
700
|
+
"detect_focus_version",
|
|
701
|
+
"output_filename_for",
|
|
702
|
+
"read_csv_rows",
|
|
703
|
+
"rows_to_csv_bytes",
|
|
704
|
+
"write_result",
|
|
705
|
+
]
|
|
706
|
+
|
|
707
|
+
# Imported last (streaming imports names from this module, which are now all defined).
|
|
708
|
+
from focus_data_toolkit.convert.streaming import convert_files # noqa: E402
|