focus-data-toolkit 0.11.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- focus_data_toolkit/__init__.py +69 -0
- focus_data_toolkit/__main__.py +6 -0
- focus_data_toolkit/_version.py +8 -0
- focus_data_toolkit/cli.py +968 -0
- focus_data_toolkit/context/__init__.py +88 -0
- focus_data_toolkit/context/billing.py +54 -0
- focus_data_toolkit/context/provider.py +90 -0
- focus_data_toolkit/convert/__init__.py +708 -0
- focus_data_toolkit/convert/billing_period.py +65 -0
- focus_data_toolkit/convert/contract_applied.py +235 -0
- focus_data_toolkit/convert/contract_commitment.py +182 -0
- focus_data_toolkit/convert/cost_and_usage.py +179 -0
- focus_data_toolkit/convert/detect.py +39 -0
- focus_data_toolkit/convert/invoice_detail.py +199 -0
- focus_data_toolkit/convert/streaming.py +1030 -0
- focus_data_toolkit/errors.py +145 -0
- focus_data_toolkit/focus_json.py +68 -0
- focus_data_toolkit/generators/__init__.py +61 -0
- focus_data_toolkit/generators/_shim.py +43 -0
- focus_data_toolkit/generators/engine/__init__.py +14 -0
- focus_data_toolkit/generators/engine/context.py +12 -0
- focus_data_toolkit/generators/engine/determinism.py +117 -0
- focus_data_toolkit/generators/engine/json_focus.py +63 -0
- focus_data_toolkit/generators/engine/ladder.py +71 -0
- focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
- focus_data_toolkit/generators/engine/serialize.py +151 -0
- focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
- focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
- focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
- focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
- focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
- focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
- focus_data_toolkit/generators/providers/__init__.py +29 -0
- focus_data_toolkit/generators/providers/aws.py +186 -0
- focus_data_toolkit/generators/providers/azure.py +191 -0
- focus_data_toolkit/generators/providers/gcp.py +194 -0
- focus_data_toolkit/generators/providers/profile.py +123 -0
- focus_data_toolkit/generators/scenarios.py +178 -0
- focus_data_toolkit/generators/versions/__init__.py +17 -0
- focus_data_toolkit/generators/versions/adapter.py +41 -0
- focus_data_toolkit/generators/versions/v1_2.py +111 -0
- focus_data_toolkit/generators/versions/v1_3.py +154 -0
- focus_data_toolkit/io/__init__.py +1 -0
- focus_data_toolkit/io/atomic_writer.py +462 -0
- focus_data_toolkit/io/csv_io.py +128 -0
- focus_data_toolkit/io/parquet_io.py +528 -0
- focus_data_toolkit/io/records.py +92 -0
- focus_data_toolkit/io/row_source.py +117 -0
- focus_data_toolkit/lifecycle.py +342 -0
- focus_data_toolkit/manifest.py +114 -0
- focus_data_toolkit/model/__init__.py +43 -0
- focus_data_toolkit/model/capabilities.py +66 -0
- focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
- focus_data_toolkit/model/focus_1_4_model.json +1913 -0
- focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
- focus_data_toolkit/model/focus_json_keys.py +112 -0
- focus_data_toolkit/model/iso_4217_currencies.json +23 -0
- focus_data_toolkit/model/json_schema_check.py +205 -0
- focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
- focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
- focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
- focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
- focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
- focus_data_toolkit/model/model_provenance.json +58 -0
- focus_data_toolkit/model/validator.py +498 -0
- focus_data_toolkit/modes.py +18 -0
- focus_data_toolkit/official_validator.py +61 -0
- focus_data_toolkit/progress.py +89 -0
- focus_data_toolkit/provenance.py +106 -0
- focus_data_toolkit/py.typed +1 -0
- focus_data_toolkit/runtime.py +243 -0
- focus_data_toolkit/schema/__init__.py +17 -0
- focus_data_toolkit/schema/detection.py +274 -0
- focus_data_toolkit/schema/registry.py +127 -0
- focus_data_toolkit/storage/__init__.py +1 -0
- focus_data_toolkit/storage/external_index.py +99 -0
- focus_data_toolkit/storage/spill.py +150 -0
- focus_data_toolkit/studio/__init__.py +19 -0
- focus_data_toolkit/studio/app.py +467 -0
- focus_data_toolkit/studio/config.py +42 -0
- focus_data_toolkit/studio/frontend/app.js +214 -0
- focus_data_toolkit/studio/frontend/index.html +101 -0
- focus_data_toolkit/studio/frontend/style.css +60 -0
- focus_data_toolkit/studio/jobs.py +142 -0
- focus_data_toolkit/studio/preview.py +32 -0
- focus_data_toolkit/studio/security.py +125 -0
- focus_data_toolkit/studio/server.py +71 -0
- focus_data_toolkit/supplement/__init__.py +50 -0
- focus_data_toolkit/supplement/adapters/__init__.py +21 -0
- focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
- focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
- focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
- focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
- focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
- focus_data_toolkit/supplement/adapters/registry.py +215 -0
- focus_data_toolkit/supplement/apply.py +318 -0
- focus_data_toolkit/supplement/gaps.py +219 -0
- focus_data_toolkit/supplement/kinds.py +118 -0
- focus_data_toolkit/supplement/loader.py +409 -0
- focus_data_toolkit/supplement/spec.py +74 -0
- focus_data_toolkit/supplement/validate.py +215 -0
- focus_data_toolkit/validate/__init__.py +15 -0
- focus_data_toolkit/validate/allocation.py +333 -0
- focus_data_toolkit/validate/bundle.py +254 -0
- focus_data_toolkit/validate/codes.py +93 -0
- focus_data_toolkit/validate/corrections.py +245 -0
- focus_data_toolkit/validate/reconciliation.py +98 -0
- focus_data_toolkit/validate/referential.py +289 -0
- focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
- focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
- focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
- focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
- focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,462 @@
|
|
|
1
|
+
"""Atomic output directory: results appear only once everything succeeded.
|
|
2
|
+
|
|
3
|
+
Nothing is written to the destination until all files are on disk (in a temporary directory
|
|
4
|
+
on the *same* filesystem), mandatory validations have passed, and checksums + manifest are
|
|
5
|
+
written last. Publication is a single directory rename; on any error the temporary directory
|
|
6
|
+
is removed and the destination is left untouched. This prevents partial files, a stale mix of
|
|
7
|
+
old and new results, or an inconsistent manifest after a crash.
|
|
8
|
+
|
|
9
|
+
A **replace** (``on_exists=replace``) needs two renames (old aside, new in), so that swap is
|
|
10
|
+
**journaled**: a durable ``.replace-journal-<run_id>.json`` in the parent directory records
|
|
11
|
+
the (target, tmp, trash) names before the first rename and is removed after the swap
|
|
12
|
+
concludes. If the process dies inside the window, the next :class:`AtomicOutputDir` for the
|
|
13
|
+
same destination — or an explicit ``fdt clean`` — reads the journal and finishes the job:
|
|
14
|
+
roll the fully-staged new result forward, or roll the old result back, never leaving the
|
|
15
|
+
destination missing. Recovery actions are surfaced as :class:`RuntimeWarning`s.
|
|
16
|
+
|
|
17
|
+
The temporary directory name carries a run id, and an operational ``_run.json`` sidecar
|
|
18
|
+
carries the run id / timestamp / file checksums — kept out of the deterministic business
|
|
19
|
+
manifest so dataset bytes stay reproducible while operational metadata is still recorded.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import hashlib
|
|
25
|
+
import json
|
|
26
|
+
import os
|
|
27
|
+
import shutil
|
|
28
|
+
import uuid
|
|
29
|
+
import warnings
|
|
30
|
+
from collections.abc import Callable, Iterable
|
|
31
|
+
from enum import StrEnum
|
|
32
|
+
from pathlib import Path
|
|
33
|
+
|
|
34
|
+
_JOURNAL_PREFIX = ".replace-journal-"
|
|
35
|
+
_TMP_PREFIX = ".output.tmp-"
|
|
36
|
+
_TRASH_PREFIX = ".trash-"
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class OnExists(StrEnum):
|
|
40
|
+
REFUSE = "refuse" # fail fast if the destination already exists (default, safest)
|
|
41
|
+
REPLACE = "replace" # atomically swap the existing destination for the new one
|
|
42
|
+
VERSION = "version" # write into a new versioned subdirectory, never touching prior results
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class AtomicWriteError(Exception):
|
|
46
|
+
"""Raised on an atomic-write failure (e.g. destination exists under REFUSE)."""
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class DestinationExistsError(AtomicWriteError):
|
|
50
|
+
"""The destination already exists and the policy is REFUSE."""
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _fsync_file(path: Path) -> None:
|
|
54
|
+
# Durability hardening, best-effort. On POSIX, fsync of a read-only fd works and is
|
|
55
|
+
# honoured. On Windows, os.fsync requires a writable descriptor — a read-only handle
|
|
56
|
+
# raises EBADF — and there is no portable read-only file-flush, so a refused fsync must
|
|
57
|
+
# not fail an otherwise-complete publish (the atomic directory rename remains the
|
|
58
|
+
# correctness guarantee). This mirrors the best-effort handling in ``_fsync_dir``.
|
|
59
|
+
try:
|
|
60
|
+
fd = os.open(path, os.O_RDONLY)
|
|
61
|
+
except OSError:
|
|
62
|
+
return
|
|
63
|
+
try:
|
|
64
|
+
os.fsync(fd)
|
|
65
|
+
except OSError:
|
|
66
|
+
pass
|
|
67
|
+
finally:
|
|
68
|
+
os.close(fd)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _fsync_dir(path: Path) -> None:
|
|
72
|
+
# POSIX only; Windows has no directory fsync and forbids opening a dir fd.
|
|
73
|
+
try:
|
|
74
|
+
fd = os.open(path, os.O_RDONLY)
|
|
75
|
+
except OSError:
|
|
76
|
+
return
|
|
77
|
+
try:
|
|
78
|
+
os.fsync(fd)
|
|
79
|
+
except OSError:
|
|
80
|
+
pass
|
|
81
|
+
finally:
|
|
82
|
+
os.close(fd)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _is_plain_name(name: object) -> bool:
|
|
86
|
+
"""A single path component: non-empty, no separators/drive, no traversal, not absolute."""
|
|
87
|
+
if not isinstance(name, str) or not name or name in (".", ".."):
|
|
88
|
+
return False
|
|
89
|
+
if any(sep in name for sep in ("/", "\\", ":", "\x00")):
|
|
90
|
+
return False
|
|
91
|
+
return Path(name).name == name and not Path(name).is_absolute()
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _read_journal(journal: Path) -> dict | None:
|
|
95
|
+
"""Load a replace journal; None when unreadable/invalid (left for ``fdt clean``).
|
|
96
|
+
|
|
97
|
+
Every recorded name must be a plain sibling basename with the writer's own prefixes —
|
|
98
|
+
a crafted or corrupted journal (absolute paths, ``..``, foreign names) must never steer
|
|
99
|
+
recovery renames/removals outside the output parent, so it is rejected here before any
|
|
100
|
+
filesystem operation and later removed as stale by :func:`clean_leftovers`.
|
|
101
|
+
"""
|
|
102
|
+
try:
|
|
103
|
+
info = json.loads(journal.read_text(encoding="utf-8"))
|
|
104
|
+
except (OSError, ValueError):
|
|
105
|
+
return None
|
|
106
|
+
if not (isinstance(info, dict) and {"target", "tmp", "trash"} <= set(info)):
|
|
107
|
+
return None
|
|
108
|
+
target, tmp, trash = info["target"], info["tmp"], info["trash"]
|
|
109
|
+
if not all(_is_plain_name(name) for name in (target, tmp, trash)):
|
|
110
|
+
return None
|
|
111
|
+
if not (tmp.startswith(_TMP_PREFIX) and trash.startswith(_TRASH_PREFIX)):
|
|
112
|
+
return None
|
|
113
|
+
if target.startswith((_TMP_PREFIX, _TRASH_PREFIX, _JOURNAL_PREFIX)):
|
|
114
|
+
return None # a destination is never one of the writer's own scratch names
|
|
115
|
+
return info
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def recover_interrupted_replaces(
|
|
119
|
+
parent: str | os.PathLike[str], *, dest_name: str | None = None
|
|
120
|
+
) -> list[str]:
|
|
121
|
+
"""Finish (or safely undo) journaled replace swaps that a crash left half-done.
|
|
122
|
+
|
|
123
|
+
Scans ``parent`` for ``.replace-journal-*`` files — one exists only inside a replace
|
|
124
|
+
swap — and resolves each unambiguous state:
|
|
125
|
+
|
|
126
|
+
* destination missing, staged tmp present → **roll forward** (the tmp was fully written,
|
|
127
|
+
validated and fsync'd before the journal was created), then drop the old ``.trash-*``;
|
|
128
|
+
* destination missing, only ``.trash-*`` present → **roll back** the old result;
|
|
129
|
+
* destination present → the swap concluded (or never started): drop the leftover
|
|
130
|
+
``.trash-*`` and, if the run died before its first rename, the never-published tmp.
|
|
131
|
+
|
|
132
|
+
``dest_name`` restricts recovery to journals targeting that destination (what
|
|
133
|
+
:class:`AtomicOutputDir` uses on entry). Returns a description of every action taken.
|
|
134
|
+
"""
|
|
135
|
+
parent = Path(parent)
|
|
136
|
+
actions: list[str] = []
|
|
137
|
+
for journal in sorted(parent.glob(f"{_JOURNAL_PREFIX}*.json")):
|
|
138
|
+
info = _read_journal(journal)
|
|
139
|
+
if info is None or (dest_name is not None and info["target"] != dest_name):
|
|
140
|
+
continue
|
|
141
|
+
target = parent / info["target"]
|
|
142
|
+
tmp = parent / info["tmp"]
|
|
143
|
+
trash = parent / info["trash"]
|
|
144
|
+
if not target.exists() and tmp.exists():
|
|
145
|
+
os.replace(tmp, target)
|
|
146
|
+
actions.append(
|
|
147
|
+
f"rolled forward interrupted replace of {target}: published the fully "
|
|
148
|
+
f"staged result from {tmp.name}"
|
|
149
|
+
)
|
|
150
|
+
if trash.exists():
|
|
151
|
+
shutil.rmtree(trash, ignore_errors=True)
|
|
152
|
+
actions.append(f"removed superseded previous result {trash.name}")
|
|
153
|
+
elif not target.exists() and trash.exists():
|
|
154
|
+
os.replace(trash, target)
|
|
155
|
+
actions.append(
|
|
156
|
+
f"rolled back interrupted replace of {target}: restored the previous "
|
|
157
|
+
f"result from {trash.name}"
|
|
158
|
+
)
|
|
159
|
+
else:
|
|
160
|
+
if trash.exists():
|
|
161
|
+
shutil.rmtree(trash, ignore_errors=True)
|
|
162
|
+
actions.append(f"removed leftover previous result {trash.name}")
|
|
163
|
+
if tmp.exists():
|
|
164
|
+
shutil.rmtree(tmp, ignore_errors=True)
|
|
165
|
+
actions.append(
|
|
166
|
+
f"removed staged result {tmp.name} of a replace that never started "
|
|
167
|
+
f"(destination {target.name} is intact); re-run the conversion"
|
|
168
|
+
)
|
|
169
|
+
journal.unlink(missing_ok=True)
|
|
170
|
+
_fsync_dir(parent)
|
|
171
|
+
return actions
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def _remove_orphans(directory: Path) -> list[str]:
|
|
175
|
+
"""Remove leftover staging/trash directories and stale journals inside ``directory``.
|
|
176
|
+
|
|
177
|
+
Runs after :func:`recover_interrupted_replaces`, so any journal still present is
|
|
178
|
+
unreadable or invalid — removed as stale, never acted upon.
|
|
179
|
+
"""
|
|
180
|
+
actions: list[str] = []
|
|
181
|
+
for leftover in sorted(directory.iterdir()):
|
|
182
|
+
if leftover.name.startswith((_TMP_PREFIX, _TRASH_PREFIX)) and leftover.is_dir():
|
|
183
|
+
shutil.rmtree(leftover, ignore_errors=True)
|
|
184
|
+
kind = "unpublished staging" if leftover.name.startswith(_TMP_PREFIX) else "trash"
|
|
185
|
+
actions.append(f"removed orphan {kind} directory {leftover.name}")
|
|
186
|
+
elif leftover.name.startswith(_JOURNAL_PREFIX) and leftover.is_file():
|
|
187
|
+
leftover.unlink(missing_ok=True)
|
|
188
|
+
actions.append(f"removed stale replace journal {leftover.name}")
|
|
189
|
+
if actions:
|
|
190
|
+
_fsync_dir(directory)
|
|
191
|
+
return actions
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def clean_leftovers(directory: str | os.PathLike[str]) -> list[str]:
|
|
195
|
+
"""Recover journaled replaces, then remove orphan staging/trash leftovers (``fdt clean``).
|
|
196
|
+
|
|
197
|
+
``AtomicOutputDir`` stages ``.output.tmp-*`` / ``.trash-*`` / journals in the
|
|
198
|
+
**destination's parent**, so when ``directory`` is an output directory its leftovers are
|
|
199
|
+
siblings: both ``directory`` itself (as a container of outputs) and its parent are
|
|
200
|
+
swept. Journaled replaces are recovered first (any destination — this is explicit
|
|
201
|
+
maintenance), then orphans (staging from runs that died before publishing, trash from
|
|
202
|
+
concluded swaps, unreadable journals) are removed. Only call this when no conversion is
|
|
203
|
+
currently publishing here — a live run's staging directory is indistinguishable from a
|
|
204
|
+
dead one's.
|
|
205
|
+
"""
|
|
206
|
+
directory = Path(directory)
|
|
207
|
+
actions: list[str] = []
|
|
208
|
+
parent = directory.parent
|
|
209
|
+
if parent != directory and parent.is_dir():
|
|
210
|
+
actions.extend(recover_interrupted_replaces(parent))
|
|
211
|
+
actions.extend(_remove_orphans(parent))
|
|
212
|
+
if directory.is_dir():
|
|
213
|
+
actions.extend(recover_interrupted_replaces(directory))
|
|
214
|
+
actions.extend(_remove_orphans(directory))
|
|
215
|
+
return actions
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def sha256_file(path: Path) -> str:
|
|
219
|
+
"""Stream ``path`` through SHA-256 (bounded memory) and return the hex digest."""
|
|
220
|
+
digest = hashlib.sha256()
|
|
221
|
+
with open(path, "rb") as handle:
|
|
222
|
+
for chunk in iter(lambda: handle.read(1 << 20), b""):
|
|
223
|
+
digest.update(chunk)
|
|
224
|
+
return digest.hexdigest()
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
class AtomicOutputDir:
|
|
228
|
+
"""A staging directory that publishes to ``dest`` atomically on :meth:`commit`.
|
|
229
|
+
|
|
230
|
+
Usage::
|
|
231
|
+
|
|
232
|
+
with AtomicOutputDir(dest, on_exists=OnExists.REFUSE) as out:
|
|
233
|
+
out.write_bytes("a.csv", data)
|
|
234
|
+
# ... run mandatory validations; raise to abort and clean up ...
|
|
235
|
+
out.commit(final_files={"manifest.json": manifest_bytes})
|
|
236
|
+
"""
|
|
237
|
+
|
|
238
|
+
def __init__(
|
|
239
|
+
self,
|
|
240
|
+
dest: str | os.PathLike[str],
|
|
241
|
+
*,
|
|
242
|
+
on_exists: OnExists | str = OnExists.REFUSE,
|
|
243
|
+
keep_temp: bool = False,
|
|
244
|
+
run_id: str | None = None,
|
|
245
|
+
) -> None:
|
|
246
|
+
self.dest = Path(dest)
|
|
247
|
+
self.on_exists = OnExists(on_exists)
|
|
248
|
+
self.keep_temp = keep_temp
|
|
249
|
+
self.run_id = run_id or uuid.uuid4().hex[:12]
|
|
250
|
+
self._parent = self.dest.parent
|
|
251
|
+
self._tmp = self._parent / f".output.tmp-{self.run_id}"
|
|
252
|
+
self._data_files: list[Path] = []
|
|
253
|
+
self._committed = False
|
|
254
|
+
self._published_path: Path | None = None
|
|
255
|
+
|
|
256
|
+
# -- context management ----------------------------------------------------
|
|
257
|
+
def __enter__(self) -> AtomicOutputDir:
|
|
258
|
+
# A previous run may have died mid-swap: finish its journaled replace first, so the
|
|
259
|
+
# destination is in a consistent state before this run's policy is applied.
|
|
260
|
+
if self._parent.is_dir():
|
|
261
|
+
for action in recover_interrupted_replaces(self._parent, dest_name=self.dest.name):
|
|
262
|
+
warnings.warn(f"recovered interrupted publish: {action}", RuntimeWarning,
|
|
263
|
+
stacklevel=2)
|
|
264
|
+
if self.on_exists is OnExists.REFUSE and self.dest.exists():
|
|
265
|
+
raise DestinationExistsError(
|
|
266
|
+
f"destination {self.dest} already exists (on_exists=refuse)"
|
|
267
|
+
)
|
|
268
|
+
self._parent.mkdir(parents=True, exist_ok=True)
|
|
269
|
+
if self._tmp.exists(): # pragma: no cover - astronomically unlikely id clash
|
|
270
|
+
shutil.rmtree(self._tmp)
|
|
271
|
+
self._tmp.mkdir()
|
|
272
|
+
if os.stat(self._tmp).st_dev != os.stat(self._parent).st_dev: # pragma: no cover
|
|
273
|
+
shutil.rmtree(self._tmp, ignore_errors=True)
|
|
274
|
+
raise AtomicWriteError("temporary directory is not on the destination filesystem")
|
|
275
|
+
return self
|
|
276
|
+
|
|
277
|
+
def __exit__(self, exc_type, exc, tb) -> None:
|
|
278
|
+
# Return None (falsy) — never suppress the exception.
|
|
279
|
+
if not self._committed and not self.keep_temp:
|
|
280
|
+
shutil.rmtree(self._tmp, ignore_errors=True)
|
|
281
|
+
|
|
282
|
+
# -- writing ---------------------------------------------------------------
|
|
283
|
+
def path_for(self, name: str) -> Path:
|
|
284
|
+
# Reject absolute paths and '..' so a caller-supplied name cannot escape the staging
|
|
285
|
+
# directory (which would write outside the atomic flow and the checksum set). Relative
|
|
286
|
+
# subdirectories are allowed, e.g. for Parquet partitioning.
|
|
287
|
+
candidate = Path(name)
|
|
288
|
+
if candidate.is_absolute() or ".." in candidate.parts:
|
|
289
|
+
raise AtomicWriteError(
|
|
290
|
+
f"unsafe output name {name!r}: must be a relative path without '..'"
|
|
291
|
+
)
|
|
292
|
+
return self._tmp / candidate
|
|
293
|
+
|
|
294
|
+
def write_bytes(self, name: str, data: bytes, *, is_data: bool = True) -> Path:
|
|
295
|
+
"""Write ``data`` into the staging dir, flush and fsync it."""
|
|
296
|
+
path = self.path_for(name)
|
|
297
|
+
with open(path, "wb") as handle:
|
|
298
|
+
handle.write(data)
|
|
299
|
+
handle.flush()
|
|
300
|
+
os.fsync(handle.fileno())
|
|
301
|
+
if is_data:
|
|
302
|
+
self._data_files.append(path)
|
|
303
|
+
return path
|
|
304
|
+
|
|
305
|
+
def write_text(self, name: str, text: str, *, is_data: bool = True) -> Path:
|
|
306
|
+
return self.write_bytes(name, text.encode("utf-8"), is_data=is_data)
|
|
307
|
+
|
|
308
|
+
def add_data_file(self, name: str) -> Path:
|
|
309
|
+
"""Register a staging file written directly (not via :meth:`write_bytes`).
|
|
310
|
+
|
|
311
|
+
The streaming path writes large datasets through incremental file handles to keep memory
|
|
312
|
+
bounded; this fsyncs the finished file and enrolls it in the checksum/size set so it is
|
|
313
|
+
durably persisted and covered by ``SHA256SUMS`` like any other data file.
|
|
314
|
+
"""
|
|
315
|
+
path = self.path_for(name)
|
|
316
|
+
_fsync_file(path)
|
|
317
|
+
self._data_files.append(path)
|
|
318
|
+
return path
|
|
319
|
+
|
|
320
|
+
def add_data_tree(self, name: str) -> list[Path]:
|
|
321
|
+
"""Register every file under a staged directory (e.g. a partitioned Parquet dataset).
|
|
322
|
+
|
|
323
|
+
Each file is fsync'd and enrolled for checksums under its path relative to the staging
|
|
324
|
+
directory, so a partition tree is covered by ``SHA256SUMS`` file-by-file. Every directory
|
|
325
|
+
in the tree (root and each partition level) is fsync'd too, so the nested directory
|
|
326
|
+
entries are durable before publish — otherwise a crash could lose part files that the
|
|
327
|
+
manifest and ``SHA256SUMS`` already reference.
|
|
328
|
+
"""
|
|
329
|
+
root = self.path_for(name)
|
|
330
|
+
added = sorted(p for p in root.rglob("*") if p.is_file())
|
|
331
|
+
dirs: set[Path] = {root}
|
|
332
|
+
for path in added:
|
|
333
|
+
_fsync_file(path)
|
|
334
|
+
self._data_files.append(path)
|
|
335
|
+
dirs.update(path.parents) # every partition-level dir up the tree
|
|
336
|
+
# fsync deepest-first so a parent's entry for a child dir is persisted after the child.
|
|
337
|
+
for directory in sorted(dirs, key=lambda p: len(p.parts), reverse=True):
|
|
338
|
+
if root in (directory, *directory.parents): # stay within the staged tree
|
|
339
|
+
_fsync_dir(directory)
|
|
340
|
+
return added
|
|
341
|
+
|
|
342
|
+
def discard(self, name: str) -> None:
|
|
343
|
+
"""Delete a staging file (e.g. scratch state) so it is never published."""
|
|
344
|
+
self.path_for(name).unlink(missing_ok=True)
|
|
345
|
+
|
|
346
|
+
def _rel(self, path: Path) -> str:
|
|
347
|
+
"""Path relative to the staging dir — the key under which a file is published/checksummed.
|
|
348
|
+
|
|
349
|
+
Using the relative path (not just the basename) keeps partition part files distinct
|
|
350
|
+
(many partitions each have a ``part-0.parquet``); for flat files it equals the basename,
|
|
351
|
+
so single-file output is unaffected.
|
|
352
|
+
"""
|
|
353
|
+
return path.relative_to(self._tmp).as_posix()
|
|
354
|
+
|
|
355
|
+
def checksums(self) -> dict[str, str]:
|
|
356
|
+
"""SHA-256 of every data file written so far, keyed by relative path (sorted)."""
|
|
357
|
+
return {self._rel(p): sha256_file(p) for p in sorted(self._data_files, key=self._rel)}
|
|
358
|
+
|
|
359
|
+
def sizes(self) -> dict[str, int]:
|
|
360
|
+
return {self._rel(p): p.stat().st_size for p in sorted(self._data_files, key=self._rel)}
|
|
361
|
+
|
|
362
|
+
# -- publishing ------------------------------------------------------------
|
|
363
|
+
def commit(self, *, final_files: dict[str, bytes] | None = None) -> Path:
|
|
364
|
+
"""Write ``final_files`` (manifest, checksums) last, then rename atomically."""
|
|
365
|
+
for name, data in (final_files or {}).items():
|
|
366
|
+
self.write_bytes(name, data, is_data=False)
|
|
367
|
+
_fsync_dir(self._tmp)
|
|
368
|
+
target = self._resolve_target()
|
|
369
|
+
self._atomic_publish(target)
|
|
370
|
+
self._committed = True
|
|
371
|
+
self._published_path = target
|
|
372
|
+
return target
|
|
373
|
+
|
|
374
|
+
def _resolve_target(self) -> Path:
|
|
375
|
+
if self.on_exists is OnExists.VERSION:
|
|
376
|
+
self.dest.mkdir(parents=True, exist_ok=True)
|
|
377
|
+
return self.dest / f"run-{self.run_id}"
|
|
378
|
+
return self.dest
|
|
379
|
+
|
|
380
|
+
def _atomic_publish(self, target: Path) -> None:
|
|
381
|
+
# Close is implicit (files already closed). Windows forbids renaming over open handles.
|
|
382
|
+
if not target.exists():
|
|
383
|
+
os.replace(self._tmp, target)
|
|
384
|
+
_fsync_dir(target.parent)
|
|
385
|
+
return
|
|
386
|
+
# The destination was checked at __enter__, but another run may have created it while
|
|
387
|
+
# this one was staging. Re-honour the refuse policy at publish time rather than
|
|
388
|
+
# clobbering a concurrently-published result.
|
|
389
|
+
if self.on_exists is OnExists.REFUSE:
|
|
390
|
+
raise DestinationExistsError(
|
|
391
|
+
f"destination {target} appeared during staging (on_exists=refuse)"
|
|
392
|
+
)
|
|
393
|
+
if self.on_exists is OnExists.VERSION: # pragma: no cover - run-id collision
|
|
394
|
+
raise AtomicWriteError(
|
|
395
|
+
f"versioned destination {target} already exists (run-id collision); retry"
|
|
396
|
+
)
|
|
397
|
+
# target exists -> swap: move it aside, move the new one in, then delete the old.
|
|
398
|
+
# The two renames are journaled so a crash inside the window is recoverable (the
|
|
399
|
+
# journal is written and fsync'd durably *before* the destination is touched).
|
|
400
|
+
trash = self._parent / f"{_TRASH_PREFIX}{self.run_id}"
|
|
401
|
+
journal = self._parent / f"{_JOURNAL_PREFIX}{self.run_id}.json"
|
|
402
|
+
record = {
|
|
403
|
+
"run_id": self.run_id,
|
|
404
|
+
"target": target.name,
|
|
405
|
+
"tmp": self._tmp.name,
|
|
406
|
+
"trash": trash.name,
|
|
407
|
+
}
|
|
408
|
+
with open(journal, "wb") as handle:
|
|
409
|
+
handle.write(json.dumps(record, sort_keys=True).encode("utf-8"))
|
|
410
|
+
handle.flush()
|
|
411
|
+
os.fsync(handle.fileno())
|
|
412
|
+
_fsync_dir(self._parent)
|
|
413
|
+
os.replace(target, trash)
|
|
414
|
+
try:
|
|
415
|
+
os.replace(self._tmp, target)
|
|
416
|
+
except OSError: # pragma: no cover - restore on failure
|
|
417
|
+
os.replace(trash, target)
|
|
418
|
+
raise
|
|
419
|
+
finally:
|
|
420
|
+
shutil.rmtree(trash, ignore_errors=True)
|
|
421
|
+
journal.unlink(missing_ok=True)
|
|
422
|
+
_fsync_dir(target.parent)
|
|
423
|
+
|
|
424
|
+
|
|
425
|
+
def sha256sums_text(checksums: dict[str, str]) -> str:
|
|
426
|
+
"""Render a ``SHA256SUMS`` file body (``<hex> <name>`` per line, sorted)."""
|
|
427
|
+
return "".join(f"{checksums[name]} {name}\n" for name in sorted(checksums))
|
|
428
|
+
|
|
429
|
+
|
|
430
|
+
def write_files_atomically(
|
|
431
|
+
dest: str | os.PathLike[str],
|
|
432
|
+
data_files: Iterable[tuple[str, bytes]],
|
|
433
|
+
*,
|
|
434
|
+
final_files: dict[str, bytes] | None = None,
|
|
435
|
+
on_exists: OnExists | str = OnExists.REFUSE,
|
|
436
|
+
keep_temp: bool = False,
|
|
437
|
+
validate: Callable[[], None] | None = None,
|
|
438
|
+
) -> Path:
|
|
439
|
+
"""Write ``data_files`` then ``final_files`` to ``dest`` atomically.
|
|
440
|
+
|
|
441
|
+
``validate`` (if given) runs after the data files are staged and before anything is
|
|
442
|
+
published; raising from it aborts the write and removes the staging directory.
|
|
443
|
+
"""
|
|
444
|
+
with AtomicOutputDir(dest, on_exists=on_exists, keep_temp=keep_temp) as out:
|
|
445
|
+
for name, data in data_files:
|
|
446
|
+
out.write_bytes(name, data)
|
|
447
|
+
if validate is not None:
|
|
448
|
+
validate()
|
|
449
|
+
return out.commit(final_files=final_files)
|
|
450
|
+
|
|
451
|
+
|
|
452
|
+
__all__ = [
|
|
453
|
+
"AtomicOutputDir",
|
|
454
|
+
"AtomicWriteError",
|
|
455
|
+
"DestinationExistsError",
|
|
456
|
+
"OnExists",
|
|
457
|
+
"clean_leftovers",
|
|
458
|
+
"recover_interrupted_replaces",
|
|
459
|
+
"sha256_file",
|
|
460
|
+
"sha256sums_text",
|
|
461
|
+
"write_files_atomically",
|
|
462
|
+
]
|
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
"""Streaming CSV reader/writer.
|
|
2
|
+
|
|
3
|
+
``CsvRowReader`` reads a CSV (transparently gzip-decompressing ``.gz`` inputs) one row at a
|
|
4
|
+
time, owning the physical line number for actionable errors, and rejecting malformed rows
|
|
5
|
+
(wrong field count) rather than silently mangling them. ``CsvRowWriter`` writes rows in the
|
|
6
|
+
schema's column order using the same ``csv`` dialect as the eager ``rows_to_csv_bytes`` path,
|
|
7
|
+
so streaming output is byte-identical to the in-memory reference.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import csv
|
|
13
|
+
import gzip
|
|
14
|
+
import io
|
|
15
|
+
from collections.abc import Iterator, Mapping
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from typing import TextIO
|
|
18
|
+
|
|
19
|
+
from focus_data_toolkit.io.records import DatasetSchema, MalformedRecordError, Record
|
|
20
|
+
|
|
21
|
+
_GZIP_MAGIC = b"\x1f\x8b"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _open_text(path: Path, encoding: str) -> TextIO:
|
|
25
|
+
with open(path, "rb") as probe:
|
|
26
|
+
magic = probe.read(2)
|
|
27
|
+
if magic == _GZIP_MAGIC:
|
|
28
|
+
return io.TextIOWrapper(gzip.open(path, "rb"), encoding=encoding, newline="")
|
|
29
|
+
return open(path, newline="", encoding=encoding)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class CsvRowReader:
|
|
33
|
+
"""Stream ``Record``s from a CSV file (``.gz`` auto-detected)."""
|
|
34
|
+
|
|
35
|
+
def __init__(self, path: str | Path, *, encoding: str = "utf-8") -> None:
|
|
36
|
+
self._path = Path(path)
|
|
37
|
+
self._fh = _open_text(self._path, encoding)
|
|
38
|
+
self._reader = csv.reader(self._fh)
|
|
39
|
+
try:
|
|
40
|
+
header = next(self._reader)
|
|
41
|
+
except StopIteration:
|
|
42
|
+
header = []
|
|
43
|
+
self.source_columns: tuple[str, ...] = tuple(header)
|
|
44
|
+
|
|
45
|
+
def __iter__(self) -> Iterator[Record]:
|
|
46
|
+
ncols = len(self.source_columns)
|
|
47
|
+
for row in self._reader:
|
|
48
|
+
if len(row) != ncols:
|
|
49
|
+
raise MalformedRecordError(
|
|
50
|
+
f"malformed CSV record at line {self._reader.line_num}: expected {ncols} "
|
|
51
|
+
f"field(s), got {len(row)}",
|
|
52
|
+
line_number=self._reader.line_num,
|
|
53
|
+
)
|
|
54
|
+
yield Record(dict(zip(self.source_columns, row, strict=True)), self._reader.line_num)
|
|
55
|
+
|
|
56
|
+
@property
|
|
57
|
+
def bytes_total(self) -> int | None:
|
|
58
|
+
"""Size of the source file in bytes (the *compressed* size for a ``.gz`` input)."""
|
|
59
|
+
try:
|
|
60
|
+
return self._path.stat().st_size
|
|
61
|
+
except OSError:
|
|
62
|
+
return None
|
|
63
|
+
|
|
64
|
+
@property
|
|
65
|
+
def bytes_read(self) -> int | None:
|
|
66
|
+
"""Approximate position in the *compressed* byte stream, for progress reporting.
|
|
67
|
+
|
|
68
|
+
Both this and :attr:`bytes_total` are measured on the compressed stream, so the
|
|
69
|
+
ratio is consistent for gzip input. The value advances in buffered read-ahead steps
|
|
70
|
+
(monotonic, slightly ahead of the last yielded row) — fine for a progress bar. For a
|
|
71
|
+
gzip source the compressed offset lives on the wrapped raw file object, not on the
|
|
72
|
+
``GzipFile`` (whose ``tell()`` is the *uncompressed* offset); returns ``None`` if the
|
|
73
|
+
underlying stream exposes no usable position.
|
|
74
|
+
"""
|
|
75
|
+
try:
|
|
76
|
+
buffer = self._fh.buffer # the TextIOWrapper's underlying binary stream
|
|
77
|
+
except (AttributeError, ValueError):
|
|
78
|
+
return None
|
|
79
|
+
# gzip: the compressed offset is on the wrapped raw file, exposed as ``fileobj``.
|
|
80
|
+
raw = getattr(buffer, "fileobj", None)
|
|
81
|
+
target = raw if raw is not None else buffer
|
|
82
|
+
try:
|
|
83
|
+
pos = target.tell()
|
|
84
|
+
except (OSError, ValueError, AttributeError):
|
|
85
|
+
return None
|
|
86
|
+
return pos if isinstance(pos, int) and pos >= 0 else None
|
|
87
|
+
|
|
88
|
+
def close(self) -> None:
|
|
89
|
+
self._fh.close()
|
|
90
|
+
|
|
91
|
+
def __enter__(self) -> CsvRowReader:
|
|
92
|
+
return self
|
|
93
|
+
|
|
94
|
+
def __exit__(self, *exc: object) -> None:
|
|
95
|
+
self.close()
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
class CsvRowWriter:
|
|
99
|
+
"""Write rows to a text stream in ``schema.columns`` order.
|
|
100
|
+
|
|
101
|
+
The header is written lazily on the first row, so a zero-row dataset produces an empty
|
|
102
|
+
file (matching ``rows_to_csv_bytes([])``). The default ``csv`` dialect (CRLF line
|
|
103
|
+
terminator) matches the eager writer exactly.
|
|
104
|
+
"""
|
|
105
|
+
|
|
106
|
+
def __init__(self, stream: TextIO, schema: DatasetSchema) -> None:
|
|
107
|
+
self._schema = schema
|
|
108
|
+
self._writer = csv.DictWriter(stream, fieldnames=list(schema.columns), extrasaction="ignore")
|
|
109
|
+
self._header_written = False
|
|
110
|
+
|
|
111
|
+
def write(self, values: Mapping[str, str]) -> None:
|
|
112
|
+
if not self._header_written:
|
|
113
|
+
self._writer.writeheader()
|
|
114
|
+
self._header_written = True
|
|
115
|
+
self._writer.writerow({col: values.get(col, "") for col in self._schema.columns})
|
|
116
|
+
|
|
117
|
+
def close(self) -> None:
|
|
118
|
+
pass
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def open_csv_writer(path: str | Path, schema: DatasetSchema, *, encoding: str = "utf-8"):
|
|
122
|
+
"""Open ``path`` for writing and return ``(file_handle, CsvRowWriter)``.
|
|
123
|
+
|
|
124
|
+
The file is opened with ``newline=""`` so the ``csv`` module owns line endings (no OS
|
|
125
|
+
translation), matching the in-memory ``io.StringIO`` reference byte-for-byte.
|
|
126
|
+
"""
|
|
127
|
+
handle = open(path, "w", newline="", encoding=encoding)
|
|
128
|
+
return handle, CsvRowWriter(handle, schema)
|