focus-data-toolkit 0.11.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. focus_data_toolkit/__init__.py +69 -0
  2. focus_data_toolkit/__main__.py +6 -0
  3. focus_data_toolkit/_version.py +8 -0
  4. focus_data_toolkit/cli.py +968 -0
  5. focus_data_toolkit/context/__init__.py +88 -0
  6. focus_data_toolkit/context/billing.py +54 -0
  7. focus_data_toolkit/context/provider.py +90 -0
  8. focus_data_toolkit/convert/__init__.py +708 -0
  9. focus_data_toolkit/convert/billing_period.py +65 -0
  10. focus_data_toolkit/convert/contract_applied.py +235 -0
  11. focus_data_toolkit/convert/contract_commitment.py +182 -0
  12. focus_data_toolkit/convert/cost_and_usage.py +179 -0
  13. focus_data_toolkit/convert/detect.py +39 -0
  14. focus_data_toolkit/convert/invoice_detail.py +199 -0
  15. focus_data_toolkit/convert/streaming.py +1030 -0
  16. focus_data_toolkit/errors.py +145 -0
  17. focus_data_toolkit/focus_json.py +68 -0
  18. focus_data_toolkit/generators/__init__.py +61 -0
  19. focus_data_toolkit/generators/_shim.py +43 -0
  20. focus_data_toolkit/generators/engine/__init__.py +14 -0
  21. focus_data_toolkit/generators/engine/context.py +12 -0
  22. focus_data_toolkit/generators/engine/determinism.py +117 -0
  23. focus_data_toolkit/generators/engine/json_focus.py +63 -0
  24. focus_data_toolkit/generators/engine/ladder.py +71 -0
  25. focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
  26. focus_data_toolkit/generators/engine/serialize.py +151 -0
  27. focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
  28. focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
  29. focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
  30. focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
  31. focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
  32. focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
  33. focus_data_toolkit/generators/providers/__init__.py +29 -0
  34. focus_data_toolkit/generators/providers/aws.py +186 -0
  35. focus_data_toolkit/generators/providers/azure.py +191 -0
  36. focus_data_toolkit/generators/providers/gcp.py +194 -0
  37. focus_data_toolkit/generators/providers/profile.py +123 -0
  38. focus_data_toolkit/generators/scenarios.py +178 -0
  39. focus_data_toolkit/generators/versions/__init__.py +17 -0
  40. focus_data_toolkit/generators/versions/adapter.py +41 -0
  41. focus_data_toolkit/generators/versions/v1_2.py +111 -0
  42. focus_data_toolkit/generators/versions/v1_3.py +154 -0
  43. focus_data_toolkit/io/__init__.py +1 -0
  44. focus_data_toolkit/io/atomic_writer.py +462 -0
  45. focus_data_toolkit/io/csv_io.py +128 -0
  46. focus_data_toolkit/io/parquet_io.py +528 -0
  47. focus_data_toolkit/io/records.py +92 -0
  48. focus_data_toolkit/io/row_source.py +117 -0
  49. focus_data_toolkit/lifecycle.py +342 -0
  50. focus_data_toolkit/manifest.py +114 -0
  51. focus_data_toolkit/model/__init__.py +43 -0
  52. focus_data_toolkit/model/capabilities.py +66 -0
  53. focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
  54. focus_data_toolkit/model/focus_1_4_model.json +1913 -0
  55. focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
  56. focus_data_toolkit/model/focus_json_keys.py +112 -0
  57. focus_data_toolkit/model/iso_4217_currencies.json +23 -0
  58. focus_data_toolkit/model/json_schema_check.py +205 -0
  59. focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
  60. focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
  61. focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
  62. focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
  63. focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
  64. focus_data_toolkit/model/model_provenance.json +58 -0
  65. focus_data_toolkit/model/validator.py +498 -0
  66. focus_data_toolkit/modes.py +18 -0
  67. focus_data_toolkit/official_validator.py +61 -0
  68. focus_data_toolkit/progress.py +89 -0
  69. focus_data_toolkit/provenance.py +106 -0
  70. focus_data_toolkit/py.typed +1 -0
  71. focus_data_toolkit/runtime.py +243 -0
  72. focus_data_toolkit/schema/__init__.py +17 -0
  73. focus_data_toolkit/schema/detection.py +274 -0
  74. focus_data_toolkit/schema/registry.py +127 -0
  75. focus_data_toolkit/storage/__init__.py +1 -0
  76. focus_data_toolkit/storage/external_index.py +99 -0
  77. focus_data_toolkit/storage/spill.py +150 -0
  78. focus_data_toolkit/studio/__init__.py +19 -0
  79. focus_data_toolkit/studio/app.py +467 -0
  80. focus_data_toolkit/studio/config.py +42 -0
  81. focus_data_toolkit/studio/frontend/app.js +214 -0
  82. focus_data_toolkit/studio/frontend/index.html +101 -0
  83. focus_data_toolkit/studio/frontend/style.css +60 -0
  84. focus_data_toolkit/studio/jobs.py +142 -0
  85. focus_data_toolkit/studio/preview.py +32 -0
  86. focus_data_toolkit/studio/security.py +125 -0
  87. focus_data_toolkit/studio/server.py +71 -0
  88. focus_data_toolkit/supplement/__init__.py +50 -0
  89. focus_data_toolkit/supplement/adapters/__init__.py +21 -0
  90. focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
  91. focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
  92. focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
  93. focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
  94. focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
  95. focus_data_toolkit/supplement/adapters/registry.py +215 -0
  96. focus_data_toolkit/supplement/apply.py +318 -0
  97. focus_data_toolkit/supplement/gaps.py +219 -0
  98. focus_data_toolkit/supplement/kinds.py +118 -0
  99. focus_data_toolkit/supplement/loader.py +409 -0
  100. focus_data_toolkit/supplement/spec.py +74 -0
  101. focus_data_toolkit/supplement/validate.py +215 -0
  102. focus_data_toolkit/validate/__init__.py +15 -0
  103. focus_data_toolkit/validate/allocation.py +333 -0
  104. focus_data_toolkit/validate/bundle.py +254 -0
  105. focus_data_toolkit/validate/codes.py +93 -0
  106. focus_data_toolkit/validate/corrections.py +245 -0
  107. focus_data_toolkit/validate/reconciliation.py +98 -0
  108. focus_data_toolkit/validate/referential.py +289 -0
  109. focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
  110. focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
  111. focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
  112. focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
  113. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
  114. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
  115. focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
  116. focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,462 @@
1
+ """Atomic output directory: results appear only once everything succeeded.
2
+
3
+ Nothing is written to the destination until all files are on disk (in a temporary directory
4
+ on the *same* filesystem), mandatory validations have passed, and checksums + manifest are
5
+ written last. Publication is a single directory rename; on any error the temporary directory
6
+ is removed and the destination is left untouched. This prevents partial files, a stale mix of
7
+ old and new results, or an inconsistent manifest after a crash.
8
+
9
+ A **replace** (``on_exists=replace``) needs two renames (old aside, new in), so that swap is
10
+ **journaled**: a durable ``.replace-journal-<run_id>.json`` in the parent directory records
11
+ the (target, tmp, trash) names before the first rename and is removed after the swap
12
+ concludes. If the process dies inside the window, the next :class:`AtomicOutputDir` for the
13
+ same destination — or an explicit ``fdt clean`` — reads the journal and finishes the job:
14
+ roll the fully-staged new result forward, or roll the old result back, never leaving the
15
+ destination missing. Recovery actions are surfaced as :class:`RuntimeWarning`s.
16
+
17
+ The temporary directory name carries a run id, and an operational ``_run.json`` sidecar
18
+ carries the run id / timestamp / file checksums — kept out of the deterministic business
19
+ manifest so dataset bytes stay reproducible while operational metadata is still recorded.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ import hashlib
25
+ import json
26
+ import os
27
+ import shutil
28
+ import uuid
29
+ import warnings
30
+ from collections.abc import Callable, Iterable
31
+ from enum import StrEnum
32
+ from pathlib import Path
33
+
34
+ _JOURNAL_PREFIX = ".replace-journal-"
35
+ _TMP_PREFIX = ".output.tmp-"
36
+ _TRASH_PREFIX = ".trash-"
37
+
38
+
39
+ class OnExists(StrEnum):
40
+ REFUSE = "refuse" # fail fast if the destination already exists (default, safest)
41
+ REPLACE = "replace" # atomically swap the existing destination for the new one
42
+ VERSION = "version" # write into a new versioned subdirectory, never touching prior results
43
+
44
+
45
+ class AtomicWriteError(Exception):
46
+ """Raised on an atomic-write failure (e.g. destination exists under REFUSE)."""
47
+
48
+
49
+ class DestinationExistsError(AtomicWriteError):
50
+ """The destination already exists and the policy is REFUSE."""
51
+
52
+
53
+ def _fsync_file(path: Path) -> None:
54
+ # Durability hardening, best-effort. On POSIX, fsync of a read-only fd works and is
55
+ # honoured. On Windows, os.fsync requires a writable descriptor — a read-only handle
56
+ # raises EBADF — and there is no portable read-only file-flush, so a refused fsync must
57
+ # not fail an otherwise-complete publish (the atomic directory rename remains the
58
+ # correctness guarantee). This mirrors the best-effort handling in ``_fsync_dir``.
59
+ try:
60
+ fd = os.open(path, os.O_RDONLY)
61
+ except OSError:
62
+ return
63
+ try:
64
+ os.fsync(fd)
65
+ except OSError:
66
+ pass
67
+ finally:
68
+ os.close(fd)
69
+
70
+
71
+ def _fsync_dir(path: Path) -> None:
72
+ # POSIX only; Windows has no directory fsync and forbids opening a dir fd.
73
+ try:
74
+ fd = os.open(path, os.O_RDONLY)
75
+ except OSError:
76
+ return
77
+ try:
78
+ os.fsync(fd)
79
+ except OSError:
80
+ pass
81
+ finally:
82
+ os.close(fd)
83
+
84
+
85
+ def _is_plain_name(name: object) -> bool:
86
+ """A single path component: non-empty, no separators/drive, no traversal, not absolute."""
87
+ if not isinstance(name, str) or not name or name in (".", ".."):
88
+ return False
89
+ if any(sep in name for sep in ("/", "\\", ":", "\x00")):
90
+ return False
91
+ return Path(name).name == name and not Path(name).is_absolute()
92
+
93
+
94
+ def _read_journal(journal: Path) -> dict | None:
95
+ """Load a replace journal; None when unreadable/invalid (left for ``fdt clean``).
96
+
97
+ Every recorded name must be a plain sibling basename with the writer's own prefixes —
98
+ a crafted or corrupted journal (absolute paths, ``..``, foreign names) must never steer
99
+ recovery renames/removals outside the output parent, so it is rejected here before any
100
+ filesystem operation and later removed as stale by :func:`clean_leftovers`.
101
+ """
102
+ try:
103
+ info = json.loads(journal.read_text(encoding="utf-8"))
104
+ except (OSError, ValueError):
105
+ return None
106
+ if not (isinstance(info, dict) and {"target", "tmp", "trash"} <= set(info)):
107
+ return None
108
+ target, tmp, trash = info["target"], info["tmp"], info["trash"]
109
+ if not all(_is_plain_name(name) for name in (target, tmp, trash)):
110
+ return None
111
+ if not (tmp.startswith(_TMP_PREFIX) and trash.startswith(_TRASH_PREFIX)):
112
+ return None
113
+ if target.startswith((_TMP_PREFIX, _TRASH_PREFIX, _JOURNAL_PREFIX)):
114
+ return None # a destination is never one of the writer's own scratch names
115
+ return info
116
+
117
+
118
+ def recover_interrupted_replaces(
119
+ parent: str | os.PathLike[str], *, dest_name: str | None = None
120
+ ) -> list[str]:
121
+ """Finish (or safely undo) journaled replace swaps that a crash left half-done.
122
+
123
+ Scans ``parent`` for ``.replace-journal-*`` files — one exists only inside a replace
124
+ swap — and resolves each unambiguous state:
125
+
126
+ * destination missing, staged tmp present → **roll forward** (the tmp was fully written,
127
+ validated and fsync'd before the journal was created), then drop the old ``.trash-*``;
128
+ * destination missing, only ``.trash-*`` present → **roll back** the old result;
129
+ * destination present → the swap concluded (or never started): drop the leftover
130
+ ``.trash-*`` and, if the run died before its first rename, the never-published tmp.
131
+
132
+ ``dest_name`` restricts recovery to journals targeting that destination (what
133
+ :class:`AtomicOutputDir` uses on entry). Returns a description of every action taken.
134
+ """
135
+ parent = Path(parent)
136
+ actions: list[str] = []
137
+ for journal in sorted(parent.glob(f"{_JOURNAL_PREFIX}*.json")):
138
+ info = _read_journal(journal)
139
+ if info is None or (dest_name is not None and info["target"] != dest_name):
140
+ continue
141
+ target = parent / info["target"]
142
+ tmp = parent / info["tmp"]
143
+ trash = parent / info["trash"]
144
+ if not target.exists() and tmp.exists():
145
+ os.replace(tmp, target)
146
+ actions.append(
147
+ f"rolled forward interrupted replace of {target}: published the fully "
148
+ f"staged result from {tmp.name}"
149
+ )
150
+ if trash.exists():
151
+ shutil.rmtree(trash, ignore_errors=True)
152
+ actions.append(f"removed superseded previous result {trash.name}")
153
+ elif not target.exists() and trash.exists():
154
+ os.replace(trash, target)
155
+ actions.append(
156
+ f"rolled back interrupted replace of {target}: restored the previous "
157
+ f"result from {trash.name}"
158
+ )
159
+ else:
160
+ if trash.exists():
161
+ shutil.rmtree(trash, ignore_errors=True)
162
+ actions.append(f"removed leftover previous result {trash.name}")
163
+ if tmp.exists():
164
+ shutil.rmtree(tmp, ignore_errors=True)
165
+ actions.append(
166
+ f"removed staged result {tmp.name} of a replace that never started "
167
+ f"(destination {target.name} is intact); re-run the conversion"
168
+ )
169
+ journal.unlink(missing_ok=True)
170
+ _fsync_dir(parent)
171
+ return actions
172
+
173
+
174
+ def _remove_orphans(directory: Path) -> list[str]:
175
+ """Remove leftover staging/trash directories and stale journals inside ``directory``.
176
+
177
+ Runs after :func:`recover_interrupted_replaces`, so any journal still present is
178
+ unreadable or invalid — removed as stale, never acted upon.
179
+ """
180
+ actions: list[str] = []
181
+ for leftover in sorted(directory.iterdir()):
182
+ if leftover.name.startswith((_TMP_PREFIX, _TRASH_PREFIX)) and leftover.is_dir():
183
+ shutil.rmtree(leftover, ignore_errors=True)
184
+ kind = "unpublished staging" if leftover.name.startswith(_TMP_PREFIX) else "trash"
185
+ actions.append(f"removed orphan {kind} directory {leftover.name}")
186
+ elif leftover.name.startswith(_JOURNAL_PREFIX) and leftover.is_file():
187
+ leftover.unlink(missing_ok=True)
188
+ actions.append(f"removed stale replace journal {leftover.name}")
189
+ if actions:
190
+ _fsync_dir(directory)
191
+ return actions
192
+
193
+
194
+ def clean_leftovers(directory: str | os.PathLike[str]) -> list[str]:
195
+ """Recover journaled replaces, then remove orphan staging/trash leftovers (``fdt clean``).
196
+
197
+ ``AtomicOutputDir`` stages ``.output.tmp-*`` / ``.trash-*`` / journals in the
198
+ **destination's parent**, so when ``directory`` is an output directory its leftovers are
199
+ siblings: both ``directory`` itself (as a container of outputs) and its parent are
200
+ swept. Journaled replaces are recovered first (any destination — this is explicit
201
+ maintenance), then orphans (staging from runs that died before publishing, trash from
202
+ concluded swaps, unreadable journals) are removed. Only call this when no conversion is
203
+ currently publishing here — a live run's staging directory is indistinguishable from a
204
+ dead one's.
205
+ """
206
+ directory = Path(directory)
207
+ actions: list[str] = []
208
+ parent = directory.parent
209
+ if parent != directory and parent.is_dir():
210
+ actions.extend(recover_interrupted_replaces(parent))
211
+ actions.extend(_remove_orphans(parent))
212
+ if directory.is_dir():
213
+ actions.extend(recover_interrupted_replaces(directory))
214
+ actions.extend(_remove_orphans(directory))
215
+ return actions
216
+
217
+
218
+ def sha256_file(path: Path) -> str:
219
+ """Stream ``path`` through SHA-256 (bounded memory) and return the hex digest."""
220
+ digest = hashlib.sha256()
221
+ with open(path, "rb") as handle:
222
+ for chunk in iter(lambda: handle.read(1 << 20), b""):
223
+ digest.update(chunk)
224
+ return digest.hexdigest()
225
+
226
+
227
+ class AtomicOutputDir:
228
+ """A staging directory that publishes to ``dest`` atomically on :meth:`commit`.
229
+
230
+ Usage::
231
+
232
+ with AtomicOutputDir(dest, on_exists=OnExists.REFUSE) as out:
233
+ out.write_bytes("a.csv", data)
234
+ # ... run mandatory validations; raise to abort and clean up ...
235
+ out.commit(final_files={"manifest.json": manifest_bytes})
236
+ """
237
+
238
+ def __init__(
239
+ self,
240
+ dest: str | os.PathLike[str],
241
+ *,
242
+ on_exists: OnExists | str = OnExists.REFUSE,
243
+ keep_temp: bool = False,
244
+ run_id: str | None = None,
245
+ ) -> None:
246
+ self.dest = Path(dest)
247
+ self.on_exists = OnExists(on_exists)
248
+ self.keep_temp = keep_temp
249
+ self.run_id = run_id or uuid.uuid4().hex[:12]
250
+ self._parent = self.dest.parent
251
+ self._tmp = self._parent / f".output.tmp-{self.run_id}"
252
+ self._data_files: list[Path] = []
253
+ self._committed = False
254
+ self._published_path: Path | None = None
255
+
256
+ # -- context management ----------------------------------------------------
257
+ def __enter__(self) -> AtomicOutputDir:
258
+ # A previous run may have died mid-swap: finish its journaled replace first, so the
259
+ # destination is in a consistent state before this run's policy is applied.
260
+ if self._parent.is_dir():
261
+ for action in recover_interrupted_replaces(self._parent, dest_name=self.dest.name):
262
+ warnings.warn(f"recovered interrupted publish: {action}", RuntimeWarning,
263
+ stacklevel=2)
264
+ if self.on_exists is OnExists.REFUSE and self.dest.exists():
265
+ raise DestinationExistsError(
266
+ f"destination {self.dest} already exists (on_exists=refuse)"
267
+ )
268
+ self._parent.mkdir(parents=True, exist_ok=True)
269
+ if self._tmp.exists(): # pragma: no cover - astronomically unlikely id clash
270
+ shutil.rmtree(self._tmp)
271
+ self._tmp.mkdir()
272
+ if os.stat(self._tmp).st_dev != os.stat(self._parent).st_dev: # pragma: no cover
273
+ shutil.rmtree(self._tmp, ignore_errors=True)
274
+ raise AtomicWriteError("temporary directory is not on the destination filesystem")
275
+ return self
276
+
277
+ def __exit__(self, exc_type, exc, tb) -> None:
278
+ # Return None (falsy) — never suppress the exception.
279
+ if not self._committed and not self.keep_temp:
280
+ shutil.rmtree(self._tmp, ignore_errors=True)
281
+
282
+ # -- writing ---------------------------------------------------------------
283
+ def path_for(self, name: str) -> Path:
284
+ # Reject absolute paths and '..' so a caller-supplied name cannot escape the staging
285
+ # directory (which would write outside the atomic flow and the checksum set). Relative
286
+ # subdirectories are allowed, e.g. for Parquet partitioning.
287
+ candidate = Path(name)
288
+ if candidate.is_absolute() or ".." in candidate.parts:
289
+ raise AtomicWriteError(
290
+ f"unsafe output name {name!r}: must be a relative path without '..'"
291
+ )
292
+ return self._tmp / candidate
293
+
294
+ def write_bytes(self, name: str, data: bytes, *, is_data: bool = True) -> Path:
295
+ """Write ``data`` into the staging dir, flush and fsync it."""
296
+ path = self.path_for(name)
297
+ with open(path, "wb") as handle:
298
+ handle.write(data)
299
+ handle.flush()
300
+ os.fsync(handle.fileno())
301
+ if is_data:
302
+ self._data_files.append(path)
303
+ return path
304
+
305
+ def write_text(self, name: str, text: str, *, is_data: bool = True) -> Path:
306
+ return self.write_bytes(name, text.encode("utf-8"), is_data=is_data)
307
+
308
+ def add_data_file(self, name: str) -> Path:
309
+ """Register a staging file written directly (not via :meth:`write_bytes`).
310
+
311
+ The streaming path writes large datasets through incremental file handles to keep memory
312
+ bounded; this fsyncs the finished file and enrolls it in the checksum/size set so it is
313
+ durably persisted and covered by ``SHA256SUMS`` like any other data file.
314
+ """
315
+ path = self.path_for(name)
316
+ _fsync_file(path)
317
+ self._data_files.append(path)
318
+ return path
319
+
320
+ def add_data_tree(self, name: str) -> list[Path]:
321
+ """Register every file under a staged directory (e.g. a partitioned Parquet dataset).
322
+
323
+ Each file is fsync'd and enrolled for checksums under its path relative to the staging
324
+ directory, so a partition tree is covered by ``SHA256SUMS`` file-by-file. Every directory
325
+ in the tree (root and each partition level) is fsync'd too, so the nested directory
326
+ entries are durable before publish — otherwise a crash could lose part files that the
327
+ manifest and ``SHA256SUMS`` already reference.
328
+ """
329
+ root = self.path_for(name)
330
+ added = sorted(p for p in root.rglob("*") if p.is_file())
331
+ dirs: set[Path] = {root}
332
+ for path in added:
333
+ _fsync_file(path)
334
+ self._data_files.append(path)
335
+ dirs.update(path.parents) # every partition-level dir up the tree
336
+ # fsync deepest-first so a parent's entry for a child dir is persisted after the child.
337
+ for directory in sorted(dirs, key=lambda p: len(p.parts), reverse=True):
338
+ if root in (directory, *directory.parents): # stay within the staged tree
339
+ _fsync_dir(directory)
340
+ return added
341
+
342
+ def discard(self, name: str) -> None:
343
+ """Delete a staging file (e.g. scratch state) so it is never published."""
344
+ self.path_for(name).unlink(missing_ok=True)
345
+
346
+ def _rel(self, path: Path) -> str:
347
+ """Path relative to the staging dir — the key under which a file is published/checksummed.
348
+
349
+ Using the relative path (not just the basename) keeps partition part files distinct
350
+ (many partitions each have a ``part-0.parquet``); for flat files it equals the basename,
351
+ so single-file output is unaffected.
352
+ """
353
+ return path.relative_to(self._tmp).as_posix()
354
+
355
+ def checksums(self) -> dict[str, str]:
356
+ """SHA-256 of every data file written so far, keyed by relative path (sorted)."""
357
+ return {self._rel(p): sha256_file(p) for p in sorted(self._data_files, key=self._rel)}
358
+
359
+ def sizes(self) -> dict[str, int]:
360
+ return {self._rel(p): p.stat().st_size for p in sorted(self._data_files, key=self._rel)}
361
+
362
+ # -- publishing ------------------------------------------------------------
363
+ def commit(self, *, final_files: dict[str, bytes] | None = None) -> Path:
364
+ """Write ``final_files`` (manifest, checksums) last, then rename atomically."""
365
+ for name, data in (final_files or {}).items():
366
+ self.write_bytes(name, data, is_data=False)
367
+ _fsync_dir(self._tmp)
368
+ target = self._resolve_target()
369
+ self._atomic_publish(target)
370
+ self._committed = True
371
+ self._published_path = target
372
+ return target
373
+
374
+ def _resolve_target(self) -> Path:
375
+ if self.on_exists is OnExists.VERSION:
376
+ self.dest.mkdir(parents=True, exist_ok=True)
377
+ return self.dest / f"run-{self.run_id}"
378
+ return self.dest
379
+
380
+ def _atomic_publish(self, target: Path) -> None:
381
+ # Close is implicit (files already closed). Windows forbids renaming over open handles.
382
+ if not target.exists():
383
+ os.replace(self._tmp, target)
384
+ _fsync_dir(target.parent)
385
+ return
386
+ # The destination was checked at __enter__, but another run may have created it while
387
+ # this one was staging. Re-honour the refuse policy at publish time rather than
388
+ # clobbering a concurrently-published result.
389
+ if self.on_exists is OnExists.REFUSE:
390
+ raise DestinationExistsError(
391
+ f"destination {target} appeared during staging (on_exists=refuse)"
392
+ )
393
+ if self.on_exists is OnExists.VERSION: # pragma: no cover - run-id collision
394
+ raise AtomicWriteError(
395
+ f"versioned destination {target} already exists (run-id collision); retry"
396
+ )
397
+ # target exists -> swap: move it aside, move the new one in, then delete the old.
398
+ # The two renames are journaled so a crash inside the window is recoverable (the
399
+ # journal is written and fsync'd durably *before* the destination is touched).
400
+ trash = self._parent / f"{_TRASH_PREFIX}{self.run_id}"
401
+ journal = self._parent / f"{_JOURNAL_PREFIX}{self.run_id}.json"
402
+ record = {
403
+ "run_id": self.run_id,
404
+ "target": target.name,
405
+ "tmp": self._tmp.name,
406
+ "trash": trash.name,
407
+ }
408
+ with open(journal, "wb") as handle:
409
+ handle.write(json.dumps(record, sort_keys=True).encode("utf-8"))
410
+ handle.flush()
411
+ os.fsync(handle.fileno())
412
+ _fsync_dir(self._parent)
413
+ os.replace(target, trash)
414
+ try:
415
+ os.replace(self._tmp, target)
416
+ except OSError: # pragma: no cover - restore on failure
417
+ os.replace(trash, target)
418
+ raise
419
+ finally:
420
+ shutil.rmtree(trash, ignore_errors=True)
421
+ journal.unlink(missing_ok=True)
422
+ _fsync_dir(target.parent)
423
+
424
+
425
+ def sha256sums_text(checksums: dict[str, str]) -> str:
426
+ """Render a ``SHA256SUMS`` file body (``<hex> <name>`` per line, sorted)."""
427
+ return "".join(f"{checksums[name]} {name}\n" for name in sorted(checksums))
428
+
429
+
430
+ def write_files_atomically(
431
+ dest: str | os.PathLike[str],
432
+ data_files: Iterable[tuple[str, bytes]],
433
+ *,
434
+ final_files: dict[str, bytes] | None = None,
435
+ on_exists: OnExists | str = OnExists.REFUSE,
436
+ keep_temp: bool = False,
437
+ validate: Callable[[], None] | None = None,
438
+ ) -> Path:
439
+ """Write ``data_files`` then ``final_files`` to ``dest`` atomically.
440
+
441
+ ``validate`` (if given) runs after the data files are staged and before anything is
442
+ published; raising from it aborts the write and removes the staging directory.
443
+ """
444
+ with AtomicOutputDir(dest, on_exists=on_exists, keep_temp=keep_temp) as out:
445
+ for name, data in data_files:
446
+ out.write_bytes(name, data)
447
+ if validate is not None:
448
+ validate()
449
+ return out.commit(final_files=final_files)
450
+
451
+
452
+ __all__ = [
453
+ "AtomicOutputDir",
454
+ "AtomicWriteError",
455
+ "DestinationExistsError",
456
+ "OnExists",
457
+ "clean_leftovers",
458
+ "recover_interrupted_replaces",
459
+ "sha256_file",
460
+ "sha256sums_text",
461
+ "write_files_atomically",
462
+ ]
@@ -0,0 +1,128 @@
1
+ """Streaming CSV reader/writer.
2
+
3
+ ``CsvRowReader`` reads a CSV (transparently gzip-decompressing ``.gz`` inputs) one row at a
4
+ time, owning the physical line number for actionable errors, and rejecting malformed rows
5
+ (wrong field count) rather than silently mangling them. ``CsvRowWriter`` writes rows in the
6
+ schema's column order using the same ``csv`` dialect as the eager ``rows_to_csv_bytes`` path,
7
+ so streaming output is byte-identical to the in-memory reference.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import csv
13
+ import gzip
14
+ import io
15
+ from collections.abc import Iterator, Mapping
16
+ from pathlib import Path
17
+ from typing import TextIO
18
+
19
+ from focus_data_toolkit.io.records import DatasetSchema, MalformedRecordError, Record
20
+
21
+ _GZIP_MAGIC = b"\x1f\x8b"
22
+
23
+
24
+ def _open_text(path: Path, encoding: str) -> TextIO:
25
+ with open(path, "rb") as probe:
26
+ magic = probe.read(2)
27
+ if magic == _GZIP_MAGIC:
28
+ return io.TextIOWrapper(gzip.open(path, "rb"), encoding=encoding, newline="")
29
+ return open(path, newline="", encoding=encoding)
30
+
31
+
32
+ class CsvRowReader:
33
+ """Stream ``Record``s from a CSV file (``.gz`` auto-detected)."""
34
+
35
+ def __init__(self, path: str | Path, *, encoding: str = "utf-8") -> None:
36
+ self._path = Path(path)
37
+ self._fh = _open_text(self._path, encoding)
38
+ self._reader = csv.reader(self._fh)
39
+ try:
40
+ header = next(self._reader)
41
+ except StopIteration:
42
+ header = []
43
+ self.source_columns: tuple[str, ...] = tuple(header)
44
+
45
+ def __iter__(self) -> Iterator[Record]:
46
+ ncols = len(self.source_columns)
47
+ for row in self._reader:
48
+ if len(row) != ncols:
49
+ raise MalformedRecordError(
50
+ f"malformed CSV record at line {self._reader.line_num}: expected {ncols} "
51
+ f"field(s), got {len(row)}",
52
+ line_number=self._reader.line_num,
53
+ )
54
+ yield Record(dict(zip(self.source_columns, row, strict=True)), self._reader.line_num)
55
+
56
+ @property
57
+ def bytes_total(self) -> int | None:
58
+ """Size of the source file in bytes (the *compressed* size for a ``.gz`` input)."""
59
+ try:
60
+ return self._path.stat().st_size
61
+ except OSError:
62
+ return None
63
+
64
+ @property
65
+ def bytes_read(self) -> int | None:
66
+ """Approximate position in the *compressed* byte stream, for progress reporting.
67
+
68
+ Both this and :attr:`bytes_total` are measured on the compressed stream, so the
69
+ ratio is consistent for gzip input. The value advances in buffered read-ahead steps
70
+ (monotonic, slightly ahead of the last yielded row) — fine for a progress bar. For a
71
+ gzip source the compressed offset lives on the wrapped raw file object, not on the
72
+ ``GzipFile`` (whose ``tell()`` is the *uncompressed* offset); returns ``None`` if the
73
+ underlying stream exposes no usable position.
74
+ """
75
+ try:
76
+ buffer = self._fh.buffer # the TextIOWrapper's underlying binary stream
77
+ except (AttributeError, ValueError):
78
+ return None
79
+ # gzip: the compressed offset is on the wrapped raw file, exposed as ``fileobj``.
80
+ raw = getattr(buffer, "fileobj", None)
81
+ target = raw if raw is not None else buffer
82
+ try:
83
+ pos = target.tell()
84
+ except (OSError, ValueError, AttributeError):
85
+ return None
86
+ return pos if isinstance(pos, int) and pos >= 0 else None
87
+
88
+ def close(self) -> None:
89
+ self._fh.close()
90
+
91
+ def __enter__(self) -> CsvRowReader:
92
+ return self
93
+
94
+ def __exit__(self, *exc: object) -> None:
95
+ self.close()
96
+
97
+
98
+ class CsvRowWriter:
99
+ """Write rows to a text stream in ``schema.columns`` order.
100
+
101
+ The header is written lazily on the first row, so a zero-row dataset produces an empty
102
+ file (matching ``rows_to_csv_bytes([])``). The default ``csv`` dialect (CRLF line
103
+ terminator) matches the eager writer exactly.
104
+ """
105
+
106
+ def __init__(self, stream: TextIO, schema: DatasetSchema) -> None:
107
+ self._schema = schema
108
+ self._writer = csv.DictWriter(stream, fieldnames=list(schema.columns), extrasaction="ignore")
109
+ self._header_written = False
110
+
111
+ def write(self, values: Mapping[str, str]) -> None:
112
+ if not self._header_written:
113
+ self._writer.writeheader()
114
+ self._header_written = True
115
+ self._writer.writerow({col: values.get(col, "") for col in self._schema.columns})
116
+
117
+ def close(self) -> None:
118
+ pass
119
+
120
+
121
+ def open_csv_writer(path: str | Path, schema: DatasetSchema, *, encoding: str = "utf-8"):
122
+ """Open ``path`` for writing and return ``(file_handle, CsvRowWriter)``.
123
+
124
+ The file is opened with ``newline=""`` so the ``csv`` module owns line endings (no OS
125
+ translation), matching the in-memory ``io.StringIO`` reference byte-for-byte.
126
+ """
127
+ handle = open(path, "w", newline="", encoding=encoding)
128
+ return handle, CsvRowWriter(handle, schema)