crowdb-tpc-loader 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,3 @@
1
+ """CrowDB TPC Loader. Importing this module performs no I/O."""
2
+
3
+ __version__ = "0.1.0"
@@ -0,0 +1,3 @@
1
+ from .cli import main
2
+
3
+ raise SystemExit(main())
@@ -0,0 +1,90 @@
1
+ """A read-only Arrow filesystem bridge for independent PyIceberg HTTP scans.
2
+
3
+ Metadata/data writes still use HttpFileIO's conditional upload streams. Explicit
4
+ paths are supported; recursive object listing and filesystem mutations are not.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from urllib.parse import urlsplit
10
+
11
+ from .errors import CompatibilityError
12
+
13
+
14
+ def make_filesystem(fileio, scheme: str, netloc: str):
15
+ import pyarrow as pa
16
+ from pyarrow.fs import FileInfo, FileSystemHandler, FileType, PyFileSystem
17
+ from .http_fileio import origin
18
+
19
+ expected_origin = origin(f"{scheme}://{netloc}")
20
+
21
+ class Handler(FileSystemHandler):
22
+ def _uri(self, path):
23
+ if urlsplit(path).scheme in {"http", "https"}:
24
+ uri = path
25
+ else:
26
+ path = path.lstrip("/")
27
+ # PyIceberg's parse_location returns `netloc/path` for HTTP.
28
+ uri = f"{scheme}://{path}" if path.startswith(netloc + "/") else f"{scheme}://{netloc}/{path}"
29
+ if origin(uri) != expected_origin:
30
+ raise CompatibilityError("Arrow HTTP filesystem path changed origin")
31
+ return uri
32
+
33
+ def get_type_name(self):
34
+ return "crowdb-tpc-http"
35
+
36
+ def normalize_path(self, path):
37
+ self._uri(path)
38
+ return path
39
+
40
+ def get_file_info(self, paths):
41
+ result = []
42
+ for path in paths:
43
+ try:
44
+ size = len(fileio.new_input(self._uri(path)))
45
+ result.append(FileInfo(path, FileType.File, size=size))
46
+ except FileNotFoundError:
47
+ result.append(FileInfo(path, FileType.NotFound))
48
+ return result
49
+
50
+ def get_file_info_selector(self, selector):
51
+ raise NotImplementedError(
52
+ "HTTP object listing is not supported; Iceberg supplies explicit manifest file paths"
53
+ )
54
+
55
+ def open_input_file(self, path):
56
+ return pa.PythonFile(fileio.new_input(self._uri(path)).open(), mode="r")
57
+
58
+ def open_input_stream(self, path):
59
+ return self.open_input_file(path)
60
+
61
+ def create_dir(self, path, recursive=True):
62
+ raise NotImplementedError("Use HttpFileIO output streams, not directory operations")
63
+
64
+ def delete_dir(self, path):
65
+ raise NotImplementedError("Recursive deletion is deliberately unavailable")
66
+
67
+ def delete_dir_contents(self, path, missing_dir_ok=False):
68
+ raise NotImplementedError("Recursive deletion is deliberately unavailable")
69
+
70
+ def delete_root_dir_contents(self):
71
+ raise NotImplementedError("Recursive deletion is deliberately unavailable")
72
+
73
+ def delete_file(self, path):
74
+ raise NotImplementedError(
75
+ "Use explicit HttpFileIO.delete for independently verified uncommitted objects"
76
+ )
77
+
78
+ def move(self, src, dest):
79
+ raise NotImplementedError("HTTP object move is unavailable")
80
+
81
+ def copy_file(self, src, dest):
82
+ raise NotImplementedError("Use bounded FileIO streams to copy objects")
83
+
84
+ def open_output_stream(self, path, metadata=None):
85
+ raise NotImplementedError("Use HttpFileIO.new_output().create() for conditional writes")
86
+
87
+ def open_append_stream(self, path, metadata=None):
88
+ raise NotImplementedError("Appending to immutable HTTP objects is unavailable")
89
+
90
+ return PyFileSystem(Handler())
@@ -0,0 +1,296 @@
1
+ """PyIceberg adapter. All data paths and FileIO properties come from the catalog.
2
+
3
+ The only protected API access is the staged table of CreateTableTransaction.
4
+ It is isolated below and guarded so incompatible clients fail before data generation.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import importlib.metadata
10
+ import time
11
+ from dataclasses import dataclass
12
+ from typing import Any
13
+
14
+ from .errors import CompatibilityError, ExistingTablesError, LoadError
15
+ from .models import Options, TableData
16
+ from .security import Redactor
17
+ from .util import remote_uri
18
+
19
+ CROWDB_FILE_IO = "crowdb_tpc_loader.crowdb_fileio.CrowdbFileIO"
20
+
21
+
22
+ @dataclass(frozen=True)
23
+ class RemotePart:
24
+ uri: str
25
+ rows: int
26
+ size_bytes: int
27
+
28
+
29
+ @dataclass(frozen=True)
30
+ class Inventory:
31
+ snapshot_id: int | None
32
+ files: dict[str, RemotePart]
33
+ table_uuid: str
34
+
35
+
36
+ def data_location(table: Any, filename: str) -> str:
37
+ """Honor the catalog's table location AND Iceberg location provider properties."""
38
+ provider_getter = getattr(table, "location_provider", None)
39
+ if callable(provider_getter):
40
+ provider = provider_getter()
41
+ else:
42
+ from pyiceberg.table.locations import load_location_provider
43
+
44
+ provider = load_location_provider(table.location(), table.metadata.properties)
45
+ try:
46
+ return remote_uri(provider.new_data_location(filename))
47
+ except CompatibilityError:
48
+ raise
49
+ except Exception as exc:
50
+ raise CompatibilityError(
51
+ f"Cannot obtain a durable data location from the catalog's table metadata: {exc}"
52
+ ) from exc
53
+
54
+
55
+ def table_uuid(table: Any) -> str:
56
+ return str(table.metadata.table_uuid)
57
+
58
+
59
+ def inspect_inventory(table: Any) -> Inventory:
60
+ """Read live manifest entries, not a data scan: empty Parquet files must count too."""
61
+ snapshot = table.current_snapshot()
62
+ files: dict[str, RemotePart] = {}
63
+ if snapshot is not None:
64
+ for manifest in snapshot.manifests(table.io):
65
+ for entry in manifest.fetch_manifest_entry(table.io, discard_deleted=True):
66
+ item = entry.data_file
67
+ if int(item.content) != 0:
68
+ raise LoadError("Unexpected delete file in a newly created unpartitioned benchmark table")
69
+ uri = str(item.file_path)
70
+ if uri in files:
71
+ raise LoadError("Duplicate data URI found in the current snapshot's live manifests")
72
+ files[uri] = RemotePart(uri, int(item.record_count), int(item.file_size_in_bytes))
73
+ return Inventory(snapshot.snapshot_id if snapshot else None, files, table_uuid(table))
74
+
75
+
76
+ def schema_compatible(table: Any, data: TableData) -> None:
77
+ """Validate the returned Iceberg schema before any benchmark upload."""
78
+ from pyiceberg.io.pyarrow import schema_to_pyarrow
79
+ import pyarrow as pa
80
+
81
+ actual = schema_to_pyarrow(table.schema(), include_field_ids=False)
82
+ if actual.names != data.schema.names:
83
+ raise CompatibilityError(f"{data.name}: catalog returned different column names/order")
84
+ for left, right in zip(actual, data.schema):
85
+ both_strings = (pa.types.is_string(left.type) or pa.types.is_large_string(left.type)) and (
86
+ pa.types.is_string(right.type) or pa.types.is_large_string(right.type)
87
+ )
88
+ if left.type != right.type and not both_strings:
89
+ raise CompatibilityError(
90
+ f"{data.name}.{left.name}: Iceberg type {left.type} differs from Parquet type {right.type}"
91
+ )
92
+ if not left.nullable and right.nullable:
93
+ raise CompatibilityError(
94
+ f"{data.name}.{left.name}: catalog strengthened a nullable column to required"
95
+ )
96
+ if table.spec().fields:
97
+ raise CompatibilityError("First-release benchmark tables must be unpartitioned")
98
+
99
+
100
+ class IcebergBackend:
101
+ def __init__(self, options: Options, redactor: Redactor):
102
+ self.options, self.redactor = options, redactor
103
+ self.catalog: Any = None
104
+ self.probe_cleanup_supported = True
105
+ self.staging = None
106
+
107
+ def connect(self) -> None:
108
+ try:
109
+ from packaging.version import Version
110
+ from .rest_catalog import create_catalog
111
+
112
+ version = importlib.metadata.version("pyiceberg")
113
+ if not Version("0.10") <= Version(version) < Version("0.11"):
114
+ raise CompatibilityError(
115
+ f"PyIceberg {version} is outside this release's 0.10.x adapter range"
116
+ )
117
+ properties = {"py-io-impl": CROWDB_FILE_IO}
118
+ properties.update(self.options.catalog_properties)
119
+ properties.update({"uri": self.options.catalog_uri, "http.timeout": str(self.options.timeout)})
120
+ if self.options.token is not None:
121
+ properties["token"] = self.options.token
122
+ self.redactor.learn(properties)
123
+ self.catalog = create_catalog("crowdb_tpc_loader", self.options.timeout, properties)
124
+ self.probe_cleanup_supported = self.catalog.properties.get("py-io-impl") != CROWDB_FILE_IO
125
+ self.redactor.learn(self.catalog.properties)
126
+ except CompatibilityError:
127
+ raise
128
+ except ImportError as exc:
129
+ raise CompatibilityError("PyIceberg and its FileIO dependencies are required for load") from exc
130
+ except Exception as exc:
131
+ raise LoadError(
132
+ f"REST Catalog connection failed: {exc}. Check --catalog-uri/ICEBERG_URI and credentials."
133
+ ) from exc
134
+
135
+ def fork(self) -> "IcebergBackend":
136
+ from .security import Redactor
137
+
138
+ other = IcebergBackend(self.options, Redactor(tuple(self.redactor.secrets)))
139
+ other.connect()
140
+ if self.staging is not None:
141
+ other.set_staging(self.staging)
142
+ return other
143
+
144
+ def existing(self, names: tuple[str, ...]) -> set[str]:
145
+ from pyiceberg.exceptions import NoSuchNamespaceError
146
+
147
+ try:
148
+ identifiers = self.catalog.list_tables(self.options.namespace)
149
+ except NoSuchNamespaceError:
150
+ return set()
151
+ except Exception as exc:
152
+ raise LoadError(f"Cannot list target tables (credentials/permissions/network): {exc}") from exc
153
+ return {identifier[-1] for identifier in identifiers if identifier[-1] in names}
154
+
155
+ def ensure_namespace(self) -> None:
156
+ from pyiceberg.exceptions import NamespaceAlreadyExistsError
157
+
158
+ for length in range(1, len(self.options.namespace) + 1):
159
+ try:
160
+ self.catalog.create_namespace(self.options.namespace[:length])
161
+ except NamespaceAlreadyExistsError:
162
+ continue
163
+ except Exception as exc:
164
+ raise LoadError(
165
+ f"Cannot create namespace {'.'.join(self.options.namespace[:length])}: {exc}"
166
+ ) from exc
167
+
168
+ def set_staging(self, scratch) -> None:
169
+ self.staging = scratch
170
+ self.catalog.properties.setdefault("http.spool-directory", str(scratch))
171
+
172
+ def stage_probe(self, run_id: str):
173
+ """Ask the REST server for an UNCOMMITTED location; never publish a probe table."""
174
+ import pyarrow as pa
175
+
176
+ try:
177
+ transaction = self.catalog.create_table_transaction(
178
+ (*self.options.namespace, f"__crowdb_tpc_probe_{run_id}"),
179
+ schema=pa.schema([pa.field("probe", pa.int64(), nullable=True)]),
180
+ properties={"format-version": "2"},
181
+ )
182
+ table = getattr(transaction, "_table", None)
183
+ if table is None:
184
+ raise CompatibilityError("PyIceberg staged table API is unavailable")
185
+ self._learn(table)
186
+ # Deliberately no `with transaction`, commit_transaction or add_files here.
187
+ return table
188
+ except CompatibilityError:
189
+ raise
190
+ except Exception as exc:
191
+ raise CompatibilityError(
192
+ f"Cannot obtain an uncommitted FileIO probe location using REST stage-create: {exc}. "
193
+ "This release requires staged table creation for side-effect-safe preflight. "
194
+ "Check CrowDB/PyIceberg compatibility and the configured py-io-impl; generate remains usable. "
195
+ "No benchmark table has been created."
196
+ ) from exc
197
+
198
+ def _learn(self, table: Any) -> None:
199
+ self.redactor.learn(getattr(table.io, "properties", {}))
200
+ self.redactor.learn(getattr(table, "config", {}))
201
+ self.redactor.learn(table.metadata.properties)
202
+ for method in ("new_input", "new_output", "delete"):
203
+ if not callable(getattr(table.io, method, None)):
204
+ raise CompatibilityError(f"Catalog FileIO does not implement {method}")
205
+
206
+ def create(self, data: TableData, run_id: str, generator: dict[str, Any]):
207
+ from pyiceberg.exceptions import ServiceUnavailableError, TableAlreadyExistsError
208
+
209
+ properties = {
210
+ "format-version": "2",
211
+ "commit.retry.num-retries": "0",
212
+ "crowdb-tpc-loader.run-id": run_id,
213
+ "crowdb-tpc-loader.benchmark": self.options.benchmark,
214
+ "crowdb-tpc-loader.scale-factor": str(self.options.sf),
215
+ "crowdb-tpc-loader.generator": str(generator["implementation"]),
216
+ "crowdb-tpc-loader.generator-version": str(generator["version"]),
217
+ }
218
+ try:
219
+ for attempt in range(6):
220
+ try:
221
+ table = self.catalog.create_table(
222
+ (*self.options.namespace, data.name), schema=data.schema, properties=properties
223
+ )
224
+ break
225
+ except ServiceUnavailableError:
226
+ if attempt == 5:
227
+ raise
228
+ time.sleep(0.25 * (attempt + 1))
229
+ except TableAlreadyExistsError:
230
+ if attempt == 0:
231
+ raise
232
+ table = self.catalog.load_table((*self.options.namespace, data.name))
233
+ if table.metadata.properties.get("crowdb-tpc-loader.run-id") != run_id:
234
+ raise
235
+ break
236
+ self._learn(table)
237
+ schema_compatible(table, data)
238
+ return table
239
+ except TableAlreadyExistsError as exc:
240
+ if self.options.on_exists == "skip":
241
+ return None
242
+ raise ExistingTablesError(
243
+ f"Table {data.name} appeared during this run; refusing to modify it"
244
+ ) from exc
245
+ except (CompatibilityError, LoadError):
246
+ raise
247
+ except Exception as exc:
248
+ raise LoadError(f"Could not create/validate table {data.name}: {exc}") from exc
249
+
250
+ def validate_import(self, table: Any, uris: list[str]) -> None:
251
+ """Run the exact PyIceberg Parquet-to-DataFile conversion without committing."""
252
+ try:
253
+ from pyiceberg.io.pyarrow import parquet_files_to_data_files
254
+
255
+ for _ in parquet_files_to_data_files(table.io, table.metadata, iter(uris)):
256
+ pass
257
+ except Exception as exc:
258
+ raise CompatibilityError(
259
+ f"This FileIO/PyIceberg combination cannot inspect/register remote Parquet: {exc}"
260
+ ) from exc
261
+
262
+ def reload_inventory(self, table: Any) -> tuple[Any, Inventory]:
263
+ refreshed = self.catalog.load_table(table.name())
264
+ self._learn(refreshed)
265
+ if table_uuid(refreshed) != table_uuid(table):
266
+ raise LoadError(
267
+ "Table identity changed concurrently; refusing to treat a replacement table as this run's table"
268
+ )
269
+ return refreshed, inspect_inventory(refreshed)
270
+
271
+ def assert_empty(self, table: Any) -> Any:
272
+ refreshed, state = self.reload_inventory(table)
273
+ if state.snapshot_id is not None or state.files:
274
+ raise LoadError("New benchmark table was modified by another writer; refusing to append")
275
+ return refreshed
276
+
277
+ def register(self, table: Any, uris: list[str], run_id: str) -> None:
278
+ for uri in uris:
279
+ remote_uri(uri)
280
+ table.add_files(
281
+ file_paths=uris,
282
+ check_duplicate_files=True,
283
+ snapshot_properties={"crowdb-tpc-loader.run-id": run_id},
284
+ )
285
+
286
+ @staticmethod
287
+ def definite_rejection(error: Exception) -> bool:
288
+ # A transport failure / generic server exception is never assumed to be a rejected commit.
289
+ from pyiceberg.exceptions import (
290
+ BadRequestError,
291
+ CommitFailedException,
292
+ ForbiddenError,
293
+ UnauthorizedError,
294
+ )
295
+
296
+ return isinstance(error, (BadRequestError, CommitFailedException, ForbiddenError, UnauthorizedError))