crowdb-tpc-loader 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- crowdb_tpc_loader/__init__.py +3 -0
- crowdb_tpc_loader/__main__.py +3 -0
- crowdb_tpc_loader/arrow_http_bridge.py +90 -0
- crowdb_tpc_loader/backend.py +296 -0
- crowdb_tpc_loader/cli.py +325 -0
- crowdb_tpc_loader/crowdb_fileio.py +40 -0
- crowdb_tpc_loader/errors.py +37 -0
- crowdb_tpc_loader/generators/__init__.py +18 -0
- crowdb_tpc_loader/generators/binary.py +191 -0
- crowdb_tpc_loader/generators/tpcds.py +134 -0
- crowdb_tpc_loader/generators/tpch.py +153 -0
- crowdb_tpc_loader/http_fileio.py +303 -0
- crowdb_tpc_loader/loader.py +287 -0
- crowdb_tpc_loader/models.py +70 -0
- crowdb_tpc_loader/report.py +140 -0
- crowdb_tpc_loader/rest_catalog.py +49 -0
- crowdb_tpc_loader/runner.py +236 -0
- crowdb_tpc_loader/s3_upload.py +146 -0
- crowdb_tpc_loader/schemas.py +197 -0
- crowdb_tpc_loader/security.py +98 -0
- crowdb_tpc_loader/transfers.py +99 -0
- crowdb_tpc_loader/util.py +217 -0
- crowdb_tpc_loader/validation.py +172 -0
- crowdb_tpc_loader-0.1.0.dist-info/METADATA +96 -0
- crowdb_tpc_loader-0.1.0.dist-info/RECORD +29 -0
- crowdb_tpc_loader-0.1.0.dist-info/WHEEL +5 -0
- crowdb_tpc_loader-0.1.0.dist-info/entry_points.txt +2 -0
- crowdb_tpc_loader-0.1.0.dist-info/licenses/LICENSE +201 -0
- crowdb_tpc_loader-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
"""A read-only Arrow filesystem bridge for independent PyIceberg HTTP scans.
|
|
2
|
+
|
|
3
|
+
Metadata/data writes still use HttpFileIO's conditional upload streams. Explicit
|
|
4
|
+
paths are supported; recursive object listing and filesystem mutations are not.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from urllib.parse import urlsplit
|
|
10
|
+
|
|
11
|
+
from .errors import CompatibilityError
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def make_filesystem(fileio, scheme: str, netloc: str):
|
|
15
|
+
import pyarrow as pa
|
|
16
|
+
from pyarrow.fs import FileInfo, FileSystemHandler, FileType, PyFileSystem
|
|
17
|
+
from .http_fileio import origin
|
|
18
|
+
|
|
19
|
+
expected_origin = origin(f"{scheme}://{netloc}")
|
|
20
|
+
|
|
21
|
+
class Handler(FileSystemHandler):
|
|
22
|
+
def _uri(self, path):
|
|
23
|
+
if urlsplit(path).scheme in {"http", "https"}:
|
|
24
|
+
uri = path
|
|
25
|
+
else:
|
|
26
|
+
path = path.lstrip("/")
|
|
27
|
+
# PyIceberg's parse_location returns `netloc/path` for HTTP.
|
|
28
|
+
uri = f"{scheme}://{path}" if path.startswith(netloc + "/") else f"{scheme}://{netloc}/{path}"
|
|
29
|
+
if origin(uri) != expected_origin:
|
|
30
|
+
raise CompatibilityError("Arrow HTTP filesystem path changed origin")
|
|
31
|
+
return uri
|
|
32
|
+
|
|
33
|
+
def get_type_name(self):
|
|
34
|
+
return "crowdb-tpc-http"
|
|
35
|
+
|
|
36
|
+
def normalize_path(self, path):
|
|
37
|
+
self._uri(path)
|
|
38
|
+
return path
|
|
39
|
+
|
|
40
|
+
def get_file_info(self, paths):
|
|
41
|
+
result = []
|
|
42
|
+
for path in paths:
|
|
43
|
+
try:
|
|
44
|
+
size = len(fileio.new_input(self._uri(path)))
|
|
45
|
+
result.append(FileInfo(path, FileType.File, size=size))
|
|
46
|
+
except FileNotFoundError:
|
|
47
|
+
result.append(FileInfo(path, FileType.NotFound))
|
|
48
|
+
return result
|
|
49
|
+
|
|
50
|
+
def get_file_info_selector(self, selector):
|
|
51
|
+
raise NotImplementedError(
|
|
52
|
+
"HTTP object listing is not supported; Iceberg supplies explicit manifest file paths"
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
def open_input_file(self, path):
|
|
56
|
+
return pa.PythonFile(fileio.new_input(self._uri(path)).open(), mode="r")
|
|
57
|
+
|
|
58
|
+
def open_input_stream(self, path):
|
|
59
|
+
return self.open_input_file(path)
|
|
60
|
+
|
|
61
|
+
def create_dir(self, path, recursive=True):
|
|
62
|
+
raise NotImplementedError("Use HttpFileIO output streams, not directory operations")
|
|
63
|
+
|
|
64
|
+
def delete_dir(self, path):
|
|
65
|
+
raise NotImplementedError("Recursive deletion is deliberately unavailable")
|
|
66
|
+
|
|
67
|
+
def delete_dir_contents(self, path, missing_dir_ok=False):
|
|
68
|
+
raise NotImplementedError("Recursive deletion is deliberately unavailable")
|
|
69
|
+
|
|
70
|
+
def delete_root_dir_contents(self):
|
|
71
|
+
raise NotImplementedError("Recursive deletion is deliberately unavailable")
|
|
72
|
+
|
|
73
|
+
def delete_file(self, path):
|
|
74
|
+
raise NotImplementedError(
|
|
75
|
+
"Use explicit HttpFileIO.delete for independently verified uncommitted objects"
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
def move(self, src, dest):
|
|
79
|
+
raise NotImplementedError("HTTP object move is unavailable")
|
|
80
|
+
|
|
81
|
+
def copy_file(self, src, dest):
|
|
82
|
+
raise NotImplementedError("Use bounded FileIO streams to copy objects")
|
|
83
|
+
|
|
84
|
+
def open_output_stream(self, path, metadata=None):
|
|
85
|
+
raise NotImplementedError("Use HttpFileIO.new_output().create() for conditional writes")
|
|
86
|
+
|
|
87
|
+
def open_append_stream(self, path, metadata=None):
|
|
88
|
+
raise NotImplementedError("Appending to immutable HTTP objects is unavailable")
|
|
89
|
+
|
|
90
|
+
return PyFileSystem(Handler())
|
|
@@ -0,0 +1,296 @@
|
|
|
1
|
+
"""PyIceberg adapter. All data paths and FileIO properties come from the catalog.
|
|
2
|
+
|
|
3
|
+
The only protected API access is the staged table of CreateTableTransaction.
|
|
4
|
+
It is isolated below and guarded so incompatible clients fail before data generation.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import importlib.metadata
|
|
10
|
+
import time
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
from .errors import CompatibilityError, ExistingTablesError, LoadError
|
|
15
|
+
from .models import Options, TableData
|
|
16
|
+
from .security import Redactor
|
|
17
|
+
from .util import remote_uri
|
|
18
|
+
|
|
19
|
+
CROWDB_FILE_IO = "crowdb_tpc_loader.crowdb_fileio.CrowdbFileIO"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass(frozen=True)
|
|
23
|
+
class RemotePart:
|
|
24
|
+
uri: str
|
|
25
|
+
rows: int
|
|
26
|
+
size_bytes: int
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass(frozen=True)
|
|
30
|
+
class Inventory:
|
|
31
|
+
snapshot_id: int | None
|
|
32
|
+
files: dict[str, RemotePart]
|
|
33
|
+
table_uuid: str
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def data_location(table: Any, filename: str) -> str:
|
|
37
|
+
"""Honor the catalog's table location AND Iceberg location provider properties."""
|
|
38
|
+
provider_getter = getattr(table, "location_provider", None)
|
|
39
|
+
if callable(provider_getter):
|
|
40
|
+
provider = provider_getter()
|
|
41
|
+
else:
|
|
42
|
+
from pyiceberg.table.locations import load_location_provider
|
|
43
|
+
|
|
44
|
+
provider = load_location_provider(table.location(), table.metadata.properties)
|
|
45
|
+
try:
|
|
46
|
+
return remote_uri(provider.new_data_location(filename))
|
|
47
|
+
except CompatibilityError:
|
|
48
|
+
raise
|
|
49
|
+
except Exception as exc:
|
|
50
|
+
raise CompatibilityError(
|
|
51
|
+
f"Cannot obtain a durable data location from the catalog's table metadata: {exc}"
|
|
52
|
+
) from exc
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def table_uuid(table: Any) -> str:
|
|
56
|
+
return str(table.metadata.table_uuid)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def inspect_inventory(table: Any) -> Inventory:
|
|
60
|
+
"""Read live manifest entries, not a data scan: empty Parquet files must count too."""
|
|
61
|
+
snapshot = table.current_snapshot()
|
|
62
|
+
files: dict[str, RemotePart] = {}
|
|
63
|
+
if snapshot is not None:
|
|
64
|
+
for manifest in snapshot.manifests(table.io):
|
|
65
|
+
for entry in manifest.fetch_manifest_entry(table.io, discard_deleted=True):
|
|
66
|
+
item = entry.data_file
|
|
67
|
+
if int(item.content) != 0:
|
|
68
|
+
raise LoadError("Unexpected delete file in a newly created unpartitioned benchmark table")
|
|
69
|
+
uri = str(item.file_path)
|
|
70
|
+
if uri in files:
|
|
71
|
+
raise LoadError("Duplicate data URI found in the current snapshot's live manifests")
|
|
72
|
+
files[uri] = RemotePart(uri, int(item.record_count), int(item.file_size_in_bytes))
|
|
73
|
+
return Inventory(snapshot.snapshot_id if snapshot else None, files, table_uuid(table))
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def schema_compatible(table: Any, data: TableData) -> None:
|
|
77
|
+
"""Validate the returned Iceberg schema before any benchmark upload."""
|
|
78
|
+
from pyiceberg.io.pyarrow import schema_to_pyarrow
|
|
79
|
+
import pyarrow as pa
|
|
80
|
+
|
|
81
|
+
actual = schema_to_pyarrow(table.schema(), include_field_ids=False)
|
|
82
|
+
if actual.names != data.schema.names:
|
|
83
|
+
raise CompatibilityError(f"{data.name}: catalog returned different column names/order")
|
|
84
|
+
for left, right in zip(actual, data.schema):
|
|
85
|
+
both_strings = (pa.types.is_string(left.type) or pa.types.is_large_string(left.type)) and (
|
|
86
|
+
pa.types.is_string(right.type) or pa.types.is_large_string(right.type)
|
|
87
|
+
)
|
|
88
|
+
if left.type != right.type and not both_strings:
|
|
89
|
+
raise CompatibilityError(
|
|
90
|
+
f"{data.name}.{left.name}: Iceberg type {left.type} differs from Parquet type {right.type}"
|
|
91
|
+
)
|
|
92
|
+
if not left.nullable and right.nullable:
|
|
93
|
+
raise CompatibilityError(
|
|
94
|
+
f"{data.name}.{left.name}: catalog strengthened a nullable column to required"
|
|
95
|
+
)
|
|
96
|
+
if table.spec().fields:
|
|
97
|
+
raise CompatibilityError("First-release benchmark tables must be unpartitioned")
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
class IcebergBackend:
|
|
101
|
+
def __init__(self, options: Options, redactor: Redactor):
|
|
102
|
+
self.options, self.redactor = options, redactor
|
|
103
|
+
self.catalog: Any = None
|
|
104
|
+
self.probe_cleanup_supported = True
|
|
105
|
+
self.staging = None
|
|
106
|
+
|
|
107
|
+
def connect(self) -> None:
|
|
108
|
+
try:
|
|
109
|
+
from packaging.version import Version
|
|
110
|
+
from .rest_catalog import create_catalog
|
|
111
|
+
|
|
112
|
+
version = importlib.metadata.version("pyiceberg")
|
|
113
|
+
if not Version("0.10") <= Version(version) < Version("0.11"):
|
|
114
|
+
raise CompatibilityError(
|
|
115
|
+
f"PyIceberg {version} is outside this release's 0.10.x adapter range"
|
|
116
|
+
)
|
|
117
|
+
properties = {"py-io-impl": CROWDB_FILE_IO}
|
|
118
|
+
properties.update(self.options.catalog_properties)
|
|
119
|
+
properties.update({"uri": self.options.catalog_uri, "http.timeout": str(self.options.timeout)})
|
|
120
|
+
if self.options.token is not None:
|
|
121
|
+
properties["token"] = self.options.token
|
|
122
|
+
self.redactor.learn(properties)
|
|
123
|
+
self.catalog = create_catalog("crowdb_tpc_loader", self.options.timeout, properties)
|
|
124
|
+
self.probe_cleanup_supported = self.catalog.properties.get("py-io-impl") != CROWDB_FILE_IO
|
|
125
|
+
self.redactor.learn(self.catalog.properties)
|
|
126
|
+
except CompatibilityError:
|
|
127
|
+
raise
|
|
128
|
+
except ImportError as exc:
|
|
129
|
+
raise CompatibilityError("PyIceberg and its FileIO dependencies are required for load") from exc
|
|
130
|
+
except Exception as exc:
|
|
131
|
+
raise LoadError(
|
|
132
|
+
f"REST Catalog connection failed: {exc}. Check --catalog-uri/ICEBERG_URI and credentials."
|
|
133
|
+
) from exc
|
|
134
|
+
|
|
135
|
+
def fork(self) -> "IcebergBackend":
|
|
136
|
+
from .security import Redactor
|
|
137
|
+
|
|
138
|
+
other = IcebergBackend(self.options, Redactor(tuple(self.redactor.secrets)))
|
|
139
|
+
other.connect()
|
|
140
|
+
if self.staging is not None:
|
|
141
|
+
other.set_staging(self.staging)
|
|
142
|
+
return other
|
|
143
|
+
|
|
144
|
+
def existing(self, names: tuple[str, ...]) -> set[str]:
|
|
145
|
+
from pyiceberg.exceptions import NoSuchNamespaceError
|
|
146
|
+
|
|
147
|
+
try:
|
|
148
|
+
identifiers = self.catalog.list_tables(self.options.namespace)
|
|
149
|
+
except NoSuchNamespaceError:
|
|
150
|
+
return set()
|
|
151
|
+
except Exception as exc:
|
|
152
|
+
raise LoadError(f"Cannot list target tables (credentials/permissions/network): {exc}") from exc
|
|
153
|
+
return {identifier[-1] for identifier in identifiers if identifier[-1] in names}
|
|
154
|
+
|
|
155
|
+
def ensure_namespace(self) -> None:
|
|
156
|
+
from pyiceberg.exceptions import NamespaceAlreadyExistsError
|
|
157
|
+
|
|
158
|
+
for length in range(1, len(self.options.namespace) + 1):
|
|
159
|
+
try:
|
|
160
|
+
self.catalog.create_namespace(self.options.namespace[:length])
|
|
161
|
+
except NamespaceAlreadyExistsError:
|
|
162
|
+
continue
|
|
163
|
+
except Exception as exc:
|
|
164
|
+
raise LoadError(
|
|
165
|
+
f"Cannot create namespace {'.'.join(self.options.namespace[:length])}: {exc}"
|
|
166
|
+
) from exc
|
|
167
|
+
|
|
168
|
+
def set_staging(self, scratch) -> None:
|
|
169
|
+
self.staging = scratch
|
|
170
|
+
self.catalog.properties.setdefault("http.spool-directory", str(scratch))
|
|
171
|
+
|
|
172
|
+
def stage_probe(self, run_id: str):
|
|
173
|
+
"""Ask the REST server for an UNCOMMITTED location; never publish a probe table."""
|
|
174
|
+
import pyarrow as pa
|
|
175
|
+
|
|
176
|
+
try:
|
|
177
|
+
transaction = self.catalog.create_table_transaction(
|
|
178
|
+
(*self.options.namespace, f"__crowdb_tpc_probe_{run_id}"),
|
|
179
|
+
schema=pa.schema([pa.field("probe", pa.int64(), nullable=True)]),
|
|
180
|
+
properties={"format-version": "2"},
|
|
181
|
+
)
|
|
182
|
+
table = getattr(transaction, "_table", None)
|
|
183
|
+
if table is None:
|
|
184
|
+
raise CompatibilityError("PyIceberg staged table API is unavailable")
|
|
185
|
+
self._learn(table)
|
|
186
|
+
# Deliberately no `with transaction`, commit_transaction or add_files here.
|
|
187
|
+
return table
|
|
188
|
+
except CompatibilityError:
|
|
189
|
+
raise
|
|
190
|
+
except Exception as exc:
|
|
191
|
+
raise CompatibilityError(
|
|
192
|
+
f"Cannot obtain an uncommitted FileIO probe location using REST stage-create: {exc}. "
|
|
193
|
+
"This release requires staged table creation for side-effect-safe preflight. "
|
|
194
|
+
"Check CrowDB/PyIceberg compatibility and the configured py-io-impl; generate remains usable. "
|
|
195
|
+
"No benchmark table has been created."
|
|
196
|
+
) from exc
|
|
197
|
+
|
|
198
|
+
def _learn(self, table: Any) -> None:
|
|
199
|
+
self.redactor.learn(getattr(table.io, "properties", {}))
|
|
200
|
+
self.redactor.learn(getattr(table, "config", {}))
|
|
201
|
+
self.redactor.learn(table.metadata.properties)
|
|
202
|
+
for method in ("new_input", "new_output", "delete"):
|
|
203
|
+
if not callable(getattr(table.io, method, None)):
|
|
204
|
+
raise CompatibilityError(f"Catalog FileIO does not implement {method}")
|
|
205
|
+
|
|
206
|
+
def create(self, data: TableData, run_id: str, generator: dict[str, Any]):
|
|
207
|
+
from pyiceberg.exceptions import ServiceUnavailableError, TableAlreadyExistsError
|
|
208
|
+
|
|
209
|
+
properties = {
|
|
210
|
+
"format-version": "2",
|
|
211
|
+
"commit.retry.num-retries": "0",
|
|
212
|
+
"crowdb-tpc-loader.run-id": run_id,
|
|
213
|
+
"crowdb-tpc-loader.benchmark": self.options.benchmark,
|
|
214
|
+
"crowdb-tpc-loader.scale-factor": str(self.options.sf),
|
|
215
|
+
"crowdb-tpc-loader.generator": str(generator["implementation"]),
|
|
216
|
+
"crowdb-tpc-loader.generator-version": str(generator["version"]),
|
|
217
|
+
}
|
|
218
|
+
try:
|
|
219
|
+
for attempt in range(6):
|
|
220
|
+
try:
|
|
221
|
+
table = self.catalog.create_table(
|
|
222
|
+
(*self.options.namespace, data.name), schema=data.schema, properties=properties
|
|
223
|
+
)
|
|
224
|
+
break
|
|
225
|
+
except ServiceUnavailableError:
|
|
226
|
+
if attempt == 5:
|
|
227
|
+
raise
|
|
228
|
+
time.sleep(0.25 * (attempt + 1))
|
|
229
|
+
except TableAlreadyExistsError:
|
|
230
|
+
if attempt == 0:
|
|
231
|
+
raise
|
|
232
|
+
table = self.catalog.load_table((*self.options.namespace, data.name))
|
|
233
|
+
if table.metadata.properties.get("crowdb-tpc-loader.run-id") != run_id:
|
|
234
|
+
raise
|
|
235
|
+
break
|
|
236
|
+
self._learn(table)
|
|
237
|
+
schema_compatible(table, data)
|
|
238
|
+
return table
|
|
239
|
+
except TableAlreadyExistsError as exc:
|
|
240
|
+
if self.options.on_exists == "skip":
|
|
241
|
+
return None
|
|
242
|
+
raise ExistingTablesError(
|
|
243
|
+
f"Table {data.name} appeared during this run; refusing to modify it"
|
|
244
|
+
) from exc
|
|
245
|
+
except (CompatibilityError, LoadError):
|
|
246
|
+
raise
|
|
247
|
+
except Exception as exc:
|
|
248
|
+
raise LoadError(f"Could not create/validate table {data.name}: {exc}") from exc
|
|
249
|
+
|
|
250
|
+
def validate_import(self, table: Any, uris: list[str]) -> None:
|
|
251
|
+
"""Run the exact PyIceberg Parquet-to-DataFile conversion without committing."""
|
|
252
|
+
try:
|
|
253
|
+
from pyiceberg.io.pyarrow import parquet_files_to_data_files
|
|
254
|
+
|
|
255
|
+
for _ in parquet_files_to_data_files(table.io, table.metadata, iter(uris)):
|
|
256
|
+
pass
|
|
257
|
+
except Exception as exc:
|
|
258
|
+
raise CompatibilityError(
|
|
259
|
+
f"This FileIO/PyIceberg combination cannot inspect/register remote Parquet: {exc}"
|
|
260
|
+
) from exc
|
|
261
|
+
|
|
262
|
+
def reload_inventory(self, table: Any) -> tuple[Any, Inventory]:
|
|
263
|
+
refreshed = self.catalog.load_table(table.name())
|
|
264
|
+
self._learn(refreshed)
|
|
265
|
+
if table_uuid(refreshed) != table_uuid(table):
|
|
266
|
+
raise LoadError(
|
|
267
|
+
"Table identity changed concurrently; refusing to treat a replacement table as this run's table"
|
|
268
|
+
)
|
|
269
|
+
return refreshed, inspect_inventory(refreshed)
|
|
270
|
+
|
|
271
|
+
def assert_empty(self, table: Any) -> Any:
|
|
272
|
+
refreshed, state = self.reload_inventory(table)
|
|
273
|
+
if state.snapshot_id is not None or state.files:
|
|
274
|
+
raise LoadError("New benchmark table was modified by another writer; refusing to append")
|
|
275
|
+
return refreshed
|
|
276
|
+
|
|
277
|
+
def register(self, table: Any, uris: list[str], run_id: str) -> None:
|
|
278
|
+
for uri in uris:
|
|
279
|
+
remote_uri(uri)
|
|
280
|
+
table.add_files(
|
|
281
|
+
file_paths=uris,
|
|
282
|
+
check_duplicate_files=True,
|
|
283
|
+
snapshot_properties={"crowdb-tpc-loader.run-id": run_id},
|
|
284
|
+
)
|
|
285
|
+
|
|
286
|
+
@staticmethod
|
|
287
|
+
def definite_rejection(error: Exception) -> bool:
|
|
288
|
+
# A transport failure / generic server exception is never assumed to be a rejected commit.
|
|
289
|
+
from pyiceberg.exceptions import (
|
|
290
|
+
BadRequestError,
|
|
291
|
+
CommitFailedException,
|
|
292
|
+
ForbiddenError,
|
|
293
|
+
UnauthorizedError,
|
|
294
|
+
)
|
|
295
|
+
|
|
296
|
+
return isinstance(error, (BadRequestError, CommitFailedException, ForbiddenError, UnauthorizedError))
|