crowdb-tpc-loader 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- crowdb_tpc_loader-0.1.0/CHANGELOG.md +9 -0
- crowdb_tpc_loader-0.1.0/LICENSE +201 -0
- crowdb_tpc_loader-0.1.0/MANIFEST.in +5 -0
- crowdb_tpc_loader-0.1.0/PKG-INFO +96 -0
- crowdb_tpc_loader-0.1.0/README.md +63 -0
- crowdb_tpc_loader-0.1.0/docs/COMPATIBILITY.md +13 -0
- crowdb_tpc_loader-0.1.0/docs/RECOVERY.md +49 -0
- crowdb_tpc_loader-0.1.0/docs/TESTING.md +37 -0
- crowdb_tpc_loader-0.1.0/docs/TEST_REPORT.md +10 -0
- crowdb_tpc_loader-0.1.0/pyproject.toml +64 -0
- crowdb_tpc_loader-0.1.0/requirements-integration.txt +11 -0
- crowdb_tpc_loader-0.1.0/scripts/smoke_test.py +71 -0
- crowdb_tpc_loader-0.1.0/scripts/verify_crowdb.py +157 -0
- crowdb_tpc_loader-0.1.0/setup.cfg +4 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/__init__.py +3 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/__main__.py +3 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/arrow_http_bridge.py +90 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/backend.py +296 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/cli.py +325 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/crowdb_fileio.py +40 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/errors.py +37 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/generators/__init__.py +18 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/generators/binary.py +191 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/generators/tpcds.py +134 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/generators/tpch.py +153 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/http_fileio.py +303 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/loader.py +287 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/models.py +70 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/report.py +140 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/rest_catalog.py +49 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/runner.py +236 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/s3_upload.py +146 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/schemas.py +197 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/security.py +98 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/transfers.py +99 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/util.py +217 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/validation.py +172 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader.egg-info/PKG-INFO +96 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader.egg-info/SOURCES.txt +61 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader.egg-info/dependency_links.txt +1 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader.egg-info/entry_points.txt +2 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader.egg-info/requires.txt +15 -0
- crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader.egg-info/top_level.txt +1 -0
- crowdb_tpc_loader-0.1.0/tests/__init__.py +0 -0
- crowdb_tpc_loader-0.1.0/tests/conftest.py +173 -0
- crowdb_tpc_loader-0.1.0/tests/integration/__init__.py +0 -0
- crowdb_tpc_loader-0.1.0/tests/integration/test_crowdb.py +62 -0
- crowdb_tpc_loader-0.1.0/tests/integration/test_generators.py +45 -0
- crowdb_tpc_loader-0.1.0/tests/integration/test_iceberg_http.py +66 -0
- crowdb_tpc_loader-0.1.0/tests/integration/test_parquet.py +71 -0
- crowdb_tpc_loader-0.1.0/tests/unit/__init__.py +0 -0
- crowdb_tpc_loader-0.1.0/tests/unit/test_binary_generator.py +127 -0
- crowdb_tpc_loader-0.1.0/tests/unit/test_cli_security.py +205 -0
- crowdb_tpc_loader-0.1.0/tests/unit/test_commits.py +260 -0
- crowdb_tpc_loader-0.1.0/tests/unit/test_crowdb_fileio.py +29 -0
- crowdb_tpc_loader-0.1.0/tests/unit/test_files_reports.py +273 -0
- crowdb_tpc_loader-0.1.0/tests/unit/test_http_fileio.py +277 -0
- crowdb_tpc_loader-0.1.0/tests/unit/test_rest_session.py +25 -0
- crowdb_tpc_loader-0.1.0/tests/unit/test_runner.py +219 -0
- crowdb_tpc_loader-0.1.0/tests/unit/test_s3_upload.py +83 -0
- crowdb_tpc_loader-0.1.0/tests/unit/test_tpcds_adapter.py +125 -0
- crowdb_tpc_loader-0.1.0/tests/unit/test_transfers.py +104 -0
- crowdb_tpc_loader-0.1.0/tests/unit/test_validation_logic.py +166 -0
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
## 0.1.0
|
|
4
|
+
|
|
5
|
+
- Load TPC-H and TPC-DS datasets into Iceberg tables, with validation and recoverable reports.
|
|
6
|
+
- Write separate tables concurrently, with 8 workers by default and a configurable 1–24 range.
|
|
7
|
+
- Retry transient Catalog 503 responses during table creation without retrying ambiguous table commits.
|
|
8
|
+
- Verify the published CROWDB Iceberg image with a small TPC-H load and DuckDB Q1 read.
|
|
9
|
+
- Add CI checks and a manually triggered PyPI release workflow.
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
Apache License
|
|
2
|
+
Version 2.0, January 2004
|
|
3
|
+
http://www.apache.org/licenses/
|
|
4
|
+
|
|
5
|
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
|
6
|
+
|
|
7
|
+
1. Definitions.
|
|
8
|
+
|
|
9
|
+
"License" shall mean the terms and conditions for use, reproduction,
|
|
10
|
+
and distribution as defined by Sections 1 through 9 of this document.
|
|
11
|
+
|
|
12
|
+
"Licensor" shall mean the copyright owner or entity authorized by
|
|
13
|
+
the copyright owner that is granting the License.
|
|
14
|
+
|
|
15
|
+
"Legal Entity" shall mean the union of the acting entity and all
|
|
16
|
+
other entities that control, are controlled by, or are under common
|
|
17
|
+
control with that entity. For the purposes of this definition,
|
|
18
|
+
"control" means (i) the power, direct or indirect, to cause the
|
|
19
|
+
direction or management of such entity, whether by contract or
|
|
20
|
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
|
21
|
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
|
22
|
+
|
|
23
|
+
"You" (or "Your") shall mean an individual or Legal Entity
|
|
24
|
+
exercising permissions granted by this License.
|
|
25
|
+
|
|
26
|
+
"Source" form shall mean the preferred form for making modifications,
|
|
27
|
+
including but not limited to software source code, documentation
|
|
28
|
+
source, and configuration files.
|
|
29
|
+
|
|
30
|
+
"Object" form shall mean any form resulting from mechanical
|
|
31
|
+
transformation or translation of a Source form, including but
|
|
32
|
+
not limited to compiled object code, generated documentation,
|
|
33
|
+
and conversions to other media types.
|
|
34
|
+
|
|
35
|
+
"Work" shall mean the work of authorship, whether in Source or
|
|
36
|
+
Object form, made available under the License, as indicated by a
|
|
37
|
+
copyright notice that is included in or attached to the work
|
|
38
|
+
(an example is provided in the Appendix below).
|
|
39
|
+
|
|
40
|
+
"Derivative Works" shall mean any work, whether in Source or Object
|
|
41
|
+
form, that is based on (or derived from) the Work and for which the
|
|
42
|
+
editorial revisions, annotations, elaborations, or other modifications
|
|
43
|
+
represent, as a whole, an original work of authorship. For the purposes
|
|
44
|
+
of this License, Derivative Works shall not include works that remain
|
|
45
|
+
separable from, or merely link (or bind by name) to the interfaces of,
|
|
46
|
+
the Work and Derivative Works thereof.
|
|
47
|
+
|
|
48
|
+
"Contribution" shall mean any work of authorship, including
|
|
49
|
+
the original version of the Work and any modifications or additions
|
|
50
|
+
to that Work or Derivative Works thereof, that is intentionally
|
|
51
|
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
|
52
|
+
or by an individual or Legal Entity authorized to submit on behalf of
|
|
53
|
+
the copyright owner. For the purposes of this definition, "submitted"
|
|
54
|
+
means any form of electronic, verbal, or written communication sent
|
|
55
|
+
to the Licensor or its representatives, including but not limited to
|
|
56
|
+
communication on electronic mailing lists, source code control systems,
|
|
57
|
+
and issue tracking systems that are managed by, or on behalf of, the
|
|
58
|
+
Licensor for the purpose of discussing and improving the Work, but
|
|
59
|
+
excluding communication that is conspicuously marked or otherwise
|
|
60
|
+
designated in writing by the copyright owner as "Not a Contribution."
|
|
61
|
+
|
|
62
|
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
|
63
|
+
on behalf of whom a Contribution has been received by Licensor and
|
|
64
|
+
subsequently incorporated within the Work.
|
|
65
|
+
|
|
66
|
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
|
67
|
+
this License, each Contributor hereby grants to You a perpetual,
|
|
68
|
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
|
69
|
+
copyright license to reproduce, prepare Derivative Works of,
|
|
70
|
+
publicly display, publicly perform, sublicense, and distribute the
|
|
71
|
+
Work and such Derivative Works in Source or Object form.
|
|
72
|
+
|
|
73
|
+
3. Grant of Patent License. Subject to the terms and conditions of
|
|
74
|
+
this License, each Contributor hereby grants to You a perpetual,
|
|
75
|
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
|
76
|
+
(except as stated in this section) patent license to make, have made,
|
|
77
|
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
|
78
|
+
where such license applies only to those patent claims licensable
|
|
79
|
+
by such Contributor that are necessarily infringed by their
|
|
80
|
+
Contribution(s) alone or by combination of their Contribution(s)
|
|
81
|
+
with the Work to which such Contribution(s) was submitted. If You
|
|
82
|
+
institute patent litigation against any entity (including a
|
|
83
|
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
|
84
|
+
or a Contribution incorporated within the Work constitutes direct
|
|
85
|
+
or contributory patent infringement, then any patent licenses
|
|
86
|
+
granted to You under this License for that Work shall terminate
|
|
87
|
+
as of the date such litigation is filed.
|
|
88
|
+
|
|
89
|
+
4. Redistribution. You may reproduce and distribute copies of the
|
|
90
|
+
Work or Derivative Works thereof in any medium, with or without
|
|
91
|
+
modifications, and in Source or Object form, provided that You
|
|
92
|
+
meet the following conditions:
|
|
93
|
+
|
|
94
|
+
(a) You must give any other recipients of the Work or
|
|
95
|
+
Derivative Works a copy of this License; and
|
|
96
|
+
|
|
97
|
+
(b) You must cause any modified files to carry prominent notices
|
|
98
|
+
stating that You changed the files; and
|
|
99
|
+
|
|
100
|
+
(c) You must retain, in the Source form of any Derivative Works
|
|
101
|
+
that You distribute, all copyright, patent, trademark, and
|
|
102
|
+
attribution notices from the Source form of the Work,
|
|
103
|
+
excluding those notices that do not pertain to any part of
|
|
104
|
+
the Derivative Works; and
|
|
105
|
+
|
|
106
|
+
(d) If the Work includes a "NOTICE" text file as part of its
|
|
107
|
+
distribution, then any Derivative Works that You distribute must
|
|
108
|
+
include a readable copy of the attribution notices contained
|
|
109
|
+
within such NOTICE file, excluding those notices that do not
|
|
110
|
+
pertain to any part of the Derivative Works, in at least one
|
|
111
|
+
of the following places: within a NOTICE text file distributed
|
|
112
|
+
as part of the Derivative Works; within the Source form or
|
|
113
|
+
documentation, if provided along with the Derivative Works; or,
|
|
114
|
+
within a display generated by the Derivative Works, if and
|
|
115
|
+
wherever such third-party notices normally appear. The contents
|
|
116
|
+
of the NOTICE file are for informational purposes only and
|
|
117
|
+
do not modify the License. You may add Your own attribution
|
|
118
|
+
notices within Derivative Works that You distribute, alongside
|
|
119
|
+
or as an addendum to the NOTICE text from the Work, provided
|
|
120
|
+
that such additional attribution notices cannot be construed
|
|
121
|
+
as modifying the License.
|
|
122
|
+
|
|
123
|
+
You may add Your own copyright statement to Your modifications and
|
|
124
|
+
may provide additional or different license terms and conditions
|
|
125
|
+
for use, reproduction, or distribution of Your modifications, or
|
|
126
|
+
for any such Derivative Works as a whole, provided Your use,
|
|
127
|
+
reproduction, and distribution of the Work otherwise complies with
|
|
128
|
+
the conditions stated in this License.
|
|
129
|
+
|
|
130
|
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
|
131
|
+
any Contribution intentionally submitted for inclusion in the Work
|
|
132
|
+
by You to the Licensor shall be under the terms and conditions of
|
|
133
|
+
this License, without any additional terms or conditions.
|
|
134
|
+
Notwithstanding the above, nothing herein shall supersede or modify
|
|
135
|
+
the terms of any separate license agreement you may have executed
|
|
136
|
+
with Licensor regarding such Contributions.
|
|
137
|
+
|
|
138
|
+
6. Trademarks. This License does not grant permission to use the trade
|
|
139
|
+
names, trademarks, service marks, or product names of the Licensor,
|
|
140
|
+
except as required for reasonable and customary use in describing the
|
|
141
|
+
origin of the Work and reproducing the content of the NOTICE file.
|
|
142
|
+
|
|
143
|
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
|
144
|
+
agreed to in writing, Licensor provides the Work (and each
|
|
145
|
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
|
146
|
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
|
147
|
+
implied, including, without limitation, any warranties or conditions
|
|
148
|
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
|
149
|
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
|
150
|
+
appropriateness of using or redistributing the Work and assume any
|
|
151
|
+
risks associated with Your exercise of permissions under this License.
|
|
152
|
+
|
|
153
|
+
8. Limitation of Liability. In no event and under no legal theory,
|
|
154
|
+
whether in tort (including negligence), contract, or otherwise,
|
|
155
|
+
unless required by applicable law (such as deliberate and grossly
|
|
156
|
+
negligent acts) or agreed to in writing, shall any Contributor be
|
|
157
|
+
liable to You for damages, including any direct, indirect, special,
|
|
158
|
+
incidental, or consequential damages of any character arising as a
|
|
159
|
+
result of this License or out of the use or inability to use the
|
|
160
|
+
Work (including but not limited to damages for loss of goodwill,
|
|
161
|
+
work stoppage, computer failure or malfunction, or any and all
|
|
162
|
+
other commercial damages or losses), even if such Contributor
|
|
163
|
+
has been advised of the possibility of such damages.
|
|
164
|
+
|
|
165
|
+
9. Accepting Warranty or Additional Liability. While redistributing
|
|
166
|
+
the Work or Derivative Works thereof, You may choose to offer,
|
|
167
|
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
|
168
|
+
or other liability obligations and/or rights consistent with this
|
|
169
|
+
License. However, in accepting such obligations, You may act only
|
|
170
|
+
on Your own behalf and on Your sole responsibility, not on behalf
|
|
171
|
+
of any other Contributor, and only if You agree to indemnify,
|
|
172
|
+
defend, and hold each Contributor harmless for any liability
|
|
173
|
+
incurred by, or claims asserted against, such Contributor by reason
|
|
174
|
+
of your accepting any such warranty or additional liability.
|
|
175
|
+
|
|
176
|
+
END OF TERMS AND CONDITIONS
|
|
177
|
+
|
|
178
|
+
APPENDIX: How to apply the Apache License to your work.
|
|
179
|
+
|
|
180
|
+
To apply the Apache License to your work, attach the following
|
|
181
|
+
boilerplate notice, with the fields enclosed by brackets "[]"
|
|
182
|
+
replaced with your own identifying information. (Don't include
|
|
183
|
+
the brackets!) The text should be enclosed in the appropriate
|
|
184
|
+
comment syntax for the file format. We also recommend that a
|
|
185
|
+
file or class name and description of purpose be included on the
|
|
186
|
+
same "printed page" as the copyright notice for easier
|
|
187
|
+
identification within third-party archives.
|
|
188
|
+
|
|
189
|
+
Copyright [yyyy] [name of copyright owner]
|
|
190
|
+
|
|
191
|
+
Licensed under the Apache License, Version 2.0 (the "License");
|
|
192
|
+
you may not use this file except in compliance with the License.
|
|
193
|
+
You may obtain a copy of the License at
|
|
194
|
+
|
|
195
|
+
http://www.apache.org/licenses/LICENSE-2.0
|
|
196
|
+
|
|
197
|
+
Unless required by applicable law or agreed to in writing, software
|
|
198
|
+
distributed under the License is distributed on an "AS IS" BASIS,
|
|
199
|
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
200
|
+
See the License for the specific language governing permissions and
|
|
201
|
+
limitations under the License.
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: crowdb-tpc-loader
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Generate and safely load TPC-H and TPC-DS Parquet datasets into an Iceberg REST catalog
|
|
5
|
+
Author: CrowDB contributors
|
|
6
|
+
License-Expression: Apache-2.0
|
|
7
|
+
Project-URL: Repository, https://github.com/buzzcrow/crowdb-tpc-loader
|
|
8
|
+
Project-URL: Issues, https://github.com/buzzcrow/crowdb-tpc-loader/issues
|
|
9
|
+
Classifier: Development Status :: 4 - Beta
|
|
10
|
+
Classifier: Environment :: Console
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Classifier: Topic :: Database
|
|
16
|
+
Requires-Python: <3.13,>=3.10
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
License-File: LICENSE
|
|
19
|
+
Requires-Dist: botocore<2,>=1.34
|
|
20
|
+
Requires-Dist: pyarrow<24,>=18
|
|
21
|
+
Requires-Dist: duckdb<1.6,>=1.4
|
|
22
|
+
Requires-Dist: pyiceberg[pyarrow]<0.11,>=0.10
|
|
23
|
+
Requires-Dist: packaging<27,>=24
|
|
24
|
+
Provides-Extra: dev
|
|
25
|
+
Requires-Dist: pytest<10,>=8; extra == "dev"
|
|
26
|
+
Requires-Dist: pytest-cov<8,>=5; extra == "dev"
|
|
27
|
+
Requires-Dist: build<2,>=1.2; extra == "dev"
|
|
28
|
+
Requires-Dist: ruff<1,>=0.11; extra == "dev"
|
|
29
|
+
Requires-Dist: twine<7,>=6; extra == "dev"
|
|
30
|
+
Provides-Extra: sql-test
|
|
31
|
+
Requires-Dist: sqlalchemy<3,>=2; extra == "sql-test"
|
|
32
|
+
Dynamic: license-file
|
|
33
|
+
|
|
34
|
+
# CrowDB TPC Loader
|
|
35
|
+
|
|
36
|
+
Generate TPC-H or TPC-DS Parquet, upload it to CROWDB Iceberg, and register complete tables through the REST Catalog. The loader creates 8 TPC-H or 24 TPC-DS tables. It does not run benchmark SQL queries or claim TPC certification.
|
|
37
|
+
|
|
38
|
+
- Repository: [buzzcrow/crowdb-tpc-loader](https://github.com/buzzcrow/crowdb-tpc-loader)
|
|
39
|
+
- End-to-end guide: [CROWDB TPC loader documentation](https://crowdb.dev/docs/tpc-loader/)
|
|
40
|
+
|
|
41
|
+
## Install
|
|
42
|
+
|
|
43
|
+
Python 3.10–3.12 is required. After the first PyPI release:
|
|
44
|
+
|
|
45
|
+
```sh
|
|
46
|
+
python3 -m venv .venv
|
|
47
|
+
. .venv/bin/activate
|
|
48
|
+
python -m pip install crowdb-tpc-loader
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
Until PyPI publication, install from a checkout with `python -m pip install --only-binary=:all: -e .`. The first TPC-H run may download `tpchgen-cli` 3.0.0; TPC-DS may download DuckDB's `tpcds` extension. Use `--no-download` and provide these components ahead of time for an offline run.
|
|
52
|
+
|
|
53
|
+
## Load
|
|
54
|
+
|
|
55
|
+
Start `crowdb/crowdb-iceberg:latest` and export the `ICEBERG_URI` and `ICEBERG_TOKEN` values printed by `docker exec <container> crowdb-monitor credentials show --format env`. Keep the token private. Use a new namespace for each run:
|
|
56
|
+
|
|
57
|
+
```sh
|
|
58
|
+
crowdb-tpc-loader load --benchmark tpch --sf 0.01 \
|
|
59
|
+
--namespace tpch_demo --report-file ./tpch-demo.json
|
|
60
|
+
crowdb-tpc-loader load --benchmark tpcds --sf 0.01 \
|
|
61
|
+
--namespace tpcds_demo --report-file ./tpcds-demo.json
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
The loader validates the entire generated dataset before creating a table. It writes different tables concurrently, with 8 workers by default. Use `--upload-workers N` to control concurrent Iceberg table writes (1–24). Each table's files are uploaded and registered in one snapshot, with a durable report checkpoint before each remote side effect. One table's failure does not roll back tables that already succeeded. The report identifies committed, unregistered, and uncertain files; see [recovery](docs/RECOVERY.md) before retrying. An existing table stops the default load; `--on-exists skip` leaves it unchanged without verifying it.
|
|
65
|
+
|
|
66
|
+
For local Parquet only, use `crowdb-tpc-loader generate --benchmark tpch --sf 0.01 --output-dir ./tpch-001`.
|
|
67
|
+
|
|
68
|
+
## Check the result
|
|
69
|
+
|
|
70
|
+
Run the read-only verifier from a checkout after loading:
|
|
71
|
+
|
|
72
|
+
```sh
|
|
73
|
+
python scripts/verify_crowdb.py ./tpch-demo.json --require-complete --iceberg-scan
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
With DuckDB's `iceberg` and `httpfs` extensions, attach the REST Catalog using its token and query `tpch_demo.region` or run TPC-H Q1 against `tpch_demo.lineitem`. The [website guide](https://crowdb.dev/docs/tpc-loader/) has the SQL. A published `latest` image and the local DuckDB 1.5.6 CLI passed an SF 0.01 TPC-H import, independent table verification, and Q1 read on October 1, 2026. This is an integration check, not a performance result. See [compatibility](docs/COMPATIBILITY.md) and the [test record](docs/TEST_REPORT.md).
|
|
77
|
+
|
|
78
|
+
## Develop and publish
|
|
79
|
+
|
|
80
|
+
```sh
|
|
81
|
+
python -m pip install --only-binary=:all: -e '.[dev,sql-test]'
|
|
82
|
+
ruff check src tests scripts
|
|
83
|
+
ruff format --check src tests scripts
|
|
84
|
+
python -m pytest
|
|
85
|
+
python -m build
|
|
86
|
+
python -m twine check dist/*
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
CI runs lint, format, tests, and package checks. Publishing is manual through [the PyPI workflow](.github/workflows/publish.yml) from a `release/v<version>` branch after configuring PyPI Trusted Publishing. See [testing](docs/TESTING.md) for optional real generator and CROWDB runs. The package is Apache-2.0; third-party generators and libraries keep their own licenses.
|
|
90
|
+
|
|
91
|
+
To publish `0.1.0`:
|
|
92
|
+
|
|
93
|
+
1. In PyPI, create a pending Trusted Publisher for project `crowdb-tpc-loader`: GitHub owner `buzzcrow`, repository `crowdb-tpc-loader`, workflow `publish.yml`, environment `pypi`. Create the `pypi` environment in GitHub.
|
|
94
|
+
2. After CI is green, create and push branch `release/v0.1.0` from the commit to publish.
|
|
95
|
+
3. In GitHub Actions, open **Publish to PyPI**, click **Run workflow**, select branch `release/v0.1.0`, then run it. The workflow verifies the branch name against the package version, reruns checks, builds distributions, and publishes through OIDC. No PyPI API token is stored in GitHub.
|
|
96
|
+
4. Confirm the files on [PyPI](https://pypi.org/project/crowdb-tpc-loader/), then test `python -m pip install --no-cache-dir crowdb-tpc-loader==0.1.0` in a clean environment and run `crowdb-tpc-loader --version`.
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
# CrowDB TPC Loader
|
|
2
|
+
|
|
3
|
+
Generate TPC-H or TPC-DS Parquet, upload it to CROWDB Iceberg, and register complete tables through the REST Catalog. The loader creates 8 TPC-H or 24 TPC-DS tables. It does not run benchmark SQL queries or claim TPC certification.
|
|
4
|
+
|
|
5
|
+
- Repository: [buzzcrow/crowdb-tpc-loader](https://github.com/buzzcrow/crowdb-tpc-loader)
|
|
6
|
+
- End-to-end guide: [CROWDB TPC loader documentation](https://crowdb.dev/docs/tpc-loader/)
|
|
7
|
+
|
|
8
|
+
## Install
|
|
9
|
+
|
|
10
|
+
Python 3.10–3.12 is required. After the first PyPI release:
|
|
11
|
+
|
|
12
|
+
```sh
|
|
13
|
+
python3 -m venv .venv
|
|
14
|
+
. .venv/bin/activate
|
|
15
|
+
python -m pip install crowdb-tpc-loader
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
Until PyPI publication, install from a checkout with `python -m pip install --only-binary=:all: -e .`. The first TPC-H run may download `tpchgen-cli` 3.0.0; TPC-DS may download DuckDB's `tpcds` extension. Use `--no-download` and provide these components ahead of time for an offline run.
|
|
19
|
+
|
|
20
|
+
## Load
|
|
21
|
+
|
|
22
|
+
Start `crowdb/crowdb-iceberg:latest` and export the `ICEBERG_URI` and `ICEBERG_TOKEN` values printed by `docker exec <container> crowdb-monitor credentials show --format env`. Keep the token private. Use a new namespace for each run:
|
|
23
|
+
|
|
24
|
+
```sh
|
|
25
|
+
crowdb-tpc-loader load --benchmark tpch --sf 0.01 \
|
|
26
|
+
--namespace tpch_demo --report-file ./tpch-demo.json
|
|
27
|
+
crowdb-tpc-loader load --benchmark tpcds --sf 0.01 \
|
|
28
|
+
--namespace tpcds_demo --report-file ./tpcds-demo.json
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
The loader validates the entire generated dataset before creating a table. It writes different tables concurrently, with 8 workers by default. Use `--upload-workers N` to control concurrent Iceberg table writes (1–24). Each table's files are uploaded and registered in one snapshot, with a durable report checkpoint before each remote side effect. One table's failure does not roll back tables that already succeeded. The report identifies committed, unregistered, and uncertain files; see [recovery](docs/RECOVERY.md) before retrying. An existing table stops the default load; `--on-exists skip` leaves it unchanged without verifying it.
|
|
32
|
+
|
|
33
|
+
For local Parquet only, use `crowdb-tpc-loader generate --benchmark tpch --sf 0.01 --output-dir ./tpch-001`.
|
|
34
|
+
|
|
35
|
+
## Check the result
|
|
36
|
+
|
|
37
|
+
Run the read-only verifier from a checkout after loading:
|
|
38
|
+
|
|
39
|
+
```sh
|
|
40
|
+
python scripts/verify_crowdb.py ./tpch-demo.json --require-complete --iceberg-scan
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
With DuckDB's `iceberg` and `httpfs` extensions, attach the REST Catalog using its token and query `tpch_demo.region` or run TPC-H Q1 against `tpch_demo.lineitem`. The [website guide](https://crowdb.dev/docs/tpc-loader/) has the SQL. A published `latest` image and the local DuckDB 1.5.6 CLI passed an SF 0.01 TPC-H import, independent table verification, and Q1 read on October 1, 2026. This is an integration check, not a performance result. See [compatibility](docs/COMPATIBILITY.md) and the [test record](docs/TEST_REPORT.md).
|
|
44
|
+
|
|
45
|
+
## Develop and publish
|
|
46
|
+
|
|
47
|
+
```sh
|
|
48
|
+
python -m pip install --only-binary=:all: -e '.[dev,sql-test]'
|
|
49
|
+
ruff check src tests scripts
|
|
50
|
+
ruff format --check src tests scripts
|
|
51
|
+
python -m pytest
|
|
52
|
+
python -m build
|
|
53
|
+
python -m twine check dist/*
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
CI runs lint, format, tests, and package checks. Publishing is manual through [the PyPI workflow](.github/workflows/publish.yml) from a `release/v<version>` branch after configuring PyPI Trusted Publishing. See [testing](docs/TESTING.md) for optional real generator and CROWDB runs. The package is Apache-2.0; third-party generators and libraries keep their own licenses.
|
|
57
|
+
|
|
58
|
+
To publish `0.1.0`:
|
|
59
|
+
|
|
60
|
+
1. In PyPI, create a pending Trusted Publisher for project `crowdb-tpc-loader`: GitHub owner `buzzcrow`, repository `crowdb-tpc-loader`, workflow `publish.yml`, environment `pypi`. Create the `pypi` environment in GitHub.
|
|
61
|
+
2. After CI is green, create and push branch `release/v0.1.0` from the commit to publish.
|
|
62
|
+
3. In GitHub Actions, open **Publish to PyPI**, click **Run workflow**, select branch `release/v0.1.0`, then run it. The workflow verifies the branch name against the package version, reruns checks, builds distributions, and publishes through OIDC. No PyPI API token is stored in GitHub.
|
|
63
|
+
4. Confirm the files on [PyPI](https://pypi.org/project/crowdb-tpc-loader/), then test `python -m pip install --no-cache-dir crowdb-tpc-loader==0.1.0` in a clean environment and run `crowdb-tpc-loader --version`.
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
# Compatibility
|
|
2
|
+
|
|
3
|
+
The package requires Python 3.10–3.12, PyArrow `>=18,<24`, PyIceberg `>=0.10,<0.11`, and DuckDB `>=1.4,<1.6`. TPC-H uses `tpchgen-cli` 3.x; automatic download pins 3.0.0. These ranges are package constraints, not a claim that every version combination passed. The [test record](TEST_REPORT.md) names the observed local combination.
|
|
4
|
+
|
|
5
|
+
`load` requires an Iceberg REST Catalog and working remote FileIO. It reads table locations and credentials from the catalog; it never registers a local path. The default `crowdb_tpc_loader.crowdb_fileio.CrowdbFileIO` uses exact-object opens on CROWDB's native Iceberg endpoint, which does not require S3 prefix listing. Other deployments can provide a trusted FileIO class with `--py-io-impl`.
|
|
6
|
+
|
|
7
|
+
Before generation, the loader checks catalog access, existing tables, namespace creation, and a staged FileIO write probe where cleanup is supported. CROWDB's native FileIO has no remote file delete, so the first real table upload acts as its write test. The entire generated dataset passes schema and footer checks before any benchmark table is created.
|
|
8
|
+
|
|
9
|
+
The catalog can return transient HTTP 503 when its bounded operation capacity is busy. Table creation retries up to six attempts with short backoff. If a retry finds a table created by an earlier attempt in the same run, it verifies the run ID before continuing. Registration (`add_files`) is never blindly retried after an ambiguous response; the loader reads the current snapshot to reconcile the outcome. Keep the JSON report for [failure recovery](RECOVERY.md).
|
|
10
|
+
|
|
11
|
+
Each worker owns a separate Catalog client and one table. Eight workers run by default; `--upload-workers` accepts 1–24. Files within a table upload in sequence, then that table commits once. Other tables can upload or commit at the same time. This can expose server backpressure; choose fewer workers for a small deployment if bounded retries are still exhausted. A failure in one table does not roll back another table's successful snapshot.
|
|
12
|
+
|
|
13
|
+
TPC-DS generation uses DuckDB's `tpcds` extension and a disk database. `--memory-limit` controls DuckDB memory, not total process memory. TPC-H defaults to `ceil(SF / 10)` parts. Reports record the actual dependency and generator versions for each run. Distributed deployments and the full TPC query suites need separate acceptance.
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
# Failure recovery: determine commit status before handling files
|
|
2
|
+
|
|
3
|
+
The tool does not provide automatic resume, overwrite, drop, purge, or remote garbage collection. There is no atomic transaction across all 8 or 24 tables. With concurrent table writes, each table uploads its files and commits one snapshot independently. An upload or commit failure in one table does not roll back successful snapshots in other tables.
|
|
4
|
+
|
|
5
|
+
## Reports and directories
|
|
6
|
+
|
|
7
|
+
Keep the JSON report and working subdirectory shown in the console. A persistent `--report-file` path is the most reliable choice. Before uploading, the tool records the URI, run ID, and file status on disk. Before a commit, it marks the file `commit_unknown` and persists that state. If the process crashes between upload and commit, the unknown result will not be treated as definitely uncommitted.
|
|
8
|
+
|
|
9
|
+
Forced process termination, sudden power loss, and disk damage can still interrupt the latest checkpoint. Atomic replacement, fsync, and reports in two locations reduce this risk but do not provide a distributed transaction guarantee.
|
|
10
|
+
|
|
11
|
+
| Status | Meaning | Handling |
|
|
12
|
+
|---|---|---|
|
|
13
|
+
| `local` | No target remote URI yet | A locally generated file; this does not imply upload |
|
|
14
|
+
| `upload_started` | Target URI recorded; the object may be absent, partly uploaded, or complete | Check whether the object exists; do not infer its size from status alone |
|
|
15
|
+
| `uploaded_unregistered` | Upload accepted by the server; commit not attempted, or explicitly rejected and verified unregistered | Check other snapshots and references before cleanup |
|
|
16
|
+
| `commit_unknown` | Commit started or may have started; result cannot be confirmed | Keep the file; do not blindly repeat `add_files` |
|
|
17
|
+
| `registered` | URI observed in the verified current snapshot | Do not delete directly; follow the Iceberg snapshot and metadata lifecycle |
|
|
18
|
+
|
|
19
|
+
`summary.unregistered_uploads` lists **candidates for inspection**, not objects that can be deleted in bulk. Keep `summary.uncertain_uploads` as a priority. `summary.tables_requiring_inspection` includes tables that may have been created but not fully verified, or for which writes failed. `table_creation_state=unknown_or_unvalidated` does not mean the table is definitely absent.
|
|
20
|
+
|
|
21
|
+
## Common cases
|
|
22
|
+
|
|
23
|
+
### Generation or schema validation fails
|
|
24
|
+
|
|
25
|
+
This release creates benchmark tables only after validating the entire Parquet dataset. Check the generator version, actual schema, missing tables, and source-versus-export row counts in the report. Run again with a new output directory or namespace; do not overwrite the original files.
|
|
26
|
+
|
|
27
|
+
### FileIO preflight fails
|
|
28
|
+
|
|
29
|
+
A reachable REST endpoint does not prove that storage is writable. Check whether the returned storage address is reachable from the client, whether authentication needs a separate token, whether the correct FileIO was selected, and whether the HTTP service supports the required operations. Preflight uses an uncommitted staged table and does not fall back to a local path.
|
|
30
|
+
|
|
31
|
+
### Some tables succeed and a later table fails
|
|
32
|
+
|
|
33
|
+
First run the independent verifier on tables marked `succeeded` in the report. The failed table may exist but be empty, or it may have a committed or uncertain snapshot. A default rerun stops on existing tables as a protective measure.
|
|
34
|
+
|
|
35
|
+
`--on-exists skip` does not repair a failed table and may skip an empty one. Do not use it to claim that the full dataset has been completed. The easiest way to verify a fresh run is to use a new namespace, then handle old data according to Iceberg's rules. If the old namespace must be kept, inspect every snapshot, branch, and reference for the table before deciding how to handle empty tables or orphaned objects. The tool does not perform these destructive operations.
|
|
36
|
+
|
|
37
|
+
### Commit request times out
|
|
38
|
+
|
|
39
|
+
The tool makes up to three rounds of read-only verification without calling `add_files` again. It records success only if the current snapshot's file set, row counts, and sizes match exactly. Otherwise, it records `uncertain` and marks that table uncertain; other in-flight tables may still finish.
|
|
40
|
+
|
|
41
|
+
Another writer may subsequently change the final snapshot. Files recorded as `registered` may also be referenced by historical snapshots; absence from the current snapshot does not prove absence from history. Use the namespace, table UUID, run ID, and all file URIs to inspect server logs, snapshots, and branch references before proceeding.
|
|
42
|
+
|
|
43
|
+
### Insufficient local space or process interruption
|
|
44
|
+
|
|
45
|
+
Check the exit code, external report, and working path. The working directory is kept after a failure, and the tool does not delete remote benchmark objects. If local cleanup fails after a successful run, the tool records a cleanup warning without changing a verified successful remote commit into a failure. Before manually deleting a local directory, confirm that it is this run's working subdirectory, not the `--work-dir` parent.
|
|
46
|
+
|
|
47
|
+
## Information to include in a bug report
|
|
48
|
+
|
|
49
|
+
Keep the command with the token removed, Python/dependency/generator versions, run ID, error from the report, failed table, and FileIO type. Inspect the redacted report manually before sharing it. Do not include tokens, storage keys, signed URLs, complete environment variables, or private business data in an issue.
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
# Testing
|
|
2
|
+
|
|
3
|
+
CI runs `ruff check src tests scripts`, `ruff format --check src tests scripts`, and `python -m pytest` on Python 3.10, 3.11, and 3.12. It builds and checks the wheel and sdist on Python 3.12. The default tests include unit and local PyArrow/PyIceberg integration tests. Real generators and a CROWDB instance are opt-in.
|
|
4
|
+
|
|
5
|
+
## Local checks
|
|
6
|
+
|
|
7
|
+
```sh
|
|
8
|
+
python -m pip install --only-binary=:all: -e '.[dev,sql-test]'
|
|
9
|
+
ruff check src tests scripts
|
|
10
|
+
ruff format --check src tests scripts
|
|
11
|
+
python -m pytest -q
|
|
12
|
+
python -m build
|
|
13
|
+
python -m twine check dist/*
|
|
14
|
+
```
|
|
15
|
+
|
|
16
|
+
Check a built wheel in a fresh virtual environment before release. Confirm both `crowdb-tpc-loader --version` and `crowdb-tpc-loader load --help` work.
|
|
17
|
+
|
|
18
|
+
## Real generators
|
|
19
|
+
|
|
20
|
+
These may download generator binaries or extensions and need disk space:
|
|
21
|
+
|
|
22
|
+
```sh
|
|
23
|
+
CROWDB_TPC_GENERATOR_TESTS=1 python -m pytest tests/integration/test_generators.py -ra
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
Set `CROWDB_TPC_SF1_TESTS=1` as well to include full SF 1 datasets. Or run `python scripts/smoke_test.py --benchmark tpch --sf 0.01` for local generation only.
|
|
27
|
+
|
|
28
|
+
## Real CROWDB load
|
|
29
|
+
|
|
30
|
+
Use a dedicated disposable instance with `ICEBERG_URI` and `ICEBERG_TOKEN` exported. The tests create fresh remote namespaces and do not delete them:
|
|
31
|
+
|
|
32
|
+
```sh
|
|
33
|
+
CROWDB_TPC_CROWDB_TESTS=1 CROWDB_TPC_SF1_TESTS=1 \
|
|
34
|
+
python -m pytest tests/integration/test_crowdb.py -s -ra
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
For a smaller remote run, use `python scripts/smoke_test.py --benchmark tpch --sf 0.01 --load`. Preserve the JSON report on failure. A skipped integration test is not evidence of remote acceptance.
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
# Test record
|
|
2
|
+
|
|
3
|
+
## Published image, October 1, 2026
|
|
4
|
+
|
|
5
|
+
- Image: `crowdb/crowdb-iceberg:latest`, digest `sha256:e4e7f80367094a35ab54c2bf1b0a9780811f9a668bc6037cfe7ced1b23373718`, Linux amd64 single-node.
|
|
6
|
+
- Loader: eight concurrent table writes (`--upload-workers 8`), TPC-H SF 0.01, fresh namespace. All 8 tables committed; independent verifier checked 86,805 rows in manifests and remote Parquet footers, plus sample Iceberg scans after local staging was removed.
|
|
7
|
+
- DuckDB CLI: locally built 1.5.6 with `iceberg` and `httpfs`. `region` returned `(1, AMERICA)`. Standard TPC-H Q1 on the imported `lineitem` returned four groups: `(A,F,14876)`, `(N,F,348)`, `(N,O,29181)`, `(R,F,14902)` for `(returnflag, linestatus, count_order)`. The CLI reported 0.075 seconds for that small warm-host run. No baseline or repeated trials were taken.
|
|
8
|
+
- Initial 8-worker attempt saw four transient Catalog HTTP 503 responses and partial table success. The loader now retries bounded 503 create responses; a fresh namespace completed. Both reports are local test artifacts, not included in the package.
|
|
9
|
+
|
|
10
|
+
This proves one integration path and one Q1 execution. It does not establish the full TPC-H suite, benchmark performance, distributed behavior, or TPC certification.
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77,<83"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "crowdb-tpc-loader"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Generate and safely load TPC-H and TPC-DS Parquet datasets into an Iceberg REST catalog"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10,<3.13"
|
|
11
|
+
license = "Apache-2.0"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [{name = "CrowDB contributors"}]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 4 - Beta",
|
|
16
|
+
"Environment :: Console",
|
|
17
|
+
"Programming Language :: Python :: 3",
|
|
18
|
+
"Programming Language :: Python :: 3.10",
|
|
19
|
+
"Programming Language :: Python :: 3.11",
|
|
20
|
+
"Programming Language :: Python :: 3.12",
|
|
21
|
+
"Topic :: Database",
|
|
22
|
+
]
|
|
23
|
+
dependencies = [
|
|
24
|
+
"botocore>=1.34,<2",
|
|
25
|
+
"pyarrow>=18,<24",
|
|
26
|
+
"duckdb>=1.4,<1.6",
|
|
27
|
+
"pyiceberg[pyarrow]>=0.10,<0.11",
|
|
28
|
+
"packaging>=24,<27",
|
|
29
|
+
]
|
|
30
|
+
|
|
31
|
+
[project.optional-dependencies]
|
|
32
|
+
dev = ["pytest>=8,<10", "pytest-cov>=5,<8", "build>=1.2,<2", "ruff>=0.11,<1", "twine>=6,<7"]
|
|
33
|
+
sql-test = ["sqlalchemy>=2,<3"]
|
|
34
|
+
|
|
35
|
+
[project.scripts]
|
|
36
|
+
crowdb-tpc-loader = "crowdb_tpc_loader.cli:main"
|
|
37
|
+
|
|
38
|
+
[project.urls]
|
|
39
|
+
Repository = "https://github.com/buzzcrow/crowdb-tpc-loader"
|
|
40
|
+
Issues = "https://github.com/buzzcrow/crowdb-tpc-loader/issues"
|
|
41
|
+
|
|
42
|
+
[tool.setuptools.packages.find]
|
|
43
|
+
where = ["src"]
|
|
44
|
+
|
|
45
|
+
[tool.pytest.ini_options]
|
|
46
|
+
testpaths = ["tests"]
|
|
47
|
+
pythonpath = ["src"]
|
|
48
|
+
addopts = "-ra"
|
|
49
|
+
markers = [
|
|
50
|
+
"integration: requires installed runtime dependencies",
|
|
51
|
+
"generator: invokes the real generator (may download binaries/extensions)",
|
|
52
|
+
"crowdb: writes a fresh namespace in an explicitly configured CrowDB instance",
|
|
53
|
+
]
|
|
54
|
+
|
|
55
|
+
[tool.ruff]
|
|
56
|
+
line-length = 110
|
|
57
|
+
target-version = "py310"
|
|
58
|
+
|
|
59
|
+
[tool.ruff.lint]
|
|
60
|
+
select = ["E4", "E7", "E9", "F"]
|
|
61
|
+
|
|
62
|
+
[tool.ruff.lint.per-file-ignores]
|
|
63
|
+
"tests/integration/test_iceberg_http.py" = ["E402", "F811"]
|
|
64
|
+
"tests/integration/test_parquet.py" = ["E402"]
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
# Candidate reproducibility matrix, NOT an already-tested dependency lock.
|
|
2
|
+
# Install with --only-binary=:all: and run the real integration suite.
|
|
3
|
+
pyarrow==22.0.0
|
|
4
|
+
pyiceberg[pyarrow]==0.10.0
|
|
5
|
+
duckdb==1.5.6
|
|
6
|
+
packaging==25.0
|
|
7
|
+
sqlalchemy==2.0.43
|
|
8
|
+
pytest==8.4.2
|
|
9
|
+
pytest-cov==6.2.1
|
|
10
|
+
# Optional: use the loader's binary-only bootstrap instead of installing this.
|
|
11
|
+
# tpchgen-cli==3.0.0
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Run a local generation smoke test, or an explicitly requested remote load.
|
|
3
|
+
|
|
4
|
+
No remote writes unless --load is present. Uses a unique output directory and,
|
|
5
|
+
for loads, a unique namespace. It never drops remote data.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import argparse
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
import subprocess
|
|
13
|
+
import sys
|
|
14
|
+
import uuid
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def main() -> int:
|
|
18
|
+
parser = argparse.ArgumentParser(description=__doc__)
|
|
19
|
+
parser.add_argument("--benchmark", choices=("tpch", "tpcds"), required=True)
|
|
20
|
+
parser.add_argument("--sf", default="0.01")
|
|
21
|
+
parser.add_argument("--output-root", type=Path, default=Path("smoke-results"))
|
|
22
|
+
parser.add_argument(
|
|
23
|
+
"--load", action="store_true", help="explicitly enable writes to ICEBERG_URI using ICEBERG_TOKEN"
|
|
24
|
+
)
|
|
25
|
+
parser.add_argument("--py-io-impl")
|
|
26
|
+
parser.add_argument("--tpchgen", type=Path)
|
|
27
|
+
parser.add_argument("--full-read", action="store_true")
|
|
28
|
+
args = parser.parse_args()
|
|
29
|
+
run = args.output_root / (args.benchmark + "-" + uuid.uuid4().hex[:12])
|
|
30
|
+
run.mkdir(parents=True, exist_ok=False)
|
|
31
|
+
command = [
|
|
32
|
+
sys.executable,
|
|
33
|
+
"-m",
|
|
34
|
+
"crowdb_tpc_loader",
|
|
35
|
+
"load" if args.load else "generate",
|
|
36
|
+
"--benchmark",
|
|
37
|
+
args.benchmark,
|
|
38
|
+
"--sf",
|
|
39
|
+
args.sf,
|
|
40
|
+
"--report-file",
|
|
41
|
+
str(run / "report.json"),
|
|
42
|
+
]
|
|
43
|
+
if args.tpchgen:
|
|
44
|
+
command += ["--tpchgen", str(args.tpchgen)]
|
|
45
|
+
if args.load:
|
|
46
|
+
namespace = "smoke_" + args.benchmark + "_" + uuid.uuid4().hex[:12]
|
|
47
|
+
command += ["--namespace", namespace, "--work-dir", str(run / "staging")]
|
|
48
|
+
if args.py_io_impl:
|
|
49
|
+
command += ["--py-io-impl", args.py_io_impl]
|
|
50
|
+
else:
|
|
51
|
+
command += ["--output-dir", str(run / "generated")]
|
|
52
|
+
print(f"Smoke output: {run.resolve()}", flush=True)
|
|
53
|
+
result = subprocess.run(command, check=False)
|
|
54
|
+
if result.returncode != 0 or not args.load:
|
|
55
|
+
return result.returncode
|
|
56
|
+
verify = [
|
|
57
|
+
sys.executable,
|
|
58
|
+
str(Path(__file__).with_name("verify_crowdb.py")),
|
|
59
|
+
str(run / "report.json"),
|
|
60
|
+
"--iceberg-scan",
|
|
61
|
+
"--require-complete",
|
|
62
|
+
]
|
|
63
|
+
if args.py_io_impl:
|
|
64
|
+
verify += ["--py-io-impl", args.py_io_impl]
|
|
65
|
+
if args.full_read:
|
|
66
|
+
verify += ["--full-read"]
|
|
67
|
+
return subprocess.run(verify, check=False).returncode
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
if __name__ == "__main__":
|
|
71
|
+
raise SystemExit(main())
|