crowdb-tpc-loader 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. crowdb_tpc_loader-0.1.0/CHANGELOG.md +9 -0
  2. crowdb_tpc_loader-0.1.0/LICENSE +201 -0
  3. crowdb_tpc_loader-0.1.0/MANIFEST.in +5 -0
  4. crowdb_tpc_loader-0.1.0/PKG-INFO +96 -0
  5. crowdb_tpc_loader-0.1.0/README.md +63 -0
  6. crowdb_tpc_loader-0.1.0/docs/COMPATIBILITY.md +13 -0
  7. crowdb_tpc_loader-0.1.0/docs/RECOVERY.md +49 -0
  8. crowdb_tpc_loader-0.1.0/docs/TESTING.md +37 -0
  9. crowdb_tpc_loader-0.1.0/docs/TEST_REPORT.md +10 -0
  10. crowdb_tpc_loader-0.1.0/pyproject.toml +64 -0
  11. crowdb_tpc_loader-0.1.0/requirements-integration.txt +11 -0
  12. crowdb_tpc_loader-0.1.0/scripts/smoke_test.py +71 -0
  13. crowdb_tpc_loader-0.1.0/scripts/verify_crowdb.py +157 -0
  14. crowdb_tpc_loader-0.1.0/setup.cfg +4 -0
  15. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/__init__.py +3 -0
  16. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/__main__.py +3 -0
  17. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/arrow_http_bridge.py +90 -0
  18. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/backend.py +296 -0
  19. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/cli.py +325 -0
  20. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/crowdb_fileio.py +40 -0
  21. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/errors.py +37 -0
  22. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/generators/__init__.py +18 -0
  23. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/generators/binary.py +191 -0
  24. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/generators/tpcds.py +134 -0
  25. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/generators/tpch.py +153 -0
  26. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/http_fileio.py +303 -0
  27. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/loader.py +287 -0
  28. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/models.py +70 -0
  29. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/report.py +140 -0
  30. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/rest_catalog.py +49 -0
  31. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/runner.py +236 -0
  32. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/s3_upload.py +146 -0
  33. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/schemas.py +197 -0
  34. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/security.py +98 -0
  35. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/transfers.py +99 -0
  36. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/util.py +217 -0
  37. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader/validation.py +172 -0
  38. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader.egg-info/PKG-INFO +96 -0
  39. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader.egg-info/SOURCES.txt +61 -0
  40. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader.egg-info/dependency_links.txt +1 -0
  41. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader.egg-info/entry_points.txt +2 -0
  42. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader.egg-info/requires.txt +15 -0
  43. crowdb_tpc_loader-0.1.0/src/crowdb_tpc_loader.egg-info/top_level.txt +1 -0
  44. crowdb_tpc_loader-0.1.0/tests/__init__.py +0 -0
  45. crowdb_tpc_loader-0.1.0/tests/conftest.py +173 -0
  46. crowdb_tpc_loader-0.1.0/tests/integration/__init__.py +0 -0
  47. crowdb_tpc_loader-0.1.0/tests/integration/test_crowdb.py +62 -0
  48. crowdb_tpc_loader-0.1.0/tests/integration/test_generators.py +45 -0
  49. crowdb_tpc_loader-0.1.0/tests/integration/test_iceberg_http.py +66 -0
  50. crowdb_tpc_loader-0.1.0/tests/integration/test_parquet.py +71 -0
  51. crowdb_tpc_loader-0.1.0/tests/unit/__init__.py +0 -0
  52. crowdb_tpc_loader-0.1.0/tests/unit/test_binary_generator.py +127 -0
  53. crowdb_tpc_loader-0.1.0/tests/unit/test_cli_security.py +205 -0
  54. crowdb_tpc_loader-0.1.0/tests/unit/test_commits.py +260 -0
  55. crowdb_tpc_loader-0.1.0/tests/unit/test_crowdb_fileio.py +29 -0
  56. crowdb_tpc_loader-0.1.0/tests/unit/test_files_reports.py +273 -0
  57. crowdb_tpc_loader-0.1.0/tests/unit/test_http_fileio.py +277 -0
  58. crowdb_tpc_loader-0.1.0/tests/unit/test_rest_session.py +25 -0
  59. crowdb_tpc_loader-0.1.0/tests/unit/test_runner.py +219 -0
  60. crowdb_tpc_loader-0.1.0/tests/unit/test_s3_upload.py +83 -0
  61. crowdb_tpc_loader-0.1.0/tests/unit/test_tpcds_adapter.py +125 -0
  62. crowdb_tpc_loader-0.1.0/tests/unit/test_transfers.py +104 -0
  63. crowdb_tpc_loader-0.1.0/tests/unit/test_validation_logic.py +166 -0
@@ -0,0 +1,9 @@
1
+ # Changelog
2
+
3
+ ## 0.1.0
4
+
5
+ - Load TPC-H and TPC-DS datasets into Iceberg tables, with validation and recoverable reports.
6
+ - Write separate tables concurrently, with 8 workers by default and a configurable 1–24 range.
7
+ - Retry transient Catalog 503 responses during table creation without retrying ambiguous table commits.
8
+ - Verify the published CROWDB Iceberg image with a small TPC-H load and DuckDB Q1 read.
9
+ - Add CI checks and a manually triggered PyPI release workflow.
@@ -0,0 +1,201 @@
1
+ Apache License
2
+ Version 2.0, January 2004
3
+ http://www.apache.org/licenses/
4
+
5
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
6
+
7
+ 1. Definitions.
8
+
9
+ "License" shall mean the terms and conditions for use, reproduction,
10
+ and distribution as defined by Sections 1 through 9 of this document.
11
+
12
+ "Licensor" shall mean the copyright owner or entity authorized by
13
+ the copyright owner that is granting the License.
14
+
15
+ "Legal Entity" shall mean the union of the acting entity and all
16
+ other entities that control, are controlled by, or are under common
17
+ control with that entity. For the purposes of this definition,
18
+ "control" means (i) the power, direct or indirect, to cause the
19
+ direction or management of such entity, whether by contract or
20
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
21
+ outstanding shares, or (iii) beneficial ownership of such entity.
22
+
23
+ "You" (or "Your") shall mean an individual or Legal Entity
24
+ exercising permissions granted by this License.
25
+
26
+ "Source" form shall mean the preferred form for making modifications,
27
+ including but not limited to software source code, documentation
28
+ source, and configuration files.
29
+
30
+ "Object" form shall mean any form resulting from mechanical
31
+ transformation or translation of a Source form, including but
32
+ not limited to compiled object code, generated documentation,
33
+ and conversions to other media types.
34
+
35
+ "Work" shall mean the work of authorship, whether in Source or
36
+ Object form, made available under the License, as indicated by a
37
+ copyright notice that is included in or attached to the work
38
+ (an example is provided in the Appendix below).
39
+
40
+ "Derivative Works" shall mean any work, whether in Source or Object
41
+ form, that is based on (or derived from) the Work and for which the
42
+ editorial revisions, annotations, elaborations, or other modifications
43
+ represent, as a whole, an original work of authorship. For the purposes
44
+ of this License, Derivative Works shall not include works that remain
45
+ separable from, or merely link (or bind by name) to the interfaces of,
46
+ the Work and Derivative Works thereof.
47
+
48
+ "Contribution" shall mean any work of authorship, including
49
+ the original version of the Work and any modifications or additions
50
+ to that Work or Derivative Works thereof, that is intentionally
51
+ submitted to Licensor for inclusion in the Work by the copyright owner
52
+ or by an individual or Legal Entity authorized to submit on behalf of
53
+ the copyright owner. For the purposes of this definition, "submitted"
54
+ means any form of electronic, verbal, or written communication sent
55
+ to the Licensor or its representatives, including but not limited to
56
+ communication on electronic mailing lists, source code control systems,
57
+ and issue tracking systems that are managed by, or on behalf of, the
58
+ Licensor for the purpose of discussing and improving the Work, but
59
+ excluding communication that is conspicuously marked or otherwise
60
+ designated in writing by the copyright owner as "Not a Contribution."
61
+
62
+ "Contributor" shall mean Licensor and any individual or Legal Entity
63
+ on behalf of whom a Contribution has been received by Licensor and
64
+ subsequently incorporated within the Work.
65
+
66
+ 2. Grant of Copyright License. Subject to the terms and conditions of
67
+ this License, each Contributor hereby grants to You a perpetual,
68
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
69
+ copyright license to reproduce, prepare Derivative Works of,
70
+ publicly display, publicly perform, sublicense, and distribute the
71
+ Work and such Derivative Works in Source or Object form.
72
+
73
+ 3. Grant of Patent License. Subject to the terms and conditions of
74
+ this License, each Contributor hereby grants to You a perpetual,
75
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
76
+ (except as stated in this section) patent license to make, have made,
77
+ use, offer to sell, sell, import, and otherwise transfer the Work,
78
+ where such license applies only to those patent claims licensable
79
+ by such Contributor that are necessarily infringed by their
80
+ Contribution(s) alone or by combination of their Contribution(s)
81
+ with the Work to which such Contribution(s) was submitted. If You
82
+ institute patent litigation against any entity (including a
83
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
84
+ or a Contribution incorporated within the Work constitutes direct
85
+ or contributory patent infringement, then any patent licenses
86
+ granted to You under this License for that Work shall terminate
87
+ as of the date such litigation is filed.
88
+
89
+ 4. Redistribution. You may reproduce and distribute copies of the
90
+ Work or Derivative Works thereof in any medium, with or without
91
+ modifications, and in Source or Object form, provided that You
92
+ meet the following conditions:
93
+
94
+ (a) You must give any other recipients of the Work or
95
+ Derivative Works a copy of this License; and
96
+
97
+ (b) You must cause any modified files to carry prominent notices
98
+ stating that You changed the files; and
99
+
100
+ (c) You must retain, in the Source form of any Derivative Works
101
+ that You distribute, all copyright, patent, trademark, and
102
+ attribution notices from the Source form of the Work,
103
+ excluding those notices that do not pertain to any part of
104
+ the Derivative Works; and
105
+
106
+ (d) If the Work includes a "NOTICE" text file as part of its
107
+ distribution, then any Derivative Works that You distribute must
108
+ include a readable copy of the attribution notices contained
109
+ within such NOTICE file, excluding those notices that do not
110
+ pertain to any part of the Derivative Works, in at least one
111
+ of the following places: within a NOTICE text file distributed
112
+ as part of the Derivative Works; within the Source form or
113
+ documentation, if provided along with the Derivative Works; or,
114
+ within a display generated by the Derivative Works, if and
115
+ wherever such third-party notices normally appear. The contents
116
+ of the NOTICE file are for informational purposes only and
117
+ do not modify the License. You may add Your own attribution
118
+ notices within Derivative Works that You distribute, alongside
119
+ or as an addendum to the NOTICE text from the Work, provided
120
+ that such additional attribution notices cannot be construed
121
+ as modifying the License.
122
+
123
+ You may add Your own copyright statement to Your modifications and
124
+ may provide additional or different license terms and conditions
125
+ for use, reproduction, or distribution of Your modifications, or
126
+ for any such Derivative Works as a whole, provided Your use,
127
+ reproduction, and distribution of the Work otherwise complies with
128
+ the conditions stated in this License.
129
+
130
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
131
+ any Contribution intentionally submitted for inclusion in the Work
132
+ by You to the Licensor shall be under the terms and conditions of
133
+ this License, without any additional terms or conditions.
134
+ Notwithstanding the above, nothing herein shall supersede or modify
135
+ the terms of any separate license agreement you may have executed
136
+ with Licensor regarding such Contributions.
137
+
138
+ 6. Trademarks. This License does not grant permission to use the trade
139
+ names, trademarks, service marks, or product names of the Licensor,
140
+ except as required for reasonable and customary use in describing the
141
+ origin of the Work and reproducing the content of the NOTICE file.
142
+
143
+ 7. Disclaimer of Warranty. Unless required by applicable law or
144
+ agreed to in writing, Licensor provides the Work (and each
145
+ Contributor provides its Contributions) on an "AS IS" BASIS,
146
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
147
+ implied, including, without limitation, any warranties or conditions
148
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
149
+ PARTICULAR PURPOSE. You are solely responsible for determining the
150
+ appropriateness of using or redistributing the Work and assume any
151
+ risks associated with Your exercise of permissions under this License.
152
+
153
+ 8. Limitation of Liability. In no event and under no legal theory,
154
+ whether in tort (including negligence), contract, or otherwise,
155
+ unless required by applicable law (such as deliberate and grossly
156
+ negligent acts) or agreed to in writing, shall any Contributor be
157
+ liable to You for damages, including any direct, indirect, special,
158
+ incidental, or consequential damages of any character arising as a
159
+ result of this License or out of the use or inability to use the
160
+ Work (including but not limited to damages for loss of goodwill,
161
+ work stoppage, computer failure or malfunction, or any and all
162
+ other commercial damages or losses), even if such Contributor
163
+ has been advised of the possibility of such damages.
164
+
165
+ 9. Accepting Warranty or Additional Liability. While redistributing
166
+ the Work or Derivative Works thereof, You may choose to offer,
167
+ and charge a fee for, acceptance of support, warranty, indemnity,
168
+ or other liability obligations and/or rights consistent with this
169
+ License. However, in accepting such obligations, You may act only
170
+ on Your own behalf and on Your sole responsibility, not on behalf
171
+ of any other Contributor, and only if You agree to indemnify,
172
+ defend, and hold each Contributor harmless for any liability
173
+ incurred by, or claims asserted against, such Contributor by reason
174
+ of your accepting any such warranty or additional liability.
175
+
176
+ END OF TERMS AND CONDITIONS
177
+
178
+ APPENDIX: How to apply the Apache License to your work.
179
+
180
+ To apply the Apache License to your work, attach the following
181
+ boilerplate notice, with the fields enclosed by brackets "[]"
182
+ replaced with your own identifying information. (Don't include
183
+ the brackets!) The text should be enclosed in the appropriate
184
+ comment syntax for the file format. We also recommend that a
185
+ file or class name and description of purpose be included on the
186
+ same "printed page" as the copyright notice for easier
187
+ identification within third-party archives.
188
+
189
+ Copyright [yyyy] [name of copyright owner]
190
+
191
+ Licensed under the Apache License, Version 2.0 (the "License");
192
+ you may not use this file except in compliance with the License.
193
+ You may obtain a copy of the License at
194
+
195
+ http://www.apache.org/licenses/LICENSE-2.0
196
+
197
+ Unless required by applicable law or agreed to in writing, software
198
+ distributed under the License is distributed on an "AS IS" BASIS,
199
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
200
+ See the License for the specific language governing permissions and
201
+ limitations under the License.
@@ -0,0 +1,5 @@
1
+ include LICENSE README.md CHANGELOG.md
2
+ recursive-include docs *.md
3
+ recursive-include scripts *.py
4
+ recursive-include tests *.py
5
+ include requirements-integration.txt
@@ -0,0 +1,96 @@
1
+ Metadata-Version: 2.4
2
+ Name: crowdb-tpc-loader
3
+ Version: 0.1.0
4
+ Summary: Generate and safely load TPC-H and TPC-DS Parquet datasets into an Iceberg REST catalog
5
+ Author: CrowDB contributors
6
+ License-Expression: Apache-2.0
7
+ Project-URL: Repository, https://github.com/buzzcrow/crowdb-tpc-loader
8
+ Project-URL: Issues, https://github.com/buzzcrow/crowdb-tpc-loader/issues
9
+ Classifier: Development Status :: 4 - Beta
10
+ Classifier: Environment :: Console
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Programming Language :: Python :: 3.10
13
+ Classifier: Programming Language :: Python :: 3.11
14
+ Classifier: Programming Language :: Python :: 3.12
15
+ Classifier: Topic :: Database
16
+ Requires-Python: <3.13,>=3.10
17
+ Description-Content-Type: text/markdown
18
+ License-File: LICENSE
19
+ Requires-Dist: botocore<2,>=1.34
20
+ Requires-Dist: pyarrow<24,>=18
21
+ Requires-Dist: duckdb<1.6,>=1.4
22
+ Requires-Dist: pyiceberg[pyarrow]<0.11,>=0.10
23
+ Requires-Dist: packaging<27,>=24
24
+ Provides-Extra: dev
25
+ Requires-Dist: pytest<10,>=8; extra == "dev"
26
+ Requires-Dist: pytest-cov<8,>=5; extra == "dev"
27
+ Requires-Dist: build<2,>=1.2; extra == "dev"
28
+ Requires-Dist: ruff<1,>=0.11; extra == "dev"
29
+ Requires-Dist: twine<7,>=6; extra == "dev"
30
+ Provides-Extra: sql-test
31
+ Requires-Dist: sqlalchemy<3,>=2; extra == "sql-test"
32
+ Dynamic: license-file
33
+
34
+ # CrowDB TPC Loader
35
+
36
+ Generate TPC-H or TPC-DS Parquet, upload it to CROWDB Iceberg, and register complete tables through the REST Catalog. The loader creates 8 TPC-H or 24 TPC-DS tables. It does not run benchmark SQL queries or claim TPC certification.
37
+
38
+ - Repository: [buzzcrow/crowdb-tpc-loader](https://github.com/buzzcrow/crowdb-tpc-loader)
39
+ - End-to-end guide: [CROWDB TPC loader documentation](https://crowdb.dev/docs/tpc-loader/)
40
+
41
+ ## Install
42
+
43
+ Python 3.10–3.12 is required. After the first PyPI release:
44
+
45
+ ```sh
46
+ python3 -m venv .venv
47
+ . .venv/bin/activate
48
+ python -m pip install crowdb-tpc-loader
49
+ ```
50
+
51
+ Until PyPI publication, install from a checkout with `python -m pip install --only-binary=:all: -e .`. The first TPC-H run may download `tpchgen-cli` 3.0.0; TPC-DS may download DuckDB's `tpcds` extension. Use `--no-download` and provide these components ahead of time for an offline run.
52
+
53
+ ## Load
54
+
55
+ Start `crowdb/crowdb-iceberg:latest` and export the `ICEBERG_URI` and `ICEBERG_TOKEN` values printed by `docker exec <container> crowdb-monitor credentials show --format env`. Keep the token private. Use a new namespace for each run:
56
+
57
+ ```sh
58
+ crowdb-tpc-loader load --benchmark tpch --sf 0.01 \
59
+ --namespace tpch_demo --report-file ./tpch-demo.json
60
+ crowdb-tpc-loader load --benchmark tpcds --sf 0.01 \
61
+ --namespace tpcds_demo --report-file ./tpcds-demo.json
62
+ ```
63
+
64
+ The loader validates the entire generated dataset before creating a table. It writes different tables concurrently, with 8 workers by default. Use `--upload-workers N` to control concurrent Iceberg table writes (1–24). Each table's files are uploaded and registered in one snapshot, with a durable report checkpoint before each remote side effect. One table's failure does not roll back tables that already succeeded. The report identifies committed, unregistered, and uncertain files; see [recovery](docs/RECOVERY.md) before retrying. An existing table stops the default load; `--on-exists skip` leaves it unchanged without verifying it.
65
+
66
+ For local Parquet only, use `crowdb-tpc-loader generate --benchmark tpch --sf 0.01 --output-dir ./tpch-001`.
67
+
68
+ ## Check the result
69
+
70
+ Run the read-only verifier from a checkout after loading:
71
+
72
+ ```sh
73
+ python scripts/verify_crowdb.py ./tpch-demo.json --require-complete --iceberg-scan
74
+ ```
75
+
76
+ With DuckDB's `iceberg` and `httpfs` extensions, attach the REST Catalog using its token and query `tpch_demo.region` or run TPC-H Q1 against `tpch_demo.lineitem`. The [website guide](https://crowdb.dev/docs/tpc-loader/) has the SQL. A published `latest` image and the local DuckDB 1.5.6 CLI passed an SF 0.01 TPC-H import, independent table verification, and Q1 read on October 1, 2026. This is an integration check, not a performance result. See [compatibility](docs/COMPATIBILITY.md) and the [test record](docs/TEST_REPORT.md).
77
+
78
+ ## Develop and publish
79
+
80
+ ```sh
81
+ python -m pip install --only-binary=:all: -e '.[dev,sql-test]'
82
+ ruff check src tests scripts
83
+ ruff format --check src tests scripts
84
+ python -m pytest
85
+ python -m build
86
+ python -m twine check dist/*
87
+ ```
88
+
89
+ CI runs lint, format, tests, and package checks. Publishing is manual through [the PyPI workflow](.github/workflows/publish.yml) from a `release/v<version>` branch after configuring PyPI Trusted Publishing. See [testing](docs/TESTING.md) for optional real generator and CROWDB runs. The package is Apache-2.0; third-party generators and libraries keep their own licenses.
90
+
91
+ To publish `0.1.0`:
92
+
93
+ 1. In PyPI, create a pending Trusted Publisher for project `crowdb-tpc-loader`: GitHub owner `buzzcrow`, repository `crowdb-tpc-loader`, workflow `publish.yml`, environment `pypi`. Create the `pypi` environment in GitHub.
94
+ 2. After CI is green, create and push branch `release/v0.1.0` from the commit to publish.
95
+ 3. In GitHub Actions, open **Publish to PyPI**, click **Run workflow**, select branch `release/v0.1.0`, then run it. The workflow verifies the branch name against the package version, reruns checks, builds distributions, and publishes through OIDC. No PyPI API token is stored in GitHub.
96
+ 4. Confirm the files on [PyPI](https://pypi.org/project/crowdb-tpc-loader/), then test `python -m pip install --no-cache-dir crowdb-tpc-loader==0.1.0` in a clean environment and run `crowdb-tpc-loader --version`.
@@ -0,0 +1,63 @@
1
+ # CrowDB TPC Loader
2
+
3
+ Generate TPC-H or TPC-DS Parquet, upload it to CROWDB Iceberg, and register complete tables through the REST Catalog. The loader creates 8 TPC-H or 24 TPC-DS tables. It does not run benchmark SQL queries or claim TPC certification.
4
+
5
+ - Repository: [buzzcrow/crowdb-tpc-loader](https://github.com/buzzcrow/crowdb-tpc-loader)
6
+ - End-to-end guide: [CROWDB TPC loader documentation](https://crowdb.dev/docs/tpc-loader/)
7
+
8
+ ## Install
9
+
10
+ Python 3.10–3.12 is required. After the first PyPI release:
11
+
12
+ ```sh
13
+ python3 -m venv .venv
14
+ . .venv/bin/activate
15
+ python -m pip install crowdb-tpc-loader
16
+ ```
17
+
18
+ Until PyPI publication, install from a checkout with `python -m pip install --only-binary=:all: -e .`. The first TPC-H run may download `tpchgen-cli` 3.0.0; TPC-DS may download DuckDB's `tpcds` extension. Use `--no-download` and provide these components ahead of time for an offline run.
19
+
20
+ ## Load
21
+
22
+ Start `crowdb/crowdb-iceberg:latest` and export the `ICEBERG_URI` and `ICEBERG_TOKEN` values printed by `docker exec <container> crowdb-monitor credentials show --format env`. Keep the token private. Use a new namespace for each run:
23
+
24
+ ```sh
25
+ crowdb-tpc-loader load --benchmark tpch --sf 0.01 \
26
+ --namespace tpch_demo --report-file ./tpch-demo.json
27
+ crowdb-tpc-loader load --benchmark tpcds --sf 0.01 \
28
+ --namespace tpcds_demo --report-file ./tpcds-demo.json
29
+ ```
30
+
31
+ The loader validates the entire generated dataset before creating a table. It writes different tables concurrently, with 8 workers by default. Use `--upload-workers N` to control concurrent Iceberg table writes (1–24). Each table's files are uploaded and registered in one snapshot, with a durable report checkpoint before each remote side effect. One table's failure does not roll back tables that already succeeded. The report identifies committed, unregistered, and uncertain files; see [recovery](docs/RECOVERY.md) before retrying. An existing table stops the default load; `--on-exists skip` leaves it unchanged without verifying it.
32
+
33
+ For local Parquet only, use `crowdb-tpc-loader generate --benchmark tpch --sf 0.01 --output-dir ./tpch-001`.
34
+
35
+ ## Check the result
36
+
37
+ Run the read-only verifier from a checkout after loading:
38
+
39
+ ```sh
40
+ python scripts/verify_crowdb.py ./tpch-demo.json --require-complete --iceberg-scan
41
+ ```
42
+
43
+ With DuckDB's `iceberg` and `httpfs` extensions, attach the REST Catalog using its token and query `tpch_demo.region` or run TPC-H Q1 against `tpch_demo.lineitem`. The [website guide](https://crowdb.dev/docs/tpc-loader/) has the SQL. A published `latest` image and the local DuckDB 1.5.6 CLI passed an SF 0.01 TPC-H import, independent table verification, and Q1 read on October 1, 2026. This is an integration check, not a performance result. See [compatibility](docs/COMPATIBILITY.md) and the [test record](docs/TEST_REPORT.md).
44
+
45
+ ## Develop and publish
46
+
47
+ ```sh
48
+ python -m pip install --only-binary=:all: -e '.[dev,sql-test]'
49
+ ruff check src tests scripts
50
+ ruff format --check src tests scripts
51
+ python -m pytest
52
+ python -m build
53
+ python -m twine check dist/*
54
+ ```
55
+
56
+ CI runs lint, format, tests, and package checks. Publishing is manual through [the PyPI workflow](.github/workflows/publish.yml) from a `release/v<version>` branch after configuring PyPI Trusted Publishing. See [testing](docs/TESTING.md) for optional real generator and CROWDB runs. The package is Apache-2.0; third-party generators and libraries keep their own licenses.
57
+
58
+ To publish `0.1.0`:
59
+
60
+ 1. In PyPI, create a pending Trusted Publisher for project `crowdb-tpc-loader`: GitHub owner `buzzcrow`, repository `crowdb-tpc-loader`, workflow `publish.yml`, environment `pypi`. Create the `pypi` environment in GitHub.
61
+ 2. After CI is green, create and push branch `release/v0.1.0` from the commit to publish.
62
+ 3. In GitHub Actions, open **Publish to PyPI**, click **Run workflow**, select branch `release/v0.1.0`, then run it. The workflow verifies the branch name against the package version, reruns checks, builds distributions, and publishes through OIDC. No PyPI API token is stored in GitHub.
63
+ 4. Confirm the files on [PyPI](https://pypi.org/project/crowdb-tpc-loader/), then test `python -m pip install --no-cache-dir crowdb-tpc-loader==0.1.0` in a clean environment and run `crowdb-tpc-loader --version`.
@@ -0,0 +1,13 @@
1
+ # Compatibility
2
+
3
+ The package requires Python 3.10–3.12, PyArrow `>=18,<24`, PyIceberg `>=0.10,<0.11`, and DuckDB `>=1.4,<1.6`. TPC-H uses `tpchgen-cli` 3.x; automatic download pins 3.0.0. These ranges are package constraints, not a claim that every version combination passed. The [test record](TEST_REPORT.md) names the observed local combination.
4
+
5
+ `load` requires an Iceberg REST Catalog and working remote FileIO. It reads table locations and credentials from the catalog; it never registers a local path. The default `crowdb_tpc_loader.crowdb_fileio.CrowdbFileIO` uses exact-object opens on CROWDB's native Iceberg endpoint, which does not require S3 prefix listing. Other deployments can provide a trusted FileIO class with `--py-io-impl`.
6
+
7
+ Before generation, the loader checks catalog access, existing tables, namespace creation, and a staged FileIO write probe where cleanup is supported. CROWDB's native FileIO has no remote file delete, so the first real table upload acts as its write test. The entire generated dataset passes schema and footer checks before any benchmark table is created.
8
+
9
+ The catalog can return transient HTTP 503 when its bounded operation capacity is busy. Table creation retries up to six attempts with short backoff. If a retry finds a table created by an earlier attempt in the same run, it verifies the run ID before continuing. Registration (`add_files`) is never blindly retried after an ambiguous response; the loader reads the current snapshot to reconcile the outcome. Keep the JSON report for [failure recovery](RECOVERY.md).
10
+
11
+ Each worker owns a separate Catalog client and one table. Eight workers run by default; `--upload-workers` accepts 1–24. Files within a table upload in sequence, then that table commits once. Other tables can upload or commit at the same time. This can expose server backpressure; choose fewer workers for a small deployment if bounded retries are still exhausted. A failure in one table does not roll back another table's successful snapshot.
12
+
13
+ TPC-DS generation uses DuckDB's `tpcds` extension and a disk database. `--memory-limit` controls DuckDB memory, not total process memory. TPC-H defaults to `ceil(SF / 10)` parts. Reports record the actual dependency and generator versions for each run. Distributed deployments and the full TPC query suites need separate acceptance.
@@ -0,0 +1,49 @@
1
+ # Failure recovery: determine commit status before handling files
2
+
3
+ The tool does not provide automatic resume, overwrite, drop, purge, or remote garbage collection. There is no atomic transaction across all 8 or 24 tables. With concurrent table writes, each table uploads its files and commits one snapshot independently. An upload or commit failure in one table does not roll back successful snapshots in other tables.
4
+
5
+ ## Reports and directories
6
+
7
+ Keep the JSON report and working subdirectory shown in the console. A persistent `--report-file` path is the most reliable choice. Before uploading, the tool records the URI, run ID, and file status on disk. Before a commit, it marks the file `commit_unknown` and persists that state. If the process crashes between upload and commit, the unknown result will not be treated as definitely uncommitted.
8
+
9
+ Forced process termination, sudden power loss, and disk damage can still interrupt the latest checkpoint. Atomic replacement, fsync, and reports in two locations reduce this risk but do not provide a distributed transaction guarantee.
10
+
11
+ | Status | Meaning | Handling |
12
+ |---|---|---|
13
+ | `local` | No target remote URI yet | A locally generated file; this does not imply upload |
14
+ | `upload_started` | Target URI recorded; the object may be absent, partly uploaded, or complete | Check whether the object exists; do not infer its size from status alone |
15
+ | `uploaded_unregistered` | Upload accepted by the server; commit not attempted, or explicitly rejected and verified unregistered | Check other snapshots and references before cleanup |
16
+ | `commit_unknown` | Commit started or may have started; result cannot be confirmed | Keep the file; do not blindly repeat `add_files` |
17
+ | `registered` | URI observed in the verified current snapshot | Do not delete directly; follow the Iceberg snapshot and metadata lifecycle |
18
+
19
+ `summary.unregistered_uploads` lists **candidates for inspection**, not objects that can be deleted in bulk. Keep `summary.uncertain_uploads` as a priority. `summary.tables_requiring_inspection` includes tables that may have been created but not fully verified, or for which writes failed. `table_creation_state=unknown_or_unvalidated` does not mean the table is definitely absent.
20
+
21
+ ## Common cases
22
+
23
+ ### Generation or schema validation fails
24
+
25
+ This release creates benchmark tables only after validating the entire Parquet dataset. Check the generator version, actual schema, missing tables, and source-versus-export row counts in the report. Run again with a new output directory or namespace; do not overwrite the original files.
26
+
27
+ ### FileIO preflight fails
28
+
29
+ A reachable REST endpoint does not prove that storage is writable. Check whether the returned storage address is reachable from the client, whether authentication needs a separate token, whether the correct FileIO was selected, and whether the HTTP service supports the required operations. Preflight uses an uncommitted staged table and does not fall back to a local path.
30
+
31
+ ### Some tables succeed and a later table fails
32
+
33
+ First run the independent verifier on tables marked `succeeded` in the report. The failed table may exist but be empty, or it may have a committed or uncertain snapshot. A default rerun stops on existing tables as a protective measure.
34
+
35
+ `--on-exists skip` does not repair a failed table and may skip an empty one. Do not use it to claim that the full dataset has been completed. The easiest way to verify a fresh run is to use a new namespace, then handle old data according to Iceberg's rules. If the old namespace must be kept, inspect every snapshot, branch, and reference for the table before deciding how to handle empty tables or orphaned objects. The tool does not perform these destructive operations.
36
+
37
+ ### Commit request times out
38
+
39
+ The tool makes up to three rounds of read-only verification without calling `add_files` again. It records success only if the current snapshot's file set, row counts, and sizes match exactly. Otherwise, it records `uncertain` and marks that table uncertain; other in-flight tables may still finish.
40
+
41
+ Another writer may subsequently change the final snapshot. Files recorded as `registered` may also be referenced by historical snapshots; absence from the current snapshot does not prove absence from history. Use the namespace, table UUID, run ID, and all file URIs to inspect server logs, snapshots, and branch references before proceeding.
42
+
43
+ ### Insufficient local space or process interruption
44
+
45
+ Check the exit code, external report, and working path. The working directory is kept after a failure, and the tool does not delete remote benchmark objects. If local cleanup fails after a successful run, the tool records a cleanup warning without changing a verified successful remote commit into a failure. Before manually deleting a local directory, confirm that it is this run's working subdirectory, not the `--work-dir` parent.
46
+
47
+ ## Information to include in a bug report
48
+
49
+ Keep the command with the token removed, Python/dependency/generator versions, run ID, error from the report, failed table, and FileIO type. Inspect the redacted report manually before sharing it. Do not include tokens, storage keys, signed URLs, complete environment variables, or private business data in an issue.
@@ -0,0 +1,37 @@
1
+ # Testing
2
+
3
+ CI runs `ruff check src tests scripts`, `ruff format --check src tests scripts`, and `python -m pytest` on Python 3.10, 3.11, and 3.12. It builds and checks the wheel and sdist on Python 3.12. The default tests include unit and local PyArrow/PyIceberg integration tests. Real generators and a CROWDB instance are opt-in.
4
+
5
+ ## Local checks
6
+
7
+ ```sh
8
+ python -m pip install --only-binary=:all: -e '.[dev,sql-test]'
9
+ ruff check src tests scripts
10
+ ruff format --check src tests scripts
11
+ python -m pytest -q
12
+ python -m build
13
+ python -m twine check dist/*
14
+ ```
15
+
16
+ Check a built wheel in a fresh virtual environment before release. Confirm both `crowdb-tpc-loader --version` and `crowdb-tpc-loader load --help` work.
17
+
18
+ ## Real generators
19
+
20
+ These may download generator binaries or extensions and need disk space:
21
+
22
+ ```sh
23
+ CROWDB_TPC_GENERATOR_TESTS=1 python -m pytest tests/integration/test_generators.py -ra
24
+ ```
25
+
26
+ Set `CROWDB_TPC_SF1_TESTS=1` as well to include full SF 1 datasets. Or run `python scripts/smoke_test.py --benchmark tpch --sf 0.01` for local generation only.
27
+
28
+ ## Real CROWDB load
29
+
30
+ Use a dedicated disposable instance with `ICEBERG_URI` and `ICEBERG_TOKEN` exported. The tests create fresh remote namespaces and do not delete them:
31
+
32
+ ```sh
33
+ CROWDB_TPC_CROWDB_TESTS=1 CROWDB_TPC_SF1_TESTS=1 \
34
+ python -m pytest tests/integration/test_crowdb.py -s -ra
35
+ ```
36
+
37
+ For a smaller remote run, use `python scripts/smoke_test.py --benchmark tpch --sf 0.01 --load`. Preserve the JSON report on failure. A skipped integration test is not evidence of remote acceptance.
@@ -0,0 +1,10 @@
1
+ # Test record
2
+
3
+ ## Published image, October 1, 2026
4
+
5
+ - Image: `crowdb/crowdb-iceberg:latest`, digest `sha256:e4e7f80367094a35ab54c2bf1b0a9780811f9a668bc6037cfe7ced1b23373718`, Linux amd64 single-node.
6
+ - Loader: eight concurrent table writes (`--upload-workers 8`), TPC-H SF 0.01, fresh namespace. All 8 tables committed; independent verifier checked 86,805 rows in manifests and remote Parquet footers, plus sample Iceberg scans after local staging was removed.
7
+ - DuckDB CLI: locally built 1.5.6 with `iceberg` and `httpfs`. `region` returned `(1, AMERICA)`. Standard TPC-H Q1 on the imported `lineitem` returned four groups: `(A,F,14876)`, `(N,F,348)`, `(N,O,29181)`, `(R,F,14902)` for `(returnflag, linestatus, count_order)`. The CLI reported 0.075 seconds for that small warm-host run. No baseline or repeated trials were taken.
8
+ - Initial 8-worker attempt saw four transient Catalog HTTP 503 responses and partial table success. The loader now retries bounded 503 create responses; a fresh namespace completed. Both reports are local test artifacts, not included in the package.
9
+
10
+ This proves one integration path and one Q1 execution. It does not establish the full TPC-H suite, benchmark performance, distributed behavior, or TPC certification.
@@ -0,0 +1,64 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77,<83"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "crowdb-tpc-loader"
7
+ version = "0.1.0"
8
+ description = "Generate and safely load TPC-H and TPC-DS Parquet datasets into an Iceberg REST catalog"
9
+ readme = "README.md"
10
+ requires-python = ">=3.10,<3.13"
11
+ license = "Apache-2.0"
12
+ license-files = ["LICENSE"]
13
+ authors = [{name = "CrowDB contributors"}]
14
+ classifiers = [
15
+ "Development Status :: 4 - Beta",
16
+ "Environment :: Console",
17
+ "Programming Language :: Python :: 3",
18
+ "Programming Language :: Python :: 3.10",
19
+ "Programming Language :: Python :: 3.11",
20
+ "Programming Language :: Python :: 3.12",
21
+ "Topic :: Database",
22
+ ]
23
+ dependencies = [
24
+ "botocore>=1.34,<2",
25
+ "pyarrow>=18,<24",
26
+ "duckdb>=1.4,<1.6",
27
+ "pyiceberg[pyarrow]>=0.10,<0.11",
28
+ "packaging>=24,<27",
29
+ ]
30
+
31
+ [project.optional-dependencies]
32
+ dev = ["pytest>=8,<10", "pytest-cov>=5,<8", "build>=1.2,<2", "ruff>=0.11,<1", "twine>=6,<7"]
33
+ sql-test = ["sqlalchemy>=2,<3"]
34
+
35
+ [project.scripts]
36
+ crowdb-tpc-loader = "crowdb_tpc_loader.cli:main"
37
+
38
+ [project.urls]
39
+ Repository = "https://github.com/buzzcrow/crowdb-tpc-loader"
40
+ Issues = "https://github.com/buzzcrow/crowdb-tpc-loader/issues"
41
+
42
+ [tool.setuptools.packages.find]
43
+ where = ["src"]
44
+
45
+ [tool.pytest.ini_options]
46
+ testpaths = ["tests"]
47
+ pythonpath = ["src"]
48
+ addopts = "-ra"
49
+ markers = [
50
+ "integration: requires installed runtime dependencies",
51
+ "generator: invokes the real generator (may download binaries/extensions)",
52
+ "crowdb: writes a fresh namespace in an explicitly configured CrowDB instance",
53
+ ]
54
+
55
+ [tool.ruff]
56
+ line-length = 110
57
+ target-version = "py310"
58
+
59
+ [tool.ruff.lint]
60
+ select = ["E4", "E7", "E9", "F"]
61
+
62
+ [tool.ruff.lint.per-file-ignores]
63
+ "tests/integration/test_iceberg_http.py" = ["E402", "F811"]
64
+ "tests/integration/test_parquet.py" = ["E402"]
@@ -0,0 +1,11 @@
1
+ # Candidate reproducibility matrix, NOT an already-tested dependency lock.
2
+ # Install with --only-binary=:all: and run the real integration suite.
3
+ pyarrow==22.0.0
4
+ pyiceberg[pyarrow]==0.10.0
5
+ duckdb==1.5.6
6
+ packaging==25.0
7
+ sqlalchemy==2.0.43
8
+ pytest==8.4.2
9
+ pytest-cov==6.2.1
10
+ # Optional: use the loader's binary-only bootstrap instead of installing this.
11
+ # tpchgen-cli==3.0.0
@@ -0,0 +1,71 @@
1
+ #!/usr/bin/env python3
2
+ """Run a local generation smoke test, or an explicitly requested remote load.
3
+
4
+ No remote writes unless --load is present. Uses a unique output directory and,
5
+ for loads, a unique namespace. It never drops remote data.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import argparse
11
+ from pathlib import Path
12
+ import subprocess
13
+ import sys
14
+ import uuid
15
+
16
+
17
+ def main() -> int:
18
+ parser = argparse.ArgumentParser(description=__doc__)
19
+ parser.add_argument("--benchmark", choices=("tpch", "tpcds"), required=True)
20
+ parser.add_argument("--sf", default="0.01")
21
+ parser.add_argument("--output-root", type=Path, default=Path("smoke-results"))
22
+ parser.add_argument(
23
+ "--load", action="store_true", help="explicitly enable writes to ICEBERG_URI using ICEBERG_TOKEN"
24
+ )
25
+ parser.add_argument("--py-io-impl")
26
+ parser.add_argument("--tpchgen", type=Path)
27
+ parser.add_argument("--full-read", action="store_true")
28
+ args = parser.parse_args()
29
+ run = args.output_root / (args.benchmark + "-" + uuid.uuid4().hex[:12])
30
+ run.mkdir(parents=True, exist_ok=False)
31
+ command = [
32
+ sys.executable,
33
+ "-m",
34
+ "crowdb_tpc_loader",
35
+ "load" if args.load else "generate",
36
+ "--benchmark",
37
+ args.benchmark,
38
+ "--sf",
39
+ args.sf,
40
+ "--report-file",
41
+ str(run / "report.json"),
42
+ ]
43
+ if args.tpchgen:
44
+ command += ["--tpchgen", str(args.tpchgen)]
45
+ if args.load:
46
+ namespace = "smoke_" + args.benchmark + "_" + uuid.uuid4().hex[:12]
47
+ command += ["--namespace", namespace, "--work-dir", str(run / "staging")]
48
+ if args.py_io_impl:
49
+ command += ["--py-io-impl", args.py_io_impl]
50
+ else:
51
+ command += ["--output-dir", str(run / "generated")]
52
+ print(f"Smoke output: {run.resolve()}", flush=True)
53
+ result = subprocess.run(command, check=False)
54
+ if result.returncode != 0 or not args.load:
55
+ return result.returncode
56
+ verify = [
57
+ sys.executable,
58
+ str(Path(__file__).with_name("verify_crowdb.py")),
59
+ str(run / "report.json"),
60
+ "--iceberg-scan",
61
+ "--require-complete",
62
+ ]
63
+ if args.py_io_impl:
64
+ verify += ["--py-io-impl", args.py_io_impl]
65
+ if args.full_read:
66
+ verify += ["--full-read"]
67
+ return subprocess.run(verify, check=False).returncode
68
+
69
+
70
+ if __name__ == "__main__":
71
+ raise SystemExit(main())