gigaxml 0.9.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. gigaxml-0.9.0/LICENSE +21 -0
  2. gigaxml-0.9.0/PKG-INFO +241 -0
  3. gigaxml-0.9.0/README.md +201 -0
  4. gigaxml-0.9.0/pyproject.toml +101 -0
  5. gigaxml-0.9.0/setup.cfg +4 -0
  6. gigaxml-0.9.0/src/gigaxml/__init__.py +5 -0
  7. gigaxml-0.9.0/src/gigaxml/checkpoint.py +417 -0
  8. gigaxml-0.9.0/src/gigaxml/cli.py +1067 -0
  9. gigaxml-0.9.0/src/gigaxml/config.py +328 -0
  10. gigaxml-0.9.0/src/gigaxml/errors.py +150 -0
  11. gigaxml-0.9.0/src/gigaxml/fields.py +464 -0
  12. gigaxml-0.9.0/src/gigaxml/generate.py +566 -0
  13. gigaxml-0.9.0/src/gigaxml/gui/__init__.py +7 -0
  14. gigaxml-0.9.0/src/gigaxml/gui/app.py +123 -0
  15. gigaxml-0.9.0/src/gigaxml/gui/batch_queue.py +184 -0
  16. gigaxml-0.9.0/src/gigaxml/gui/cli_process.py +281 -0
  17. gigaxml-0.9.0/src/gigaxml/gui/document_info.py +134 -0
  18. gigaxml-0.9.0/src/gigaxml/gui/error_advice.py +131 -0
  19. gigaxml-0.9.0/src/gigaxml/gui/field_rows.py +294 -0
  20. gigaxml-0.9.0/src/gigaxml/gui/i18n.py +439 -0
  21. gigaxml-0.9.0/src/gigaxml/gui/inspect_report.py +269 -0
  22. gigaxml-0.9.0/src/gigaxml/gui/main_window.py +469 -0
  23. gigaxml-0.9.0/src/gigaxml/gui/panels/__init__.py +5 -0
  24. gigaxml-0.9.0/src/gigaxml/gui/panels/batch.py +311 -0
  25. gigaxml-0.9.0/src/gigaxml/gui/panels/document.py +308 -0
  26. gigaxml-0.9.0/src/gigaxml/gui/panels/errors.py +212 -0
  27. gigaxml-0.9.0/src/gigaxml/gui/panels/execution.py +919 -0
  28. gigaxml-0.9.0/src/gigaxml/gui/panels/fields.py +674 -0
  29. gigaxml-0.9.0/src/gigaxml/gui/panels/preview.py +430 -0
  30. gigaxml-0.9.0/src/gigaxml/gui/panels/results.py +297 -0
  31. gigaxml-0.9.0/src/gigaxml/gui/panels/settings.py +189 -0
  32. gigaxml-0.9.0/src/gigaxml/gui/panels/structure.py +828 -0
  33. gigaxml-0.9.0/src/gigaxml/gui/progress.py +130 -0
  34. gigaxml-0.9.0/src/gigaxml/gui/recent_files.py +160 -0
  35. gigaxml-0.9.0/src/gigaxml/gui/run_report.py +285 -0
  36. gigaxml-0.9.0/src/gigaxml/gui/sampling.py +260 -0
  37. gigaxml-0.9.0/src/gigaxml/gui/saved_configs.py +145 -0
  38. gigaxml-0.9.0/src/gigaxml/gui/settings.py +184 -0
  39. gigaxml-0.9.0/src/gigaxml/inspect.py +1176 -0
  40. gigaxml-0.9.0/src/gigaxml/parser/__init__.py +25 -0
  41. gigaxml-0.9.0/src/gigaxml/parser/streaming.py +436 -0
  42. gigaxml-0.9.0/src/gigaxml/paths.py +153 -0
  43. gigaxml-0.9.0/src/gigaxml/run.py +496 -0
  44. gigaxml-0.9.0/src/gigaxml/sample.py +166 -0
  45. gigaxml-0.9.0/src/gigaxml/writers.py +656 -0
  46. gigaxml-0.9.0/src/gigaxml.egg-info/PKG-INFO +241 -0
  47. gigaxml-0.9.0/src/gigaxml.egg-info/SOURCES.txt +49 -0
  48. gigaxml-0.9.0/src/gigaxml.egg-info/dependency_links.txt +1 -0
  49. gigaxml-0.9.0/src/gigaxml.egg-info/entry_points.txt +3 -0
  50. gigaxml-0.9.0/src/gigaxml.egg-info/requires.txt +22 -0
  51. gigaxml-0.9.0/src/gigaxml.egg-info/top_level.txt +1 -0
gigaxml-0.9.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 gg320324492-lgtm
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
gigaxml-0.9.0/PKG-INFO ADDED
@@ -0,0 +1,241 @@
1
+ Metadata-Version: 2.4
2
+ Name: gigaxml
3
+ Version: 0.9.0
4
+ Summary: A production-oriented CLI toolkit for profiling, validating and extracting structured data from multi-gigabyte XML files with bounded memory usage.
5
+ License-Expression: MIT
6
+ Project-URL: Homepage, https://github.com/gg320324492-lgtm/GigaXML-Memory-Efficient-XML-Extraction-Toolkit
7
+ Keywords: xml,streaming,etl,iterparse,large-files,cli,memory-efficient
8
+ Classifier: Development Status :: 3 - Alpha
9
+ Classifier: Environment :: Console
10
+ Classifier: Intended Audience :: Developers
11
+ Classifier: Operating System :: OS Independent
12
+ Classifier: Programming Language :: Python :: 3
13
+ Classifier: Programming Language :: Python :: 3.11
14
+ Classifier: Programming Language :: Python :: 3.12
15
+ Classifier: Programming Language :: Python :: 3.13
16
+ Classifier: Topic :: Text Processing :: Markup :: XML
17
+ Classifier: Typing :: Typed
18
+ Requires-Python: <3.14,>=3.11
19
+ Description-Content-Type: text/markdown
20
+ License-File: LICENSE
21
+ Requires-Dist: lxml>=5.0
22
+ Requires-Dist: pyyaml>=6.0
23
+ Provides-Extra: parquet
24
+ Requires-Dist: pyarrow>=14.0; extra == "parquet"
25
+ Provides-Extra: pretty
26
+ Requires-Dist: rich>=13.0; extra == "pretty"
27
+ Provides-Extra: gui
28
+ Requires-Dist: PySide6>=6.6; extra == "gui"
29
+ Requires-Dist: pytest-qt>=4.4; extra == "gui"
30
+ Provides-Extra: dev
31
+ Requires-Dist: pytest>=8.0; extra == "dev"
32
+ Requires-Dist: pytest-cov>=5.0; extra == "dev"
33
+ Requires-Dist: ruff>=0.6; extra == "dev"
34
+ Requires-Dist: psutil>=5.9; extra == "dev"
35
+ Requires-Dist: pyarrow>=14.0; extra == "dev"
36
+ Requires-Dist: rich>=13.0; extra == "dev"
37
+ Requires-Dist: pyinstaller>=6.10; extra == "dev"
38
+ Requires-Dist: pillow>=10.0; extra == "dev"
39
+ Dynamic: license-file
40
+
41
+ # GigaXML
42
+
43
+ A production-oriented CLI toolkit for profiling, validating and extracting structured data
44
+ from multi-gigabyte XML files with bounded memory usage.
45
+
46
+ ## What this is
47
+
48
+ The point of this project is not "it can parse XML" — plenty of tools can. The point is
49
+ **constant, bounded peak memory while extracting from 4 GB / 10 GB files**, and being able
50
+ to prove it with a reproducible measurement harness.
51
+
52
+ Measured on this machine, on a generated 4.05 GiB file holding 11,915,264 records:
53
+
54
+ | | |
55
+ |---|---|
56
+ | **Input** | 4142.72 MiB, 11,915,264 records |
57
+ | **Time** | 284.67 s (14.6 MiB/s, 41,856 records/s) |
58
+ | **Peak RSS** | 33.703 MiB |
59
+ | **Increase over the post-import baseline** | **5.059 MiB** |
60
+
61
+ The same run at 1 GiB (2,978,816 records) added **5.105 MiB** — four times the input and
62
+ the increase did not move. Something that accumulated per record would make the four
63
+ gigabyte figure four times the one gigabyte figure; it is flat to within a percent. Every
64
+ number here comes from a script in this repository; see [Benchmarks](#benchmarks).
65
+
66
+ The configuration used above reads six fields, including a nested path and a type
67
+ conversion — the shapes a real config uses:
68
+
69
+ ```yaml
70
+ record: /catalog/products/product
71
+ fields:
72
+ id: {path: '@id'}
73
+ type: {path: '@type'}
74
+ name: {path: name}
75
+ category: {path: category}
76
+ price: {path: price, type: float}
77
+ manufacturer: {path: manufacturer/name}
78
+ ```
79
+
80
+ ## Install
81
+
82
+ From PyPI:
83
+
84
+ ```bash
85
+ pip install gigaxml # the command-line toolkit
86
+ pip install "gigaxml[gui]" # and the desktop application (pulls PySide6: 640 MiB installed, measured on Windows)
87
+ ```
88
+
89
+ Parquet output needs the `parquet` extra (`pip install "gigaxml[parquet]"`); CSV and JSONL
90
+ do not. The desktop application is also packaged per platform — an unsigned Windows
91
+ build, an Apple Silicon `.dmg` and an x86_64 AppImage — under
92
+ [Releases](https://github.com/gg320324492-lgtm/GigaXML-Memory-Efficient-XML-Extraction-Toolkit/releases);
93
+ its release notes say what each build runs on and what the unsigned warnings mean.
94
+
95
+ From source, if you would rather:
96
+
97
+ ```bash
98
+ git clone https://github.com/gg320324492-lgtm/GigaXML-Memory-Efficient-XML-Extraction-Toolkit.git
99
+ cd GigaXML-Memory-Efficient-XML-Extraction-Toolkit
100
+ python -m venv .venv
101
+ .venv/Scripts/python -m pip install -e ".[dev]" # POSIX: .venv/bin/python
102
+ ```
103
+
104
+ `lxml` and `pyyaml` are the only required dependencies.
105
+
106
+ ## Thirty seconds
107
+
108
+ Point it at a document you know nothing about, let it propose a config, then run it:
109
+
110
+ ```bash
111
+ # 1. What is in this file?
112
+ gigaxml inspect big.xml
113
+
114
+ # 2. Write a starting-point config for the highest-ranked record candidate
115
+ gigaxml inspect big.xml --generate-config config.yaml --infer-types
116
+
117
+ # 3. Extract
118
+ gigaxml extract big.xml -c config.yaml -o out.csv
119
+ ```
120
+
121
+ `inspect` does not read the whole document — it reports the structure and the repeating
122
+ paths it found, and ranks them. The config it writes is explicitly a **starting point**,
123
+ not a conclusion; read the comments in it.
124
+
125
+ For a quick look at the data before committing to a full run:
126
+
127
+ ```bash
128
+ gigaxml sample big.xml -c config.yaml -n 20 -o first20.jsonl
129
+ ```
130
+
131
+ ### What it looks like
132
+
133
+ ![gigaxml inspect reading the structure of a document](assets/inspect.gif)
134
+
135
+ *Finding the records in a document that opens with a licence comment.*
136
+
137
+ ![gigaxml extract processing a four-gigabyte file](assets/extract-4g.gif)
138
+
139
+ *Extracting 11.9 million records from 4.05 GiB.*
140
+
141
+ > **Both of these are animations rendered from the tools' real output, not screen
142
+ > recordings.** The text is what the tools actually printed and the timings are the
143
+ > measured ones, but the frames are drawn rather than captured — this machine's sandbox
144
+ > does not permit screen capture. Each frame carries the same note.
145
+
146
+ ## Benchmarks
147
+
148
+ Three generated datasets, two configs, measured in a subprocess with `psutil`:
149
+
150
+ ```
151
+ dataset fields input MiB records s MiB/s peak delta
152
+ ------------------------------------------------------------------------------
153
+ 100MB 6 fields 100.57 290,900 6.92 14.5 33.281 5.113
154
+ 100MB 1 field 100.57 290,900 2.69 37.4 31.203 2.617
155
+ 1GB 6 fields 1033.65 2,978,816 70.66 14.6 33.676 5.105
156
+ 1GB 1 field 1033.65 2,978,816 27.62 37.4 31.496 2.922
157
+ 4GB 6 fields 4142.72 11,915,264 284.67 14.6 33.703 5.059
158
+ 4GB 1 field 4142.72 11,915,264 112.04 37.0 31.500 2.973
159
+ ```
160
+
161
+ `peak` and `delta` are MiB; `delta` is against the same process's post-import baseline.
162
+ `peak` is `PeakWorkingSetSize` — the maximum over the process's life, not the current RSS
163
+ at the end, which reads 10–14% lower.
164
+ The one-field rows are a control, not the headline: a single-field config is the easiest
165
+ member of this family to run, and quoting it alone would overstate what a real config
166
+ costs. Field count costs about **2.5×** in throughput.
167
+
168
+ To reproduce, generate the datasets and run the harness in [`benchmarks/`](benchmarks):
169
+
170
+ ```bash
171
+ gigaxml generate --size 100MB -o data/b100m.xml
172
+ gigaxml generate --size 1GB -o data/b1g.xml
173
+ gigaxml generate --size 4GB -o data/b4g.xml
174
+ python benchmarks/bench_extraction.py
175
+ ```
176
+
177
+ `--size` is approximate: `--size 1GB` produces 1033.65 MiB, not 1024, and the sizes above
178
+ are the measured ones. Peak and delta are both reported because either alone can be
179
+ misread — peak includes about 32 MiB of interpreter and library overhead, delta is what
180
+ the workload is responsible for, and both baselines in this repository are taken after
181
+ every import so that two deltas are comparable.
182
+
183
+ ## Non-goals
184
+
185
+ - No full XPath 3.1 — XPath is evaluated only inside a single record subtree.
186
+ - No arbitrary byte-offset seek/resume — XML byte offsets are not a safe parse boundary.
187
+ - No AI/ML structure inference — confidence values are deterministic statistics.
188
+ - No real customer data — everything runs on synthetic, reproducible datasets.
189
+ - No fabricated benchmarks — every performance claim comes from a runnable script.
190
+
191
+ ## Known limitations
192
+
193
+ - **`--resume` re-parses and skips; it does not seek.** XML cannot be re-entered
194
+ mid-stream, so continuing a run means reading from the beginning and discarding the
195
+ records already accounted for. On a 403 MB file that costs 8.7 s against 17.1 s to
196
+ extract, so resuming saves roughly half of what you had already done. `--help` says so
197
+ too.
198
+ - **`inspect` is slower than `extract`** — 17.3 MiB/s against 38.2 MiB/s on the same
199
+ 1 GB file. It maintains several parallel bookkeeping stacks per element. It is also the
200
+ command you run once on a document, not in a loop.
201
+ - **An inferred config treats containers as leaves.** `--generate-config` proposes direct
202
+ children and attributes; a field whose element has children of its own is read as
203
+ concatenated text, so `<tags><tag>a</tag><tag>b</tag></tags>` becomes `ab`. Nested
204
+ paths (`manufacturer/name`) have to be written by hand, as the generated comments say.
205
+ - **Types are inferred from a sample**, and `decimal` is never inferred. If a field is
206
+ money, set `type: decimal` yourself — `float` cannot represent 49.90 exactly.
207
+ - **Parsing limits are not configurable.** Entities are never expanded, the network is
208
+ never touched, and no DTD is loaded; documents nested deeper than 256 levels, carrying a
209
+ single text node over about 10 MB, or amplified by entities are refused rather than
210
+ partially read. These are deliberate and there are no flags to turn them off.
211
+ - **`--checkpoint-every` verifies the parts on disk before resuming**, which costs one
212
+ pass over the output at about **200 MiB/s** — about 10 ms for 2 MiB of CSV, negligible for
213
+ Parquet, whose row counts come from file metadata. It grows with the size of the
214
+ output, not the input. Measured by `benchmarks/bench_resident.py`.
215
+ - **A resume is dominated by starting the process, not by checking the output.** Against
216
+ an already-complete manifest on a 403 MiB source, the command takes about 770 ms: some
217
+ 400 ms of that is interpreter startup and imports, and most of the rest is hashing the
218
+ source to confirm it has not changed. The part check itself is about 10 ms.
219
+ - **One field value that is itself gigabytes is held in memory.** Records stream, but
220
+ there is no streaming mode for a single value, because there is nothing to stream it
221
+ into.
222
+ - **`sample` reads what it samples into memory.** It is meant for looking at a file, not
223
+ for measuring one.
224
+ - **Input is a local file, never a URL.** A `.xml.gz` file is fine; a document that lives
225
+ behind HTTP is out of scope.
226
+
227
+ ## Development
228
+
229
+ ```bash
230
+ pytest -q # unit + integration
231
+ pytest -q tests/performance # memory and throughput; not in CI
232
+ ruff check .
233
+ ruff format --check .
234
+ ```
235
+
236
+ Performance tests are excluded from CI: they measure memory and throughput, take minutes,
237
+ and are not a pass/fail signal.
238
+
239
+ ## License
240
+
241
+ MIT — see [LICENSE](LICENSE).
@@ -0,0 +1,201 @@
1
+ # GigaXML
2
+
3
+ A production-oriented CLI toolkit for profiling, validating and extracting structured data
4
+ from multi-gigabyte XML files with bounded memory usage.
5
+
6
+ ## What this is
7
+
8
+ The point of this project is not "it can parse XML" — plenty of tools can. The point is
9
+ **constant, bounded peak memory while extracting from 4 GB / 10 GB files**, and being able
10
+ to prove it with a reproducible measurement harness.
11
+
12
+ Measured on this machine, on a generated 4.05 GiB file holding 11,915,264 records:
13
+
14
+ | | |
15
+ |---|---|
16
+ | **Input** | 4142.72 MiB, 11,915,264 records |
17
+ | **Time** | 284.67 s (14.6 MiB/s, 41,856 records/s) |
18
+ | **Peak RSS** | 33.703 MiB |
19
+ | **Increase over the post-import baseline** | **5.059 MiB** |
20
+
21
+ The same run at 1 GiB (2,978,816 records) added **5.105 MiB** — four times the input and
22
+ the increase did not move. Something that accumulated per record would make the four
23
+ gigabyte figure four times the one gigabyte figure; it is flat to within a percent. Every
24
+ number here comes from a script in this repository; see [Benchmarks](#benchmarks).
25
+
26
+ The configuration used above reads six fields, including a nested path and a type
27
+ conversion — the shapes a real config uses:
28
+
29
+ ```yaml
30
+ record: /catalog/products/product
31
+ fields:
32
+ id: {path: '@id'}
33
+ type: {path: '@type'}
34
+ name: {path: name}
35
+ category: {path: category}
36
+ price: {path: price, type: float}
37
+ manufacturer: {path: manufacturer/name}
38
+ ```
39
+
40
+ ## Install
41
+
42
+ From PyPI:
43
+
44
+ ```bash
45
+ pip install gigaxml # the command-line toolkit
46
+ pip install "gigaxml[gui]" # and the desktop application (pulls PySide6: 640 MiB installed, measured on Windows)
47
+ ```
48
+
49
+ Parquet output needs the `parquet` extra (`pip install "gigaxml[parquet]"`); CSV and JSONL
50
+ do not. The desktop application is also packaged per platform — an unsigned Windows
51
+ build, an Apple Silicon `.dmg` and an x86_64 AppImage — under
52
+ [Releases](https://github.com/gg320324492-lgtm/GigaXML-Memory-Efficient-XML-Extraction-Toolkit/releases);
53
+ its release notes say what each build runs on and what the unsigned warnings mean.
54
+
55
+ From source, if you would rather:
56
+
57
+ ```bash
58
+ git clone https://github.com/gg320324492-lgtm/GigaXML-Memory-Efficient-XML-Extraction-Toolkit.git
59
+ cd GigaXML-Memory-Efficient-XML-Extraction-Toolkit
60
+ python -m venv .venv
61
+ .venv/Scripts/python -m pip install -e ".[dev]" # POSIX: .venv/bin/python
62
+ ```
63
+
64
+ `lxml` and `pyyaml` are the only required dependencies.
65
+
66
+ ## Thirty seconds
67
+
68
+ Point it at a document you know nothing about, let it propose a config, then run it:
69
+
70
+ ```bash
71
+ # 1. What is in this file?
72
+ gigaxml inspect big.xml
73
+
74
+ # 2. Write a starting-point config for the highest-ranked record candidate
75
+ gigaxml inspect big.xml --generate-config config.yaml --infer-types
76
+
77
+ # 3. Extract
78
+ gigaxml extract big.xml -c config.yaml -o out.csv
79
+ ```
80
+
81
+ `inspect` does not read the whole document — it reports the structure and the repeating
82
+ paths it found, and ranks them. The config it writes is explicitly a **starting point**,
83
+ not a conclusion; read the comments in it.
84
+
85
+ For a quick look at the data before committing to a full run:
86
+
87
+ ```bash
88
+ gigaxml sample big.xml -c config.yaml -n 20 -o first20.jsonl
89
+ ```
90
+
91
+ ### What it looks like
92
+
93
+ ![gigaxml inspect reading the structure of a document](assets/inspect.gif)
94
+
95
+ *Finding the records in a document that opens with a licence comment.*
96
+
97
+ ![gigaxml extract processing a four-gigabyte file](assets/extract-4g.gif)
98
+
99
+ *Extracting 11.9 million records from 4.05 GiB.*
100
+
101
+ > **Both of these are animations rendered from the tools' real output, not screen
102
+ > recordings.** The text is what the tools actually printed and the timings are the
103
+ > measured ones, but the frames are drawn rather than captured — this machine's sandbox
104
+ > does not permit screen capture. Each frame carries the same note.
105
+
106
+ ## Benchmarks
107
+
108
+ Three generated datasets, two configs, measured in a subprocess with `psutil`:
109
+
110
+ ```
111
+ dataset fields input MiB records s MiB/s peak delta
112
+ ------------------------------------------------------------------------------
113
+ 100MB 6 fields 100.57 290,900 6.92 14.5 33.281 5.113
114
+ 100MB 1 field 100.57 290,900 2.69 37.4 31.203 2.617
115
+ 1GB 6 fields 1033.65 2,978,816 70.66 14.6 33.676 5.105
116
+ 1GB 1 field 1033.65 2,978,816 27.62 37.4 31.496 2.922
117
+ 4GB 6 fields 4142.72 11,915,264 284.67 14.6 33.703 5.059
118
+ 4GB 1 field 4142.72 11,915,264 112.04 37.0 31.500 2.973
119
+ ```
120
+
121
+ `peak` and `delta` are MiB; `delta` is against the same process's post-import baseline.
122
+ `peak` is `PeakWorkingSetSize` — the maximum over the process's life, not the current RSS
123
+ at the end, which reads 10–14% lower.
124
+ The one-field rows are a control, not the headline: a single-field config is the easiest
125
+ member of this family to run, and quoting it alone would overstate what a real config
126
+ costs. Field count costs about **2.5×** in throughput.
127
+
128
+ To reproduce, generate the datasets and run the harness in [`benchmarks/`](benchmarks):
129
+
130
+ ```bash
131
+ gigaxml generate --size 100MB -o data/b100m.xml
132
+ gigaxml generate --size 1GB -o data/b1g.xml
133
+ gigaxml generate --size 4GB -o data/b4g.xml
134
+ python benchmarks/bench_extraction.py
135
+ ```
136
+
137
+ `--size` is approximate: `--size 1GB` produces 1033.65 MiB, not 1024, and the sizes above
138
+ are the measured ones. Peak and delta are both reported because either alone can be
139
+ misread — peak includes about 32 MiB of interpreter and library overhead, delta is what
140
+ the workload is responsible for, and both baselines in this repository are taken after
141
+ every import so that two deltas are comparable.
142
+
143
+ ## Non-goals
144
+
145
+ - No full XPath 3.1 — XPath is evaluated only inside a single record subtree.
146
+ - No arbitrary byte-offset seek/resume — XML byte offsets are not a safe parse boundary.
147
+ - No AI/ML structure inference — confidence values are deterministic statistics.
148
+ - No real customer data — everything runs on synthetic, reproducible datasets.
149
+ - No fabricated benchmarks — every performance claim comes from a runnable script.
150
+
151
+ ## Known limitations
152
+
153
+ - **`--resume` re-parses and skips; it does not seek.** XML cannot be re-entered
154
+ mid-stream, so continuing a run means reading from the beginning and discarding the
155
+ records already accounted for. On a 403 MB file that costs 8.7 s against 17.1 s to
156
+ extract, so resuming saves roughly half of what you had already done. `--help` says so
157
+ too.
158
+ - **`inspect` is slower than `extract`** — 17.3 MiB/s against 38.2 MiB/s on the same
159
+ 1 GB file. It maintains several parallel bookkeeping stacks per element. It is also the
160
+ command you run once on a document, not in a loop.
161
+ - **An inferred config treats containers as leaves.** `--generate-config` proposes direct
162
+ children and attributes; a field whose element has children of its own is read as
163
+ concatenated text, so `<tags><tag>a</tag><tag>b</tag></tags>` becomes `ab`. Nested
164
+ paths (`manufacturer/name`) have to be written by hand, as the generated comments say.
165
+ - **Types are inferred from a sample**, and `decimal` is never inferred. If a field is
166
+ money, set `type: decimal` yourself — `float` cannot represent 49.90 exactly.
167
+ - **Parsing limits are not configurable.** Entities are never expanded, the network is
168
+ never touched, and no DTD is loaded; documents nested deeper than 256 levels, carrying a
169
+ single text node over about 10 MB, or amplified by entities are refused rather than
170
+ partially read. These are deliberate and there are no flags to turn them off.
171
+ - **`--checkpoint-every` verifies the parts on disk before resuming**, which costs one
172
+ pass over the output at about **200 MiB/s** — about 10 ms for 2 MiB of CSV, negligible for
173
+ Parquet, whose row counts come from file metadata. It grows with the size of the
174
+ output, not the input. Measured by `benchmarks/bench_resident.py`.
175
+ - **A resume is dominated by starting the process, not by checking the output.** Against
176
+ an already-complete manifest on a 403 MiB source, the command takes about 770 ms: some
177
+ 400 ms of that is interpreter startup and imports, and most of the rest is hashing the
178
+ source to confirm it has not changed. The part check itself is about 10 ms.
179
+ - **One field value that is itself gigabytes is held in memory.** Records stream, but
180
+ there is no streaming mode for a single value, because there is nothing to stream it
181
+ into.
182
+ - **`sample` reads what it samples into memory.** It is meant for looking at a file, not
183
+ for measuring one.
184
+ - **Input is a local file, never a URL.** A `.xml.gz` file is fine; a document that lives
185
+ behind HTTP is out of scope.
186
+
187
+ ## Development
188
+
189
+ ```bash
190
+ pytest -q # unit + integration
191
+ pytest -q tests/performance # memory and throughput; not in CI
192
+ ruff check .
193
+ ruff format --check .
194
+ ```
195
+
196
+ Performance tests are excluded from CI: they measure memory and throughput, take minutes,
197
+ and are not a pass/fail signal.
198
+
199
+ ## License
200
+
201
+ MIT — see [LICENSE](LICENSE).
@@ -0,0 +1,101 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "gigaxml"
7
+ version = "0.9.0"
8
+ description = "A production-oriented CLI toolkit for profiling, validating and extracting structured data from multi-gigabyte XML files with bounded memory usage."
9
+ readme = "README.md"
10
+ requires-python = ">=3.11,<3.14"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ keywords = ["xml", "streaming", "etl", "iterparse", "large-files", "cli", "memory-efficient"]
14
+ classifiers = [
15
+ "Development Status :: 3 - Alpha",
16
+ "Environment :: Console",
17
+ "Intended Audience :: Developers",
18
+ "Operating System :: OS Independent",
19
+ "Programming Language :: Python :: 3",
20
+ "Programming Language :: Python :: 3.11",
21
+ "Programming Language :: Python :: 3.12",
22
+ "Programming Language :: Python :: 3.13",
23
+ "Topic :: Text Processing :: Markup :: XML",
24
+ "Typing :: Typed",
25
+ ]
26
+ dependencies = [
27
+ "lxml>=5.0",
28
+ "pyyaml>=6.0",
29
+ ]
30
+
31
+ [project.optional-dependencies]
32
+ parquet = ["pyarrow>=14.0"]
33
+ pretty = ["rich>=13.0"]
34
+ # The desktop application. Optional on purpose: the CLI is the product, and Qt is a
35
+ # 150-200 MB dependency that only a user of the window needs. Nothing in gigaxml outside
36
+ # gigaxml.gui may import it.
37
+ #
38
+ # pytest-qt belongs here rather than in dev, and that is not tidiness. It refuses to load
39
+ # without a Qt binding, and it is loaded as a plugin -- so with it in dev, a plain
40
+ # `pip install -e ".[dev]"` gives a pytest that dies before collecting anything:
41
+ # ERROR: pytest-qt requires either PySide6, PyQt5 or PyQt6 installed.
42
+ # Keeping it with the binding means `.[dev]` runs the suite and skips the desktop tests,
43
+ # while `.[dev,gui]` runs them. Both configurations work, which is the point.
44
+ gui = ["PySide6>=6.6", "pytest-qt>=4.4"]
45
+ dev = [
46
+ "pytest>=8.0",
47
+ "pytest-cov>=5.0",
48
+ "ruff>=0.6",
49
+ "psutil>=5.9",
50
+ "pyarrow>=14.0",
51
+ "rich>=13.0",
52
+ "pyinstaller>=6.10",
53
+ # Drawn by tools/make_icon, which the packaging pipeline runs on every platform. A
54
+ # dependency that only a script's author happened to have installed is a build that only
55
+ # that author's machine can run -- and the icon step is the first thing the pipeline
56
+ # calls, so its absence fails the build before anything is compiled. Build-time only,
57
+ # never a runtime one: nothing in `src/` imports it.
58
+ "pillow>=10.0",
59
+ ]
60
+
61
+ [project.scripts]
62
+ gigaxml = "gigaxml.cli:main"
63
+ gigaxml-gui = "gigaxml.gui.app:main"
64
+
65
+ [project.urls]
66
+ Homepage = "https://github.com/gg320324492-lgtm/GigaXML-Memory-Efficient-XML-Extraction-Toolkit"
67
+
68
+ [tool.setuptools.packages.find]
69
+ where = ["src"]
70
+
71
+ [tool.pytest.ini_options]
72
+ testpaths = ["tests"]
73
+ pythonpath = ["."]
74
+ addopts = "-ra"
75
+ markers = [
76
+ "performance: memory / throughput measurement (excluded from CI, see ROADMAP A11)",
77
+ ]
78
+
79
+ [tool.ruff]
80
+ line-length = 100
81
+ target-version = "py311"
82
+ src = ["src", "tests"]
83
+ # `docs/` holds the human-authored spec documents. Ruff >= 0.16 reformats fenced
84
+ # Python code blocks inside Markdown, which would rewrite the spec text itself.
85
+ # The spec is an input artifact, not source code, so it is excluded from linting.
86
+ extend-exclude = ["docs"]
87
+
88
+ [tool.ruff.lint]
89
+ select = ["E", "F", "W", "I", "N", "UP", "B", "C4", "SIM", "PTH", "RUF", "ANN", "ARG"]
90
+ # RUF001/2/3 exist to catch a lookalike character typed by accident where ASCII was meant.
91
+ # The interface's Chinese text in gigaxml/gui/i18n.py uses fullwidth punctuation as
92
+ # content -- that is what written Chinese is punctuated with -- so the family is declared
93
+ # intentional rather than silenced with a noqa on every line of the table. Anything outside
94
+ # this list still trips the rule, which is the part of it worth keeping.
95
+ allowed-confusables = [",", "。", ":", ";", "!", "?", "、", "(", ")"]
96
+
97
+ [tool.ruff.lint.isort]
98
+ known-first-party = ["gigaxml"]
99
+
100
+ [tool.ruff.format]
101
+ quote-style = "double"
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,5 @@
1
+ """gigaxml — memory-efficient XML extraction toolkit."""
2
+
3
+ __version__ = "0.9.0"
4
+
5
+ __all__ = ["__version__"]