gigaxml 0.9.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gigaxml-0.9.0/LICENSE +21 -0
- gigaxml-0.9.0/PKG-INFO +241 -0
- gigaxml-0.9.0/README.md +201 -0
- gigaxml-0.9.0/pyproject.toml +101 -0
- gigaxml-0.9.0/setup.cfg +4 -0
- gigaxml-0.9.0/src/gigaxml/__init__.py +5 -0
- gigaxml-0.9.0/src/gigaxml/checkpoint.py +417 -0
- gigaxml-0.9.0/src/gigaxml/cli.py +1067 -0
- gigaxml-0.9.0/src/gigaxml/config.py +328 -0
- gigaxml-0.9.0/src/gigaxml/errors.py +150 -0
- gigaxml-0.9.0/src/gigaxml/fields.py +464 -0
- gigaxml-0.9.0/src/gigaxml/generate.py +566 -0
- gigaxml-0.9.0/src/gigaxml/gui/__init__.py +7 -0
- gigaxml-0.9.0/src/gigaxml/gui/app.py +123 -0
- gigaxml-0.9.0/src/gigaxml/gui/batch_queue.py +184 -0
- gigaxml-0.9.0/src/gigaxml/gui/cli_process.py +281 -0
- gigaxml-0.9.0/src/gigaxml/gui/document_info.py +134 -0
- gigaxml-0.9.0/src/gigaxml/gui/error_advice.py +131 -0
- gigaxml-0.9.0/src/gigaxml/gui/field_rows.py +294 -0
- gigaxml-0.9.0/src/gigaxml/gui/i18n.py +439 -0
- gigaxml-0.9.0/src/gigaxml/gui/inspect_report.py +269 -0
- gigaxml-0.9.0/src/gigaxml/gui/main_window.py +469 -0
- gigaxml-0.9.0/src/gigaxml/gui/panels/__init__.py +5 -0
- gigaxml-0.9.0/src/gigaxml/gui/panels/batch.py +311 -0
- gigaxml-0.9.0/src/gigaxml/gui/panels/document.py +308 -0
- gigaxml-0.9.0/src/gigaxml/gui/panels/errors.py +212 -0
- gigaxml-0.9.0/src/gigaxml/gui/panels/execution.py +919 -0
- gigaxml-0.9.0/src/gigaxml/gui/panels/fields.py +674 -0
- gigaxml-0.9.0/src/gigaxml/gui/panels/preview.py +430 -0
- gigaxml-0.9.0/src/gigaxml/gui/panels/results.py +297 -0
- gigaxml-0.9.0/src/gigaxml/gui/panels/settings.py +189 -0
- gigaxml-0.9.0/src/gigaxml/gui/panels/structure.py +828 -0
- gigaxml-0.9.0/src/gigaxml/gui/progress.py +130 -0
- gigaxml-0.9.0/src/gigaxml/gui/recent_files.py +160 -0
- gigaxml-0.9.0/src/gigaxml/gui/run_report.py +285 -0
- gigaxml-0.9.0/src/gigaxml/gui/sampling.py +260 -0
- gigaxml-0.9.0/src/gigaxml/gui/saved_configs.py +145 -0
- gigaxml-0.9.0/src/gigaxml/gui/settings.py +184 -0
- gigaxml-0.9.0/src/gigaxml/inspect.py +1176 -0
- gigaxml-0.9.0/src/gigaxml/parser/__init__.py +25 -0
- gigaxml-0.9.0/src/gigaxml/parser/streaming.py +436 -0
- gigaxml-0.9.0/src/gigaxml/paths.py +153 -0
- gigaxml-0.9.0/src/gigaxml/run.py +496 -0
- gigaxml-0.9.0/src/gigaxml/sample.py +166 -0
- gigaxml-0.9.0/src/gigaxml/writers.py +656 -0
- gigaxml-0.9.0/src/gigaxml.egg-info/PKG-INFO +241 -0
- gigaxml-0.9.0/src/gigaxml.egg-info/SOURCES.txt +49 -0
- gigaxml-0.9.0/src/gigaxml.egg-info/dependency_links.txt +1 -0
- gigaxml-0.9.0/src/gigaxml.egg-info/entry_points.txt +3 -0
- gigaxml-0.9.0/src/gigaxml.egg-info/requires.txt +22 -0
- gigaxml-0.9.0/src/gigaxml.egg-info/top_level.txt +1 -0
gigaxml-0.9.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 gg320324492-lgtm
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
gigaxml-0.9.0/PKG-INFO
ADDED
|
@@ -0,0 +1,241 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: gigaxml
|
|
3
|
+
Version: 0.9.0
|
|
4
|
+
Summary: A production-oriented CLI toolkit for profiling, validating and extracting structured data from multi-gigabyte XML files with bounded memory usage.
|
|
5
|
+
License-Expression: MIT
|
|
6
|
+
Project-URL: Homepage, https://github.com/gg320324492-lgtm/GigaXML-Memory-Efficient-XML-Extraction-Toolkit
|
|
7
|
+
Keywords: xml,streaming,etl,iterparse,large-files,cli,memory-efficient
|
|
8
|
+
Classifier: Development Status :: 3 - Alpha
|
|
9
|
+
Classifier: Environment :: Console
|
|
10
|
+
Classifier: Intended Audience :: Developers
|
|
11
|
+
Classifier: Operating System :: OS Independent
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
16
|
+
Classifier: Topic :: Text Processing :: Markup :: XML
|
|
17
|
+
Classifier: Typing :: Typed
|
|
18
|
+
Requires-Python: <3.14,>=3.11
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
License-File: LICENSE
|
|
21
|
+
Requires-Dist: lxml>=5.0
|
|
22
|
+
Requires-Dist: pyyaml>=6.0
|
|
23
|
+
Provides-Extra: parquet
|
|
24
|
+
Requires-Dist: pyarrow>=14.0; extra == "parquet"
|
|
25
|
+
Provides-Extra: pretty
|
|
26
|
+
Requires-Dist: rich>=13.0; extra == "pretty"
|
|
27
|
+
Provides-Extra: gui
|
|
28
|
+
Requires-Dist: PySide6>=6.6; extra == "gui"
|
|
29
|
+
Requires-Dist: pytest-qt>=4.4; extra == "gui"
|
|
30
|
+
Provides-Extra: dev
|
|
31
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
32
|
+
Requires-Dist: pytest-cov>=5.0; extra == "dev"
|
|
33
|
+
Requires-Dist: ruff>=0.6; extra == "dev"
|
|
34
|
+
Requires-Dist: psutil>=5.9; extra == "dev"
|
|
35
|
+
Requires-Dist: pyarrow>=14.0; extra == "dev"
|
|
36
|
+
Requires-Dist: rich>=13.0; extra == "dev"
|
|
37
|
+
Requires-Dist: pyinstaller>=6.10; extra == "dev"
|
|
38
|
+
Requires-Dist: pillow>=10.0; extra == "dev"
|
|
39
|
+
Dynamic: license-file
|
|
40
|
+
|
|
41
|
+
# GigaXML
|
|
42
|
+
|
|
43
|
+
A production-oriented CLI toolkit for profiling, validating and extracting structured data
|
|
44
|
+
from multi-gigabyte XML files with bounded memory usage.
|
|
45
|
+
|
|
46
|
+
## What this is
|
|
47
|
+
|
|
48
|
+
The point of this project is not "it can parse XML" — plenty of tools can. The point is
|
|
49
|
+
**constant, bounded peak memory while extracting from 4 GB / 10 GB files**, and being able
|
|
50
|
+
to prove it with a reproducible measurement harness.
|
|
51
|
+
|
|
52
|
+
Measured on this machine, on a generated 4.05 GiB file holding 11,915,264 records:
|
|
53
|
+
|
|
54
|
+
| | |
|
|
55
|
+
|---|---|
|
|
56
|
+
| **Input** | 4142.72 MiB, 11,915,264 records |
|
|
57
|
+
| **Time** | 284.67 s (14.6 MiB/s, 41,856 records/s) |
|
|
58
|
+
| **Peak RSS** | 33.703 MiB |
|
|
59
|
+
| **Increase over the post-import baseline** | **5.059 MiB** |
|
|
60
|
+
|
|
61
|
+
The same run at 1 GiB (2,978,816 records) added **5.105 MiB** — four times the input and
|
|
62
|
+
the increase did not move. Something that accumulated per record would make the four
|
|
63
|
+
gigabyte figure four times the one gigabyte figure; it is flat to within a percent. Every
|
|
64
|
+
number here comes from a script in this repository; see [Benchmarks](#benchmarks).
|
|
65
|
+
|
|
66
|
+
The configuration used above reads six fields, including a nested path and a type
|
|
67
|
+
conversion — the shapes a real config uses:
|
|
68
|
+
|
|
69
|
+
```yaml
|
|
70
|
+
record: /catalog/products/product
|
|
71
|
+
fields:
|
|
72
|
+
id: {path: '@id'}
|
|
73
|
+
type: {path: '@type'}
|
|
74
|
+
name: {path: name}
|
|
75
|
+
category: {path: category}
|
|
76
|
+
price: {path: price, type: float}
|
|
77
|
+
manufacturer: {path: manufacturer/name}
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
## Install
|
|
81
|
+
|
|
82
|
+
From PyPI:
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
pip install gigaxml # the command-line toolkit
|
|
86
|
+
pip install "gigaxml[gui]" # and the desktop application (pulls PySide6: 640 MiB installed, measured on Windows)
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
Parquet output needs the `parquet` extra (`pip install "gigaxml[parquet]"`); CSV and JSONL
|
|
90
|
+
do not. The desktop application is also packaged per platform — an unsigned Windows
|
|
91
|
+
build, an Apple Silicon `.dmg` and an x86_64 AppImage — under
|
|
92
|
+
[Releases](https://github.com/gg320324492-lgtm/GigaXML-Memory-Efficient-XML-Extraction-Toolkit/releases);
|
|
93
|
+
its release notes say what each build runs on and what the unsigned warnings mean.
|
|
94
|
+
|
|
95
|
+
From source, if you would rather:
|
|
96
|
+
|
|
97
|
+
```bash
|
|
98
|
+
git clone https://github.com/gg320324492-lgtm/GigaXML-Memory-Efficient-XML-Extraction-Toolkit.git
|
|
99
|
+
cd GigaXML-Memory-Efficient-XML-Extraction-Toolkit
|
|
100
|
+
python -m venv .venv
|
|
101
|
+
.venv/Scripts/python -m pip install -e ".[dev]" # POSIX: .venv/bin/python
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
`lxml` and `pyyaml` are the only required dependencies.
|
|
105
|
+
|
|
106
|
+
## Thirty seconds
|
|
107
|
+
|
|
108
|
+
Point it at a document you know nothing about, let it propose a config, then run it:
|
|
109
|
+
|
|
110
|
+
```bash
|
|
111
|
+
# 1. What is in this file?
|
|
112
|
+
gigaxml inspect big.xml
|
|
113
|
+
|
|
114
|
+
# 2. Write a starting-point config for the highest-ranked record candidate
|
|
115
|
+
gigaxml inspect big.xml --generate-config config.yaml --infer-types
|
|
116
|
+
|
|
117
|
+
# 3. Extract
|
|
118
|
+
gigaxml extract big.xml -c config.yaml -o out.csv
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
`inspect` does not read the whole document — it reports the structure and the repeating
|
|
122
|
+
paths it found, and ranks them. The config it writes is explicitly a **starting point**,
|
|
123
|
+
not a conclusion; read the comments in it.
|
|
124
|
+
|
|
125
|
+
For a quick look at the data before committing to a full run:
|
|
126
|
+
|
|
127
|
+
```bash
|
|
128
|
+
gigaxml sample big.xml -c config.yaml -n 20 -o first20.jsonl
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
### What it looks like
|
|
132
|
+
|
|
133
|
+

|
|
134
|
+
|
|
135
|
+
*Finding the records in a document that opens with a licence comment.*
|
|
136
|
+
|
|
137
|
+

|
|
138
|
+
|
|
139
|
+
*Extracting 11.9 million records from 4.05 GiB.*
|
|
140
|
+
|
|
141
|
+
> **Both of these are animations rendered from the tools' real output, not screen
|
|
142
|
+
> recordings.** The text is what the tools actually printed and the timings are the
|
|
143
|
+
> measured ones, but the frames are drawn rather than captured — this machine's sandbox
|
|
144
|
+
> does not permit screen capture. Each frame carries the same note.
|
|
145
|
+
|
|
146
|
+
## Benchmarks
|
|
147
|
+
|
|
148
|
+
Three generated datasets, two configs, measured in a subprocess with `psutil`:
|
|
149
|
+
|
|
150
|
+
```
|
|
151
|
+
dataset fields input MiB records s MiB/s peak delta
|
|
152
|
+
------------------------------------------------------------------------------
|
|
153
|
+
100MB 6 fields 100.57 290,900 6.92 14.5 33.281 5.113
|
|
154
|
+
100MB 1 field 100.57 290,900 2.69 37.4 31.203 2.617
|
|
155
|
+
1GB 6 fields 1033.65 2,978,816 70.66 14.6 33.676 5.105
|
|
156
|
+
1GB 1 field 1033.65 2,978,816 27.62 37.4 31.496 2.922
|
|
157
|
+
4GB 6 fields 4142.72 11,915,264 284.67 14.6 33.703 5.059
|
|
158
|
+
4GB 1 field 4142.72 11,915,264 112.04 37.0 31.500 2.973
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
`peak` and `delta` are MiB; `delta` is against the same process's post-import baseline.
|
|
162
|
+
`peak` is `PeakWorkingSetSize` — the maximum over the process's life, not the current RSS
|
|
163
|
+
at the end, which reads 10–14% lower.
|
|
164
|
+
The one-field rows are a control, not the headline: a single-field config is the easiest
|
|
165
|
+
member of this family to run, and quoting it alone would overstate what a real config
|
|
166
|
+
costs. Field count costs about **2.5×** in throughput.
|
|
167
|
+
|
|
168
|
+
To reproduce, generate the datasets and run the harness in [`benchmarks/`](benchmarks):
|
|
169
|
+
|
|
170
|
+
```bash
|
|
171
|
+
gigaxml generate --size 100MB -o data/b100m.xml
|
|
172
|
+
gigaxml generate --size 1GB -o data/b1g.xml
|
|
173
|
+
gigaxml generate --size 4GB -o data/b4g.xml
|
|
174
|
+
python benchmarks/bench_extraction.py
|
|
175
|
+
```
|
|
176
|
+
|
|
177
|
+
`--size` is approximate: `--size 1GB` produces 1033.65 MiB, not 1024, and the sizes above
|
|
178
|
+
are the measured ones. Peak and delta are both reported because either alone can be
|
|
179
|
+
misread — peak includes about 32 MiB of interpreter and library overhead, delta is what
|
|
180
|
+
the workload is responsible for, and both baselines in this repository are taken after
|
|
181
|
+
every import so that two deltas are comparable.
|
|
182
|
+
|
|
183
|
+
## Non-goals
|
|
184
|
+
|
|
185
|
+
- No full XPath 3.1 — XPath is evaluated only inside a single record subtree.
|
|
186
|
+
- No arbitrary byte-offset seek/resume — XML byte offsets are not a safe parse boundary.
|
|
187
|
+
- No AI/ML structure inference — confidence values are deterministic statistics.
|
|
188
|
+
- No real customer data — everything runs on synthetic, reproducible datasets.
|
|
189
|
+
- No fabricated benchmarks — every performance claim comes from a runnable script.
|
|
190
|
+
|
|
191
|
+
## Known limitations
|
|
192
|
+
|
|
193
|
+
- **`--resume` re-parses and skips; it does not seek.** XML cannot be re-entered
|
|
194
|
+
mid-stream, so continuing a run means reading from the beginning and discarding the
|
|
195
|
+
records already accounted for. On a 403 MB file that costs 8.7 s against 17.1 s to
|
|
196
|
+
extract, so resuming saves roughly half of what you had already done. `--help` says so
|
|
197
|
+
too.
|
|
198
|
+
- **`inspect` is slower than `extract`** — 17.3 MiB/s against 38.2 MiB/s on the same
|
|
199
|
+
1 GB file. It maintains several parallel bookkeeping stacks per element. It is also the
|
|
200
|
+
command you run once on a document, not in a loop.
|
|
201
|
+
- **An inferred config treats containers as leaves.** `--generate-config` proposes direct
|
|
202
|
+
children and attributes; a field whose element has children of its own is read as
|
|
203
|
+
concatenated text, so `<tags><tag>a</tag><tag>b</tag></tags>` becomes `ab`. Nested
|
|
204
|
+
paths (`manufacturer/name`) have to be written by hand, as the generated comments say.
|
|
205
|
+
- **Types are inferred from a sample**, and `decimal` is never inferred. If a field is
|
|
206
|
+
money, set `type: decimal` yourself — `float` cannot represent 49.90 exactly.
|
|
207
|
+
- **Parsing limits are not configurable.** Entities are never expanded, the network is
|
|
208
|
+
never touched, and no DTD is loaded; documents nested deeper than 256 levels, carrying a
|
|
209
|
+
single text node over about 10 MB, or amplified by entities are refused rather than
|
|
210
|
+
partially read. These are deliberate and there are no flags to turn them off.
|
|
211
|
+
- **`--checkpoint-every` verifies the parts on disk before resuming**, which costs one
|
|
212
|
+
pass over the output at about **200 MiB/s** — about 10 ms for 2 MiB of CSV, negligible for
|
|
213
|
+
Parquet, whose row counts come from file metadata. It grows with the size of the
|
|
214
|
+
output, not the input. Measured by `benchmarks/bench_resident.py`.
|
|
215
|
+
- **A resume is dominated by starting the process, not by checking the output.** Against
|
|
216
|
+
an already-complete manifest on a 403 MiB source, the command takes about 770 ms: some
|
|
217
|
+
400 ms of that is interpreter startup and imports, and most of the rest is hashing the
|
|
218
|
+
source to confirm it has not changed. The part check itself is about 10 ms.
|
|
219
|
+
- **One field value that is itself gigabytes is held in memory.** Records stream, but
|
|
220
|
+
there is no streaming mode for a single value, because there is nothing to stream it
|
|
221
|
+
into.
|
|
222
|
+
- **`sample` reads what it samples into memory.** It is meant for looking at a file, not
|
|
223
|
+
for measuring one.
|
|
224
|
+
- **Input is a local file, never a URL.** A `.xml.gz` file is fine; a document that lives
|
|
225
|
+
behind HTTP is out of scope.
|
|
226
|
+
|
|
227
|
+
## Development
|
|
228
|
+
|
|
229
|
+
```bash
|
|
230
|
+
pytest -q # unit + integration
|
|
231
|
+
pytest -q tests/performance # memory and throughput; not in CI
|
|
232
|
+
ruff check .
|
|
233
|
+
ruff format --check .
|
|
234
|
+
```
|
|
235
|
+
|
|
236
|
+
Performance tests are excluded from CI: they measure memory and throughput, take minutes,
|
|
237
|
+
and are not a pass/fail signal.
|
|
238
|
+
|
|
239
|
+
## License
|
|
240
|
+
|
|
241
|
+
MIT — see [LICENSE](LICENSE).
|
gigaxml-0.9.0/README.md
ADDED
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
# GigaXML
|
|
2
|
+
|
|
3
|
+
A production-oriented CLI toolkit for profiling, validating and extracting structured data
|
|
4
|
+
from multi-gigabyte XML files with bounded memory usage.
|
|
5
|
+
|
|
6
|
+
## What this is
|
|
7
|
+
|
|
8
|
+
The point of this project is not "it can parse XML" — plenty of tools can. The point is
|
|
9
|
+
**constant, bounded peak memory while extracting from 4 GB / 10 GB files**, and being able
|
|
10
|
+
to prove it with a reproducible measurement harness.
|
|
11
|
+
|
|
12
|
+
Measured on this machine, on a generated 4.05 GiB file holding 11,915,264 records:
|
|
13
|
+
|
|
14
|
+
| | |
|
|
15
|
+
|---|---|
|
|
16
|
+
| **Input** | 4142.72 MiB, 11,915,264 records |
|
|
17
|
+
| **Time** | 284.67 s (14.6 MiB/s, 41,856 records/s) |
|
|
18
|
+
| **Peak RSS** | 33.703 MiB |
|
|
19
|
+
| **Increase over the post-import baseline** | **5.059 MiB** |
|
|
20
|
+
|
|
21
|
+
The same run at 1 GiB (2,978,816 records) added **5.105 MiB** — four times the input and
|
|
22
|
+
the increase did not move. Something that accumulated per record would make the four
|
|
23
|
+
gigabyte figure four times the one gigabyte figure; it is flat to within a percent. Every
|
|
24
|
+
number here comes from a script in this repository; see [Benchmarks](#benchmarks).
|
|
25
|
+
|
|
26
|
+
The configuration used above reads six fields, including a nested path and a type
|
|
27
|
+
conversion — the shapes a real config uses:
|
|
28
|
+
|
|
29
|
+
```yaml
|
|
30
|
+
record: /catalog/products/product
|
|
31
|
+
fields:
|
|
32
|
+
id: {path: '@id'}
|
|
33
|
+
type: {path: '@type'}
|
|
34
|
+
name: {path: name}
|
|
35
|
+
category: {path: category}
|
|
36
|
+
price: {path: price, type: float}
|
|
37
|
+
manufacturer: {path: manufacturer/name}
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
## Install
|
|
41
|
+
|
|
42
|
+
From PyPI:
|
|
43
|
+
|
|
44
|
+
```bash
|
|
45
|
+
pip install gigaxml # the command-line toolkit
|
|
46
|
+
pip install "gigaxml[gui]" # and the desktop application (pulls PySide6: 640 MiB installed, measured on Windows)
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
Parquet output needs the `parquet` extra (`pip install "gigaxml[parquet]"`); CSV and JSONL
|
|
50
|
+
do not. The desktop application is also packaged per platform — an unsigned Windows
|
|
51
|
+
build, an Apple Silicon `.dmg` and an x86_64 AppImage — under
|
|
52
|
+
[Releases](https://github.com/gg320324492-lgtm/GigaXML-Memory-Efficient-XML-Extraction-Toolkit/releases);
|
|
53
|
+
its release notes say what each build runs on and what the unsigned warnings mean.
|
|
54
|
+
|
|
55
|
+
From source, if you would rather:
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
git clone https://github.com/gg320324492-lgtm/GigaXML-Memory-Efficient-XML-Extraction-Toolkit.git
|
|
59
|
+
cd GigaXML-Memory-Efficient-XML-Extraction-Toolkit
|
|
60
|
+
python -m venv .venv
|
|
61
|
+
.venv/Scripts/python -m pip install -e ".[dev]" # POSIX: .venv/bin/python
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
`lxml` and `pyyaml` are the only required dependencies.
|
|
65
|
+
|
|
66
|
+
## Thirty seconds
|
|
67
|
+
|
|
68
|
+
Point it at a document you know nothing about, let it propose a config, then run it:
|
|
69
|
+
|
|
70
|
+
```bash
|
|
71
|
+
# 1. What is in this file?
|
|
72
|
+
gigaxml inspect big.xml
|
|
73
|
+
|
|
74
|
+
# 2. Write a starting-point config for the highest-ranked record candidate
|
|
75
|
+
gigaxml inspect big.xml --generate-config config.yaml --infer-types
|
|
76
|
+
|
|
77
|
+
# 3. Extract
|
|
78
|
+
gigaxml extract big.xml -c config.yaml -o out.csv
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
`inspect` does not read the whole document — it reports the structure and the repeating
|
|
82
|
+
paths it found, and ranks them. The config it writes is explicitly a **starting point**,
|
|
83
|
+
not a conclusion; read the comments in it.
|
|
84
|
+
|
|
85
|
+
For a quick look at the data before committing to a full run:
|
|
86
|
+
|
|
87
|
+
```bash
|
|
88
|
+
gigaxml sample big.xml -c config.yaml -n 20 -o first20.jsonl
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
### What it looks like
|
|
92
|
+
|
|
93
|
+

|
|
94
|
+
|
|
95
|
+
*Finding the records in a document that opens with a licence comment.*
|
|
96
|
+
|
|
97
|
+

|
|
98
|
+
|
|
99
|
+
*Extracting 11.9 million records from 4.05 GiB.*
|
|
100
|
+
|
|
101
|
+
> **Both of these are animations rendered from the tools' real output, not screen
|
|
102
|
+
> recordings.** The text is what the tools actually printed and the timings are the
|
|
103
|
+
> measured ones, but the frames are drawn rather than captured — this machine's sandbox
|
|
104
|
+
> does not permit screen capture. Each frame carries the same note.
|
|
105
|
+
|
|
106
|
+
## Benchmarks
|
|
107
|
+
|
|
108
|
+
Three generated datasets, two configs, measured in a subprocess with `psutil`:
|
|
109
|
+
|
|
110
|
+
```
|
|
111
|
+
dataset fields input MiB records s MiB/s peak delta
|
|
112
|
+
------------------------------------------------------------------------------
|
|
113
|
+
100MB 6 fields 100.57 290,900 6.92 14.5 33.281 5.113
|
|
114
|
+
100MB 1 field 100.57 290,900 2.69 37.4 31.203 2.617
|
|
115
|
+
1GB 6 fields 1033.65 2,978,816 70.66 14.6 33.676 5.105
|
|
116
|
+
1GB 1 field 1033.65 2,978,816 27.62 37.4 31.496 2.922
|
|
117
|
+
4GB 6 fields 4142.72 11,915,264 284.67 14.6 33.703 5.059
|
|
118
|
+
4GB 1 field 4142.72 11,915,264 112.04 37.0 31.500 2.973
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
`peak` and `delta` are MiB; `delta` is against the same process's post-import baseline.
|
|
122
|
+
`peak` is `PeakWorkingSetSize` — the maximum over the process's life, not the current RSS
|
|
123
|
+
at the end, which reads 10–14% lower.
|
|
124
|
+
The one-field rows are a control, not the headline: a single-field config is the easiest
|
|
125
|
+
member of this family to run, and quoting it alone would overstate what a real config
|
|
126
|
+
costs. Field count costs about **2.5×** in throughput.
|
|
127
|
+
|
|
128
|
+
To reproduce, generate the datasets and run the harness in [`benchmarks/`](benchmarks):
|
|
129
|
+
|
|
130
|
+
```bash
|
|
131
|
+
gigaxml generate --size 100MB -o data/b100m.xml
|
|
132
|
+
gigaxml generate --size 1GB -o data/b1g.xml
|
|
133
|
+
gigaxml generate --size 4GB -o data/b4g.xml
|
|
134
|
+
python benchmarks/bench_extraction.py
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
`--size` is approximate: `--size 1GB` produces 1033.65 MiB, not 1024, and the sizes above
|
|
138
|
+
are the measured ones. Peak and delta are both reported because either alone can be
|
|
139
|
+
misread — peak includes about 32 MiB of interpreter and library overhead, delta is what
|
|
140
|
+
the workload is responsible for, and both baselines in this repository are taken after
|
|
141
|
+
every import so that two deltas are comparable.
|
|
142
|
+
|
|
143
|
+
## Non-goals
|
|
144
|
+
|
|
145
|
+
- No full XPath 3.1 — XPath is evaluated only inside a single record subtree.
|
|
146
|
+
- No arbitrary byte-offset seek/resume — XML byte offsets are not a safe parse boundary.
|
|
147
|
+
- No AI/ML structure inference — confidence values are deterministic statistics.
|
|
148
|
+
- No real customer data — everything runs on synthetic, reproducible datasets.
|
|
149
|
+
- No fabricated benchmarks — every performance claim comes from a runnable script.
|
|
150
|
+
|
|
151
|
+
## Known limitations
|
|
152
|
+
|
|
153
|
+
- **`--resume` re-parses and skips; it does not seek.** XML cannot be re-entered
|
|
154
|
+
mid-stream, so continuing a run means reading from the beginning and discarding the
|
|
155
|
+
records already accounted for. On a 403 MB file that costs 8.7 s against 17.1 s to
|
|
156
|
+
extract, so resuming saves roughly half of what you had already done. `--help` says so
|
|
157
|
+
too.
|
|
158
|
+
- **`inspect` is slower than `extract`** — 17.3 MiB/s against 38.2 MiB/s on the same
|
|
159
|
+
1 GB file. It maintains several parallel bookkeeping stacks per element. It is also the
|
|
160
|
+
command you run once on a document, not in a loop.
|
|
161
|
+
- **An inferred config treats containers as leaves.** `--generate-config` proposes direct
|
|
162
|
+
children and attributes; a field whose element has children of its own is read as
|
|
163
|
+
concatenated text, so `<tags><tag>a</tag><tag>b</tag></tags>` becomes `ab`. Nested
|
|
164
|
+
paths (`manufacturer/name`) have to be written by hand, as the generated comments say.
|
|
165
|
+
- **Types are inferred from a sample**, and `decimal` is never inferred. If a field is
|
|
166
|
+
money, set `type: decimal` yourself — `float` cannot represent 49.90 exactly.
|
|
167
|
+
- **Parsing limits are not configurable.** Entities are never expanded, the network is
|
|
168
|
+
never touched, and no DTD is loaded; documents nested deeper than 256 levels, carrying a
|
|
169
|
+
single text node over about 10 MB, or amplified by entities are refused rather than
|
|
170
|
+
partially read. These are deliberate and there are no flags to turn them off.
|
|
171
|
+
- **`--checkpoint-every` verifies the parts on disk before resuming**, which costs one
|
|
172
|
+
pass over the output at about **200 MiB/s** — about 10 ms for 2 MiB of CSV, negligible for
|
|
173
|
+
Parquet, whose row counts come from file metadata. It grows with the size of the
|
|
174
|
+
output, not the input. Measured by `benchmarks/bench_resident.py`.
|
|
175
|
+
- **A resume is dominated by starting the process, not by checking the output.** Against
|
|
176
|
+
an already-complete manifest on a 403 MiB source, the command takes about 770 ms: some
|
|
177
|
+
400 ms of that is interpreter startup and imports, and most of the rest is hashing the
|
|
178
|
+
source to confirm it has not changed. The part check itself is about 10 ms.
|
|
179
|
+
- **One field value that is itself gigabytes is held in memory.** Records stream, but
|
|
180
|
+
there is no streaming mode for a single value, because there is nothing to stream it
|
|
181
|
+
into.
|
|
182
|
+
- **`sample` reads what it samples into memory.** It is meant for looking at a file, not
|
|
183
|
+
for measuring one.
|
|
184
|
+
- **Input is a local file, never a URL.** A `.xml.gz` file is fine; a document that lives
|
|
185
|
+
behind HTTP is out of scope.
|
|
186
|
+
|
|
187
|
+
## Development
|
|
188
|
+
|
|
189
|
+
```bash
|
|
190
|
+
pytest -q # unit + integration
|
|
191
|
+
pytest -q tests/performance # memory and throughput; not in CI
|
|
192
|
+
ruff check .
|
|
193
|
+
ruff format --check .
|
|
194
|
+
```
|
|
195
|
+
|
|
196
|
+
Performance tests are excluded from CI: they measure memory and throughput, take minutes,
|
|
197
|
+
and are not a pass/fail signal.
|
|
198
|
+
|
|
199
|
+
## License
|
|
200
|
+
|
|
201
|
+
MIT — see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "gigaxml"
|
|
7
|
+
version = "0.9.0"
|
|
8
|
+
description = "A production-oriented CLI toolkit for profiling, validating and extracting structured data from multi-gigabyte XML files with bounded memory usage."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.11,<3.14"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
keywords = ["xml", "streaming", "etl", "iterparse", "large-files", "cli", "memory-efficient"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 3 - Alpha",
|
|
16
|
+
"Environment :: Console",
|
|
17
|
+
"Intended Audience :: Developers",
|
|
18
|
+
"Operating System :: OS Independent",
|
|
19
|
+
"Programming Language :: Python :: 3",
|
|
20
|
+
"Programming Language :: Python :: 3.11",
|
|
21
|
+
"Programming Language :: Python :: 3.12",
|
|
22
|
+
"Programming Language :: Python :: 3.13",
|
|
23
|
+
"Topic :: Text Processing :: Markup :: XML",
|
|
24
|
+
"Typing :: Typed",
|
|
25
|
+
]
|
|
26
|
+
dependencies = [
|
|
27
|
+
"lxml>=5.0",
|
|
28
|
+
"pyyaml>=6.0",
|
|
29
|
+
]
|
|
30
|
+
|
|
31
|
+
[project.optional-dependencies]
|
|
32
|
+
parquet = ["pyarrow>=14.0"]
|
|
33
|
+
pretty = ["rich>=13.0"]
|
|
34
|
+
# The desktop application. Optional on purpose: the CLI is the product, and Qt is a
|
|
35
|
+
# 150-200 MB dependency that only a user of the window needs. Nothing in gigaxml outside
|
|
36
|
+
# gigaxml.gui may import it.
|
|
37
|
+
#
|
|
38
|
+
# pytest-qt belongs here rather than in dev, and that is not tidiness. It refuses to load
|
|
39
|
+
# without a Qt binding, and it is loaded as a plugin -- so with it in dev, a plain
|
|
40
|
+
# `pip install -e ".[dev]"` gives a pytest that dies before collecting anything:
|
|
41
|
+
# ERROR: pytest-qt requires either PySide6, PyQt5 or PyQt6 installed.
|
|
42
|
+
# Keeping it with the binding means `.[dev]` runs the suite and skips the desktop tests,
|
|
43
|
+
# while `.[dev,gui]` runs them. Both configurations work, which is the point.
|
|
44
|
+
gui = ["PySide6>=6.6", "pytest-qt>=4.4"]
|
|
45
|
+
dev = [
|
|
46
|
+
"pytest>=8.0",
|
|
47
|
+
"pytest-cov>=5.0",
|
|
48
|
+
"ruff>=0.6",
|
|
49
|
+
"psutil>=5.9",
|
|
50
|
+
"pyarrow>=14.0",
|
|
51
|
+
"rich>=13.0",
|
|
52
|
+
"pyinstaller>=6.10",
|
|
53
|
+
# Drawn by tools/make_icon, which the packaging pipeline runs on every platform. A
|
|
54
|
+
# dependency that only a script's author happened to have installed is a build that only
|
|
55
|
+
# that author's machine can run -- and the icon step is the first thing the pipeline
|
|
56
|
+
# calls, so its absence fails the build before anything is compiled. Build-time only,
|
|
57
|
+
# never a runtime one: nothing in `src/` imports it.
|
|
58
|
+
"pillow>=10.0",
|
|
59
|
+
]
|
|
60
|
+
|
|
61
|
+
[project.scripts]
|
|
62
|
+
gigaxml = "gigaxml.cli:main"
|
|
63
|
+
gigaxml-gui = "gigaxml.gui.app:main"
|
|
64
|
+
|
|
65
|
+
[project.urls]
|
|
66
|
+
Homepage = "https://github.com/gg320324492-lgtm/GigaXML-Memory-Efficient-XML-Extraction-Toolkit"
|
|
67
|
+
|
|
68
|
+
[tool.setuptools.packages.find]
|
|
69
|
+
where = ["src"]
|
|
70
|
+
|
|
71
|
+
[tool.pytest.ini_options]
|
|
72
|
+
testpaths = ["tests"]
|
|
73
|
+
pythonpath = ["."]
|
|
74
|
+
addopts = "-ra"
|
|
75
|
+
markers = [
|
|
76
|
+
"performance: memory / throughput measurement (excluded from CI, see ROADMAP A11)",
|
|
77
|
+
]
|
|
78
|
+
|
|
79
|
+
[tool.ruff]
|
|
80
|
+
line-length = 100
|
|
81
|
+
target-version = "py311"
|
|
82
|
+
src = ["src", "tests"]
|
|
83
|
+
# `docs/` holds the human-authored spec documents. Ruff >= 0.16 reformats fenced
|
|
84
|
+
# Python code blocks inside Markdown, which would rewrite the spec text itself.
|
|
85
|
+
# The spec is an input artifact, not source code, so it is excluded from linting.
|
|
86
|
+
extend-exclude = ["docs"]
|
|
87
|
+
|
|
88
|
+
[tool.ruff.lint]
|
|
89
|
+
select = ["E", "F", "W", "I", "N", "UP", "B", "C4", "SIM", "PTH", "RUF", "ANN", "ARG"]
|
|
90
|
+
# RUF001/2/3 exist to catch a lookalike character typed by accident where ASCII was meant.
|
|
91
|
+
# The interface's Chinese text in gigaxml/gui/i18n.py uses fullwidth punctuation as
|
|
92
|
+
# content -- that is what written Chinese is punctuated with -- so the family is declared
|
|
93
|
+
# intentional rather than silenced with a noqa on every line of the table. Anything outside
|
|
94
|
+
# this list still trips the rule, which is the part of it worth keeping.
|
|
95
|
+
allowed-confusables = [",", "。", ":", ";", "!", "?", "、", "(", ")"]
|
|
96
|
+
|
|
97
|
+
[tool.ruff.lint.isort]
|
|
98
|
+
known-first-party = ["gigaxml"]
|
|
99
|
+
|
|
100
|
+
[tool.ruff.format]
|
|
101
|
+
quote-style = "double"
|
gigaxml-0.9.0/setup.cfg
ADDED