vortex-rdflib 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- vortex_rdflib-0.1.0/LICENSE +21 -0
- vortex_rdflib-0.1.0/PKG-INFO +299 -0
- vortex_rdflib-0.1.0/README.md +269 -0
- vortex_rdflib-0.1.0/pyproject.toml +85 -0
- vortex_rdflib-0.1.0/src/vortex_rdflib/__init__.py +17 -0
- vortex_rdflib-0.1.0/src/vortex_rdflib/filters.py +895 -0
- vortex_rdflib-0.1.0/src/vortex_rdflib/pushdown.py +1712 -0
- vortex_rdflib-0.1.0/src/vortex_rdflib/py.typed +0 -0
- vortex_rdflib-0.1.0/src/vortex_rdflib/store.py +572 -0
- vortex_rdflib-0.1.0/src/vortex_rdflib/terms.py +178 -0
- vortex_rdflib-0.1.0/tests/conftest.py +65 -0
- vortex_rdflib-0.1.0/tests/test_bench_dataset.py +95 -0
- vortex_rdflib-0.1.0/tests/test_filters.py +352 -0
- vortex_rdflib-0.1.0/tests/test_pushdown.py +1357 -0
- vortex_rdflib-0.1.0/tests/test_store.py +272 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Vortex RDF
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,299 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: vortex-rdflib
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: rdflib Store implementation for Vortex-RDF, a columnar zero-copy RDF store
|
|
5
|
+
Keywords: rdf,rdflib,vortex,columnar,quad-store,sparql
|
|
6
|
+
Author: Julián Rojas
|
|
7
|
+
Author-email: Julián Rojas <julianandres.rojasmelendez@ugent.be>
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: Operating System :: OS Independent
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
19
|
+
Classifier: Topic :: Database
|
|
20
|
+
Classifier: Topic :: Scientific/Engineering :: Information Analysis
|
|
21
|
+
Classifier: Typing :: Typed
|
|
22
|
+
Requires-Dist: vortex-rdf>=0.10,<0.11
|
|
23
|
+
Requires-Dist: rdflib>=7,<8
|
|
24
|
+
Requires-Python: >=3.11
|
|
25
|
+
Project-URL: Homepage, https://github.com/vortex-rdf/vortex-rdflib
|
|
26
|
+
Project-URL: Issues, https://github.com/vortex-rdf/vortex-rdflib/issues
|
|
27
|
+
Project-URL: Repository, https://github.com/vortex-rdf/vortex-rdflib
|
|
28
|
+
Project-URL: Vortex-RDF, https://github.com/vortex-rdf/vortex-rdf
|
|
29
|
+
Description-Content-Type: text/markdown
|
|
30
|
+
|
|
31
|
+
# vortex-rdflib
|
|
32
|
+
|
|
33
|
+
[](https://github.com/vortex-rdf/vortex-rdflib/actions/workflows/ci.yml)
|
|
34
|
+
[](https://app.codspeed.io/vortex-rdf/vortex-rdflib?utm_source=badge)
|
|
35
|
+
[](https://pypi.org/project/vortex-rdflib/)
|
|
36
|
+
[](https://pypi.org/project/vortex-rdflib/)
|
|
37
|
+
[](LICENSE)
|
|
38
|
+
|
|
39
|
+
An [rdflib](https://rdflib.readthedocs.io/) `Store` implementation for
|
|
40
|
+
[Vortex-RDF](https://github.com/vortex-rdf/vortex-rdf), a columnar
|
|
41
|
+
zero-copy RDF serialization format — so `.vortex` files can be queried with
|
|
42
|
+
SPARQL.
|
|
43
|
+
|
|
44
|
+
The native layer is the [vortex-rdf](https://pypi.org/project/vortex-rdf/)
|
|
45
|
+
package (PyO3 bindings over the `vortex-rdf-core` Rust crate), pulled in as
|
|
46
|
+
a dependency: stores are **opened lazily from `.vortex` files** by default,
|
|
47
|
+
and queried in place without loading the dataset into memory. However, a `.vortex`
|
|
48
|
+
file can also be fully loaded in memory, with exactly the same data
|
|
49
|
+
structure and queried.
|
|
50
|
+
|
|
51
|
+
## Install
|
|
52
|
+
|
|
53
|
+
```bash
|
|
54
|
+
pip install vortex-rdflib
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
Python 3.11+. The `vortex-rdf` dependency ships prebuilt wheels for Linux
|
|
58
|
+
(x86_64, aarch64), macOS (x86_64, arm64) and Windows (x64); on other
|
|
59
|
+
platforms it builds from source.
|
|
60
|
+
|
|
61
|
+
## Usage
|
|
62
|
+
|
|
63
|
+
```python
|
|
64
|
+
from rdflib import Graph
|
|
65
|
+
from vortex_rdflib import VortexRdflibStore
|
|
66
|
+
|
|
67
|
+
graph = Graph(store=VortexRdflibStore("data.vortex"))
|
|
68
|
+
for row in graph.query("""
|
|
69
|
+
SELECT ?s ?o WHERE {
|
|
70
|
+
?s <http://xmlns.com/foaf/0.1/name> ?o
|
|
71
|
+
}
|
|
72
|
+
LIMIT 10
|
|
73
|
+
"""):
|
|
74
|
+
print(row.s, row.o)
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
SPARQL evaluation is rdflib's engine; the store serves quad patterns from
|
|
78
|
+
the Vortex file. The store is read-only so far; mutation support is on the roadmap.
|
|
79
|
+
|
|
80
|
+
A `.vortex` file holds quads, so the store is context-aware. A `Dataset` gives
|
|
81
|
+
the named graphs, and `GRAPH` works in SPARQL:
|
|
82
|
+
|
|
83
|
+
```python
|
|
84
|
+
from rdflib import Dataset
|
|
85
|
+
from vortex_rdflib import VortexRdflibStore
|
|
86
|
+
|
|
87
|
+
# default_union=True makes the SPARQL default graph the union of every graph
|
|
88
|
+
dataset = Dataset(store=VortexRdflibStore("data.vortex"), default_union=True)
|
|
89
|
+
|
|
90
|
+
for graph in dataset.graphs():
|
|
91
|
+
print(graph.identifier, len(graph))
|
|
92
|
+
|
|
93
|
+
for row in dataset.query("""
|
|
94
|
+
SELECT ?g (COUNT(*) AS ?n) WHERE {
|
|
95
|
+
GRAPH ?g { ?s ?p ?o }
|
|
96
|
+
}
|
|
97
|
+
GROUP BY ?g
|
|
98
|
+
"""):
|
|
99
|
+
print(row.g, row.n)
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
A plain `Graph(store=VortexRdflibStore(path))` — as in the example above — is
|
|
103
|
+
the view over the **whole file**, every graph included. It is a multiset
|
|
104
|
+
view — a triple in two graphs is yielded twice: the store streams the quads
|
|
105
|
+
it holds rather than building the RDF merge.
|
|
106
|
+
|
|
107
|
+
To produce a `.vortex` file from an RDF file, use the binding layer directly
|
|
108
|
+
(or the [vortex-rdf CLI](https://github.com/vortex-rdf/vortex-rdf)):
|
|
109
|
+
|
|
110
|
+
```python
|
|
111
|
+
from vortex_rdf import serialize_rdf
|
|
112
|
+
|
|
113
|
+
serialize_rdf("data.nq", "data.vortex", format="nquads", layout="dictionary")
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
`layout` accepts `"default"`, `"typed-object"` and `"dictionary"`
|
|
117
|
+
([described here](https://github.com/vortex-rdf/vortex-rdf/blob/main/docs/file-format.md#4-the-quad-table));
|
|
118
|
+
opening
|
|
119
|
+
auto-detects the layout. The `"dictionary"` layout is the fastest to query
|
|
120
|
+
from Python — it enables the SPARQL pushdowns described below.
|
|
121
|
+
|
|
122
|
+
## How it works
|
|
123
|
+
|
|
124
|
+
**Term codes instead of strings.** For Dictionary-layout stores, matched rows
|
|
125
|
+
cross the native boundary as zero-copy `u32` term-code columns
|
|
126
|
+
(`vortex_rdf.VortexRdfStore.match_codes`), and each distinct code is decoded
|
|
127
|
+
to an rdflib term once — in one GIL-released `TermDict.decode_many` call per
|
|
128
|
+
batch — and cached for the store's lifetime. Other layouts fall back to
|
|
129
|
+
N-Triples string columns, parsing each distinct term once.
|
|
130
|
+
|
|
131
|
+
**SPARQL pushdown.** Constructing a `VortexRdflibStore` registers an rdflib
|
|
132
|
+
`CUSTOM_EVALS` hook that answers the algebra operators it understands over
|
|
133
|
+
vortex term codes instead of leaving them to rdflib's per-row evaluation: basic
|
|
134
|
+
graph patterns, `FILTER`, `OPTIONAL`, `MINUS`, `FILTER (NOT) EXISTS`, nested groups
|
|
135
|
+
and `VALUES`, projection, `DISTINCT`, `ORDER BY`, `LIMIT`/`OFFSET`,
|
|
136
|
+
`ASK` and `COUNT` aggregates above them. Anything else is evaluated by
|
|
137
|
+
rdflib. Each pushdown is described, with an example and numbers, in
|
|
138
|
+
[docs/pushdown.md](docs/pushdown.md); the switches to disable or narrow it
|
|
139
|
+
are in the table below.
|
|
140
|
+
|
|
141
|
+
**File-backed vs in-memory.** The default open is lazy and file-backed.
|
|
142
|
+
`VortexRdflibStore(path, in_memory=True)` (or env `VORTEX_RDF_IN_MEMORY=1`) loads
|
|
143
|
+
the store into memory once, so queries skip the per-call file-read pipeline.
|
|
144
|
+
That helps mainly point lookups and joins; the scan-dominated queries are
|
|
145
|
+
bound by rdflib's own result handling either way.
|
|
146
|
+
|
|
147
|
+
**Secondary indexes.** `serialize_rdf(..., indexes=["secondary-by-copy"])`
|
|
148
|
+
(or `"secondary-by-reference"`) writes index components into the `.vortex`
|
|
149
|
+
file, for a modest increase in build time and file size. They pay off on
|
|
150
|
+
file-backed stores answering single-pattern lookups, where they largely erase
|
|
151
|
+
the file-backed penalty for object and predicate-object lookups. On an
|
|
152
|
+
in-memory store they change nothing measurable, since the rows are resident
|
|
153
|
+
already, and on multi-pattern joins the run-to-run spread is wider than any
|
|
154
|
+
effect they have. Enable them for lookup-heavy file-backed workloads.
|
|
155
|
+
|
|
156
|
+
For Dictionary-layout files, the term dictionary is held in memory when it
|
|
157
|
+
fits the residency budget; pass `VortexRdflibStore(path, max_resident_bytes=...)`
|
|
158
|
+
(the dictionary's compressed size in bytes) to raise the budget
|
|
159
|
+
(recommended for large stores).
|
|
160
|
+
|
|
161
|
+
## Environment variables
|
|
162
|
+
|
|
163
|
+
| Variable | Effect |
|
|
164
|
+
| --- | --- |
|
|
165
|
+
| `VORTEX_RDF_IN_MEMORY=1` | Load stores into memory instead of file-backed lazy open |
|
|
166
|
+
| `VORTEX_RDF_DISABLE_CODE_PATH=1` | Force the N-Triples string path instead of `u32` codes |
|
|
167
|
+
| `VORTEX_RDF_DISABLE_PUSHDOWN=1` | Keep rdflib's default evaluator for every operator (see [docs/pushdown.md](docs/pushdown.md)) |
|
|
168
|
+
| `VORTEX_RDF_PUSHDOWN_OPS=<list>` | Only push down the listed algebra nodes (`bgp` = basic graph patterns only) |
|
|
169
|
+
| `VORTEX_RDF_FILTER_FAST=0` | Evaluate every FILTER value through rdflib's expression evaluator (still once per distinct value) |
|
|
170
|
+
| `VORTEX_RDF_TRACE_TRIPLES=1` | Print every `triples()` pattern (debugging) |
|
|
171
|
+
|
|
172
|
+
## Benchmarks
|
|
173
|
+
|
|
174
|
+
A comparative benchmark — `VortexRdflibStore` against rdflib's in-memory `Memory`
|
|
175
|
+
store, [oxrdflib](https://github.com/oxigraph/oxrdflib) (Oxigraph),
|
|
176
|
+
[pycottas](https://github.com/cottas-rdf/pycottas) (COTTAS) and
|
|
177
|
+
[rdflib-hdt](https://pypi.org/project/rdflib-hdt/) (HDT) — runs on every
|
|
178
|
+
push to `main` and publishes the current numbers to GitHub Pages:
|
|
179
|
+
**<https://vortex-rdf.github.io/vortex-rdflib/>**. That dashboard is the
|
|
180
|
+
reference for how these variants actually compare; timings vary with machine
|
|
181
|
+
and dataset.
|
|
182
|
+
|
|
183
|
+
It executes a synthetic representative SPARQL set (lookups/scans, star and
|
|
184
|
+
chain joins, FILTER/DISTINCT/ORDER BY/GROUP BY, and the shapes that name a
|
|
185
|
+
graph) and records per-store peak RSS; each store's full lifecycle runs in its
|
|
186
|
+
own process. SPARQL evaluation is rdflib's engine for every store, so the
|
|
187
|
+
store serving quad patterns is the only variable — a store's own SPARQL engine
|
|
188
|
+
is out of scope, since it skips rdflib's parse and algebra and is not
|
|
189
|
+
measuring the same work.
|
|
190
|
+
|
|
191
|
+
The dataset is a set of quads — every statement about a subject goes into
|
|
192
|
+
one graph, so the union of the graphs is exactly the triple set — and each
|
|
193
|
+
store loads it as an rdflib `Dataset` whose default graph is that union. HDT and COTTAS cannot
|
|
194
|
+
serve named graphs through rdflib — HDT's format has none, and pycottas'
|
|
195
|
+
`COTTASStore` does not expose the ones COTTAS files can hold — so those two
|
|
196
|
+
rows load the flattened N-Triples, the same statements, and are not asked the
|
|
197
|
+
`graphs` group, whose cells stay empty for them.
|
|
198
|
+
|
|
199
|
+
The Vortex rows are all Dictionary layout — the layout that enables the term codes
|
|
200
|
+
path — crossed over the two axes that change how a store answers: residency
|
|
201
|
+
(file-backed vs in-memory) and secondary index (none, by-copy, by-reference).
|
|
202
|
+
|
|
203
|
+
`pycottas` and `rdflib-hdt` pin dependencies that cannot share the project
|
|
204
|
+
environment — pycottas an exact `pyoxigraph`, rdflib-hdt an exact `rdflib` —
|
|
205
|
+
so `run_bench` builds each a throwaway virtualenv and runs that worker with
|
|
206
|
+
its interpreter. rdflib-hdt reads HDT but cannot write it, so the HDT file is
|
|
207
|
+
built by the Rust crate's CLI, which the refresh script installs on demand.
|
|
208
|
+
|
|
209
|
+
Run it locally with `scripts/refresh.sh` — it syncs the contenders, ensures
|
|
210
|
+
the HDT builder, measures, and re-renders the dashboard:
|
|
211
|
+
|
|
212
|
+
```bash
|
|
213
|
+
scripts/refresh.sh # every stage, at the 250k CI scale
|
|
214
|
+
BENCH_TRIPLES=20000 scripts/refresh.sh # scale down
|
|
215
|
+
scripts/refresh.sh --only render # template-only edits: no re-measurement
|
|
216
|
+
```
|
|
217
|
+
|
|
218
|
+
### Regression tracking (CodSpeed)
|
|
219
|
+
|
|
220
|
+
The dashboard answers "how does this compare?"; it cannot answer "did this
|
|
221
|
+
commit make things slower?", because wall-clock numbers from a shared CI
|
|
222
|
+
runner move on their own. `bench/test_codspeed.py` covers that: the **same**
|
|
223
|
+
dataset generator and the **same** query set, measured per commit under
|
|
224
|
+
CodSpeed's CPU simulation so every task gets a deterministic instruction
|
|
225
|
+
count. Every pull request gets a report at
|
|
226
|
+
<https://app.codspeed.io/vortex-rdf/vortex-rdflib>, so a change that costs
|
|
227
|
+
instructions is visible before it lands.
|
|
228
|
+
|
|
229
|
+
Only the vortex variants are measured: another library's instruction count
|
|
230
|
+
moves when *it* releases, which is not a signal this repo can act on. And
|
|
231
|
+
since instruction counts are deterministic, the suite does not run the full
|
|
232
|
+
configurations × queries cross product; the whole query set runs on the
|
|
233
|
+
primary configuration (Dictionary layout, in-memory, pushdown on) and
|
|
234
|
+
each other axis is isolated on the queries where it can move the number —
|
|
235
|
+
the pushdown A/B on the join queries, file-backed opens and secondary
|
|
236
|
+
indexes on the lookups they target, the `Store.triples()` service per pattern
|
|
237
|
+
selectivity, the u32 term codes path against the N-Triples string fallback, and
|
|
238
|
+
each residency's open cost.
|
|
239
|
+
|
|
240
|
+
The suite is not part of `uv run pytest` (which runs `tests/` only); run it
|
|
241
|
+
explicitly. 32,768 triples by default — small enough for Valgrind, and the
|
|
242
|
+
size the vortex-rdf Rust and JS suites share, so a shared-core regression
|
|
243
|
+
lands in every tab at comparable magnitude — override with
|
|
244
|
+
`CODSPEED_BENCH_TRIPLES`:
|
|
245
|
+
|
|
246
|
+
```bash
|
|
247
|
+
uv run pytest bench/test_codspeed.py --codspeed # wall-clock, no instrumentation
|
|
248
|
+
CODSPEED_BENCH_TRIPLES=5000 uv run pytest bench/test_codspeed.py --codspeed
|
|
249
|
+
```
|
|
250
|
+
|
|
251
|
+
## Development
|
|
252
|
+
|
|
253
|
+
The repo is managed with [uv](https://docs.astral.sh/uv/); `uv.lock` pins the
|
|
254
|
+
development environment.
|
|
255
|
+
|
|
256
|
+
```bash
|
|
257
|
+
uv sync # create .venv and install deps + dev group
|
|
258
|
+
uv run pytest
|
|
259
|
+
uv run ruff format # format (check mode in CI)
|
|
260
|
+
uv run ruff check # lint
|
|
261
|
+
uv run ty check # type check
|
|
262
|
+
uv build # sdist + wheel into dist/
|
|
263
|
+
```
|
|
264
|
+
|
|
265
|
+
One-time setup after cloning — install the git hooks so `git push` runs the
|
|
266
|
+
same checks as CI first (and commit messages follow
|
|
267
|
+
[Conventional Commits](https://www.conventionalcommits.org)):
|
|
268
|
+
|
|
269
|
+
```bash
|
|
270
|
+
./scripts/install-git-hooks.sh
|
|
271
|
+
```
|
|
272
|
+
|
|
273
|
+
Run the checks manually with `./scripts/ci-check.sh`; skip a hook once with
|
|
274
|
+
`git commit --no-verify` / `git push --no-verify`.
|
|
275
|
+
|
|
276
|
+
## Releasing
|
|
277
|
+
|
|
278
|
+
`.github/workflows/release.yml` builds the sdist and wheel and uploads them
|
|
279
|
+
to PyPI via [Trusted Publishing](https://docs.pypi.org/trusted-publishers/) —
|
|
280
|
+
no API token is stored in the repository. To cut a release:
|
|
281
|
+
|
|
282
|
+
1. Bump `version` in `pyproject.toml` (and run `uv lock` to sync the lockfile).
|
|
283
|
+
2. Update the changelog: `scripts/update-changelog.sh v<version>` stamps the
|
|
284
|
+
`[Unreleased]` section (regenerated from Conventional Commits via
|
|
285
|
+
[git-cliff](https://git-cliff.org/); refresh anytime with
|
|
286
|
+
`scripts/update-changelog.sh`).
|
|
287
|
+
3. Commit, then push a matching `vX.Y.Z` tag.
|
|
288
|
+
|
|
289
|
+
The full CI matrix runs on the tagged commit and must pass before anything is
|
|
290
|
+
built; the workflow refuses to publish if the tag, `pyproject.toml` and
|
|
291
|
+
`uv.lock` disagree on the version; and the wheel is smoke-tested against the
|
|
292
|
+
test suite before upload. Running the workflow by hand (Actions → Release →
|
|
293
|
+
Run workflow) is a dry run — build, validate, smoke-test, publish nothing —
|
|
294
|
+
unless the "Publish to PyPI" toggle is on, which is how to retry a release
|
|
295
|
+
whose publish step failed.
|
|
296
|
+
|
|
297
|
+
## License
|
|
298
|
+
|
|
299
|
+
MIT — see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,269 @@
|
|
|
1
|
+
# vortex-rdflib
|
|
2
|
+
|
|
3
|
+
[](https://github.com/vortex-rdf/vortex-rdflib/actions/workflows/ci.yml)
|
|
4
|
+
[](https://app.codspeed.io/vortex-rdf/vortex-rdflib?utm_source=badge)
|
|
5
|
+
[](https://pypi.org/project/vortex-rdflib/)
|
|
6
|
+
[](https://pypi.org/project/vortex-rdflib/)
|
|
7
|
+
[](LICENSE)
|
|
8
|
+
|
|
9
|
+
An [rdflib](https://rdflib.readthedocs.io/) `Store` implementation for
|
|
10
|
+
[Vortex-RDF](https://github.com/vortex-rdf/vortex-rdf), a columnar
|
|
11
|
+
zero-copy RDF serialization format — so `.vortex` files can be queried with
|
|
12
|
+
SPARQL.
|
|
13
|
+
|
|
14
|
+
The native layer is the [vortex-rdf](https://pypi.org/project/vortex-rdf/)
|
|
15
|
+
package (PyO3 bindings over the `vortex-rdf-core` Rust crate), pulled in as
|
|
16
|
+
a dependency: stores are **opened lazily from `.vortex` files** by default,
|
|
17
|
+
and queried in place without loading the dataset into memory. However, a `.vortex`
|
|
18
|
+
file can also be fully loaded in memory, with exactly the same data
|
|
19
|
+
structure and queried.
|
|
20
|
+
|
|
21
|
+
## Install
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
pip install vortex-rdflib
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
Python 3.11+. The `vortex-rdf` dependency ships prebuilt wheels for Linux
|
|
28
|
+
(x86_64, aarch64), macOS (x86_64, arm64) and Windows (x64); on other
|
|
29
|
+
platforms it builds from source.
|
|
30
|
+
|
|
31
|
+
## Usage
|
|
32
|
+
|
|
33
|
+
```python
|
|
34
|
+
from rdflib import Graph
|
|
35
|
+
from vortex_rdflib import VortexRdflibStore
|
|
36
|
+
|
|
37
|
+
graph = Graph(store=VortexRdflibStore("data.vortex"))
|
|
38
|
+
for row in graph.query("""
|
|
39
|
+
SELECT ?s ?o WHERE {
|
|
40
|
+
?s <http://xmlns.com/foaf/0.1/name> ?o
|
|
41
|
+
}
|
|
42
|
+
LIMIT 10
|
|
43
|
+
"""):
|
|
44
|
+
print(row.s, row.o)
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
SPARQL evaluation is rdflib's engine; the store serves quad patterns from
|
|
48
|
+
the Vortex file. The store is read-only so far; mutation support is on the roadmap.
|
|
49
|
+
|
|
50
|
+
A `.vortex` file holds quads, so the store is context-aware. A `Dataset` gives
|
|
51
|
+
the named graphs, and `GRAPH` works in SPARQL:
|
|
52
|
+
|
|
53
|
+
```python
|
|
54
|
+
from rdflib import Dataset
|
|
55
|
+
from vortex_rdflib import VortexRdflibStore
|
|
56
|
+
|
|
57
|
+
# default_union=True makes the SPARQL default graph the union of every graph
|
|
58
|
+
dataset = Dataset(store=VortexRdflibStore("data.vortex"), default_union=True)
|
|
59
|
+
|
|
60
|
+
for graph in dataset.graphs():
|
|
61
|
+
print(graph.identifier, len(graph))
|
|
62
|
+
|
|
63
|
+
for row in dataset.query("""
|
|
64
|
+
SELECT ?g (COUNT(*) AS ?n) WHERE {
|
|
65
|
+
GRAPH ?g { ?s ?p ?o }
|
|
66
|
+
}
|
|
67
|
+
GROUP BY ?g
|
|
68
|
+
"""):
|
|
69
|
+
print(row.g, row.n)
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
A plain `Graph(store=VortexRdflibStore(path))` — as in the example above — is
|
|
73
|
+
the view over the **whole file**, every graph included. It is a multiset
|
|
74
|
+
view — a triple in two graphs is yielded twice: the store streams the quads
|
|
75
|
+
it holds rather than building the RDF merge.
|
|
76
|
+
|
|
77
|
+
To produce a `.vortex` file from an RDF file, use the binding layer directly
|
|
78
|
+
(or the [vortex-rdf CLI](https://github.com/vortex-rdf/vortex-rdf)):
|
|
79
|
+
|
|
80
|
+
```python
|
|
81
|
+
from vortex_rdf import serialize_rdf
|
|
82
|
+
|
|
83
|
+
serialize_rdf("data.nq", "data.vortex", format="nquads", layout="dictionary")
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
`layout` accepts `"default"`, `"typed-object"` and `"dictionary"`
|
|
87
|
+
([described here](https://github.com/vortex-rdf/vortex-rdf/blob/main/docs/file-format.md#4-the-quad-table));
|
|
88
|
+
opening
|
|
89
|
+
auto-detects the layout. The `"dictionary"` layout is the fastest to query
|
|
90
|
+
from Python — it enables the SPARQL pushdowns described below.
|
|
91
|
+
|
|
92
|
+
## How it works
|
|
93
|
+
|
|
94
|
+
**Term codes instead of strings.** For Dictionary-layout stores, matched rows
|
|
95
|
+
cross the native boundary as zero-copy `u32` term-code columns
|
|
96
|
+
(`vortex_rdf.VortexRdfStore.match_codes`), and each distinct code is decoded
|
|
97
|
+
to an rdflib term once — in one GIL-released `TermDict.decode_many` call per
|
|
98
|
+
batch — and cached for the store's lifetime. Other layouts fall back to
|
|
99
|
+
N-Triples string columns, parsing each distinct term once.
|
|
100
|
+
|
|
101
|
+
**SPARQL pushdown.** Constructing a `VortexRdflibStore` registers an rdflib
|
|
102
|
+
`CUSTOM_EVALS` hook that answers the algebra operators it understands over
|
|
103
|
+
vortex term codes instead of leaving them to rdflib's per-row evaluation: basic
|
|
104
|
+
graph patterns, `FILTER`, `OPTIONAL`, `MINUS`, `FILTER (NOT) EXISTS`, nested groups
|
|
105
|
+
and `VALUES`, projection, `DISTINCT`, `ORDER BY`, `LIMIT`/`OFFSET`,
|
|
106
|
+
`ASK` and `COUNT` aggregates above them. Anything else is evaluated by
|
|
107
|
+
rdflib. Each pushdown is described, with an example and numbers, in
|
|
108
|
+
[docs/pushdown.md](docs/pushdown.md); the switches to disable or narrow it
|
|
109
|
+
are in the table below.
|
|
110
|
+
|
|
111
|
+
**File-backed vs in-memory.** The default open is lazy and file-backed.
|
|
112
|
+
`VortexRdflibStore(path, in_memory=True)` (or env `VORTEX_RDF_IN_MEMORY=1`) loads
|
|
113
|
+
the store into memory once, so queries skip the per-call file-read pipeline.
|
|
114
|
+
That helps mainly point lookups and joins; the scan-dominated queries are
|
|
115
|
+
bound by rdflib's own result handling either way.
|
|
116
|
+
|
|
117
|
+
**Secondary indexes.** `serialize_rdf(..., indexes=["secondary-by-copy"])`
|
|
118
|
+
(or `"secondary-by-reference"`) writes index components into the `.vortex`
|
|
119
|
+
file, for a modest increase in build time and file size. They pay off on
|
|
120
|
+
file-backed stores answering single-pattern lookups, where they largely erase
|
|
121
|
+
the file-backed penalty for object and predicate-object lookups. On an
|
|
122
|
+
in-memory store they change nothing measurable, since the rows are resident
|
|
123
|
+
already, and on multi-pattern joins the run-to-run spread is wider than any
|
|
124
|
+
effect they have. Enable them for lookup-heavy file-backed workloads.
|
|
125
|
+
|
|
126
|
+
For Dictionary-layout files, the term dictionary is held in memory when it
|
|
127
|
+
fits the residency budget; pass `VortexRdflibStore(path, max_resident_bytes=...)`
|
|
128
|
+
(the dictionary's compressed size in bytes) to raise the budget
|
|
129
|
+
(recommended for large stores).
|
|
130
|
+
|
|
131
|
+
## Environment variables
|
|
132
|
+
|
|
133
|
+
| Variable | Effect |
|
|
134
|
+
| --- | --- |
|
|
135
|
+
| `VORTEX_RDF_IN_MEMORY=1` | Load stores into memory instead of file-backed lazy open |
|
|
136
|
+
| `VORTEX_RDF_DISABLE_CODE_PATH=1` | Force the N-Triples string path instead of `u32` codes |
|
|
137
|
+
| `VORTEX_RDF_DISABLE_PUSHDOWN=1` | Keep rdflib's default evaluator for every operator (see [docs/pushdown.md](docs/pushdown.md)) |
|
|
138
|
+
| `VORTEX_RDF_PUSHDOWN_OPS=<list>` | Only push down the listed algebra nodes (`bgp` = basic graph patterns only) |
|
|
139
|
+
| `VORTEX_RDF_FILTER_FAST=0` | Evaluate every FILTER value through rdflib's expression evaluator (still once per distinct value) |
|
|
140
|
+
| `VORTEX_RDF_TRACE_TRIPLES=1` | Print every `triples()` pattern (debugging) |
|
|
141
|
+
|
|
142
|
+
## Benchmarks
|
|
143
|
+
|
|
144
|
+
A comparative benchmark — `VortexRdflibStore` against rdflib's in-memory `Memory`
|
|
145
|
+
store, [oxrdflib](https://github.com/oxigraph/oxrdflib) (Oxigraph),
|
|
146
|
+
[pycottas](https://github.com/cottas-rdf/pycottas) (COTTAS) and
|
|
147
|
+
[rdflib-hdt](https://pypi.org/project/rdflib-hdt/) (HDT) — runs on every
|
|
148
|
+
push to `main` and publishes the current numbers to GitHub Pages:
|
|
149
|
+
**<https://vortex-rdf.github.io/vortex-rdflib/>**. That dashboard is the
|
|
150
|
+
reference for how these variants actually compare; timings vary with machine
|
|
151
|
+
and dataset.
|
|
152
|
+
|
|
153
|
+
It executes a synthetic representative SPARQL set (lookups/scans, star and
|
|
154
|
+
chain joins, FILTER/DISTINCT/ORDER BY/GROUP BY, and the shapes that name a
|
|
155
|
+
graph) and records per-store peak RSS; each store's full lifecycle runs in its
|
|
156
|
+
own process. SPARQL evaluation is rdflib's engine for every store, so the
|
|
157
|
+
store serving quad patterns is the only variable — a store's own SPARQL engine
|
|
158
|
+
is out of scope, since it skips rdflib's parse and algebra and is not
|
|
159
|
+
measuring the same work.
|
|
160
|
+
|
|
161
|
+
The dataset is a set of quads — every statement about a subject goes into
|
|
162
|
+
one graph, so the union of the graphs is exactly the triple set — and each
|
|
163
|
+
store loads it as an rdflib `Dataset` whose default graph is that union. HDT and COTTAS cannot
|
|
164
|
+
serve named graphs through rdflib — HDT's format has none, and pycottas'
|
|
165
|
+
`COTTASStore` does not expose the ones COTTAS files can hold — so those two
|
|
166
|
+
rows load the flattened N-Triples, the same statements, and are not asked the
|
|
167
|
+
`graphs` group, whose cells stay empty for them.
|
|
168
|
+
|
|
169
|
+
The Vortex rows are all Dictionary layout — the layout that enables the term codes
|
|
170
|
+
path — crossed over the two axes that change how a store answers: residency
|
|
171
|
+
(file-backed vs in-memory) and secondary index (none, by-copy, by-reference).
|
|
172
|
+
|
|
173
|
+
`pycottas` and `rdflib-hdt` pin dependencies that cannot share the project
|
|
174
|
+
environment — pycottas an exact `pyoxigraph`, rdflib-hdt an exact `rdflib` —
|
|
175
|
+
so `run_bench` builds each a throwaway virtualenv and runs that worker with
|
|
176
|
+
its interpreter. rdflib-hdt reads HDT but cannot write it, so the HDT file is
|
|
177
|
+
built by the Rust crate's CLI, which the refresh script installs on demand.
|
|
178
|
+
|
|
179
|
+
Run it locally with `scripts/refresh.sh` — it syncs the contenders, ensures
|
|
180
|
+
the HDT builder, measures, and re-renders the dashboard:
|
|
181
|
+
|
|
182
|
+
```bash
|
|
183
|
+
scripts/refresh.sh # every stage, at the 250k CI scale
|
|
184
|
+
BENCH_TRIPLES=20000 scripts/refresh.sh # scale down
|
|
185
|
+
scripts/refresh.sh --only render # template-only edits: no re-measurement
|
|
186
|
+
```
|
|
187
|
+
|
|
188
|
+
### Regression tracking (CodSpeed)
|
|
189
|
+
|
|
190
|
+
The dashboard answers "how does this compare?"; it cannot answer "did this
|
|
191
|
+
commit make things slower?", because wall-clock numbers from a shared CI
|
|
192
|
+
runner move on their own. `bench/test_codspeed.py` covers that: the **same**
|
|
193
|
+
dataset generator and the **same** query set, measured per commit under
|
|
194
|
+
CodSpeed's CPU simulation so every task gets a deterministic instruction
|
|
195
|
+
count. Every pull request gets a report at
|
|
196
|
+
<https://app.codspeed.io/vortex-rdf/vortex-rdflib>, so a change that costs
|
|
197
|
+
instructions is visible before it lands.
|
|
198
|
+
|
|
199
|
+
Only the vortex variants are measured: another library's instruction count
|
|
200
|
+
moves when *it* releases, which is not a signal this repo can act on. And
|
|
201
|
+
since instruction counts are deterministic, the suite does not run the full
|
|
202
|
+
configurations × queries cross product; the whole query set runs on the
|
|
203
|
+
primary configuration (Dictionary layout, in-memory, pushdown on) and
|
|
204
|
+
each other axis is isolated on the queries where it can move the number —
|
|
205
|
+
the pushdown A/B on the join queries, file-backed opens and secondary
|
|
206
|
+
indexes on the lookups they target, the `Store.triples()` service per pattern
|
|
207
|
+
selectivity, the u32 term codes path against the N-Triples string fallback, and
|
|
208
|
+
each residency's open cost.
|
|
209
|
+
|
|
210
|
+
The suite is not part of `uv run pytest` (which runs `tests/` only); run it
|
|
211
|
+
explicitly. 32,768 triples by default — small enough for Valgrind, and the
|
|
212
|
+
size the vortex-rdf Rust and JS suites share, so a shared-core regression
|
|
213
|
+
lands in every tab at comparable magnitude — override with
|
|
214
|
+
`CODSPEED_BENCH_TRIPLES`:
|
|
215
|
+
|
|
216
|
+
```bash
|
|
217
|
+
uv run pytest bench/test_codspeed.py --codspeed # wall-clock, no instrumentation
|
|
218
|
+
CODSPEED_BENCH_TRIPLES=5000 uv run pytest bench/test_codspeed.py --codspeed
|
|
219
|
+
```
|
|
220
|
+
|
|
221
|
+
## Development
|
|
222
|
+
|
|
223
|
+
The repo is managed with [uv](https://docs.astral.sh/uv/); `uv.lock` pins the
|
|
224
|
+
development environment.
|
|
225
|
+
|
|
226
|
+
```bash
|
|
227
|
+
uv sync # create .venv and install deps + dev group
|
|
228
|
+
uv run pytest
|
|
229
|
+
uv run ruff format # format (check mode in CI)
|
|
230
|
+
uv run ruff check # lint
|
|
231
|
+
uv run ty check # type check
|
|
232
|
+
uv build # sdist + wheel into dist/
|
|
233
|
+
```
|
|
234
|
+
|
|
235
|
+
One-time setup after cloning — install the git hooks so `git push` runs the
|
|
236
|
+
same checks as CI first (and commit messages follow
|
|
237
|
+
[Conventional Commits](https://www.conventionalcommits.org)):
|
|
238
|
+
|
|
239
|
+
```bash
|
|
240
|
+
./scripts/install-git-hooks.sh
|
|
241
|
+
```
|
|
242
|
+
|
|
243
|
+
Run the checks manually with `./scripts/ci-check.sh`; skip a hook once with
|
|
244
|
+
`git commit --no-verify` / `git push --no-verify`.
|
|
245
|
+
|
|
246
|
+
## Releasing
|
|
247
|
+
|
|
248
|
+
`.github/workflows/release.yml` builds the sdist and wheel and uploads them
|
|
249
|
+
to PyPI via [Trusted Publishing](https://docs.pypi.org/trusted-publishers/) —
|
|
250
|
+
no API token is stored in the repository. To cut a release:
|
|
251
|
+
|
|
252
|
+
1. Bump `version` in `pyproject.toml` (and run `uv lock` to sync the lockfile).
|
|
253
|
+
2. Update the changelog: `scripts/update-changelog.sh v<version>` stamps the
|
|
254
|
+
`[Unreleased]` section (regenerated from Conventional Commits via
|
|
255
|
+
[git-cliff](https://git-cliff.org/); refresh anytime with
|
|
256
|
+
`scripts/update-changelog.sh`).
|
|
257
|
+
3. Commit, then push a matching `vX.Y.Z` tag.
|
|
258
|
+
|
|
259
|
+
The full CI matrix runs on the tagged commit and must pass before anything is
|
|
260
|
+
built; the workflow refuses to publish if the tag, `pyproject.toml` and
|
|
261
|
+
`uv.lock` disagree on the version; and the wheel is smoke-tested against the
|
|
262
|
+
test suite before upload. Running the workflow by hand (Actions → Release →
|
|
263
|
+
Run workflow) is a dry run — build, validate, smoke-test, publish nothing —
|
|
264
|
+
unless the "Publish to PyPI" toggle is on, which is how to retry a release
|
|
265
|
+
whose publish step failed.
|
|
266
|
+
|
|
267
|
+
## License
|
|
268
|
+
|
|
269
|
+
MIT — see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["uv_build>=0.8.14,<0.9.0"]
|
|
3
|
+
build-backend = "uv_build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "vortex-rdflib"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "rdflib Store implementation for Vortex-RDF, a columnar zero-copy RDF store"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
license-files = ["LICENSE"]
|
|
12
|
+
requires-python = ">=3.11"
|
|
13
|
+
authors = [
|
|
14
|
+
{ name = "Julián Rojas", email = "julianandres.rojasmelendez@ugent.be" },
|
|
15
|
+
]
|
|
16
|
+
keywords = ["rdf", "rdflib", "vortex", "columnar", "quad-store", "sparql"]
|
|
17
|
+
classifiers = [
|
|
18
|
+
"Development Status :: 4 - Beta",
|
|
19
|
+
"Intended Audience :: Developers",
|
|
20
|
+
"Intended Audience :: Science/Research",
|
|
21
|
+
"Operating System :: OS Independent",
|
|
22
|
+
"Programming Language :: Python :: 3",
|
|
23
|
+
"Programming Language :: Python :: 3.11",
|
|
24
|
+
"Programming Language :: Python :: 3.12",
|
|
25
|
+
"Programming Language :: Python :: 3.13",
|
|
26
|
+
"Programming Language :: Python :: 3.14",
|
|
27
|
+
"Topic :: Database",
|
|
28
|
+
"Topic :: Scientific/Engineering :: Information Analysis",
|
|
29
|
+
"Typing :: Typed",
|
|
30
|
+
]
|
|
31
|
+
dependencies = [
|
|
32
|
+
# The native binding layer (PyO3 extension over vortex-rdf-core). This
|
|
33
|
+
# package is the rdflib integration built on top of it.
|
|
34
|
+
"vortex-rdf>=0.10,<0.11",
|
|
35
|
+
"rdflib>=7,<8",
|
|
36
|
+
]
|
|
37
|
+
|
|
38
|
+
[dependency-groups]
|
|
39
|
+
# pytest-codspeed instruments bench/test_codspeed.py. It sits in dev rather
|
|
40
|
+
# than the bench group so `uv sync` alone can run the instrumented suite —
|
|
41
|
+
# the bench group is for the comparative dashboard's third-party contenders,
|
|
42
|
+
# which the CodSpeed run deliberately does not need.
|
|
43
|
+
dev = [
|
|
44
|
+
"pytest>=8",
|
|
45
|
+
"pytest-codspeed>=4",
|
|
46
|
+
"ruff>=0.16",
|
|
47
|
+
"ty>=0.0.60",
|
|
48
|
+
]
|
|
49
|
+
# Comparative benchmark contenders (bench/): oxrdflib pulls pyoxigraph.
|
|
50
|
+
bench = ["oxrdflib>=0.4"]
|
|
51
|
+
|
|
52
|
+
[project.urls]
|
|
53
|
+
Homepage = "https://github.com/vortex-rdf/vortex-rdflib"
|
|
54
|
+
Repository = "https://github.com/vortex-rdf/vortex-rdflib"
|
|
55
|
+
Issues = "https://github.com/vortex-rdf/vortex-rdflib/issues"
|
|
56
|
+
"Vortex-RDF" = "https://github.com/vortex-rdf/vortex-rdf"
|
|
57
|
+
|
|
58
|
+
[tool.uv.build-backend]
|
|
59
|
+
# uv_build's sdist default is pyproject/README/LICENSE/src only; ship the
|
|
60
|
+
# test suite too so the sdist is self-verifying.
|
|
61
|
+
source-include = ["tests/**"]
|
|
62
|
+
|
|
63
|
+
[tool.pytest.ini_options]
|
|
64
|
+
testpaths = ["tests"]
|
|
65
|
+
addopts = "-q"
|
|
66
|
+
# The repo root, so `tests/` can import `bench` — the benchmark generator is
|
|
67
|
+
# not collected here, but its invariants are (tests/test_bench_dataset.py).
|
|
68
|
+
pythonpath = ["."]
|
|
69
|
+
# rdflib 7.6 deprecated `Dataset.default_context` and `Dataset.contexts`, then
|
|
70
|
+
# went on calling both from its own graph and SPARQL evaluation code. Every
|
|
71
|
+
# named-graph query raises them from inside rdflib; nothing here can avoid it.
|
|
72
|
+
filterwarnings = [
|
|
73
|
+
"ignore:Dataset.default_context is deprecated:DeprecationWarning",
|
|
74
|
+
"ignore:Dataset.contexts is deprecated:DeprecationWarning",
|
|
75
|
+
]
|
|
76
|
+
|
|
77
|
+
[tool.ruff]
|
|
78
|
+
# Matches the rustfmt width used across the vortex-rdf repos.
|
|
79
|
+
line-length = 100
|
|
80
|
+
src = ["src", "tests"]
|
|
81
|
+
|
|
82
|
+
[tool.ruff.lint]
|
|
83
|
+
# pycodestyle/pyflakes (E, F, W), import sorting (I), modern syntax (UP),
|
|
84
|
+
# bugbear's likely-bug patterns (B).
|
|
85
|
+
select = ["E", "F", "W", "I", "UP", "B"]
|