vortex-rdflib 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Vortex RDF
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,299 @@
1
+ Metadata-Version: 2.4
2
+ Name: vortex-rdflib
3
+ Version: 0.1.0
4
+ Summary: rdflib Store implementation for Vortex-RDF, a columnar zero-copy RDF store
5
+ Keywords: rdf,rdflib,vortex,columnar,quad-store,sparql
6
+ Author: Julián Rojas
7
+ Author-email: Julián Rojas <julianandres.rojasmelendez@ugent.be>
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Classifier: Development Status :: 4 - Beta
11
+ Classifier: Intended Audience :: Developers
12
+ Classifier: Intended Audience :: Science/Research
13
+ Classifier: Operating System :: OS Independent
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.11
16
+ Classifier: Programming Language :: Python :: 3.12
17
+ Classifier: Programming Language :: Python :: 3.13
18
+ Classifier: Programming Language :: Python :: 3.14
19
+ Classifier: Topic :: Database
20
+ Classifier: Topic :: Scientific/Engineering :: Information Analysis
21
+ Classifier: Typing :: Typed
22
+ Requires-Dist: vortex-rdf>=0.10,<0.11
23
+ Requires-Dist: rdflib>=7,<8
24
+ Requires-Python: >=3.11
25
+ Project-URL: Homepage, https://github.com/vortex-rdf/vortex-rdflib
26
+ Project-URL: Issues, https://github.com/vortex-rdf/vortex-rdflib/issues
27
+ Project-URL: Repository, https://github.com/vortex-rdf/vortex-rdflib
28
+ Project-URL: Vortex-RDF, https://github.com/vortex-rdf/vortex-rdf
29
+ Description-Content-Type: text/markdown
30
+
31
+ # vortex-rdflib
32
+
33
+ [![CI](https://github.com/vortex-rdf/vortex-rdflib/actions/workflows/ci.yml/badge.svg)](https://github.com/vortex-rdf/vortex-rdflib/actions/workflows/ci.yml)
34
+ [![CodSpeed](https://img.shields.io/endpoint?url=https://codspeed.io/badge.json)](https://app.codspeed.io/vortex-rdf/vortex-rdflib?utm_source=badge)
35
+ [![PyPI](https://img.shields.io/pypi/v/vortex-rdflib.svg)](https://pypi.org/project/vortex-rdflib/)
36
+ [![Python versions](https://img.shields.io/pypi/pyversions/vortex-rdflib.svg)](https://pypi.org/project/vortex-rdflib/)
37
+ [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
38
+
39
+ An [rdflib](https://rdflib.readthedocs.io/) `Store` implementation for
40
+ [Vortex-RDF](https://github.com/vortex-rdf/vortex-rdf), a columnar
41
+ zero-copy RDF serialization format — so `.vortex` files can be queried with
42
+ SPARQL.
43
+
44
+ The native layer is the [vortex-rdf](https://pypi.org/project/vortex-rdf/)
45
+ package (PyO3 bindings over the `vortex-rdf-core` Rust crate), pulled in as
46
+ a dependency: stores are **opened lazily from `.vortex` files** by default,
47
+ and queried in place without loading the dataset into memory. However, a `.vortex`
48
+ file can also be fully loaded in memory, with exactly the same data
49
+ structure and queried.
50
+
51
+ ## Install
52
+
53
+ ```bash
54
+ pip install vortex-rdflib
55
+ ```
56
+
57
+ Python 3.11+. The `vortex-rdf` dependency ships prebuilt wheels for Linux
58
+ (x86_64, aarch64), macOS (x86_64, arm64) and Windows (x64); on other
59
+ platforms it builds from source.
60
+
61
+ ## Usage
62
+
63
+ ```python
64
+ from rdflib import Graph
65
+ from vortex_rdflib import VortexRdflibStore
66
+
67
+ graph = Graph(store=VortexRdflibStore("data.vortex"))
68
+ for row in graph.query("""
69
+ SELECT ?s ?o WHERE {
70
+ ?s <http://xmlns.com/foaf/0.1/name> ?o
71
+ }
72
+ LIMIT 10
73
+ """):
74
+ print(row.s, row.o)
75
+ ```
76
+
77
+ SPARQL evaluation is rdflib's engine; the store serves quad patterns from
78
+ the Vortex file. The store is read-only so far; mutation support is on the roadmap.
79
+
80
+ A `.vortex` file holds quads, so the store is context-aware. A `Dataset` gives
81
+ the named graphs, and `GRAPH` works in SPARQL:
82
+
83
+ ```python
84
+ from rdflib import Dataset
85
+ from vortex_rdflib import VortexRdflibStore
86
+
87
+ # default_union=True makes the SPARQL default graph the union of every graph
88
+ dataset = Dataset(store=VortexRdflibStore("data.vortex"), default_union=True)
89
+
90
+ for graph in dataset.graphs():
91
+ print(graph.identifier, len(graph))
92
+
93
+ for row in dataset.query("""
94
+ SELECT ?g (COUNT(*) AS ?n) WHERE {
95
+ GRAPH ?g { ?s ?p ?o }
96
+ }
97
+ GROUP BY ?g
98
+ """):
99
+ print(row.g, row.n)
100
+ ```
101
+
102
+ A plain `Graph(store=VortexRdflibStore(path))` — as in the example above — is
103
+ the view over the **whole file**, every graph included. It is a multiset
104
+ view — a triple in two graphs is yielded twice: the store streams the quads
105
+ it holds rather than building the RDF merge.
106
+
107
+ To produce a `.vortex` file from an RDF file, use the binding layer directly
108
+ (or the [vortex-rdf CLI](https://github.com/vortex-rdf/vortex-rdf)):
109
+
110
+ ```python
111
+ from vortex_rdf import serialize_rdf
112
+
113
+ serialize_rdf("data.nq", "data.vortex", format="nquads", layout="dictionary")
114
+ ```
115
+
116
+ `layout` accepts `"default"`, `"typed-object"` and `"dictionary"`
117
+ ([described here](https://github.com/vortex-rdf/vortex-rdf/blob/main/docs/file-format.md#4-the-quad-table));
118
+ opening
119
+ auto-detects the layout. The `"dictionary"` layout is the fastest to query
120
+ from Python — it enables the SPARQL pushdowns described below.
121
+
122
+ ## How it works
123
+
124
+ **Term codes instead of strings.** For Dictionary-layout stores, matched rows
125
+ cross the native boundary as zero-copy `u32` term-code columns
126
+ (`vortex_rdf.VortexRdfStore.match_codes`), and each distinct code is decoded
127
+ to an rdflib term once — in one GIL-released `TermDict.decode_many` call per
128
+ batch — and cached for the store's lifetime. Other layouts fall back to
129
+ N-Triples string columns, parsing each distinct term once.
130
+
131
+ **SPARQL pushdown.** Constructing a `VortexRdflibStore` registers an rdflib
132
+ `CUSTOM_EVALS` hook that answers the algebra operators it understands over
133
+ vortex term codes instead of leaving them to rdflib's per-row evaluation: basic
134
+ graph patterns, `FILTER`, `OPTIONAL`, `MINUS`, `FILTER (NOT) EXISTS`, nested groups
135
+ and `VALUES`, projection, `DISTINCT`, `ORDER BY`, `LIMIT`/`OFFSET`,
136
+ `ASK` and `COUNT` aggregates above them. Anything else is evaluated by
137
+ rdflib. Each pushdown is described, with an example and numbers, in
138
+ [docs/pushdown.md](docs/pushdown.md); the switches to disable or narrow it
139
+ are in the table below.
140
+
141
+ **File-backed vs in-memory.** The default open is lazy and file-backed.
142
+ `VortexRdflibStore(path, in_memory=True)` (or env `VORTEX_RDF_IN_MEMORY=1`) loads
143
+ the store into memory once, so queries skip the per-call file-read pipeline.
144
+ That helps mainly point lookups and joins; the scan-dominated queries are
145
+ bound by rdflib's own result handling either way.
146
+
147
+ **Secondary indexes.** `serialize_rdf(..., indexes=["secondary-by-copy"])`
148
+ (or `"secondary-by-reference"`) writes index components into the `.vortex`
149
+ file, for a modest increase in build time and file size. They pay off on
150
+ file-backed stores answering single-pattern lookups, where they largely erase
151
+ the file-backed penalty for object and predicate-object lookups. On an
152
+ in-memory store they change nothing measurable, since the rows are resident
153
+ already, and on multi-pattern joins the run-to-run spread is wider than any
154
+ effect they have. Enable them for lookup-heavy file-backed workloads.
155
+
156
+ For Dictionary-layout files, the term dictionary is held in memory when it
157
+ fits the residency budget; pass `VortexRdflibStore(path, max_resident_bytes=...)`
158
+ (the dictionary's compressed size in bytes) to raise the budget
159
+ (recommended for large stores).
160
+
161
+ ## Environment variables
162
+
163
+ | Variable | Effect |
164
+ | --- | --- |
165
+ | `VORTEX_RDF_IN_MEMORY=1` | Load stores into memory instead of file-backed lazy open |
166
+ | `VORTEX_RDF_DISABLE_CODE_PATH=1` | Force the N-Triples string path instead of `u32` codes |
167
+ | `VORTEX_RDF_DISABLE_PUSHDOWN=1` | Keep rdflib's default evaluator for every operator (see [docs/pushdown.md](docs/pushdown.md)) |
168
+ | `VORTEX_RDF_PUSHDOWN_OPS=<list>` | Only push down the listed algebra nodes (`bgp` = basic graph patterns only) |
169
+ | `VORTEX_RDF_FILTER_FAST=0` | Evaluate every FILTER value through rdflib's expression evaluator (still once per distinct value) |
170
+ | `VORTEX_RDF_TRACE_TRIPLES=1` | Print every `triples()` pattern (debugging) |
171
+
172
+ ## Benchmarks
173
+
174
+ A comparative benchmark — `VortexRdflibStore` against rdflib's in-memory `Memory`
175
+ store, [oxrdflib](https://github.com/oxigraph/oxrdflib) (Oxigraph),
176
+ [pycottas](https://github.com/cottas-rdf/pycottas) (COTTAS) and
177
+ [rdflib-hdt](https://pypi.org/project/rdflib-hdt/) (HDT) — runs on every
178
+ push to `main` and publishes the current numbers to GitHub Pages:
179
+ **<https://vortex-rdf.github.io/vortex-rdflib/>**. That dashboard is the
180
+ reference for how these variants actually compare; timings vary with machine
181
+ and dataset.
182
+
183
+ It executes a synthetic representative SPARQL set (lookups/scans, star and
184
+ chain joins, FILTER/DISTINCT/ORDER BY/GROUP BY, and the shapes that name a
185
+ graph) and records per-store peak RSS; each store's full lifecycle runs in its
186
+ own process. SPARQL evaluation is rdflib's engine for every store, so the
187
+ store serving quad patterns is the only variable — a store's own SPARQL engine
188
+ is out of scope, since it skips rdflib's parse and algebra and is not
189
+ measuring the same work.
190
+
191
+ The dataset is a set of quads — every statement about a subject goes into
192
+ one graph, so the union of the graphs is exactly the triple set — and each
193
+ store loads it as an rdflib `Dataset` whose default graph is that union. HDT and COTTAS cannot
194
+ serve named graphs through rdflib — HDT's format has none, and pycottas'
195
+ `COTTASStore` does not expose the ones COTTAS files can hold — so those two
196
+ rows load the flattened N-Triples, the same statements, and are not asked the
197
+ `graphs` group, whose cells stay empty for them.
198
+
199
+ The Vortex rows are all Dictionary layout — the layout that enables the term codes
200
+ path — crossed over the two axes that change how a store answers: residency
201
+ (file-backed vs in-memory) and secondary index (none, by-copy, by-reference).
202
+
203
+ `pycottas` and `rdflib-hdt` pin dependencies that cannot share the project
204
+ environment — pycottas an exact `pyoxigraph`, rdflib-hdt an exact `rdflib` —
205
+ so `run_bench` builds each a throwaway virtualenv and runs that worker with
206
+ its interpreter. rdflib-hdt reads HDT but cannot write it, so the HDT file is
207
+ built by the Rust crate's CLI, which the refresh script installs on demand.
208
+
209
+ Run it locally with `scripts/refresh.sh` — it syncs the contenders, ensures
210
+ the HDT builder, measures, and re-renders the dashboard:
211
+
212
+ ```bash
213
+ scripts/refresh.sh # every stage, at the 250k CI scale
214
+ BENCH_TRIPLES=20000 scripts/refresh.sh # scale down
215
+ scripts/refresh.sh --only render # template-only edits: no re-measurement
216
+ ```
217
+
218
+ ### Regression tracking (CodSpeed)
219
+
220
+ The dashboard answers "how does this compare?"; it cannot answer "did this
221
+ commit make things slower?", because wall-clock numbers from a shared CI
222
+ runner move on their own. `bench/test_codspeed.py` covers that: the **same**
223
+ dataset generator and the **same** query set, measured per commit under
224
+ CodSpeed's CPU simulation so every task gets a deterministic instruction
225
+ count. Every pull request gets a report at
226
+ <https://app.codspeed.io/vortex-rdf/vortex-rdflib>, so a change that costs
227
+ instructions is visible before it lands.
228
+
229
+ Only the vortex variants are measured: another library's instruction count
230
+ moves when *it* releases, which is not a signal this repo can act on. And
231
+ since instruction counts are deterministic, the suite does not run the full
232
+ configurations × queries cross product; the whole query set runs on the
233
+ primary configuration (Dictionary layout, in-memory, pushdown on) and
234
+ each other axis is isolated on the queries where it can move the number —
235
+ the pushdown A/B on the join queries, file-backed opens and secondary
236
+ indexes on the lookups they target, the `Store.triples()` service per pattern
237
+ selectivity, the u32 term codes path against the N-Triples string fallback, and
238
+ each residency's open cost.
239
+
240
+ The suite is not part of `uv run pytest` (which runs `tests/` only); run it
241
+ explicitly. 32,768 triples by default — small enough for Valgrind, and the
242
+ size the vortex-rdf Rust and JS suites share, so a shared-core regression
243
+ lands in every tab at comparable magnitude — override with
244
+ `CODSPEED_BENCH_TRIPLES`:
245
+
246
+ ```bash
247
+ uv run pytest bench/test_codspeed.py --codspeed # wall-clock, no instrumentation
248
+ CODSPEED_BENCH_TRIPLES=5000 uv run pytest bench/test_codspeed.py --codspeed
249
+ ```
250
+
251
+ ## Development
252
+
253
+ The repo is managed with [uv](https://docs.astral.sh/uv/); `uv.lock` pins the
254
+ development environment.
255
+
256
+ ```bash
257
+ uv sync # create .venv and install deps + dev group
258
+ uv run pytest
259
+ uv run ruff format # format (check mode in CI)
260
+ uv run ruff check # lint
261
+ uv run ty check # type check
262
+ uv build # sdist + wheel into dist/
263
+ ```
264
+
265
+ One-time setup after cloning — install the git hooks so `git push` runs the
266
+ same checks as CI first (and commit messages follow
267
+ [Conventional Commits](https://www.conventionalcommits.org)):
268
+
269
+ ```bash
270
+ ./scripts/install-git-hooks.sh
271
+ ```
272
+
273
+ Run the checks manually with `./scripts/ci-check.sh`; skip a hook once with
274
+ `git commit --no-verify` / `git push --no-verify`.
275
+
276
+ ## Releasing
277
+
278
+ `.github/workflows/release.yml` builds the sdist and wheel and uploads them
279
+ to PyPI via [Trusted Publishing](https://docs.pypi.org/trusted-publishers/) —
280
+ no API token is stored in the repository. To cut a release:
281
+
282
+ 1. Bump `version` in `pyproject.toml` (and run `uv lock` to sync the lockfile).
283
+ 2. Update the changelog: `scripts/update-changelog.sh v<version>` stamps the
284
+ `[Unreleased]` section (regenerated from Conventional Commits via
285
+ [git-cliff](https://git-cliff.org/); refresh anytime with
286
+ `scripts/update-changelog.sh`).
287
+ 3. Commit, then push a matching `vX.Y.Z` tag.
288
+
289
+ The full CI matrix runs on the tagged commit and must pass before anything is
290
+ built; the workflow refuses to publish if the tag, `pyproject.toml` and
291
+ `uv.lock` disagree on the version; and the wheel is smoke-tested against the
292
+ test suite before upload. Running the workflow by hand (Actions → Release →
293
+ Run workflow) is a dry run — build, validate, smoke-test, publish nothing —
294
+ unless the "Publish to PyPI" toggle is on, which is how to retry a release
295
+ whose publish step failed.
296
+
297
+ ## License
298
+
299
+ MIT — see [LICENSE](LICENSE).
@@ -0,0 +1,269 @@
1
+ # vortex-rdflib
2
+
3
+ [![CI](https://github.com/vortex-rdf/vortex-rdflib/actions/workflows/ci.yml/badge.svg)](https://github.com/vortex-rdf/vortex-rdflib/actions/workflows/ci.yml)
4
+ [![CodSpeed](https://img.shields.io/endpoint?url=https://codspeed.io/badge.json)](https://app.codspeed.io/vortex-rdf/vortex-rdflib?utm_source=badge)
5
+ [![PyPI](https://img.shields.io/pypi/v/vortex-rdflib.svg)](https://pypi.org/project/vortex-rdflib/)
6
+ [![Python versions](https://img.shields.io/pypi/pyversions/vortex-rdflib.svg)](https://pypi.org/project/vortex-rdflib/)
7
+ [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
8
+
9
+ An [rdflib](https://rdflib.readthedocs.io/) `Store` implementation for
10
+ [Vortex-RDF](https://github.com/vortex-rdf/vortex-rdf), a columnar
11
+ zero-copy RDF serialization format — so `.vortex` files can be queried with
12
+ SPARQL.
13
+
14
+ The native layer is the [vortex-rdf](https://pypi.org/project/vortex-rdf/)
15
+ package (PyO3 bindings over the `vortex-rdf-core` Rust crate), pulled in as
16
+ a dependency: stores are **opened lazily from `.vortex` files** by default,
17
+ and queried in place without loading the dataset into memory. However, a `.vortex`
18
+ file can also be fully loaded in memory, with exactly the same data
19
+ structure and queried.
20
+
21
+ ## Install
22
+
23
+ ```bash
24
+ pip install vortex-rdflib
25
+ ```
26
+
27
+ Python 3.11+. The `vortex-rdf` dependency ships prebuilt wheels for Linux
28
+ (x86_64, aarch64), macOS (x86_64, arm64) and Windows (x64); on other
29
+ platforms it builds from source.
30
+
31
+ ## Usage
32
+
33
+ ```python
34
+ from rdflib import Graph
35
+ from vortex_rdflib import VortexRdflibStore
36
+
37
+ graph = Graph(store=VortexRdflibStore("data.vortex"))
38
+ for row in graph.query("""
39
+ SELECT ?s ?o WHERE {
40
+ ?s <http://xmlns.com/foaf/0.1/name> ?o
41
+ }
42
+ LIMIT 10
43
+ """):
44
+ print(row.s, row.o)
45
+ ```
46
+
47
+ SPARQL evaluation is rdflib's engine; the store serves quad patterns from
48
+ the Vortex file. The store is read-only so far; mutation support is on the roadmap.
49
+
50
+ A `.vortex` file holds quads, so the store is context-aware. A `Dataset` gives
51
+ the named graphs, and `GRAPH` works in SPARQL:
52
+
53
+ ```python
54
+ from rdflib import Dataset
55
+ from vortex_rdflib import VortexRdflibStore
56
+
57
+ # default_union=True makes the SPARQL default graph the union of every graph
58
+ dataset = Dataset(store=VortexRdflibStore("data.vortex"), default_union=True)
59
+
60
+ for graph in dataset.graphs():
61
+ print(graph.identifier, len(graph))
62
+
63
+ for row in dataset.query("""
64
+ SELECT ?g (COUNT(*) AS ?n) WHERE {
65
+ GRAPH ?g { ?s ?p ?o }
66
+ }
67
+ GROUP BY ?g
68
+ """):
69
+ print(row.g, row.n)
70
+ ```
71
+
72
+ A plain `Graph(store=VortexRdflibStore(path))` — as in the example above — is
73
+ the view over the **whole file**, every graph included. It is a multiset
74
+ view — a triple in two graphs is yielded twice: the store streams the quads
75
+ it holds rather than building the RDF merge.
76
+
77
+ To produce a `.vortex` file from an RDF file, use the binding layer directly
78
+ (or the [vortex-rdf CLI](https://github.com/vortex-rdf/vortex-rdf)):
79
+
80
+ ```python
81
+ from vortex_rdf import serialize_rdf
82
+
83
+ serialize_rdf("data.nq", "data.vortex", format="nquads", layout="dictionary")
84
+ ```
85
+
86
+ `layout` accepts `"default"`, `"typed-object"` and `"dictionary"`
87
+ ([described here](https://github.com/vortex-rdf/vortex-rdf/blob/main/docs/file-format.md#4-the-quad-table));
88
+ opening
89
+ auto-detects the layout. The `"dictionary"` layout is the fastest to query
90
+ from Python — it enables the SPARQL pushdowns described below.
91
+
92
+ ## How it works
93
+
94
+ **Term codes instead of strings.** For Dictionary-layout stores, matched rows
95
+ cross the native boundary as zero-copy `u32` term-code columns
96
+ (`vortex_rdf.VortexRdfStore.match_codes`), and each distinct code is decoded
97
+ to an rdflib term once — in one GIL-released `TermDict.decode_many` call per
98
+ batch — and cached for the store's lifetime. Other layouts fall back to
99
+ N-Triples string columns, parsing each distinct term once.
100
+
101
+ **SPARQL pushdown.** Constructing a `VortexRdflibStore` registers an rdflib
102
+ `CUSTOM_EVALS` hook that answers the algebra operators it understands over
103
+ vortex term codes instead of leaving them to rdflib's per-row evaluation: basic
104
+ graph patterns, `FILTER`, `OPTIONAL`, `MINUS`, `FILTER (NOT) EXISTS`, nested groups
105
+ and `VALUES`, projection, `DISTINCT`, `ORDER BY`, `LIMIT`/`OFFSET`,
106
+ `ASK` and `COUNT` aggregates above them. Anything else is evaluated by
107
+ rdflib. Each pushdown is described, with an example and numbers, in
108
+ [docs/pushdown.md](docs/pushdown.md); the switches to disable or narrow it
109
+ are in the table below.
110
+
111
+ **File-backed vs in-memory.** The default open is lazy and file-backed.
112
+ `VortexRdflibStore(path, in_memory=True)` (or env `VORTEX_RDF_IN_MEMORY=1`) loads
113
+ the store into memory once, so queries skip the per-call file-read pipeline.
114
+ That helps mainly point lookups and joins; the scan-dominated queries are
115
+ bound by rdflib's own result handling either way.
116
+
117
+ **Secondary indexes.** `serialize_rdf(..., indexes=["secondary-by-copy"])`
118
+ (or `"secondary-by-reference"`) writes index components into the `.vortex`
119
+ file, for a modest increase in build time and file size. They pay off on
120
+ file-backed stores answering single-pattern lookups, where they largely erase
121
+ the file-backed penalty for object and predicate-object lookups. On an
122
+ in-memory store they change nothing measurable, since the rows are resident
123
+ already, and on multi-pattern joins the run-to-run spread is wider than any
124
+ effect they have. Enable them for lookup-heavy file-backed workloads.
125
+
126
+ For Dictionary-layout files, the term dictionary is held in memory when it
127
+ fits the residency budget; pass `VortexRdflibStore(path, max_resident_bytes=...)`
128
+ (the dictionary's compressed size in bytes) to raise the budget
129
+ (recommended for large stores).
130
+
131
+ ## Environment variables
132
+
133
+ | Variable | Effect |
134
+ | --- | --- |
135
+ | `VORTEX_RDF_IN_MEMORY=1` | Load stores into memory instead of file-backed lazy open |
136
+ | `VORTEX_RDF_DISABLE_CODE_PATH=1` | Force the N-Triples string path instead of `u32` codes |
137
+ | `VORTEX_RDF_DISABLE_PUSHDOWN=1` | Keep rdflib's default evaluator for every operator (see [docs/pushdown.md](docs/pushdown.md)) |
138
+ | `VORTEX_RDF_PUSHDOWN_OPS=<list>` | Only push down the listed algebra nodes (`bgp` = basic graph patterns only) |
139
+ | `VORTEX_RDF_FILTER_FAST=0` | Evaluate every FILTER value through rdflib's expression evaluator (still once per distinct value) |
140
+ | `VORTEX_RDF_TRACE_TRIPLES=1` | Print every `triples()` pattern (debugging) |
141
+
142
+ ## Benchmarks
143
+
144
+ A comparative benchmark — `VortexRdflibStore` against rdflib's in-memory `Memory`
145
+ store, [oxrdflib](https://github.com/oxigraph/oxrdflib) (Oxigraph),
146
+ [pycottas](https://github.com/cottas-rdf/pycottas) (COTTAS) and
147
+ [rdflib-hdt](https://pypi.org/project/rdflib-hdt/) (HDT) — runs on every
148
+ push to `main` and publishes the current numbers to GitHub Pages:
149
+ **<https://vortex-rdf.github.io/vortex-rdflib/>**. That dashboard is the
150
+ reference for how these variants actually compare; timings vary with machine
151
+ and dataset.
152
+
153
+ It executes a synthetic representative SPARQL set (lookups/scans, star and
154
+ chain joins, FILTER/DISTINCT/ORDER BY/GROUP BY, and the shapes that name a
155
+ graph) and records per-store peak RSS; each store's full lifecycle runs in its
156
+ own process. SPARQL evaluation is rdflib's engine for every store, so the
157
+ store serving quad patterns is the only variable — a store's own SPARQL engine
158
+ is out of scope, since it skips rdflib's parse and algebra and is not
159
+ measuring the same work.
160
+
161
+ The dataset is a set of quads — every statement about a subject goes into
162
+ one graph, so the union of the graphs is exactly the triple set — and each
163
+ store loads it as an rdflib `Dataset` whose default graph is that union. HDT and COTTAS cannot
164
+ serve named graphs through rdflib — HDT's format has none, and pycottas'
165
+ `COTTASStore` does not expose the ones COTTAS files can hold — so those two
166
+ rows load the flattened N-Triples, the same statements, and are not asked the
167
+ `graphs` group, whose cells stay empty for them.
168
+
169
+ The Vortex rows are all Dictionary layout — the layout that enables the term codes
170
+ path — crossed over the two axes that change how a store answers: residency
171
+ (file-backed vs in-memory) and secondary index (none, by-copy, by-reference).
172
+
173
+ `pycottas` and `rdflib-hdt` pin dependencies that cannot share the project
174
+ environment — pycottas an exact `pyoxigraph`, rdflib-hdt an exact `rdflib` —
175
+ so `run_bench` builds each a throwaway virtualenv and runs that worker with
176
+ its interpreter. rdflib-hdt reads HDT but cannot write it, so the HDT file is
177
+ built by the Rust crate's CLI, which the refresh script installs on demand.
178
+
179
+ Run it locally with `scripts/refresh.sh` — it syncs the contenders, ensures
180
+ the HDT builder, measures, and re-renders the dashboard:
181
+
182
+ ```bash
183
+ scripts/refresh.sh # every stage, at the 250k CI scale
184
+ BENCH_TRIPLES=20000 scripts/refresh.sh # scale down
185
+ scripts/refresh.sh --only render # template-only edits: no re-measurement
186
+ ```
187
+
188
+ ### Regression tracking (CodSpeed)
189
+
190
+ The dashboard answers "how does this compare?"; it cannot answer "did this
191
+ commit make things slower?", because wall-clock numbers from a shared CI
192
+ runner move on their own. `bench/test_codspeed.py` covers that: the **same**
193
+ dataset generator and the **same** query set, measured per commit under
194
+ CodSpeed's CPU simulation so every task gets a deterministic instruction
195
+ count. Every pull request gets a report at
196
+ <https://app.codspeed.io/vortex-rdf/vortex-rdflib>, so a change that costs
197
+ instructions is visible before it lands.
198
+
199
+ Only the vortex variants are measured: another library's instruction count
200
+ moves when *it* releases, which is not a signal this repo can act on. And
201
+ since instruction counts are deterministic, the suite does not run the full
202
+ configurations × queries cross product; the whole query set runs on the
203
+ primary configuration (Dictionary layout, in-memory, pushdown on) and
204
+ each other axis is isolated on the queries where it can move the number —
205
+ the pushdown A/B on the join queries, file-backed opens and secondary
206
+ indexes on the lookups they target, the `Store.triples()` service per pattern
207
+ selectivity, the u32 term codes path against the N-Triples string fallback, and
208
+ each residency's open cost.
209
+
210
+ The suite is not part of `uv run pytest` (which runs `tests/` only); run it
211
+ explicitly. 32,768 triples by default — small enough for Valgrind, and the
212
+ size the vortex-rdf Rust and JS suites share, so a shared-core regression
213
+ lands in every tab at comparable magnitude — override with
214
+ `CODSPEED_BENCH_TRIPLES`:
215
+
216
+ ```bash
217
+ uv run pytest bench/test_codspeed.py --codspeed # wall-clock, no instrumentation
218
+ CODSPEED_BENCH_TRIPLES=5000 uv run pytest bench/test_codspeed.py --codspeed
219
+ ```
220
+
221
+ ## Development
222
+
223
+ The repo is managed with [uv](https://docs.astral.sh/uv/); `uv.lock` pins the
224
+ development environment.
225
+
226
+ ```bash
227
+ uv sync # create .venv and install deps + dev group
228
+ uv run pytest
229
+ uv run ruff format # format (check mode in CI)
230
+ uv run ruff check # lint
231
+ uv run ty check # type check
232
+ uv build # sdist + wheel into dist/
233
+ ```
234
+
235
+ One-time setup after cloning — install the git hooks so `git push` runs the
236
+ same checks as CI first (and commit messages follow
237
+ [Conventional Commits](https://www.conventionalcommits.org)):
238
+
239
+ ```bash
240
+ ./scripts/install-git-hooks.sh
241
+ ```
242
+
243
+ Run the checks manually with `./scripts/ci-check.sh`; skip a hook once with
244
+ `git commit --no-verify` / `git push --no-verify`.
245
+
246
+ ## Releasing
247
+
248
+ `.github/workflows/release.yml` builds the sdist and wheel and uploads them
249
+ to PyPI via [Trusted Publishing](https://docs.pypi.org/trusted-publishers/) —
250
+ no API token is stored in the repository. To cut a release:
251
+
252
+ 1. Bump `version` in `pyproject.toml` (and run `uv lock` to sync the lockfile).
253
+ 2. Update the changelog: `scripts/update-changelog.sh v<version>` stamps the
254
+ `[Unreleased]` section (regenerated from Conventional Commits via
255
+ [git-cliff](https://git-cliff.org/); refresh anytime with
256
+ `scripts/update-changelog.sh`).
257
+ 3. Commit, then push a matching `vX.Y.Z` tag.
258
+
259
+ The full CI matrix runs on the tagged commit and must pass before anything is
260
+ built; the workflow refuses to publish if the tag, `pyproject.toml` and
261
+ `uv.lock` disagree on the version; and the wheel is smoke-tested against the
262
+ test suite before upload. Running the workflow by hand (Actions → Release →
263
+ Run workflow) is a dry run — build, validate, smoke-test, publish nothing —
264
+ unless the "Publish to PyPI" toggle is on, which is how to retry a release
265
+ whose publish step failed.
266
+
267
+ ## License
268
+
269
+ MIT — see [LICENSE](LICENSE).
@@ -0,0 +1,85 @@
1
+ [build-system]
2
+ requires = ["uv_build>=0.8.14,<0.9.0"]
3
+ build-backend = "uv_build"
4
+
5
+ [project]
6
+ name = "vortex-rdflib"
7
+ version = "0.1.0"
8
+ description = "rdflib Store implementation for Vortex-RDF, a columnar zero-copy RDF store"
9
+ readme = "README.md"
10
+ license = "MIT"
11
+ license-files = ["LICENSE"]
12
+ requires-python = ">=3.11"
13
+ authors = [
14
+ { name = "Julián Rojas", email = "julianandres.rojasmelendez@ugent.be" },
15
+ ]
16
+ keywords = ["rdf", "rdflib", "vortex", "columnar", "quad-store", "sparql"]
17
+ classifiers = [
18
+ "Development Status :: 4 - Beta",
19
+ "Intended Audience :: Developers",
20
+ "Intended Audience :: Science/Research",
21
+ "Operating System :: OS Independent",
22
+ "Programming Language :: Python :: 3",
23
+ "Programming Language :: Python :: 3.11",
24
+ "Programming Language :: Python :: 3.12",
25
+ "Programming Language :: Python :: 3.13",
26
+ "Programming Language :: Python :: 3.14",
27
+ "Topic :: Database",
28
+ "Topic :: Scientific/Engineering :: Information Analysis",
29
+ "Typing :: Typed",
30
+ ]
31
+ dependencies = [
32
+ # The native binding layer (PyO3 extension over vortex-rdf-core). This
33
+ # package is the rdflib integration built on top of it.
34
+ "vortex-rdf>=0.10,<0.11",
35
+ "rdflib>=7,<8",
36
+ ]
37
+
38
+ [dependency-groups]
39
+ # pytest-codspeed instruments bench/test_codspeed.py. It sits in dev rather
40
+ # than the bench group so `uv sync` alone can run the instrumented suite —
41
+ # the bench group is for the comparative dashboard's third-party contenders,
42
+ # which the CodSpeed run deliberately does not need.
43
+ dev = [
44
+ "pytest>=8",
45
+ "pytest-codspeed>=4",
46
+ "ruff>=0.16",
47
+ "ty>=0.0.60",
48
+ ]
49
+ # Comparative benchmark contenders (bench/): oxrdflib pulls pyoxigraph.
50
+ bench = ["oxrdflib>=0.4"]
51
+
52
+ [project.urls]
53
+ Homepage = "https://github.com/vortex-rdf/vortex-rdflib"
54
+ Repository = "https://github.com/vortex-rdf/vortex-rdflib"
55
+ Issues = "https://github.com/vortex-rdf/vortex-rdflib/issues"
56
+ "Vortex-RDF" = "https://github.com/vortex-rdf/vortex-rdf"
57
+
58
+ [tool.uv.build-backend]
59
+ # uv_build's sdist default is pyproject/README/LICENSE/src only; ship the
60
+ # test suite too so the sdist is self-verifying.
61
+ source-include = ["tests/**"]
62
+
63
+ [tool.pytest.ini_options]
64
+ testpaths = ["tests"]
65
+ addopts = "-q"
66
+ # The repo root, so `tests/` can import `bench` — the benchmark generator is
67
+ # not collected here, but its invariants are (tests/test_bench_dataset.py).
68
+ pythonpath = ["."]
69
+ # rdflib 7.6 deprecated `Dataset.default_context` and `Dataset.contexts`, then
70
+ # went on calling both from its own graph and SPARQL evaluation code. Every
71
+ # named-graph query raises them from inside rdflib; nothing here can avoid it.
72
+ filterwarnings = [
73
+ "ignore:Dataset.default_context is deprecated:DeprecationWarning",
74
+ "ignore:Dataset.contexts is deprecated:DeprecationWarning",
75
+ ]
76
+
77
+ [tool.ruff]
78
+ # Matches the rustfmt width used across the vortex-rdf repos.
79
+ line-length = 100
80
+ src = ["src", "tests"]
81
+
82
+ [tool.ruff.lint]
83
+ # pycodestyle/pyflakes (E, F, W), import sorting (I), modern syntax (UP),
84
+ # bugbear's likely-bug patterns (B).
85
+ select = ["E", "F", "W", "I", "UP", "B"]