tablassert 8.0.0__tar.gz → 8.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tablassert-8.0.0 → tablassert-8.1.0}/PKG-INFO +16 -30
- tablassert-8.1.0/README.md +70 -0
- {tablassert-8.0.0 → tablassert-8.1.0}/pyproject.toml +6 -2
- {tablassert-8.0.0 → tablassert-8.1.0}/rust/src/fullmap.rs +103 -51
- {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/agent.py +1055 -210
- {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/cli.py +159 -66
- {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/errors.py +0 -1
- {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/fullmap.py +41 -7
- {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/lib.py +1 -1
- {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/models.py +4 -13
- {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/qc.py +2 -2
- tablassert-8.0.0/README.md +0 -85
- {tablassert-8.0.0 → tablassert-8.1.0}/LICENSE +0 -0
- {tablassert-8.0.0 → tablassert-8.1.0}/rust/Cargo.lock +0 -0
- {tablassert-8.0.0 → tablassert-8.1.0}/rust/Cargo.toml +0 -0
- {tablassert-8.0.0 → tablassert-8.1.0}/rust/examples/count_tables.rs +0 -0
- {tablassert-8.0.0 → tablassert-8.1.0}/rust/src/json.rs +0 -0
- {tablassert-8.0.0 → tablassert-8.1.0}/rust/src/lib.rs +0 -0
- {tablassert-8.0.0 → tablassert-8.1.0}/rust/src/ndjson.rs +0 -0
- {tablassert-8.0.0 → tablassert-8.1.0}/rust/src/uuid.rs +0 -0
- {tablassert-8.0.0 → tablassert-8.1.0}/rust/tests/build_golden.rs +0 -0
- {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/__init__.py +0 -0
- {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/_lazy.py +0 -0
- {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/biolink.py +0 -0
- {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/coerce.py +0 -0
- {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/enums.py +0 -0
- {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/ingests.py +0 -0
- {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/log.py +0 -0
- {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/nlp.py +0 -0
- {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/progress.py +0 -0
- {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/rig.py +0 -0
- {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/rs.pyi +0 -0
- {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/utils.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: tablassert
|
|
3
|
-
Version: 8.
|
|
3
|
+
Version: 8.1.0
|
|
4
4
|
Classifier: License :: OSI Approved :: Apache Software License
|
|
5
5
|
Classifier: Development Status :: 5 - Production/Stable
|
|
6
6
|
Classifier: Intended Audience :: Science/Research
|
|
@@ -32,6 +32,7 @@ Requires-Dist: fastexcel>=0.20.2
|
|
|
32
32
|
Requires-Dist: smolagents>=1.26.0 ; extra == 'agent'
|
|
33
33
|
Requires-Dist: dspy>=3.2.1 ; extra == 'agent'
|
|
34
34
|
Requires-Dist: litellm>=1.93.0 ; extra == 'agent'
|
|
35
|
+
Requires-Dist: pdfminer-six>=20221105 ; extra == 'agent'
|
|
35
36
|
Requires-Dist: scikit-learn>=1.8.0 ; extra == 'qc'
|
|
36
37
|
Requires-Dist: sentence-transformers>=5.3.0 ; extra == 'qc'
|
|
37
38
|
Requires-Dist: polars[rtcompat]>=1.40.1 ; extra == 'rt'
|
|
@@ -71,16 +72,11 @@ tablassert build-kg config.yaml
|
|
|
71
72
|
pip install tablassert
|
|
72
73
|
```
|
|
73
74
|
|
|
74
|
-
The base install
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
```
|
|
80
|
-
|
|
81
|
-
Excel (`.xlsx`) inputs are read through Polars' `calamine` engine and additionally require `python-calamine` (`pip install python-calamine`).
|
|
82
|
-
|
|
83
|
-
QC is opt-in: pass `--qc` to `build-kg` to run the three-stage audit (exact → fuzzy → BioBERT). See the [CLI Reference](https://skyeav.github.io/Tablassert/cli/) for the full flag reference.
|
|
75
|
+
The base install builds knowledge graphs from CSV/TSV/Excel sources. Optional extras (`rt`, `qc`,
|
|
76
|
+
`agent`) add CPU-compatible Polars, the three-stage QC audit, and the autonomous agent — see the
|
|
77
|
+
[Installation guide](https://skyeav.github.io/Tablassert/installation/) for the full matrix. QC is opt-in
|
|
78
|
+
at build time (`build-kg --qc`); see the [CLI Reference](https://skyeav.github.io/Tablassert/cli/) for the
|
|
79
|
+
complete flag reference.
|
|
84
80
|
|
|
85
81
|
## Quick Demo
|
|
86
82
|
|
|
@@ -88,30 +84,20 @@ QC is opt-in: pass `--qc` to `build-kg` to run the three-stage audit (exact →
|
|
|
88
84
|
from pathlib import Path
|
|
89
85
|
from tablassert.lib import resolve_many
|
|
90
86
|
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
col="gene",
|
|
94
|
-
entities=["TP53", "BRCA1", "EGFR"],
|
|
95
|
-
fullmap=Path("/path/to/fullmap"),
|
|
96
|
-
taxon="9606",
|
|
97
|
-
)
|
|
98
|
-
|
|
99
|
-
for row in results:
|
|
100
|
-
print(f"{row['original_gene']} → {row['gene']} ({row['gene_name']})")
|
|
101
|
-
# TP53 → HGNC:11998 (TP53)
|
|
102
|
-
# BRCA1 → HGNC:1100 (BRCA1)
|
|
103
|
-
# EGFR → HGNC:3236 (EGFR)
|
|
87
|
+
results = resolve_many(col="gene", entities=["TP53", "BRCA1"], fullmap=Path("/path/to/fullmap"), taxon="9606")
|
|
88
|
+
# [{"original_gene": "TP53", "gene": "HGNC:11998", "gene_name": "TP53", ...}, ...]
|
|
104
89
|
```
|
|
105
90
|
|
|
106
|
-
Point `resolve_many()` at a fullmap database
|
|
91
|
+
Point `resolve_many()` at a fullmap database to resolve any iterable of entity strings to CURIEs — no
|
|
92
|
+
LazyFrame setup or NLP preprocessing required. See the
|
|
93
|
+
[Batch Resolution API](https://skyeav.github.io/Tablassert/api/lib/) for the full reference; for
|
|
94
|
+
YAML-configured pipeline builds use `tablassert build-kg config.yaml`.
|
|
107
95
|
|
|
108
96
|
## Key Features
|
|
109
97
|
|
|
110
|
-
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
- **KGX Compliance** — NCATS Translator-compatible NDJSON output
|
|
114
|
-
- **Performance** — Lazy evaluation pipelines with Polars and an embedded redb-accelerated entity resolution database
|
|
98
|
+
Declarative YAML configs, built-in entity resolution, optional three-stage QC, and KGX-compliant NDJSON
|
|
99
|
+
output — with lazy Polars pipelines over an embedded redb resolution database. See the
|
|
100
|
+
[documentation](https://skyeav.github.io/Tablassert/) for the full feature overview and use-case gallery.
|
|
115
101
|
|
|
116
102
|
## Developing
|
|
117
103
|
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
# Tablassert
|
|
2
|
+
|
|
3
|
+
[](https://pypi.org/project/tablassert/)
|
|
4
|
+
[](https://pypi.org/project/tablassert/)
|
|
5
|
+
[](https://github.com/SkyeAv/Tablassert/blob/main/LICENSE)
|
|
6
|
+
[](https://skyeav.github.io/Tablassert/)
|
|
7
|
+
|
|
8
|
+
Extract knowledge assertions from tabular data into NCATS Translator-compliant KGX NDJSON — declaratively, with entity resolution built in and optional quality control.
|
|
9
|
+
|
|
10
|
+
```bash
|
|
11
|
+
pip install tablassert
|
|
12
|
+
tablassert build-kg config.yaml
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
**[Full Documentation](https://skyeav.github.io/Tablassert/)** — installation guides, tutorials, configuration reference, and API docs.
|
|
16
|
+
|
|
17
|
+
## Installation
|
|
18
|
+
|
|
19
|
+
```bash
|
|
20
|
+
pip install tablassert
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
The base install builds knowledge graphs from CSV/TSV/Excel sources. Optional extras (`rt`, `qc`,
|
|
24
|
+
`agent`) add CPU-compatible Polars, the three-stage QC audit, and the autonomous agent — see the
|
|
25
|
+
[Installation guide](https://skyeav.github.io/Tablassert/installation/) for the full matrix. QC is opt-in
|
|
26
|
+
at build time (`build-kg --qc`); see the [CLI Reference](https://skyeav.github.io/Tablassert/cli/) for the
|
|
27
|
+
complete flag reference.
|
|
28
|
+
|
|
29
|
+
## Quick Demo
|
|
30
|
+
|
|
31
|
+
```python
|
|
32
|
+
from pathlib import Path
|
|
33
|
+
from tablassert.lib import resolve_many
|
|
34
|
+
|
|
35
|
+
results = resolve_many(col="gene", entities=["TP53", "BRCA1"], fullmap=Path("/path/to/fullmap"), taxon="9606")
|
|
36
|
+
# [{"original_gene": "TP53", "gene": "HGNC:11998", "gene_name": "TP53", ...}, ...]
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
Point `resolve_many()` at a fullmap database to resolve any iterable of entity strings to CURIEs — no
|
|
40
|
+
LazyFrame setup or NLP preprocessing required. See the
|
|
41
|
+
[Batch Resolution API](https://skyeav.github.io/Tablassert/api/lib/) for the full reference; for
|
|
42
|
+
YAML-configured pipeline builds use `tablassert build-kg config.yaml`.
|
|
43
|
+
|
|
44
|
+
## Key Features
|
|
45
|
+
|
|
46
|
+
Declarative YAML configs, built-in entity resolution, optional three-stage QC, and KGX-compliant NDJSON
|
|
47
|
+
output — with lazy Polars pipelines over an embedded redb resolution database. See the
|
|
48
|
+
[documentation](https://skyeav.github.io/Tablassert/) for the full feature overview and use-case gallery.
|
|
49
|
+
|
|
50
|
+
## Developing
|
|
51
|
+
|
|
52
|
+
```bash
|
|
53
|
+
uv sync --group dev --extra qc
|
|
54
|
+
uv run maturin develop --manifest-path rust/Cargo.toml
|
|
55
|
+
make check
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
See **[CONTRIBUTING.md](CONTRIBUTING.md)** for the full development loop, quality gates, and pull request guidelines.
|
|
59
|
+
|
|
60
|
+
## License
|
|
61
|
+
|
|
62
|
+
[Apache License 2.0](LICENSE)
|
|
63
|
+
|
|
64
|
+
## Contributors
|
|
65
|
+
|
|
66
|
+
[Skye Lane Goetz](mailto:sgoetz@isbscience.org) — Institute for Systems Biology
|
|
67
|
+
|
|
68
|
+
[Gwênlyn Glusman](mailto:gglusman@isbscience.org) — Institute for Systems Biology
|
|
69
|
+
|
|
70
|
+
Jared C. Roach — Institute for Systems Biology
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "tablassert"
|
|
3
|
-
version = "8.
|
|
3
|
+
version = "8.1.0"
|
|
4
4
|
description = "Extract knowledge assertions from tabular data into NCATS Translator-compliant KGX NDJSON — declaratively, with entity resolution and quality control built in."
|
|
5
5
|
authors = [
|
|
6
6
|
{ name = "Skye Lane Goetz", email = "sgoetz@isbscience.org" }
|
|
@@ -86,6 +86,7 @@ agent = [
|
|
|
86
86
|
"smolagents>=1.26.0",
|
|
87
87
|
"dspy>=3.2.1",
|
|
88
88
|
"litellm>=1.93.0",
|
|
89
|
+
"pdfminer.six>=20221105",
|
|
89
90
|
]
|
|
90
91
|
|
|
91
92
|
[dependency-groups]
|
|
@@ -98,11 +99,14 @@ dev = [
|
|
|
98
99
|
"pytest>=9.0.2",
|
|
99
100
|
"pytest-cov>=7.1.0",
|
|
100
101
|
"ruff>=0.15.6",
|
|
102
|
+
"pytest-xdist>=3.8.0",
|
|
101
103
|
]
|
|
102
104
|
|
|
103
105
|
[tool.pytest.ini_options]
|
|
104
106
|
testpaths = ["tests"]
|
|
105
|
-
|
|
107
|
+
# Parallel by default via pytest-xdist (~7x faster full suite, identical coverage).
|
|
108
|
+
# Disable for a single serial run with `-n 0` (e.g. debugging one test).
|
|
109
|
+
addopts = "--cov=tablassert --cov-report=term-missing -n auto"
|
|
106
110
|
markers = ["network: requires internet"]
|
|
107
111
|
|
|
108
112
|
[tool.coverage.run]
|
|
@@ -31,12 +31,13 @@ const SCHEMA_VERSION_V3: &str = "tablassert.fullmap.v3";
|
|
|
31
31
|
const SCHEMA_VERSION_V2: &str = "tablassert.fullmap.v2";
|
|
32
32
|
const SCHEMA_VERSION_V1: &str = "tablassert.fullmap.v1";
|
|
33
33
|
const FULLMAP_SOURCE_VERSION: &str = "2026jul22";
|
|
34
|
-
///
|
|
35
|
-
///
|
|
36
|
-
///
|
|
37
|
-
///
|
|
38
|
-
///
|
|
39
|
-
/// `SHARD_COUNT` used by the concurrent
|
|
34
|
+
/// Number of on-disk redb shard files the RECORDS table is hash-partitioned
|
|
35
|
+
/// across. The public build entry point always writes exactly this many shards
|
|
36
|
+
/// (no longer environment-configurable); the read path still honors the
|
|
37
|
+
/// per-database META `shards` value so databases built before the count was
|
|
38
|
+
/// pinned keep working. Must be a power of two so `term_shard` can route with a
|
|
39
|
+
/// bitmask. Distinct from the in-memory `SHARD_COUNT` used by the concurrent
|
|
40
|
+
/// build maps.
|
|
40
41
|
const SHARD_COUNT_SHARDS: usize = 16;
|
|
41
42
|
/// A normalized term grouped with its deduplicated `(curie_id, source_id)` pairs.
|
|
42
43
|
type TermPairs = (String, Vec<(u32, u8)>);
|
|
@@ -92,8 +93,8 @@ const DEFAULT_CHUNK_BYTES: usize = 8 * 1024 * 1024;
|
|
|
92
93
|
|
|
93
94
|
/// Round `n` DOWN to the nearest power of two (6->4, 3->2, 5->4); powers of two
|
|
94
95
|
/// are unchanged and any `n < 1` floors to 1. The shard mask routing in
|
|
95
|
-
/// `term_shard` requires a power-of-two shard count, so the
|
|
96
|
-
/// normalized through this before use.
|
|
96
|
+
/// `term_shard` requires a power-of-two shard count, so the shard count read from
|
|
97
|
+
/// a database's META is normalized through this before use.
|
|
97
98
|
fn round_down_pow2(n: usize) -> usize {
|
|
98
99
|
if n <= 1 {
|
|
99
100
|
return 1;
|
|
@@ -106,15 +107,6 @@ fn round_down_pow2(n: usize) -> usize {
|
|
|
106
107
|
}
|
|
107
108
|
}
|
|
108
109
|
|
|
109
|
-
/// Resolve the on-disk RECORDS shard count from `TABLASSERT_FULLMAP_SHARDS`
|
|
110
|
-
/// (default `SHARD_COUNT_SHARDS`). Non-powers-of-two round DOWN to the nearest
|
|
111
|
-
/// power of two (6->4, 3->2) and the result is clamped to `<= SHARD_COUNT_SHARDS`,
|
|
112
|
-
/// the compile-time cap on shard files (raise the const to allow more).
|
|
113
|
-
fn resolve_shard_count() -> usize {
|
|
114
|
-
let raw = env_usize("TABLASSERT_FULLMAP_SHARDS", SHARD_COUNT_SHARDS);
|
|
115
|
-
round_down_pow2(raw).clamp(1, SHARD_COUNT_SHARDS)
|
|
116
|
-
}
|
|
117
|
-
|
|
118
110
|
fn env_usize(name: &str, default: usize) -> usize {
|
|
119
111
|
std::env::var(name)
|
|
120
112
|
.ok()
|
|
@@ -2245,10 +2237,9 @@ pub fn build_fullmap_db(
|
|
|
2245
2237
|
DEFAULT_REDB_CACHE_BYTES,
|
|
2246
2238
|
);
|
|
2247
2239
|
let insert_batch = env_usize("TABLASSERT_FULLMAP_INSERT_BATCH", DEFAULT_INSERT_BATCH);
|
|
2248
|
-
// On-disk RECORDS shard count (one redb file + concurrent writer per shard)
|
|
2249
|
-
//
|
|
2250
|
-
|
|
2251
|
-
let shard_count = resolve_shard_count();
|
|
2240
|
+
// On-disk RECORDS shard count (one redb file + concurrent writer per shard),
|
|
2241
|
+
// fixed at SHARD_COUNT_SHARDS (16).
|
|
2242
|
+
let shard_count = SHARD_COUNT_SHARDS;
|
|
2252
2243
|
let spill_dir = std::env::var("TABLASSERT_FULLMAP_SPILL_DIR")
|
|
2253
2244
|
.map(PathBuf::from)
|
|
2254
2245
|
.unwrap_or_else(|_| {
|
|
@@ -2338,9 +2329,10 @@ fn open_cached_shard(primary: &Path, index: usize) -> PyResult<Arc<Database>> {
|
|
|
2338
2329
|
}
|
|
2339
2330
|
|
|
2340
2331
|
/// Read the RECORDS shard count advertised in the primary's META, defaulting to
|
|
2341
|
-
/// `SHARD_COUNT_SHARDS` when absent.
|
|
2342
|
-
/// `
|
|
2343
|
-
///
|
|
2332
|
+
/// `SHARD_COUNT_SHARDS` when absent. New builds always record
|
|
2333
|
+
/// `SHARD_COUNT_SHARDS`; reading the value back keeps the read path compatible
|
|
2334
|
+
/// with databases built before the count was pinned, opening exactly the shards
|
|
2335
|
+
/// that exist and routing with the matching mask.
|
|
2344
2336
|
fn shard_count_of(database: &Database) -> PyResult<usize> {
|
|
2345
2337
|
let read = database.begin_read().map_err(py_err)?;
|
|
2346
2338
|
let meta = read.open_table(META).map_err(py_err)?;
|
|
@@ -2349,10 +2341,10 @@ fn shard_count_of(database: &Database) -> PyResult<usize> {
|
|
|
2349
2341
|
.map_err(py_err)?
|
|
2350
2342
|
.and_then(|v| v.value().parse::<usize>().ok())
|
|
2351
2343
|
.unwrap_or(SHARD_COUNT_SHARDS);
|
|
2352
|
-
// Round down to a power of two before clamping
|
|
2353
|
-
// `
|
|
2354
|
-
//
|
|
2355
|
-
//
|
|
2344
|
+
// Round down to a power of two before clamping: the routing mask
|
|
2345
|
+
// `xxh64(term) & (count - 1)` is only correct for powers of two, so a legacy
|
|
2346
|
+
// or hand-edited non-pow2 META.shards (e.g. 3 -> mask &2) would silently
|
|
2347
|
+
// misroute/drop lookups onto a subset of shards.
|
|
2356
2348
|
// `round_down_pow2(SHARD_COUNT_SHARDS) == SHARD_COUNT_SHARDS` (16 is a pow2),
|
|
2357
2349
|
// so the default fallback above is preserved.
|
|
2358
2350
|
Ok(round_down_pow2(count).clamp(1, SHARD_COUNT_SHARDS))
|
|
@@ -2958,10 +2950,9 @@ mod tests {
|
|
|
2958
2950
|
.collect()
|
|
2959
2951
|
}
|
|
2960
2952
|
|
|
2961
|
-
///
|
|
2962
|
-
///
|
|
2963
|
-
///
|
|
2964
|
-
/// count is always a valid non-zero mask.
|
|
2953
|
+
/// Mask routing only works for powers of two; `round_down_pow2` must round
|
|
2954
|
+
/// 6->4, 3->2, 5->4, leave powers unchanged, and floor tiny/zero inputs at 1
|
|
2955
|
+
/// so a legacy META shard count is always a valid non-zero mask.
|
|
2965
2956
|
#[test]
|
|
2966
2957
|
fn round_down_pow2_rounds_to_nearest_lower_power_of_two() {
|
|
2967
2958
|
assert_eq!(round_down_pow2(0), 1);
|
|
@@ -2977,16 +2968,15 @@ mod tests {
|
|
|
2977
2968
|
}
|
|
2978
2969
|
|
|
2979
2970
|
/// The read path's `shard_count_of` must round a non-power-of-two META.shards
|
|
2980
|
-
/// DOWN to a power of two
|
|
2981
|
-
///
|
|
2982
|
-
///
|
|
2983
|
-
///
|
|
2984
|
-
///
|
|
2985
|
-
///
|
|
2986
|
-
///
|
|
2987
|
-
///
|
|
2988
|
-
///
|
|
2989
|
-
/// (`round_down_pow2(0) == 1`); 1 is still a valid mask (&0), so it is safe.
|
|
2971
|
+
/// DOWN to a power of two. WHY: lookups route with the mask
|
|
2972
|
+
/// `xxh64(term) & (count - 1)`, which is only correct for powers of two; a
|
|
2973
|
+
/// legacy or hand-edited META.shards of 3 (mask &2) would route terms only onto
|
|
2974
|
+
/// shards {0,2}, silently misrouting/dropping lookups (shard 1 opened but never
|
|
2975
|
+
/// queried, shard 3 never opened). Rounding down keeps the mask valid. A
|
|
2976
|
+
/// missing/unparseable value falls back to the default `SHARD_COUNT_SHARDS`
|
|
2977
|
+
/// (16) via `unwrap_or`, itself a power of two so the round-down leaves it
|
|
2978
|
+
/// unchanged. `"0"` PARSES (so the fallback does not fire) and rounds down to
|
|
2979
|
+
/// 1 (`round_down_pow2(0) == 1`); 1 is still a valid mask (&0), so it is safe.
|
|
2990
2980
|
#[test]
|
|
2991
2981
|
fn shard_count_of_rounds_non_pow2_meta_down() {
|
|
2992
2982
|
pyo3::Python::initialize();
|
|
@@ -3022,8 +3012,8 @@ mod tests {
|
|
|
3022
3012
|
// Missing / unparseable -> default SHARD_COUNT_SHARDS (16) via unwrap_or.
|
|
3023
3013
|
assert_eq!(check(None), SHARD_COUNT_SHARDS);
|
|
3024
3014
|
assert_eq!(check(Some("not-a-number")), SHARD_COUNT_SHARDS);
|
|
3025
|
-
// "0" parses (fallback does NOT fire) and rounds down to 1
|
|
3026
|
-
//
|
|
3015
|
+
// "0" parses (fallback does NOT fire) and rounds down to 1; 1 is a valid
|
|
3016
|
+
// mask, so this is safe, not a misroute.
|
|
3027
3017
|
assert_eq!(check(Some("0")), 1);
|
|
3028
3018
|
}
|
|
3029
3019
|
|
|
@@ -3477,12 +3467,13 @@ mod tests {
|
|
|
3477
3467
|
);
|
|
3478
3468
|
}
|
|
3479
3469
|
|
|
3480
|
-
///
|
|
3481
|
-
///
|
|
3482
|
-
///
|
|
3483
|
-
///
|
|
3484
|
-
///
|
|
3485
|
-
///
|
|
3470
|
+
/// The build and read paths agree on a parameterized shard count. A 2-shard
|
|
3471
|
+
/// build must write META.shards="2", create exactly s0+s1 (each with a RECORDS
|
|
3472
|
+
/// table, even the one that receives no terms), create no higher shard files up
|
|
3473
|
+
/// to the compile-time cap, and the read path must open exactly 2 shards (from
|
|
3474
|
+
/// META) and still resolve every term. The public entry point pins the count
|
|
3475
|
+
/// to `SHARD_COUNT_SHARDS`, but this exercises the same parameterized path the
|
|
3476
|
+
/// read path relies on for databases built with fewer shards.
|
|
3486
3477
|
#[test]
|
|
3487
3478
|
fn runtime_shard_count_builds_and_reads_fewer_shards() {
|
|
3488
3479
|
pyo3::Python::initialize();
|
|
@@ -3523,7 +3514,7 @@ mod tests {
|
|
|
3523
3514
|
)
|
|
3524
3515
|
.unwrap();
|
|
3525
3516
|
|
|
3526
|
-
// META.shards reflects the
|
|
3517
|
+
// META.shards reflects the parameterized shard count.
|
|
3527
3518
|
let database = open_cached(output.clone()).unwrap();
|
|
3528
3519
|
let read = database.begin_read().unwrap();
|
|
3529
3520
|
let meta = read.open_table(META).unwrap();
|
|
@@ -3556,6 +3547,67 @@ mod tests {
|
|
|
3556
3547
|
assert_eq!(rows.len(), 50);
|
|
3557
3548
|
}
|
|
3558
3549
|
|
|
3550
|
+
/// The public entry point ignores `TABLASSERT_FULLMAP_SHARDS`: even when it is
|
|
3551
|
+
/// set to a smaller value, `build_fullmap_db` writes exactly
|
|
3552
|
+
/// `SHARD_COUNT_SHARDS` (16) shard files and records META.shards="16". This is
|
|
3553
|
+
/// the regression guard for removing the env-var tunable — the layout tests
|
|
3554
|
+
/// above call `build_fullmap_inner` / `build_test` directly and so cannot catch
|
|
3555
|
+
/// a public path that re-reads the variable. Setting the variable here is safe
|
|
3556
|
+
/// across concurrent tests: the build no longer reads it (its only reader,
|
|
3557
|
+
/// `resolve_shard_count`, was removed), and it is removed again before any
|
|
3558
|
+
/// assertion can panic.
|
|
3559
|
+
#[test]
|
|
3560
|
+
fn public_build_ignores_shards_env_var() {
|
|
3561
|
+
pyo3::Python::initialize();
|
|
3562
|
+
let dir = tempfile::tempdir().unwrap();
|
|
3563
|
+
let synonyms = dir.path().join("HGNC.ndjson");
|
|
3564
|
+
let output = dir.path().join("fullmap.redb");
|
|
3565
|
+
let mut synonym_file = File::create(&synonyms).unwrap();
|
|
3566
|
+
for i in 0..50 {
|
|
3567
|
+
writeln!(
|
|
3568
|
+
synonym_file,
|
|
3569
|
+
r#"{{"curie":"HGNC:{i}","preferred_name":"GENE{i}","names":["GENE{i}"],"types":["Gene"],"taxa":["NCBITaxon:9606"]}}"#
|
|
3570
|
+
)
|
|
3571
|
+
.unwrap();
|
|
3572
|
+
}
|
|
3573
|
+
drop(synonym_file);
|
|
3574
|
+
|
|
3575
|
+
std::env::set_var("TABLASSERT_FULLMAP_SHARDS", "2");
|
|
3576
|
+
let built = Python::attach(|py| {
|
|
3577
|
+
build_fullmap_db(
|
|
3578
|
+
py,
|
|
3579
|
+
output.clone(),
|
|
3580
|
+
Vec::new(),
|
|
3581
|
+
vec![synonyms],
|
|
3582
|
+
Some(2),
|
|
3583
|
+
None,
|
|
3584
|
+
)
|
|
3585
|
+
});
|
|
3586
|
+
std::env::remove_var("TABLASSERT_FULLMAP_SHARDS");
|
|
3587
|
+
built.unwrap();
|
|
3588
|
+
|
|
3589
|
+
// The env var is ignored: META advertises 16 shards, all 16 sibling shard
|
|
3590
|
+
// files exist (s0..s15), and there is no s16.
|
|
3591
|
+
let database = open_cached(output.clone()).unwrap();
|
|
3592
|
+
let read = database.begin_read().unwrap();
|
|
3593
|
+
let meta = read.open_table(META).unwrap();
|
|
3594
|
+
assert_eq!(meta.get("shards").unwrap().unwrap().value(), "16");
|
|
3595
|
+
drop(meta);
|
|
3596
|
+
drop(read);
|
|
3597
|
+
for index in 0..SHARD_COUNT_SHARDS {
|
|
3598
|
+
assert!(
|
|
3599
|
+
shard_path(&output, index).exists(),
|
|
3600
|
+
"missing shard file {index}"
|
|
3601
|
+
);
|
|
3602
|
+
}
|
|
3603
|
+
assert!(!shard_path(&output, SHARD_COUNT_SHARDS).exists());
|
|
3604
|
+
|
|
3605
|
+
// Lookups still resolve across all 16 shards.
|
|
3606
|
+
let terms: Vec<String> = (0..50).map(|i| format!("gene{i}")).collect();
|
|
3607
|
+
let rows = lookup_terms(output, terms, Some(4)).unwrap();
|
|
3608
|
+
assert_eq!(rows.len(), 50);
|
|
3609
|
+
}
|
|
3610
|
+
|
|
3559
3611
|
/// Backward compatibility: databases built/advertising the former 4-shard
|
|
3560
3612
|
/// layout must continue to open exactly those four shard files and route terms
|
|
3561
3613
|
/// with a 4-shard mask, even though new default builds now write 16 shards.
|