tablassert 8.0.0__tar.gz → 8.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. {tablassert-8.0.0 → tablassert-8.1.0}/PKG-INFO +16 -30
  2. tablassert-8.1.0/README.md +70 -0
  3. {tablassert-8.0.0 → tablassert-8.1.0}/pyproject.toml +6 -2
  4. {tablassert-8.0.0 → tablassert-8.1.0}/rust/src/fullmap.rs +103 -51
  5. {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/agent.py +1055 -210
  6. {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/cli.py +159 -66
  7. {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/errors.py +0 -1
  8. {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/fullmap.py +41 -7
  9. {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/lib.py +1 -1
  10. {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/models.py +4 -13
  11. {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/qc.py +2 -2
  12. tablassert-8.0.0/README.md +0 -85
  13. {tablassert-8.0.0 → tablassert-8.1.0}/LICENSE +0 -0
  14. {tablassert-8.0.0 → tablassert-8.1.0}/rust/Cargo.lock +0 -0
  15. {tablassert-8.0.0 → tablassert-8.1.0}/rust/Cargo.toml +0 -0
  16. {tablassert-8.0.0 → tablassert-8.1.0}/rust/examples/count_tables.rs +0 -0
  17. {tablassert-8.0.0 → tablassert-8.1.0}/rust/src/json.rs +0 -0
  18. {tablassert-8.0.0 → tablassert-8.1.0}/rust/src/lib.rs +0 -0
  19. {tablassert-8.0.0 → tablassert-8.1.0}/rust/src/ndjson.rs +0 -0
  20. {tablassert-8.0.0 → tablassert-8.1.0}/rust/src/uuid.rs +0 -0
  21. {tablassert-8.0.0 → tablassert-8.1.0}/rust/tests/build_golden.rs +0 -0
  22. {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/__init__.py +0 -0
  23. {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/_lazy.py +0 -0
  24. {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/biolink.py +0 -0
  25. {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/coerce.py +0 -0
  26. {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/enums.py +0 -0
  27. {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/ingests.py +0 -0
  28. {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/log.py +0 -0
  29. {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/nlp.py +0 -0
  30. {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/progress.py +0 -0
  31. {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/rig.py +0 -0
  32. {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/rs.pyi +0 -0
  33. {tablassert-8.0.0 → tablassert-8.1.0}/src/tablassert/utils.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: tablassert
3
- Version: 8.0.0
3
+ Version: 8.1.0
4
4
  Classifier: License :: OSI Approved :: Apache Software License
5
5
  Classifier: Development Status :: 5 - Production/Stable
6
6
  Classifier: Intended Audience :: Science/Research
@@ -32,6 +32,7 @@ Requires-Dist: fastexcel>=0.20.2
32
32
  Requires-Dist: smolagents>=1.26.0 ; extra == 'agent'
33
33
  Requires-Dist: dspy>=3.2.1 ; extra == 'agent'
34
34
  Requires-Dist: litellm>=1.93.0 ; extra == 'agent'
35
+ Requires-Dist: pdfminer-six>=20221105 ; extra == 'agent'
35
36
  Requires-Dist: scikit-learn>=1.8.0 ; extra == 'qc'
36
37
  Requires-Dist: sentence-transformers>=5.3.0 ; extra == 'qc'
37
38
  Requires-Dist: polars[rtcompat]>=1.40.1 ; extra == 'rt'
@@ -71,16 +72,11 @@ tablassert build-kg config.yaml
71
72
  pip install tablassert
72
73
  ```
73
74
 
74
- The base install includes everything needed to build knowledge graphs from CSV/TSV sources. Optional extras are available for CPU compatibility and quality control:
75
-
76
- ```bash
77
- pip install "tablassert[rt]" # Polars build for CPUs without the required instructions
78
- pip install "tablassert[qc]" # Enable QC (torch + sentence-transformers BioBERT, scikit-learn)
79
- ```
80
-
81
- Excel (`.xlsx`) inputs are read through Polars' `calamine` engine and additionally require `python-calamine` (`pip install python-calamine`).
82
-
83
- QC is opt-in: pass `--qc` to `build-kg` to run the three-stage audit (exact → fuzzy → BioBERT). See the [CLI Reference](https://skyeav.github.io/Tablassert/cli/) for the full flag reference.
75
+ The base install builds knowledge graphs from CSV/TSV/Excel sources. Optional extras (`rt`, `qc`,
76
+ `agent`) add CPU-compatible Polars, the three-stage QC audit, and the autonomous agent — see the
77
+ [Installation guide](https://skyeav.github.io/Tablassert/installation/) for the full matrix. QC is opt-in
78
+ at build time (`build-kg --qc`); see the [CLI Reference](https://skyeav.github.io/Tablassert/cli/) for the
79
+ complete flag reference.
84
80
 
85
81
  ## Quick Demo
86
82
 
@@ -88,30 +84,20 @@ QC is opt-in: pass `--qc` to `build-kg` to run the three-stage audit (exact →
88
84
  from pathlib import Path
89
85
  from tablassert.lib import resolve_many
90
86
 
91
- # Resolve gene names to CURIEs against a fullmap database
92
- results = resolve_many(
93
- col="gene",
94
- entities=["TP53", "BRCA1", "EGFR"],
95
- fullmap=Path("/path/to/fullmap"),
96
- taxon="9606",
97
- )
98
-
99
- for row in results:
100
- print(f"{row['original_gene']} → {row['gene']} ({row['gene_name']})")
101
- # TP53 → HGNC:11998 (TP53)
102
- # BRCA1 → HGNC:1100 (BRCA1)
103
- # EGFR → HGNC:3236 (EGFR)
87
+ results = resolve_many(col="gene", entities=["TP53", "BRCA1"], fullmap=Path("/path/to/fullmap"), taxon="9606")
88
+ # [{"original_gene": "TP53", "gene": "HGNC:11998", "gene_name": "TP53", ...}, ...]
104
89
  ```
105
90
 
106
- Point `resolve_many()` at a fullmap database and resolve any iterable of entity strings to CURIEs — no LazyFrame setup or NLP preprocessing required. For full pipeline builds with YAML configuration, use `tablassert build-kg config.yaml`.
91
+ Point `resolve_many()` at a fullmap database to resolve any iterable of entity strings to CURIEs — no
92
+ LazyFrame setup or NLP preprocessing required. See the
93
+ [Batch Resolution API](https://skyeav.github.io/Tablassert/api/lib/) for the full reference; for
94
+ YAML-configured pipeline builds use `tablassert build-kg config.yaml`.
107
95
 
108
96
  ## Key Features
109
97
 
110
- - **Declarative Configuration** — YAML-based, no code required
111
- - **Entity Resolution** — Maps text to biological entities (genes, diseases, chemicals)
112
- - **Quality Control** — Optional three-stage validation (exact → fuzzy → BERT embeddings)
113
- - **KGX Compliance** — NCATS Translator-compatible NDJSON output
114
- - **Performance** — Lazy evaluation pipelines with Polars and an embedded redb-accelerated entity resolution database
98
+ Declarative YAML configs, built-in entity resolution, optional three-stage QC, and KGX-compliant NDJSON
99
+ output — with lazy Polars pipelines over an embedded redb resolution database. See the
100
+ [documentation](https://skyeav.github.io/Tablassert/) for the full feature overview and use-case gallery.
115
101
 
116
102
  ## Developing
117
103
 
@@ -0,0 +1,70 @@
1
+ # Tablassert
2
+
3
+ [![PyPI](https://img.shields.io/pypi/v/tablassert.svg)](https://pypi.org/project/tablassert/)
4
+ [![Python](https://img.shields.io/pypi/pyversions/tablassert.svg)](https://pypi.org/project/tablassert/)
5
+ [![License](https://img.shields.io/pypi/l/tablassert.svg)](https://github.com/SkyeAv/Tablassert/blob/main/LICENSE)
6
+ [![Docs](https://img.shields.io/github/deployments/SkyeAv/Tablassert/github-pages?label=docs)](https://skyeav.github.io/Tablassert/)
7
+
8
+ Extract knowledge assertions from tabular data into NCATS Translator-compliant KGX NDJSON — declaratively, with entity resolution built in and optional quality control.
9
+
10
+ ```bash
11
+ pip install tablassert
12
+ tablassert build-kg config.yaml
13
+ ```
14
+
15
+ **[Full Documentation](https://skyeav.github.io/Tablassert/)** — installation guides, tutorials, configuration reference, and API docs.
16
+
17
+ ## Installation
18
+
19
+ ```bash
20
+ pip install tablassert
21
+ ```
22
+
23
+ The base install builds knowledge graphs from CSV/TSV/Excel sources. Optional extras (`rt`, `qc`,
24
+ `agent`) add CPU-compatible Polars, the three-stage QC audit, and the autonomous agent — see the
25
+ [Installation guide](https://skyeav.github.io/Tablassert/installation/) for the full matrix. QC is opt-in
26
+ at build time (`build-kg --qc`); see the [CLI Reference](https://skyeav.github.io/Tablassert/cli/) for the
27
+ complete flag reference.
28
+
29
+ ## Quick Demo
30
+
31
+ ```python
32
+ from pathlib import Path
33
+ from tablassert.lib import resolve_many
34
+
35
+ results = resolve_many(col="gene", entities=["TP53", "BRCA1"], fullmap=Path("/path/to/fullmap"), taxon="9606")
36
+ # [{"original_gene": "TP53", "gene": "HGNC:11998", "gene_name": "TP53", ...}, ...]
37
+ ```
38
+
39
+ Point `resolve_many()` at a fullmap database to resolve any iterable of entity strings to CURIEs — no
40
+ LazyFrame setup or NLP preprocessing required. See the
41
+ [Batch Resolution API](https://skyeav.github.io/Tablassert/api/lib/) for the full reference; for
42
+ YAML-configured pipeline builds use `tablassert build-kg config.yaml`.
43
+
44
+ ## Key Features
45
+
46
+ Declarative YAML configs, built-in entity resolution, optional three-stage QC, and KGX-compliant NDJSON
47
+ output — with lazy Polars pipelines over an embedded redb resolution database. See the
48
+ [documentation](https://skyeav.github.io/Tablassert/) for the full feature overview and use-case gallery.
49
+
50
+ ## Developing
51
+
52
+ ```bash
53
+ uv sync --group dev --extra qc
54
+ uv run maturin develop --manifest-path rust/Cargo.toml
55
+ make check
56
+ ```
57
+
58
+ See **[CONTRIBUTING.md](CONTRIBUTING.md)** for the full development loop, quality gates, and pull request guidelines.
59
+
60
+ ## License
61
+
62
+ [Apache License 2.0](LICENSE)
63
+
64
+ ## Contributors
65
+
66
+ [Skye Lane Goetz](mailto:sgoetz@isbscience.org) — Institute for Systems Biology
67
+
68
+ [Gwênlyn Glusman](mailto:gglusman@isbscience.org) — Institute for Systems Biology
69
+
70
+ Jared C. Roach — Institute for Systems Biology
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "tablassert"
3
- version = "8.0.0"
3
+ version = "8.1.0"
4
4
  description = "Extract knowledge assertions from tabular data into NCATS Translator-compliant KGX NDJSON — declaratively, with entity resolution and quality control built in."
5
5
  authors = [
6
6
  { name = "Skye Lane Goetz", email = "sgoetz@isbscience.org" }
@@ -86,6 +86,7 @@ agent = [
86
86
  "smolagents>=1.26.0",
87
87
  "dspy>=3.2.1",
88
88
  "litellm>=1.93.0",
89
+ "pdfminer.six>=20221105",
89
90
  ]
90
91
 
91
92
  [dependency-groups]
@@ -98,11 +99,14 @@ dev = [
98
99
  "pytest>=9.0.2",
99
100
  "pytest-cov>=7.1.0",
100
101
  "ruff>=0.15.6",
102
+ "pytest-xdist>=3.8.0",
101
103
  ]
102
104
 
103
105
  [tool.pytest.ini_options]
104
106
  testpaths = ["tests"]
105
- addopts = "--cov=tablassert --cov-report=term-missing"
107
+ # Parallel by default via pytest-xdist (~7x faster full suite, identical coverage).
108
+ # Disable for a single serial run with `-n 0` (e.g. debugging one test).
109
+ addopts = "--cov=tablassert --cov-report=term-missing -n auto"
106
110
  markers = ["network: requires internet"]
107
111
 
108
112
  [tool.coverage.run]
@@ -31,12 +31,13 @@ const SCHEMA_VERSION_V3: &str = "tablassert.fullmap.v3";
31
31
  const SCHEMA_VERSION_V2: &str = "tablassert.fullmap.v2";
32
32
  const SCHEMA_VERSION_V1: &str = "tablassert.fullmap.v1";
33
33
  const FULLMAP_SOURCE_VERSION: &str = "2026jul22";
34
- /// Compile-time maximum number of on-disk redb shard files the RECORDS table is
35
- /// hash-partitioned across, and the default when `TABLASSERT_FULLMAP_SHARDS` is
36
- /// unset. Must be a power of two so `term_shard` can route with a bitmask.
37
- /// The runtime shard count (`resolve_shard_count`) is clamped to this cap; raise
38
- /// this const to allow more shard files. Distinct from the in-memory
39
- /// `SHARD_COUNT` used by the concurrent build maps.
34
+ /// Number of on-disk redb shard files the RECORDS table is hash-partitioned
35
+ /// across. The public build entry point always writes exactly this many shards
36
+ /// (no longer environment-configurable); the read path still honors the
37
+ /// per-database META `shards` value so databases built before the count was
38
+ /// pinned keep working. Must be a power of two so `term_shard` can route with a
39
+ /// bitmask. Distinct from the in-memory `SHARD_COUNT` used by the concurrent
40
+ /// build maps.
40
41
  const SHARD_COUNT_SHARDS: usize = 16;
41
42
  /// A normalized term grouped with its deduplicated `(curie_id, source_id)` pairs.
42
43
  type TermPairs = (String, Vec<(u32, u8)>);
@@ -92,8 +93,8 @@ const DEFAULT_CHUNK_BYTES: usize = 8 * 1024 * 1024;
92
93
 
93
94
  /// Round `n` DOWN to the nearest power of two (6->4, 3->2, 5->4); powers of two
94
95
  /// are unchanged and any `n < 1` floors to 1. The shard mask routing in
95
- /// `term_shard` requires a power-of-two shard count, so the runtime tunable is
96
- /// normalized through this before use.
96
+ /// `term_shard` requires a power-of-two shard count, so the shard count read from
97
+ /// a database's META is normalized through this before use.
97
98
  fn round_down_pow2(n: usize) -> usize {
98
99
  if n <= 1 {
99
100
  return 1;
@@ -106,15 +107,6 @@ fn round_down_pow2(n: usize) -> usize {
106
107
  }
107
108
  }
108
109
 
109
- /// Resolve the on-disk RECORDS shard count from `TABLASSERT_FULLMAP_SHARDS`
110
- /// (default `SHARD_COUNT_SHARDS`). Non-powers-of-two round DOWN to the nearest
111
- /// power of two (6->4, 3->2) and the result is clamped to `<= SHARD_COUNT_SHARDS`,
112
- /// the compile-time cap on shard files (raise the const to allow more).
113
- fn resolve_shard_count() -> usize {
114
- let raw = env_usize("TABLASSERT_FULLMAP_SHARDS", SHARD_COUNT_SHARDS);
115
- round_down_pow2(raw).clamp(1, SHARD_COUNT_SHARDS)
116
- }
117
-
118
110
  fn env_usize(name: &str, default: usize) -> usize {
119
111
  std::env::var(name)
120
112
  .ok()
@@ -2245,10 +2237,9 @@ pub fn build_fullmap_db(
2245
2237
  DEFAULT_REDB_CACHE_BYTES,
2246
2238
  );
2247
2239
  let insert_batch = env_usize("TABLASSERT_FULLMAP_INSERT_BATCH", DEFAULT_INSERT_BATCH);
2248
- // On-disk RECORDS shard count (one redb file + concurrent writer per shard).
2249
- // Defaults to SHARD_COUNT_SHARDS (16); non-powers-of-two round down,
2250
- // clamped to SHARD_COUNT_SHARDS.
2251
- let shard_count = resolve_shard_count();
2240
+ // On-disk RECORDS shard count (one redb file + concurrent writer per shard),
2241
+ // fixed at SHARD_COUNT_SHARDS (16).
2242
+ let shard_count = SHARD_COUNT_SHARDS;
2252
2243
  let spill_dir = std::env::var("TABLASSERT_FULLMAP_SPILL_DIR")
2253
2244
  .map(PathBuf::from)
2254
2245
  .unwrap_or_else(|_| {
@@ -2338,9 +2329,10 @@ fn open_cached_shard(primary: &Path, index: usize) -> PyResult<Arc<Database>> {
2338
2329
  }
2339
2330
 
2340
2331
  /// Read the RECORDS shard count advertised in the primary's META, defaulting to
2341
- /// `SHARD_COUNT_SHARDS` when absent. The write path sets this from the runtime
2342
- /// `TABLASSERT_FULLMAP_SHARDS` tunable, so the read path opens exactly the shards
2343
- /// that exist and routes with the matching mask.
2332
+ /// `SHARD_COUNT_SHARDS` when absent. New builds always record
2333
+ /// `SHARD_COUNT_SHARDS`; reading the value back keeps the read path compatible
2334
+ /// with databases built before the count was pinned, opening exactly the shards
2335
+ /// that exist and routing with the matching mask.
2344
2336
  fn shard_count_of(database: &Database) -> PyResult<usize> {
2345
2337
  let read = database.begin_read().map_err(py_err)?;
2346
2338
  let meta = read.open_table(META).map_err(py_err)?;
@@ -2349,10 +2341,10 @@ fn shard_count_of(database: &Database) -> PyResult<usize> {
2349
2341
  .map_err(py_err)?
2350
2342
  .and_then(|v| v.value().parse::<usize>().ok())
2351
2343
  .unwrap_or(SHARD_COUNT_SHARDS);
2352
- // Round down to a power of two before clamping, mirroring the write path's
2353
- // `resolve_shard_count`: the routing mask `xxh64(term) & (count - 1)` is only
2354
- // correct for powers of two, so a hand-edited non-pow2 META.shards (e.g. 3 ->
2355
- // mask &2) would silently misroute/drop lookups onto a subset of shards.
2344
+ // Round down to a power of two before clamping: the routing mask
2345
+ // `xxh64(term) & (count - 1)` is only correct for powers of two, so a legacy
2346
+ // or hand-edited non-pow2 META.shards (e.g. 3 -> mask &2) would silently
2347
+ // misroute/drop lookups onto a subset of shards.
2356
2348
  // `round_down_pow2(SHARD_COUNT_SHARDS) == SHARD_COUNT_SHARDS` (16 is a pow2),
2357
2349
  // so the default fallback above is preserved.
2358
2350
  Ok(round_down_pow2(count).clamp(1, SHARD_COUNT_SHARDS))
@@ -2958,10 +2950,9 @@ mod tests {
2958
2950
  .collect()
2959
2951
  }
2960
2952
 
2961
- /// `TABLASSERT_FULLMAP_SHARDS` may be any positive integer, but mask routing
2962
- /// only works for powers of two; `round_down_pow2` must round 6->4, 3->2,
2963
- /// 5->4, leave powers unchanged, and floor tiny/zero inputs at 1 so the shard
2964
- /// count is always a valid non-zero mask.
2953
+ /// Mask routing only works for powers of two; `round_down_pow2` must round
2954
+ /// 6->4, 3->2, 5->4, leave powers unchanged, and floor tiny/zero inputs at 1
2955
+ /// so a legacy META shard count is always a valid non-zero mask.
2965
2956
  #[test]
2966
2957
  fn round_down_pow2_rounds_to_nearest_lower_power_of_two() {
2967
2958
  assert_eq!(round_down_pow2(0), 1);
@@ -2977,16 +2968,15 @@ mod tests {
2977
2968
  }
2978
2969
 
2979
2970
  /// The read path's `shard_count_of` must round a non-power-of-two META.shards
2980
- /// DOWN to a power of two, exactly like the write path's `resolve_shard_count`.
2981
- /// WHY: lookups route with the mask `xxh64(term) & (count - 1)`, which is only
2982
- /// correct for powers of two; a hand-edited META.shards of 3 (mask &2) would
2983
- /// route terms only onto shards {0,2}, silently misrouting/dropping lookups
2984
- /// (shard 1 opened but never queried, shard 3 never opened). Rounding down
2985
- /// keeps the mask valid. A missing/unparseable value falls back to the default
2986
- /// `SHARD_COUNT_SHARDS` (16) via `unwrap_or`, itself a power of two so the
2987
- /// round-down leaves it unchanged. `"0"` PARSES (so the fallback does not fire)
2988
- /// and rounds down to 1, exactly mirroring the write path's `resolve_shard_count`
2989
- /// (`round_down_pow2(0) == 1`); 1 is still a valid mask (&0), so it is safe.
2971
+ /// DOWN to a power of two. WHY: lookups route with the mask
2972
+ /// `xxh64(term) & (count - 1)`, which is only correct for powers of two; a
2973
+ /// legacy or hand-edited META.shards of 3 (mask &2) would route terms only onto
2974
+ /// shards {0,2}, silently misrouting/dropping lookups (shard 1 opened but never
2975
+ /// queried, shard 3 never opened). Rounding down keeps the mask valid. A
2976
+ /// missing/unparseable value falls back to the default `SHARD_COUNT_SHARDS`
2977
+ /// (16) via `unwrap_or`, itself a power of two so the round-down leaves it
2978
+ /// unchanged. `"0"` PARSES (so the fallback does not fire) and rounds down to
2979
+ /// 1 (`round_down_pow2(0) == 1`); 1 is still a valid mask (&0), so it is safe.
2990
2980
  #[test]
2991
2981
  fn shard_count_of_rounds_non_pow2_meta_down() {
2992
2982
  pyo3::Python::initialize();
@@ -3022,8 +3012,8 @@ mod tests {
3022
3012
  // Missing / unparseable -> default SHARD_COUNT_SHARDS (16) via unwrap_or.
3023
3013
  assert_eq!(check(None), SHARD_COUNT_SHARDS);
3024
3014
  assert_eq!(check(Some("not-a-number")), SHARD_COUNT_SHARDS);
3025
- // "0" parses (fallback does NOT fire) and rounds down to 1, mirroring the
3026
- // write path; 1 is a valid mask, so this is safe, not a misroute.
3015
+ // "0" parses (fallback does NOT fire) and rounds down to 1; 1 is a valid
3016
+ // mask, so this is safe, not a misroute.
3027
3017
  assert_eq!(check(Some("0")), 1);
3028
3018
  }
3029
3019
 
@@ -3477,12 +3467,13 @@ mod tests {
3477
3467
  );
3478
3468
  }
3479
3469
 
3480
- /// `TABLASSERT_FULLMAP_SHARDS` makes the shard count runtime-configurable. A
3481
- /// 2-shard build must write META.shards="2", create exactly s0+s1 (each with a
3482
- /// RECORDS table, even the one that receives no terms), create no higher shard
3483
- /// files up to the compile-time cap, and the read path must open exactly 2
3484
- /// shards (from META) and still resolve every term — proving writer and reader
3485
- /// agree on the runtime shard mask.
3470
+ /// The build and read paths agree on a parameterized shard count. A 2-shard
3471
+ /// build must write META.shards="2", create exactly s0+s1 (each with a RECORDS
3472
+ /// table, even the one that receives no terms), create no higher shard files up
3473
+ /// to the compile-time cap, and the read path must open exactly 2 shards (from
3474
+ /// META) and still resolve every term. The public entry point pins the count
3475
+ /// to `SHARD_COUNT_SHARDS`, but this exercises the same parameterized path the
3476
+ /// read path relies on for databases built with fewer shards.
3486
3477
  #[test]
3487
3478
  fn runtime_shard_count_builds_and_reads_fewer_shards() {
3488
3479
  pyo3::Python::initialize();
@@ -3523,7 +3514,7 @@ mod tests {
3523
3514
  )
3524
3515
  .unwrap();
3525
3516
 
3526
- // META.shards reflects the runtime count.
3517
+ // META.shards reflects the parameterized shard count.
3527
3518
  let database = open_cached(output.clone()).unwrap();
3528
3519
  let read = database.begin_read().unwrap();
3529
3520
  let meta = read.open_table(META).unwrap();
@@ -3556,6 +3547,67 @@ mod tests {
3556
3547
  assert_eq!(rows.len(), 50);
3557
3548
  }
3558
3549
 
3550
+ /// The public entry point ignores `TABLASSERT_FULLMAP_SHARDS`: even when it is
3551
+ /// set to a smaller value, `build_fullmap_db` writes exactly
3552
+ /// `SHARD_COUNT_SHARDS` (16) shard files and records META.shards="16". This is
3553
+ /// the regression guard for removing the env-var tunable — the layout tests
3554
+ /// above call `build_fullmap_inner` / `build_test` directly and so cannot catch
3555
+ /// a public path that re-reads the variable. Setting the variable here is safe
3556
+ /// across concurrent tests: the build no longer reads it (its only reader,
3557
+ /// `resolve_shard_count`, was removed), and it is removed again before any
3558
+ /// assertion can panic.
3559
+ #[test]
3560
+ fn public_build_ignores_shards_env_var() {
3561
+ pyo3::Python::initialize();
3562
+ let dir = tempfile::tempdir().unwrap();
3563
+ let synonyms = dir.path().join("HGNC.ndjson");
3564
+ let output = dir.path().join("fullmap.redb");
3565
+ let mut synonym_file = File::create(&synonyms).unwrap();
3566
+ for i in 0..50 {
3567
+ writeln!(
3568
+ synonym_file,
3569
+ r#"{{"curie":"HGNC:{i}","preferred_name":"GENE{i}","names":["GENE{i}"],"types":["Gene"],"taxa":["NCBITaxon:9606"]}}"#
3570
+ )
3571
+ .unwrap();
3572
+ }
3573
+ drop(synonym_file);
3574
+
3575
+ std::env::set_var("TABLASSERT_FULLMAP_SHARDS", "2");
3576
+ let built = Python::attach(|py| {
3577
+ build_fullmap_db(
3578
+ py,
3579
+ output.clone(),
3580
+ Vec::new(),
3581
+ vec![synonyms],
3582
+ Some(2),
3583
+ None,
3584
+ )
3585
+ });
3586
+ std::env::remove_var("TABLASSERT_FULLMAP_SHARDS");
3587
+ built.unwrap();
3588
+
3589
+ // The env var is ignored: META advertises 16 shards, all 16 sibling shard
3590
+ // files exist (s0..s15), and there is no s16.
3591
+ let database = open_cached(output.clone()).unwrap();
3592
+ let read = database.begin_read().unwrap();
3593
+ let meta = read.open_table(META).unwrap();
3594
+ assert_eq!(meta.get("shards").unwrap().unwrap().value(), "16");
3595
+ drop(meta);
3596
+ drop(read);
3597
+ for index in 0..SHARD_COUNT_SHARDS {
3598
+ assert!(
3599
+ shard_path(&output, index).exists(),
3600
+ "missing shard file {index}"
3601
+ );
3602
+ }
3603
+ assert!(!shard_path(&output, SHARD_COUNT_SHARDS).exists());
3604
+
3605
+ // Lookups still resolve across all 16 shards.
3606
+ let terms: Vec<String> = (0..50).map(|i| format!("gene{i}")).collect();
3607
+ let rows = lookup_terms(output, terms, Some(4)).unwrap();
3608
+ assert_eq!(rows.len(), 50);
3609
+ }
3610
+
3559
3611
  /// Backward compatibility: databases built/advertising the former 4-shard
3560
3612
  /// layout must continue to open exactly those four shard files and route terms
3561
3613
  /// with a 4-shard mask, even though new default builds now write 16 shards.