tablassert 8.0.1__tar.gz → 8.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tablassert-8.0.1 → tablassert-8.1.0}/PKG-INFO +2 -1
- {tablassert-8.0.1 → tablassert-8.1.0}/pyproject.toml +6 -2
- {tablassert-8.0.1 → tablassert-8.1.0}/rust/src/fullmap.rs +103 -51
- {tablassert-8.0.1 → tablassert-8.1.0}/src/tablassert/agent.py +1050 -187
- {tablassert-8.0.1 → tablassert-8.1.0}/src/tablassert/cli.py +156 -53
- {tablassert-8.0.1 → tablassert-8.1.0}/src/tablassert/fullmap.py +41 -7
- {tablassert-8.0.1 → tablassert-8.1.0}/src/tablassert/qc.py +2 -2
- {tablassert-8.0.1 → tablassert-8.1.0}/LICENSE +0 -0
- {tablassert-8.0.1 → tablassert-8.1.0}/README.md +0 -0
- {tablassert-8.0.1 → tablassert-8.1.0}/rust/Cargo.lock +0 -0
- {tablassert-8.0.1 → tablassert-8.1.0}/rust/Cargo.toml +0 -0
- {tablassert-8.0.1 → tablassert-8.1.0}/rust/examples/count_tables.rs +0 -0
- {tablassert-8.0.1 → tablassert-8.1.0}/rust/src/json.rs +0 -0
- {tablassert-8.0.1 → tablassert-8.1.0}/rust/src/lib.rs +0 -0
- {tablassert-8.0.1 → tablassert-8.1.0}/rust/src/ndjson.rs +0 -0
- {tablassert-8.0.1 → tablassert-8.1.0}/rust/src/uuid.rs +0 -0
- {tablassert-8.0.1 → tablassert-8.1.0}/rust/tests/build_golden.rs +0 -0
- {tablassert-8.0.1 → tablassert-8.1.0}/src/tablassert/__init__.py +0 -0
- {tablassert-8.0.1 → tablassert-8.1.0}/src/tablassert/_lazy.py +0 -0
- {tablassert-8.0.1 → tablassert-8.1.0}/src/tablassert/biolink.py +0 -0
- {tablassert-8.0.1 → tablassert-8.1.0}/src/tablassert/coerce.py +0 -0
- {tablassert-8.0.1 → tablassert-8.1.0}/src/tablassert/enums.py +0 -0
- {tablassert-8.0.1 → tablassert-8.1.0}/src/tablassert/errors.py +0 -0
- {tablassert-8.0.1 → tablassert-8.1.0}/src/tablassert/ingests.py +0 -0
- {tablassert-8.0.1 → tablassert-8.1.0}/src/tablassert/lib.py +0 -0
- {tablassert-8.0.1 → tablassert-8.1.0}/src/tablassert/log.py +0 -0
- {tablassert-8.0.1 → tablassert-8.1.0}/src/tablassert/models.py +0 -0
- {tablassert-8.0.1 → tablassert-8.1.0}/src/tablassert/nlp.py +0 -0
- {tablassert-8.0.1 → tablassert-8.1.0}/src/tablassert/progress.py +0 -0
- {tablassert-8.0.1 → tablassert-8.1.0}/src/tablassert/rig.py +0 -0
- {tablassert-8.0.1 → tablassert-8.1.0}/src/tablassert/rs.pyi +0 -0
- {tablassert-8.0.1 → tablassert-8.1.0}/src/tablassert/utils.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: tablassert
|
|
3
|
-
Version: 8.0
|
|
3
|
+
Version: 8.1.0
|
|
4
4
|
Classifier: License :: OSI Approved :: Apache Software License
|
|
5
5
|
Classifier: Development Status :: 5 - Production/Stable
|
|
6
6
|
Classifier: Intended Audience :: Science/Research
|
|
@@ -32,6 +32,7 @@ Requires-Dist: fastexcel>=0.20.2
|
|
|
32
32
|
Requires-Dist: smolagents>=1.26.0 ; extra == 'agent'
|
|
33
33
|
Requires-Dist: dspy>=3.2.1 ; extra == 'agent'
|
|
34
34
|
Requires-Dist: litellm>=1.93.0 ; extra == 'agent'
|
|
35
|
+
Requires-Dist: pdfminer-six>=20221105 ; extra == 'agent'
|
|
35
36
|
Requires-Dist: scikit-learn>=1.8.0 ; extra == 'qc'
|
|
36
37
|
Requires-Dist: sentence-transformers>=5.3.0 ; extra == 'qc'
|
|
37
38
|
Requires-Dist: polars[rtcompat]>=1.40.1 ; extra == 'rt'
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "tablassert"
|
|
3
|
-
version = "8.0
|
|
3
|
+
version = "8.1.0"
|
|
4
4
|
description = "Extract knowledge assertions from tabular data into NCATS Translator-compliant KGX NDJSON — declaratively, with entity resolution and quality control built in."
|
|
5
5
|
authors = [
|
|
6
6
|
{ name = "Skye Lane Goetz", email = "sgoetz@isbscience.org" }
|
|
@@ -86,6 +86,7 @@ agent = [
|
|
|
86
86
|
"smolagents>=1.26.0",
|
|
87
87
|
"dspy>=3.2.1",
|
|
88
88
|
"litellm>=1.93.0",
|
|
89
|
+
"pdfminer.six>=20221105",
|
|
89
90
|
]
|
|
90
91
|
|
|
91
92
|
[dependency-groups]
|
|
@@ -98,11 +99,14 @@ dev = [
|
|
|
98
99
|
"pytest>=9.0.2",
|
|
99
100
|
"pytest-cov>=7.1.0",
|
|
100
101
|
"ruff>=0.15.6",
|
|
102
|
+
"pytest-xdist>=3.8.0",
|
|
101
103
|
]
|
|
102
104
|
|
|
103
105
|
[tool.pytest.ini_options]
|
|
104
106
|
testpaths = ["tests"]
|
|
105
|
-
|
|
107
|
+
# Parallel by default via pytest-xdist (~7x faster full suite, identical coverage).
|
|
108
|
+
# Disable for a single serial run with `-n 0` (e.g. debugging one test).
|
|
109
|
+
addopts = "--cov=tablassert --cov-report=term-missing -n auto"
|
|
106
110
|
markers = ["network: requires internet"]
|
|
107
111
|
|
|
108
112
|
[tool.coverage.run]
|
|
@@ -31,12 +31,13 @@ const SCHEMA_VERSION_V3: &str = "tablassert.fullmap.v3";
|
|
|
31
31
|
const SCHEMA_VERSION_V2: &str = "tablassert.fullmap.v2";
|
|
32
32
|
const SCHEMA_VERSION_V1: &str = "tablassert.fullmap.v1";
|
|
33
33
|
const FULLMAP_SOURCE_VERSION: &str = "2026jul22";
|
|
34
|
-
///
|
|
35
|
-
///
|
|
36
|
-
///
|
|
37
|
-
///
|
|
38
|
-
///
|
|
39
|
-
/// `SHARD_COUNT` used by the concurrent
|
|
34
|
+
/// Number of on-disk redb shard files the RECORDS table is hash-partitioned
|
|
35
|
+
/// across. The public build entry point always writes exactly this many shards
|
|
36
|
+
/// (no longer environment-configurable); the read path still honors the
|
|
37
|
+
/// per-database META `shards` value so databases built before the count was
|
|
38
|
+
/// pinned keep working. Must be a power of two so `term_shard` can route with a
|
|
39
|
+
/// bitmask. Distinct from the in-memory `SHARD_COUNT` used by the concurrent
|
|
40
|
+
/// build maps.
|
|
40
41
|
const SHARD_COUNT_SHARDS: usize = 16;
|
|
41
42
|
/// A normalized term grouped with its deduplicated `(curie_id, source_id)` pairs.
|
|
42
43
|
type TermPairs = (String, Vec<(u32, u8)>);
|
|
@@ -92,8 +93,8 @@ const DEFAULT_CHUNK_BYTES: usize = 8 * 1024 * 1024;
|
|
|
92
93
|
|
|
93
94
|
/// Round `n` DOWN to the nearest power of two (6->4, 3->2, 5->4); powers of two
|
|
94
95
|
/// are unchanged and any `n < 1` floors to 1. The shard mask routing in
|
|
95
|
-
/// `term_shard` requires a power-of-two shard count, so the
|
|
96
|
-
/// normalized through this before use.
|
|
96
|
+
/// `term_shard` requires a power-of-two shard count, so the shard count read from
|
|
97
|
+
/// a database's META is normalized through this before use.
|
|
97
98
|
fn round_down_pow2(n: usize) -> usize {
|
|
98
99
|
if n <= 1 {
|
|
99
100
|
return 1;
|
|
@@ -106,15 +107,6 @@ fn round_down_pow2(n: usize) -> usize {
|
|
|
106
107
|
}
|
|
107
108
|
}
|
|
108
109
|
|
|
109
|
-
/// Resolve the on-disk RECORDS shard count from `TABLASSERT_FULLMAP_SHARDS`
|
|
110
|
-
/// (default `SHARD_COUNT_SHARDS`). Non-powers-of-two round DOWN to the nearest
|
|
111
|
-
/// power of two (6->4, 3->2) and the result is clamped to `<= SHARD_COUNT_SHARDS`,
|
|
112
|
-
/// the compile-time cap on shard files (raise the const to allow more).
|
|
113
|
-
fn resolve_shard_count() -> usize {
|
|
114
|
-
let raw = env_usize("TABLASSERT_FULLMAP_SHARDS", SHARD_COUNT_SHARDS);
|
|
115
|
-
round_down_pow2(raw).clamp(1, SHARD_COUNT_SHARDS)
|
|
116
|
-
}
|
|
117
|
-
|
|
118
110
|
fn env_usize(name: &str, default: usize) -> usize {
|
|
119
111
|
std::env::var(name)
|
|
120
112
|
.ok()
|
|
@@ -2245,10 +2237,9 @@ pub fn build_fullmap_db(
|
|
|
2245
2237
|
DEFAULT_REDB_CACHE_BYTES,
|
|
2246
2238
|
);
|
|
2247
2239
|
let insert_batch = env_usize("TABLASSERT_FULLMAP_INSERT_BATCH", DEFAULT_INSERT_BATCH);
|
|
2248
|
-
// On-disk RECORDS shard count (one redb file + concurrent writer per shard)
|
|
2249
|
-
//
|
|
2250
|
-
|
|
2251
|
-
let shard_count = resolve_shard_count();
|
|
2240
|
+
// On-disk RECORDS shard count (one redb file + concurrent writer per shard),
|
|
2241
|
+
// fixed at SHARD_COUNT_SHARDS (16).
|
|
2242
|
+
let shard_count = SHARD_COUNT_SHARDS;
|
|
2252
2243
|
let spill_dir = std::env::var("TABLASSERT_FULLMAP_SPILL_DIR")
|
|
2253
2244
|
.map(PathBuf::from)
|
|
2254
2245
|
.unwrap_or_else(|_| {
|
|
@@ -2338,9 +2329,10 @@ fn open_cached_shard(primary: &Path, index: usize) -> PyResult<Arc<Database>> {
|
|
|
2338
2329
|
}
|
|
2339
2330
|
|
|
2340
2331
|
/// Read the RECORDS shard count advertised in the primary's META, defaulting to
|
|
2341
|
-
/// `SHARD_COUNT_SHARDS` when absent.
|
|
2342
|
-
/// `
|
|
2343
|
-
///
|
|
2332
|
+
/// `SHARD_COUNT_SHARDS` when absent. New builds always record
|
|
2333
|
+
/// `SHARD_COUNT_SHARDS`; reading the value back keeps the read path compatible
|
|
2334
|
+
/// with databases built before the count was pinned, opening exactly the shards
|
|
2335
|
+
/// that exist and routing with the matching mask.
|
|
2344
2336
|
fn shard_count_of(database: &Database) -> PyResult<usize> {
|
|
2345
2337
|
let read = database.begin_read().map_err(py_err)?;
|
|
2346
2338
|
let meta = read.open_table(META).map_err(py_err)?;
|
|
@@ -2349,10 +2341,10 @@ fn shard_count_of(database: &Database) -> PyResult<usize> {
|
|
|
2349
2341
|
.map_err(py_err)?
|
|
2350
2342
|
.and_then(|v| v.value().parse::<usize>().ok())
|
|
2351
2343
|
.unwrap_or(SHARD_COUNT_SHARDS);
|
|
2352
|
-
// Round down to a power of two before clamping
|
|
2353
|
-
// `
|
|
2354
|
-
//
|
|
2355
|
-
//
|
|
2344
|
+
// Round down to a power of two before clamping: the routing mask
|
|
2345
|
+
// `xxh64(term) & (count - 1)` is only correct for powers of two, so a legacy
|
|
2346
|
+
// or hand-edited non-pow2 META.shards (e.g. 3 -> mask &2) would silently
|
|
2347
|
+
// misroute/drop lookups onto a subset of shards.
|
|
2356
2348
|
// `round_down_pow2(SHARD_COUNT_SHARDS) == SHARD_COUNT_SHARDS` (16 is a pow2),
|
|
2357
2349
|
// so the default fallback above is preserved.
|
|
2358
2350
|
Ok(round_down_pow2(count).clamp(1, SHARD_COUNT_SHARDS))
|
|
@@ -2958,10 +2950,9 @@ mod tests {
|
|
|
2958
2950
|
.collect()
|
|
2959
2951
|
}
|
|
2960
2952
|
|
|
2961
|
-
///
|
|
2962
|
-
///
|
|
2963
|
-
///
|
|
2964
|
-
/// count is always a valid non-zero mask.
|
|
2953
|
+
/// Mask routing only works for powers of two; `round_down_pow2` must round
|
|
2954
|
+
/// 6->4, 3->2, 5->4, leave powers unchanged, and floor tiny/zero inputs at 1
|
|
2955
|
+
/// so a legacy META shard count is always a valid non-zero mask.
|
|
2965
2956
|
#[test]
|
|
2966
2957
|
fn round_down_pow2_rounds_to_nearest_lower_power_of_two() {
|
|
2967
2958
|
assert_eq!(round_down_pow2(0), 1);
|
|
@@ -2977,16 +2968,15 @@ mod tests {
|
|
|
2977
2968
|
}
|
|
2978
2969
|
|
|
2979
2970
|
/// The read path's `shard_count_of` must round a non-power-of-two META.shards
|
|
2980
|
-
/// DOWN to a power of two
|
|
2981
|
-
///
|
|
2982
|
-
///
|
|
2983
|
-
///
|
|
2984
|
-
///
|
|
2985
|
-
///
|
|
2986
|
-
///
|
|
2987
|
-
///
|
|
2988
|
-
///
|
|
2989
|
-
/// (`round_down_pow2(0) == 1`); 1 is still a valid mask (&0), so it is safe.
|
|
2971
|
+
/// DOWN to a power of two. WHY: lookups route with the mask
|
|
2972
|
+
/// `xxh64(term) & (count - 1)`, which is only correct for powers of two; a
|
|
2973
|
+
/// legacy or hand-edited META.shards of 3 (mask &2) would route terms only onto
|
|
2974
|
+
/// shards {0,2}, silently misrouting/dropping lookups (shard 1 opened but never
|
|
2975
|
+
/// queried, shard 3 never opened). Rounding down keeps the mask valid. A
|
|
2976
|
+
/// missing/unparseable value falls back to the default `SHARD_COUNT_SHARDS`
|
|
2977
|
+
/// (16) via `unwrap_or`, itself a power of two so the round-down leaves it
|
|
2978
|
+
/// unchanged. `"0"` PARSES (so the fallback does not fire) and rounds down to
|
|
2979
|
+
/// 1 (`round_down_pow2(0) == 1`); 1 is still a valid mask (&0), so it is safe.
|
|
2990
2980
|
#[test]
|
|
2991
2981
|
fn shard_count_of_rounds_non_pow2_meta_down() {
|
|
2992
2982
|
pyo3::Python::initialize();
|
|
@@ -3022,8 +3012,8 @@ mod tests {
|
|
|
3022
3012
|
// Missing / unparseable -> default SHARD_COUNT_SHARDS (16) via unwrap_or.
|
|
3023
3013
|
assert_eq!(check(None), SHARD_COUNT_SHARDS);
|
|
3024
3014
|
assert_eq!(check(Some("not-a-number")), SHARD_COUNT_SHARDS);
|
|
3025
|
-
// "0" parses (fallback does NOT fire) and rounds down to 1
|
|
3026
|
-
//
|
|
3015
|
+
// "0" parses (fallback does NOT fire) and rounds down to 1; 1 is a valid
|
|
3016
|
+
// mask, so this is safe, not a misroute.
|
|
3027
3017
|
assert_eq!(check(Some("0")), 1);
|
|
3028
3018
|
}
|
|
3029
3019
|
|
|
@@ -3477,12 +3467,13 @@ mod tests {
|
|
|
3477
3467
|
);
|
|
3478
3468
|
}
|
|
3479
3469
|
|
|
3480
|
-
///
|
|
3481
|
-
///
|
|
3482
|
-
///
|
|
3483
|
-
///
|
|
3484
|
-
///
|
|
3485
|
-
///
|
|
3470
|
+
/// The build and read paths agree on a parameterized shard count. A 2-shard
|
|
3471
|
+
/// build must write META.shards="2", create exactly s0+s1 (each with a RECORDS
|
|
3472
|
+
/// table, even the one that receives no terms), create no higher shard files up
|
|
3473
|
+
/// to the compile-time cap, and the read path must open exactly 2 shards (from
|
|
3474
|
+
/// META) and still resolve every term. The public entry point pins the count
|
|
3475
|
+
/// to `SHARD_COUNT_SHARDS`, but this exercises the same parameterized path the
|
|
3476
|
+
/// read path relies on for databases built with fewer shards.
|
|
3486
3477
|
#[test]
|
|
3487
3478
|
fn runtime_shard_count_builds_and_reads_fewer_shards() {
|
|
3488
3479
|
pyo3::Python::initialize();
|
|
@@ -3523,7 +3514,7 @@ mod tests {
|
|
|
3523
3514
|
)
|
|
3524
3515
|
.unwrap();
|
|
3525
3516
|
|
|
3526
|
-
// META.shards reflects the
|
|
3517
|
+
// META.shards reflects the parameterized shard count.
|
|
3527
3518
|
let database = open_cached(output.clone()).unwrap();
|
|
3528
3519
|
let read = database.begin_read().unwrap();
|
|
3529
3520
|
let meta = read.open_table(META).unwrap();
|
|
@@ -3556,6 +3547,67 @@ mod tests {
|
|
|
3556
3547
|
assert_eq!(rows.len(), 50);
|
|
3557
3548
|
}
|
|
3558
3549
|
|
|
3550
|
+
/// The public entry point ignores `TABLASSERT_FULLMAP_SHARDS`: even when it is
|
|
3551
|
+
/// set to a smaller value, `build_fullmap_db` writes exactly
|
|
3552
|
+
/// `SHARD_COUNT_SHARDS` (16) shard files and records META.shards="16". This is
|
|
3553
|
+
/// the regression guard for removing the env-var tunable — the layout tests
|
|
3554
|
+
/// above call `build_fullmap_inner` / `build_test` directly and so cannot catch
|
|
3555
|
+
/// a public path that re-reads the variable. Setting the variable here is safe
|
|
3556
|
+
/// across concurrent tests: the build no longer reads it (its only reader,
|
|
3557
|
+
/// `resolve_shard_count`, was removed), and it is removed again before any
|
|
3558
|
+
/// assertion can panic.
|
|
3559
|
+
#[test]
|
|
3560
|
+
fn public_build_ignores_shards_env_var() {
|
|
3561
|
+
pyo3::Python::initialize();
|
|
3562
|
+
let dir = tempfile::tempdir().unwrap();
|
|
3563
|
+
let synonyms = dir.path().join("HGNC.ndjson");
|
|
3564
|
+
let output = dir.path().join("fullmap.redb");
|
|
3565
|
+
let mut synonym_file = File::create(&synonyms).unwrap();
|
|
3566
|
+
for i in 0..50 {
|
|
3567
|
+
writeln!(
|
|
3568
|
+
synonym_file,
|
|
3569
|
+
r#"{{"curie":"HGNC:{i}","preferred_name":"GENE{i}","names":["GENE{i}"],"types":["Gene"],"taxa":["NCBITaxon:9606"]}}"#
|
|
3570
|
+
)
|
|
3571
|
+
.unwrap();
|
|
3572
|
+
}
|
|
3573
|
+
drop(synonym_file);
|
|
3574
|
+
|
|
3575
|
+
std::env::set_var("TABLASSERT_FULLMAP_SHARDS", "2");
|
|
3576
|
+
let built = Python::attach(|py| {
|
|
3577
|
+
build_fullmap_db(
|
|
3578
|
+
py,
|
|
3579
|
+
output.clone(),
|
|
3580
|
+
Vec::new(),
|
|
3581
|
+
vec![synonyms],
|
|
3582
|
+
Some(2),
|
|
3583
|
+
None,
|
|
3584
|
+
)
|
|
3585
|
+
});
|
|
3586
|
+
std::env::remove_var("TABLASSERT_FULLMAP_SHARDS");
|
|
3587
|
+
built.unwrap();
|
|
3588
|
+
|
|
3589
|
+
// The env var is ignored: META advertises 16 shards, all 16 sibling shard
|
|
3590
|
+
// files exist (s0..s15), and there is no s16.
|
|
3591
|
+
let database = open_cached(output.clone()).unwrap();
|
|
3592
|
+
let read = database.begin_read().unwrap();
|
|
3593
|
+
let meta = read.open_table(META).unwrap();
|
|
3594
|
+
assert_eq!(meta.get("shards").unwrap().unwrap().value(), "16");
|
|
3595
|
+
drop(meta);
|
|
3596
|
+
drop(read);
|
|
3597
|
+
for index in 0..SHARD_COUNT_SHARDS {
|
|
3598
|
+
assert!(
|
|
3599
|
+
shard_path(&output, index).exists(),
|
|
3600
|
+
"missing shard file {index}"
|
|
3601
|
+
);
|
|
3602
|
+
}
|
|
3603
|
+
assert!(!shard_path(&output, SHARD_COUNT_SHARDS).exists());
|
|
3604
|
+
|
|
3605
|
+
// Lookups still resolve across all 16 shards.
|
|
3606
|
+
let terms: Vec<String> = (0..50).map(|i| format!("gene{i}")).collect();
|
|
3607
|
+
let rows = lookup_terms(output, terms, Some(4)).unwrap();
|
|
3608
|
+
assert_eq!(rows.len(), 50);
|
|
3609
|
+
}
|
|
3610
|
+
|
|
3559
3611
|
/// Backward compatibility: databases built/advertising the former 4-shard
|
|
3560
3612
|
/// layout must continue to open exactly those four shard files and route terms
|
|
3561
3613
|
/// with a 4-shard mask, even though new default builds now write 16 shards.
|