tablassert 9.1.0__tar.gz → 10.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tablassert-9.1.0 → tablassert-10.1.0}/PKG-INFO +1 -1
- {tablassert-9.1.0 → tablassert-10.1.0}/pyproject.toml +1 -1
- {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/agent.py +5 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/cli.py +36 -5
- {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/enums.py +0 -1
- {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/errors.py +2 -3
- {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/fullmap.py +18 -4
- {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/lib.py +42 -13
- {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/models.py +65 -80
- tablassert-10.1.0/src/tablassert/study.py +158 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/LICENSE +0 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/README.md +0 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/rust/Cargo.lock +0 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/rust/Cargo.toml +0 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/rust/examples/count_tables.rs +0 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/rust/src/fullmap.rs +0 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/rust/src/json.rs +0 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/rust/src/lib.rs +0 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/rust/src/ndjson.rs +0 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/rust/src/uuid.rs +0 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/rust/tests/build_golden.rs +0 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/__init__.py +0 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/_lazy.py +0 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/biolink.py +0 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/coerce.py +0 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/extras.py +0 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/graph_registry.py +0 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/ingests.py +0 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/log.py +0 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/nlp.py +0 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/progress.py +0 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/qc.py +0 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/rig.py +0 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/rs.pyi +0 -0
- {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/utils.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "tablassert"
|
|
3
|
-
version = "
|
|
3
|
+
version = "10.1.0"
|
|
4
4
|
description = "Extract knowledge assertions from tabular data into NCATS Translator-compliant KGX NDJSON — declaratively, with entity resolution and quality control built in."
|
|
5
5
|
authors = [
|
|
6
6
|
{ name = "Skye Lane Goetz", email = "sgoetz@isbscience.org" }
|
|
@@ -2172,6 +2172,11 @@ qualifier and evidence slot the specific class declared. build_and_audit reports
|
|
|
2172
2172
|
description rather than emitted on the edge. `q_value`, `fold_change`, `z_score`, `beta` and
|
|
2173
2173
|
similar are not association slots at all and are folded into `supporting_text`. Prefer
|
|
2174
2174
|
`p_value`, `adjusted_p_value`, `effect_size`, `effect_type`, `has_evidence`.
|
|
2175
|
+
- MULTIVALUED slots (`has_evidence` and friends) take a real JSON array, never a joined string:
|
|
2176
|
+
declare the annotation `{method: column, encoding: <letter>, split_by: "|"}` so each cell's
|
|
2177
|
+
delimited text splits into its own per-row array. `split_by` is the ONLY multivalued encoding
|
|
2178
|
+
— there is no literal-list method, and a scalar bound for a multivalued slot ships to consumers
|
|
2179
|
+
as one unusable "a|b|c" blob.
|
|
2175
2180
|
- `effect_size` / `effect_type` are deliberate Tablassert extras pending biolink-model#1774 and
|
|
2176
2181
|
are EXEMPT from the validity score: a `biolink_valid_pct` below 1.0 is never caused by them.
|
|
2177
2182
|
- QUALIFIERS: enum-ranged qualifiers take a literal TOKEN, never a CURIE
|
|
@@ -121,13 +121,15 @@ def build_pipeline(
|
|
|
121
121
|
"""Build a knowledge graph from a YAML configuration file.
|
|
122
122
|
|
|
123
123
|
Runs the six-stage build pipeline: load tables → extract sections → build
|
|
124
|
-
Tcodes → collect instructions → build subgraphs → compile graph.
|
|
124
|
+
Tcodes → collect instructions → build subgraphs → compile graph. With ``qc``
|
|
125
|
+
enabled a seventh stage studies the final NDJSON files.
|
|
125
126
|
|
|
126
127
|
Args:
|
|
127
128
|
configuration_file: Path to the graph YAML file.
|
|
128
129
|
progress: Pipeline progress reporter.
|
|
129
130
|
release: When ``True``, emit release-mode artifacts.
|
|
130
|
-
qc: When ``True``, run quality-control audits on each section
|
|
131
|
+
qc: When ``True``, run quality-control audits on each section and assert
|
|
132
|
+
over the final NDJSON files (failing the build on any violation).
|
|
131
133
|
log: When ``True``, enable per-section verbose logging.
|
|
132
134
|
head: When ``True``, preview a random sample of up to 5 rows per section (fast schema/shape check).
|
|
133
135
|
|
|
@@ -218,9 +220,35 @@ def build_pipeline(
|
|
|
218
220
|
subgraphs, g.name, g.version, g.description, g.contributions, g.ui_explanation, g.tables, g.infores, on_phase=sub_step, on_subgraph=advance
|
|
219
221
|
)
|
|
220
222
|
|
|
223
|
+
# Stage 7/7 (only with --qc): assert over the final NDJSON files.
|
|
224
|
+
if qc:
|
|
225
|
+
progress.stage("Studying Graph")
|
|
226
|
+
study_final_ndjson(g.name, g.version)
|
|
227
|
+
|
|
221
228
|
logger.info("Built graph {name} v{version}: {n} sections", name=g.name, version=g.version, n=n)
|
|
222
229
|
|
|
223
230
|
|
|
231
|
+
def study_final_ndjson(name: str, version: str) -> None:
|
|
232
|
+
"""Run study assertions over a build's final NDJSON files (the ``--qc`` stage 7).
|
|
233
|
+
|
|
234
|
+
Args:
|
|
235
|
+
name: Graph name, used to locate ``./<name>_<version>.nodes.ndjson``.
|
|
236
|
+
version: Graph version, used to locate ``./<name>_<version>.edges.ndjson``.
|
|
237
|
+
|
|
238
|
+
Raises:
|
|
239
|
+
SystemExit: With status 1 when any study assertion is violated.
|
|
240
|
+
"""
|
|
241
|
+
from tablassert.study import format_violations, study_kgx
|
|
242
|
+
|
|
243
|
+
violations = study_kgx(Path(f"./{name}_{version}.nodes.ndjson"), Path(f"./{name}_{version}.edges.ndjson"))
|
|
244
|
+
if violations:
|
|
245
|
+
summary: str = format_violations(violations)
|
|
246
|
+
print(summary, file=sys.stderr)
|
|
247
|
+
logger.warning("study assertions failed on final NDJSON:\n{summary}", summary=summary)
|
|
248
|
+
raise SystemExit(1)
|
|
249
|
+
logger.info("study assertions passed on final NDJSON")
|
|
250
|
+
|
|
251
|
+
|
|
224
252
|
def validate_pipeline(table_configuration_file: Path, progress: PipelineProgress) -> None:
|
|
225
253
|
"""Validate section syntax from a YAML configuration file.
|
|
226
254
|
|
|
@@ -599,11 +627,14 @@ def build_kg(
|
|
|
599
627
|
|
|
600
628
|
``--qc`` requires the ``[qc]`` extra (``pip install "tablassert[qc]"``); it is
|
|
601
629
|
checked before the build starts, because the audit stage runs LAST and a missing
|
|
602
|
-
extra would otherwise surface only after entity resolution has finished.
|
|
630
|
+
extra would otherwise surface only after entity resolution has finished. It also
|
|
631
|
+
runs a final study stage that asserts over the emitted NDJSON -- no duplicate node
|
|
632
|
+
ids, no undeclared or isolated nodes, no malformed lines or stray whitespace --
|
|
633
|
+
and fails the build (non-zero exit) when any assertion is violated.
|
|
603
634
|
"""
|
|
604
635
|
if qc:
|
|
605
636
|
extras.require("qc", required_by="--qc")
|
|
606
|
-
run(6, build_pipeline, graph_configuration_file, release=release, qc=qc, log=log, head=head)
|
|
637
|
+
run(7 if qc else 6, build_pipeline, graph_configuration_file, release=release, qc=qc, log=log, head=head)
|
|
607
638
|
|
|
608
639
|
|
|
609
640
|
@APP.command(name="validate")
|
|
@@ -950,7 +981,7 @@ def _prebuilt_fullmap_urls(babel_version: str) -> tuple[str, str]:
|
|
|
950
981
|
|
|
951
982
|
RENCI publishes a prebuilt ``fullmap.tar.zst`` (and a ``sha256sum.txt``) under
|
|
952
983
|
``{BABEL_BASE}/{babel_version}/fullmap/{tablassert_version}/``, where the version
|
|
953
|
-
directory is the INSTALLED Tablassert package version (e.g. ``
|
|
984
|
+
directory is the INSTALLED Tablassert package version (e.g. ``10.1.0``) — resolved from
|
|
954
985
|
installed-package metadata, never hardcoded, so a new release looks itself up.
|
|
955
986
|
|
|
956
987
|
Args:
|
|
@@ -25,14 +25,13 @@ TablassertErrorCodes = Literal[
|
|
|
25
25
|
"provenance-bad-pmc-id",
|
|
26
26
|
"provenance-missing-publication",
|
|
27
27
|
"provenance-publication-and-override",
|
|
28
|
-
"encoding-list-
|
|
29
|
-
"encoding-list-incompatible-ops",
|
|
30
|
-
"encoding-list-annotation-only",
|
|
28
|
+
"encoding-list-method-removed",
|
|
31
29
|
"annotation-split-by-requires-column",
|
|
32
30
|
"annotation-split-by-empty",
|
|
33
31
|
"qualifier-auto-derived",
|
|
34
32
|
"qualifier-bad-value",
|
|
35
33
|
"qualifier-unsatisfiable",
|
|
34
|
+
"qualifier-nullable-literal",
|
|
36
35
|
]
|
|
37
36
|
|
|
38
37
|
|
|
@@ -473,7 +473,7 @@ def _coalesce_expr(col: str, suffix: str, base: str, prefix: str, cast_str: bool
|
|
|
473
473
|
return pl.when(pl.col(base).is_not_null()).then(l1).otherwise(l2).alias(col + suffix)
|
|
474
474
|
|
|
475
475
|
|
|
476
|
-
def join_matches(lf: pl.LazyFrame, col: str, matches: pl.DataFrame, tag: str = "_two") -> pl.LazyFrame:
|
|
476
|
+
def join_matches(lf: pl.LazyFrame, col: str, matches: pl.DataFrame, tag: str = "_two", drop_unresolved: bool = True) -> pl.LazyFrame:
|
|
477
477
|
"""Join ranked fullmap matches back into ``lf`` for one column.
|
|
478
478
|
|
|
479
479
|
Coalesces level-one and level-two hits per row (level one wins when
|
|
@@ -486,10 +486,15 @@ def join_matches(lf: pl.LazyFrame, col: str, matches: pl.DataFrame, tag: str = "
|
|
|
486
486
|
col: Column being resolved.
|
|
487
487
|
matches: Ranked matches for this column from ``filter_and_rank``.
|
|
488
488
|
tag: Suffix used to derive the level-two column name.
|
|
489
|
+
drop_unresolved: When True (default) rows whose ``col`` did not match are
|
|
490
|
+
dropped — the behavior subject/object require. When False the row is
|
|
491
|
+
kept and the column (plus its derived ``<col>_*`` columns) stays null;
|
|
492
|
+
used by ``nullable`` qualifiers so a blank/unresolvable cell keeps the
|
|
493
|
+
edge and the null-stripper omits the qualifier key.
|
|
489
494
|
|
|
490
495
|
Returns:
|
|
491
496
|
New LazyFrame with resolved columns; rows whose ``col`` did not match
|
|
492
|
-
are dropped.
|
|
497
|
+
are dropped unless ``drop_unresolved`` is False.
|
|
493
498
|
|
|
494
499
|
Notes:
|
|
495
500
|
Split out of ``resolve`` so ``resolve_batch`` can apply per-column
|
|
@@ -511,7 +516,10 @@ def join_matches(lf: pl.LazyFrame, col: str, matches: pl.DataFrame, tag: str = "
|
|
|
511
516
|
result = result.select(pl.exclude(r"^(CURIE|PREFERRED_NAME|CATEGORY_NAME|TAXON_ID|SOURCE_NAME|SOURCE_VERSION|NLP_LEVEL|PR|FREQUENCY)(_l2)?$"))
|
|
512
517
|
result = result.select(pl.exclude(col + tag))
|
|
513
518
|
result = result.with_columns(pl.col(f"{col}_taxon").replace("NCBITaxon:0", None))
|
|
514
|
-
|
|
519
|
+
if drop_unresolved:
|
|
520
|
+
# Subject/object (and strict qualifiers) drop rows that failed resolution; a
|
|
521
|
+
# nullable qualifier keeps the edge and leaves the column null for the null-stripper.
|
|
522
|
+
result = result.filter(pl.col(col).is_not_null())
|
|
515
523
|
|
|
516
524
|
return result.lazy()
|
|
517
525
|
|
|
@@ -525,6 +533,12 @@ class ResolveSpec(NamedTuple):
|
|
|
525
533
|
avoid: list[Categories] | None = None
|
|
526
534
|
exclude_prefixes: list[str] | None = None
|
|
527
535
|
exclude_regex: list[str] | None = None
|
|
536
|
+
nullable: bool = False
|
|
537
|
+
"""When True, an unresolved/blank cell keeps its row instead of dropping the edge.
|
|
538
|
+
|
|
539
|
+
Set only for ``nullable`` qualifiers; subject/object always resolve strict so an
|
|
540
|
+
edge with a missing node is dropped, never emitted.
|
|
541
|
+
"""
|
|
528
542
|
|
|
529
543
|
|
|
530
544
|
def resolve_batch(
|
|
@@ -585,7 +599,7 @@ def resolve_batch(
|
|
|
585
599
|
)
|
|
586
600
|
if log:
|
|
587
601
|
log_unmatched(spec.col, terms_by_col[spec.col], matches, section_hash, config_file)
|
|
588
|
-
result = join_matches(result, spec.col, matches, tag)
|
|
602
|
+
result = join_matches(result, spec.col, matches, tag, drop_unresolved=not spec.nullable)
|
|
589
603
|
|
|
590
604
|
return result
|
|
591
605
|
|
|
@@ -262,8 +262,10 @@ def prune_to_class(lf: pl.LazyFrame) -> pl.LazyFrame:
|
|
|
262
262
|
|
|
263
263
|
Categories vary per row within a section, so this masks per row rather than
|
|
264
264
|
dropping columns: values are nulled where the row's class rejects them, and the
|
|
265
|
-
Rust null-stripper then removes the key entirely. Scalars are wrapped
|
|
266
|
-
|
|
265
|
+
Rust null-stripper then removes the key entirely. Scalars are wrapped into
|
|
266
|
+
one-element lists when the slot is multivalued on every class that declares it
|
|
267
|
+
(a column has one dtype, so per-row wrapping is impossible; a slot Biolink
|
|
268
|
+
declares scalar on some classes and multivalued on others stays scalar).
|
|
267
269
|
|
|
268
270
|
Args:
|
|
269
271
|
lf: Edges LazyFrame carrying a resolved ``category`` column.
|
|
@@ -306,10 +308,18 @@ def prune_to_class(lf: pl.LazyFrame) -> pl.LazyFrame:
|
|
|
306
308
|
# Preserve what the class refuses rather than deleting it outright; the
|
|
307
309
|
# value is real evidence, it just has no slot on this association class.
|
|
308
310
|
rescued.append(pl.when(ok | text.is_null()).then(None).otherwise(pl.concat_str([pl.lit(f"{col}="), text])))
|
|
309
|
-
#
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
311
|
+
# A column has one dtype, so the multivalued wrap must be uniform across rows:
|
|
312
|
+
# wrap every value into a one-element list when every class declaring the slot
|
|
313
|
+
# types it multivalued. `is_multivalued` is False for classes that do not declare
|
|
314
|
+
# the slot at all, so the scan is restricted to declaring classes -- including
|
|
315
|
+
# them is what produced a spuriously mixed per-row wrap that died in
|
|
316
|
+
# strict_cast at collect. Rows whose class rejects the slot are already null and
|
|
317
|
+
# stay null. A hypothetically mixed slot keeps its scalar rather than crashing.
|
|
318
|
+
declaring: list[str] = [cat for cat in categories if accepts[cat]]
|
|
319
|
+
listed: dict[str, bool] = {cat: is_multivalued(association_class(cat), col) for cat in declaring}
|
|
320
|
+
if listed and all(listed.values()) and not isinstance(schema[col], pl.List):
|
|
321
|
+
# concat_list maps null -> [null]; the when preserves real nulls instead.
|
|
322
|
+
keep = pl.when(keep.is_null()).then(None).otherwise(pl.concat_list(keep))
|
|
313
323
|
updates.append(keep.alias(col))
|
|
314
324
|
if rescued:
|
|
315
325
|
updates.append(pl.concat_list(rescued).list.drop_nulls().alias(PRUNED_COLUMN))
|
|
@@ -632,13 +642,12 @@ def split_list(lf: pl.LazyFrame, col: str, delimiter: str) -> pl.LazyFrame:
|
|
|
632
642
|
|
|
633
643
|
A column encoding is scalar by construction, so a multivalued Biolink slot such as
|
|
634
644
|
``has_evidence`` fed from an aggregated cell would otherwise be emitted as a single
|
|
635
|
-
joined string -- and ``
|
|
645
|
+
joined string -- and ``prune_to_class`` wraps that scalar into a
|
|
636
646
|
one-element list, so the value survives Biolink validation while consumers iterate a
|
|
637
647
|
single ``"a|b|c"`` blob instead of three ids.
|
|
638
648
|
|
|
639
|
-
Same split as ``explode``, minus the fan-out: this is the
|
|
640
|
-
|
|
641
|
-
covers an array that differs on every row).
|
|
649
|
+
Same split as ``explode``, minus the fan-out: this is the one multivalued
|
|
650
|
+
encoding (``split_by``), turning each row's cell into its own JSON array.
|
|
642
651
|
|
|
643
652
|
Args:
|
|
644
653
|
lf: Source LazyFrame.
|
|
@@ -930,7 +939,7 @@ class Tcode(Section):
|
|
|
930
939
|
containing ``None`` placeholders) ready for ``clean`` to filter.
|
|
931
940
|
"""
|
|
932
941
|
return [
|
|
933
|
-
(value, (col, x.encoding)) if x.method
|
|
942
|
+
(value, (col, x.encoding)) if x.method == EncodingMethods.VALUE else None,
|
|
934
943
|
(column, (col, idxname(x.encoding))) if x.method == EncodingMethods.COLUMN else None,
|
|
935
944
|
(column, (f"original_{col}", col)) if table_literal else None,
|
|
936
945
|
(fill, (col, x.fill)) if x.fill else None,
|
|
@@ -1034,8 +1043,20 @@ class Tcode(Section):
|
|
|
1034
1043
|
*[(x, x.qualifier) for x in qualifiers if x.resolved],
|
|
1035
1044
|
]
|
|
1036
1045
|
literals: list[Qualifier] = [x for x in qualifiers if not x.resolved]
|
|
1046
|
+
# A nullable qualifier keeps its edge when the cell is blank or unresolvable (the
|
|
1047
|
+
# column stays null and the null-stripper omits the key); subject/object and
|
|
1048
|
+
# strict qualifiers drop the row as before.
|
|
1037
1049
|
specs: list[ResolveSpec] = [
|
|
1038
|
-
ResolveSpec(
|
|
1050
|
+
ResolveSpec(
|
|
1051
|
+
col,
|
|
1052
|
+
str(x.taxon) if x.taxon else None,
|
|
1053
|
+
x.prioritize,
|
|
1054
|
+
x.avoid,
|
|
1055
|
+
x.exclude_prefixes,
|
|
1056
|
+
x.exclude_regex,
|
|
1057
|
+
x.nullable if isinstance(x, Qualifier) else False,
|
|
1058
|
+
)
|
|
1059
|
+
for x, col in node_columns
|
|
1039
1060
|
]
|
|
1040
1061
|
return [
|
|
1041
1062
|
[self.node_prep(x, col) for x, col in node_columns],
|
|
@@ -1043,7 +1064,15 @@ class Tcode(Section):
|
|
|
1043
1064
|
# which exist to feed entity resolution these columns never undergo.
|
|
1044
1065
|
[self.encoding(x, x.qualifier) for x in literals],
|
|
1045
1066
|
(resolve_batch, (specs, db, self.log, self.store.stem, self.config.name, True)),
|
|
1046
|
-
|
|
1067
|
+
# QC audits only the strict columns: a nullable qualifier's nulls are expected
|
|
1068
|
+
# (blank cell / no match), not resolution errors for the audit to delete.
|
|
1069
|
+
[
|
|
1070
|
+
(fullmap_audit, (col, self.store.stem, self.config.name, "passed", True))
|
|
1071
|
+
for x, col in node_columns
|
|
1072
|
+
if not (isinstance(x, Qualifier) and x.nullable)
|
|
1073
|
+
]
|
|
1074
|
+
if self.qc
|
|
1075
|
+
else None,
|
|
1047
1076
|
]
|
|
1048
1077
|
|
|
1049
1078
|
def _provenance_ops(self: Self) -> list[Any]:
|
|
@@ -19,6 +19,7 @@ from tablassert.biolink import (
|
|
|
19
19
|
Predicates,
|
|
20
20
|
Qualifiers,
|
|
21
21
|
)
|
|
22
|
+
from tablassert.coerce import effect_size_target, effect_type_target, pvalue_target, study_size_target
|
|
22
23
|
from tablassert.enums import Comparisons, EncodingMethods, Files, FillMethods, Functions, Repositories, Tokens
|
|
23
24
|
from tablassert.errors import BiolinkRelocationWarning, TablassertErrorCodes, TablassertValidationError
|
|
24
25
|
|
|
@@ -164,67 +165,40 @@ class Math(TablaBase):
|
|
|
164
165
|
class Encoding(TablaBase):
|
|
165
166
|
method: EncodingMethods = Field(
|
|
166
167
|
EncodingMethods.VALUE,
|
|
167
|
-
description="Interpret `encoding` as a literal value
|
|
168
|
-
examples=[EncodingMethods.VALUE, EncodingMethods.COLUMN
|
|
169
|
-
)
|
|
170
|
-
encoding: str | int | float | list[str | int | float] = Field(
|
|
171
|
-
...,
|
|
172
|
-
description="Literal value, list of literal values (with `method: list`), or source column letters.",
|
|
173
|
-
examples=["A", "BRCA1", 1.0, ["EFO:0001", "EFO:0002"]],
|
|
168
|
+
description="Interpret `encoding` as a literal value or source column letters.",
|
|
169
|
+
examples=[EncodingMethods.VALUE, EncodingMethods.COLUMN],
|
|
174
170
|
)
|
|
171
|
+
encoding: str | int | float = Field(..., description="Literal value or source column letters.", examples=["A", "BRCA1", 1.0])
|
|
172
|
+
|
|
173
|
+
@model_validator(mode="before")
|
|
174
|
+
@classmethod
|
|
175
|
+
def reject_removed_list_method(cls, data: Any) -> Any:
|
|
176
|
+
"""Fail configs still declaring the removed ``method: list`` with a migration pointer.
|
|
177
|
+
|
|
178
|
+
``method: list`` (a literal list emitted as one fixed JSON array on every row)
|
|
179
|
+
was removed: ``split_by`` on a ``method: column`` annotation is the one
|
|
180
|
+
multivalued encoding now, and it covers the per-row case the literal never
|
|
181
|
+
could. A bare pydantic enum error would only say the value is invalid, so
|
|
182
|
+
this hook turns the stale config into the actionable coded error the
|
|
183
|
+
migration needs.
|
|
184
|
+
"""
|
|
185
|
+
if isinstance(data, dict) and data.get("method") == "list":
|
|
186
|
+
raise TablassertValidationError(
|
|
187
|
+
"`method: list` was removed; for a multivalued annotation use `method: column` with `split_by` "
|
|
188
|
+
"to split each cell's delimited text into a JSON array (subject/object/qualifier nodes are single entities).",
|
|
189
|
+
code="encoding-list-method-removed",
|
|
190
|
+
)
|
|
191
|
+
return data
|
|
175
192
|
|
|
176
193
|
@model_validator(mode="after")
|
|
177
194
|
def excel_style_columns(self: Self) -> Self:
|
|
178
|
-
|
|
179
|
-
# skip the Excel-letter check here so that case reports the clearer list error.
|
|
180
|
-
if self.method == EncodingMethods.COLUMN and not isinstance(self.encoding, list):
|
|
195
|
+
if self.method == EncodingMethods.COLUMN:
|
|
181
196
|
x = self.encoding
|
|
182
197
|
if not re.search(r"^[A-Z]{1,3}$", str(x)):
|
|
183
198
|
raise TablassertValidationError(f"`encoding` must be an Excel-style column name (A-ZZ), got {x!r}.", code="encoding-bad-excel-column")
|
|
184
199
|
|
|
185
200
|
return self
|
|
186
201
|
|
|
187
|
-
@model_validator(mode="after")
|
|
188
|
-
def list_method_consistency(self: Self) -> Self:
|
|
189
|
-
"""Enforce that ``method: list`` carries a literal list and no scalar string ops.
|
|
190
|
-
|
|
191
|
-
``method: list`` is the multivalued counterpart of ``method: value``: the
|
|
192
|
-
``encoding`` is a literal list emitted as a real JSON array (for multivalued
|
|
193
|
-
Biolink slots such as ``has_evidence``). The scalar string ops
|
|
194
|
-
(``regex``/``remove``/``prefix``/``suffix``/``transformations``/``fill``/``explode_by``)
|
|
195
|
-
operate on a single string per row and would mangle a list column, so they are
|
|
196
|
-
rejected here — encode the final values directly instead.
|
|
197
|
-
"""
|
|
198
|
-
is_list: bool = isinstance(self.encoding, list)
|
|
199
|
-
if self.method == EncodingMethods.LIST:
|
|
200
|
-
if not is_list:
|
|
201
|
-
raise TablassertValidationError("`method: list` requires `encoding` to be a list of values.", code="encoding-list-requires-list")
|
|
202
|
-
scalar_ops: list[str] = []
|
|
203
|
-
if self.regex:
|
|
204
|
-
scalar_ops.append("regex")
|
|
205
|
-
if self.fill is not None:
|
|
206
|
-
scalar_ops.append("fill")
|
|
207
|
-
if self.explode_by is not None:
|
|
208
|
-
scalar_ops.append("explode_by")
|
|
209
|
-
if self.remove:
|
|
210
|
-
scalar_ops.append("remove")
|
|
211
|
-
if self.prefix:
|
|
212
|
-
scalar_ops.append("prefix")
|
|
213
|
-
if self.suffix:
|
|
214
|
-
scalar_ops.append("suffix")
|
|
215
|
-
if self.transformations:
|
|
216
|
-
scalar_ops.append("transformations")
|
|
217
|
-
if scalar_ops:
|
|
218
|
-
raise TablassertValidationError(
|
|
219
|
-
f"`method: list` is a literal list and is incompatible with the scalar string ops "
|
|
220
|
-
f"({', '.join(scalar_ops)}); apply them upstream or encode the final values directly.",
|
|
221
|
-
code="encoding-list-incompatible-ops",
|
|
222
|
-
)
|
|
223
|
-
elif is_list:
|
|
224
|
-
raise TablassertValidationError("A list `encoding` requires `method: list`.", code="encoding-list-requires-list")
|
|
225
|
-
|
|
226
|
-
return self
|
|
227
|
-
|
|
228
202
|
regex: list[Regex] | None = Field(
|
|
229
203
|
None,
|
|
230
204
|
description="Ordered regex replacements applied to encoded text.",
|
|
@@ -306,23 +280,6 @@ class NodeEncoding(Encoding):
|
|
|
306
280
|
|
|
307
281
|
return exclude_regex
|
|
308
282
|
|
|
309
|
-
@model_validator(mode="after")
|
|
310
|
-
def reject_list_method(self: Self) -> Self:
|
|
311
|
-
"""Reject ``method: list`` on node encodings (subject/object/qualifiers).
|
|
312
|
-
|
|
313
|
-
``method: list`` is the multivalued counterpart of ``method: value`` and only
|
|
314
|
-
makes sense on an annotation (a multivalued Biolink slot). A subject/object/
|
|
315
|
-
qualifier is a single entity: a list node column crashes resolution deep in the
|
|
316
|
-
pipeline (a polars list-to-string cast) instead of failing at config time, and an
|
|
317
|
-
enum-ranged qualifier would silently emit a list where Biolink expects one token.
|
|
318
|
-
"""
|
|
319
|
-
if self.method == EncodingMethods.LIST:
|
|
320
|
-
raise TablassertValidationError(
|
|
321
|
-
"`method: list` is only valid on annotations (multivalued Biolink slots); subject/object/qualifier nodes are single entities.",
|
|
322
|
-
code="encoding-list-annotation-only",
|
|
323
|
-
)
|
|
324
|
-
return self
|
|
325
|
-
|
|
326
283
|
|
|
327
284
|
class Qualifier(NodeEncoding):
|
|
328
285
|
qualifier: Qualifiers = Field(
|
|
@@ -330,6 +287,15 @@ class Qualifier(NodeEncoding):
|
|
|
330
287
|
description="Qualifier predicate key used as the output qualifier column.",
|
|
331
288
|
examples=[Qualifiers.OBJECT_DIRECTION_QUALIFIER, Qualifiers.SUBJECT_CONTEXT_QUALIFIER],
|
|
332
289
|
)
|
|
290
|
+
nullable: bool = Field(
|
|
291
|
+
False,
|
|
292
|
+
description=(
|
|
293
|
+
"When True, a blank or unresolvable ``method: column`` cell keeps the edge and omits the "
|
|
294
|
+
"qualifier for that row (the column stays null and the null-stripper drops the key); when "
|
|
295
|
+
"False (default) such a row is dropped, exactly like an unresolved subject/object. Only "
|
|
296
|
+
"meaningful for ``method: column`` — a literal qualifier can never be null."
|
|
297
|
+
),
|
|
298
|
+
)
|
|
333
299
|
|
|
334
300
|
@property
|
|
335
301
|
def vocabulary(self: Self) -> frozenset[str] | None:
|
|
@@ -400,6 +366,23 @@ class Qualifier(NodeEncoding):
|
|
|
400
366
|
)
|
|
401
367
|
return self
|
|
402
368
|
|
|
369
|
+
@model_validator(mode="after")
|
|
370
|
+
def reject_nullable_literal_qualifiers(self: Self) -> Self:
|
|
371
|
+
"""Reject ``nullable: true`` on literal qualifiers.
|
|
372
|
+
|
|
373
|
+
``nullable`` only has meaning for a ``method: column`` qualifier: a blank or
|
|
374
|
+
unresolved cell keeps the edge and the qualifier is omitted for that row. A
|
|
375
|
+
``method: value`` qualifier is a config-time constant that can never be blank,
|
|
376
|
+
so ``nullable`` would be dead config that misleads the reader. Fail loudly at
|
|
377
|
+
config time instead (the removed ``method: list`` is already rejected upstream
|
|
378
|
+
by :meth:`Encoding.reject_removed_list_method`).
|
|
379
|
+
"""
|
|
380
|
+
if self.nullable and self.method != EncodingMethods.COLUMN:
|
|
381
|
+
raise TablassertValidationError(
|
|
382
|
+
"`nullable` only applies to `method: column` qualifiers; a literal qualifier can never be null.", code="qualifier-nullable-literal"
|
|
383
|
+
)
|
|
384
|
+
return self
|
|
385
|
+
|
|
403
386
|
|
|
404
387
|
class Statement(TablaBase):
|
|
405
388
|
subject: NodeEncoding = Field(..., description="Subject node encoding and mapping configuration.")
|
|
@@ -508,19 +491,15 @@ class Annotation(Encoding):
|
|
|
508
491
|
def split_by_requires_a_column(self) -> Self:
|
|
509
492
|
"""Enforce that ``split_by`` carries a real separator and a ``method: column`` encoding.
|
|
510
493
|
|
|
511
|
-
``split_by`` is the
|
|
512
|
-
|
|
513
|
-
literal
|
|
514
|
-
the same array on every row). A ``value``/``list`` encoding therefore declares its
|
|
515
|
-
members directly rather than round-tripping them through a separator.
|
|
494
|
+
``split_by`` is the one multivalued encoding: it turns each cell's own
|
|
495
|
+
delimited text into a real JSON array, per row. A ``value`` encoding is a
|
|
496
|
+
scalar literal with no per-row text to split, so it rejects ``split_by``.
|
|
516
497
|
"""
|
|
517
498
|
if self.split_by is None:
|
|
518
499
|
return self
|
|
519
500
|
if self.method != EncodingMethods.COLUMN:
|
|
520
501
|
raise TablassertValidationError(
|
|
521
|
-
"`split_by` splits a column's per-row text and requires `method: column
|
|
522
|
-
"declare a literal multivalued annotation with `method: list` instead.",
|
|
523
|
-
code="annotation-split-by-requires-column",
|
|
502
|
+
"`split_by` splits a column's per-row text and requires `method: column`.", code="annotation-split-by-requires-column"
|
|
524
503
|
)
|
|
525
504
|
if not self.split_by:
|
|
526
505
|
# An empty separator splits into individual characters -- exactly the
|
|
@@ -536,18 +515,24 @@ class Annotation(Encoding):
|
|
|
536
515
|
# is the real problem -- an author asking for `supporting_study_size` has no way to discover
|
|
537
516
|
# that Biolink attaches it to no class and the pipeline rerouted it.
|
|
538
517
|
name: str = str(self.annotation)
|
|
539
|
-
|
|
518
|
+
# Judge the coerced target, not the raw alias: the clean-phase column coercions rename
|
|
519
|
+
# statistical aliases to their canonical slot before any relocation runs, so
|
|
520
|
+
# `adjusted p value` reaches the edge as `adjusted_p_value` and warning on the alias
|
|
521
|
+
# is a false positive.
|
|
522
|
+
target: str = pvalue_target(name) or study_size_target(name) or effect_size_target(name) or effect_type_target(name) or name
|
|
523
|
+
shown: str = f"`{name}` (coerced to `{target}`)" if target != name else f"`{name}`"
|
|
524
|
+
if target in UNSATISFIABLE_EDGE_FIELDS:
|
|
540
525
|
warnings.warn(
|
|
541
|
-
f"
|
|
526
|
+
f"{shown} is declared in biolink-model {BIOLINK_VERSION} but attached to no association class, "
|
|
542
527
|
"so it cannot be emitted on an edge; its value is routed onto the inlined supporting study "
|
|
543
528
|
"instead. Use a slot a Biolink association declares (e.g. `p_value`, `adjusted_p_value`) if you "
|
|
544
529
|
"need it on the edge itself.",
|
|
545
530
|
BiolinkRelocationWarning,
|
|
546
531
|
stacklevel=2,
|
|
547
532
|
)
|
|
548
|
-
elif
|
|
533
|
+
elif target not in ALLOWED_EDGE_FIELDS:
|
|
549
534
|
warnings.warn(
|
|
550
|
-
f"
|
|
535
|
+
f"{shown} is not a Biolink association slot, so it is folded into `supporting_text` as a "
|
|
551
536
|
f'"{name}: <value>" string rather than emitted as its own edge field.',
|
|
552
537
|
BiolinkRelocationWarning,
|
|
553
538
|
stacklevel=2,
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from collections import Counter
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from tablassert.log import cat
|
|
9
|
+
|
|
10
|
+
logger = cat("QC")
|
|
11
|
+
|
|
12
|
+
# Human-readable phrasing for each assertion, used by format_violations.
|
|
13
|
+
_MESSAGES: dict[str, str] = {
|
|
14
|
+
"file-missing": "file not found",
|
|
15
|
+
"malformed-lines": "empty or malformed JSON lines",
|
|
16
|
+
"whitespace-values": "values with leading/trailing whitespace",
|
|
17
|
+
"duplicate-node-ids": "duplicate node ids",
|
|
18
|
+
"undeclared-nodes": "nodes referenced by edges but not declared in the nodes file",
|
|
19
|
+
"isolated-nodes": "declared nodes participating in no edge",
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass
|
|
24
|
+
class StudyViolation:
|
|
25
|
+
"""One failed study assertion over the final KGX NDJSON files.
|
|
26
|
+
|
|
27
|
+
Attributes:
|
|
28
|
+
check: Assertion key into ``_MESSAGES`` (e.g. ``"duplicate-node-ids"``).
|
|
29
|
+
label: Which file the violation belongs to (``"nodes"`` or ``"edges"``).
|
|
30
|
+
count: Exact number of offending records/values.
|
|
31
|
+
examples: Capped list of example offenders for the stderr summary.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
check: str
|
|
35
|
+
label: str
|
|
36
|
+
count: int
|
|
37
|
+
examples: list[str] = field(default_factory=list)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@dataclass
|
|
41
|
+
class _FileScan:
|
|
42
|
+
"""Accumulated facts from one streamed pass over an NDJSON file."""
|
|
43
|
+
|
|
44
|
+
ids: set[str]
|
|
45
|
+
duplicate_ids: Counter[str]
|
|
46
|
+
whitespace: Counter[str]
|
|
47
|
+
malformed: int
|
|
48
|
+
missing: bool
|
|
49
|
+
path: Path
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _scan_ndjson(path: Path, *, edge: bool) -> _FileScan:
|
|
53
|
+
"""Stream one NDJSON file, collecting the facts the study assertions need.
|
|
54
|
+
|
|
55
|
+
Args:
|
|
56
|
+
path: Path to a ``.nodes.ndjson`` or ``.edges.ndjson`` file.
|
|
57
|
+
edge: ``True`` to collect referenced ids from ``subject``/``object``;
|
|
58
|
+
``False`` to collect declared node ``id``s and track duplicates.
|
|
59
|
+
|
|
60
|
+
Returns:
|
|
61
|
+
A :class:`_FileScan`; ``missing`` is set (and nothing else) when the
|
|
62
|
+
file does not exist, so a typo'd path can never read as a clean pass.
|
|
63
|
+
"""
|
|
64
|
+
scan: _FileScan = _FileScan(set(), Counter(), Counter(), 0, not path.is_file(), path)
|
|
65
|
+
if scan.missing:
|
|
66
|
+
return scan
|
|
67
|
+
with path.open(encoding="utf-8") as handle:
|
|
68
|
+
for line in handle:
|
|
69
|
+
if not line.strip():
|
|
70
|
+
scan.malformed += 1
|
|
71
|
+
continue
|
|
72
|
+
try:
|
|
73
|
+
record: object = json.loads(line)
|
|
74
|
+
except json.JSONDecodeError:
|
|
75
|
+
scan.malformed += 1
|
|
76
|
+
continue
|
|
77
|
+
if not isinstance(record, dict):
|
|
78
|
+
scan.malformed += 1
|
|
79
|
+
continue
|
|
80
|
+
for key, value in record.items():
|
|
81
|
+
if isinstance(value, str) and value != value.strip():
|
|
82
|
+
scan.whitespace[key] += 1
|
|
83
|
+
if edge:
|
|
84
|
+
for role in ("subject", "object"):
|
|
85
|
+
ident: object = record.get(role)
|
|
86
|
+
if isinstance(ident, str):
|
|
87
|
+
scan.ids.add(ident)
|
|
88
|
+
else:
|
|
89
|
+
ident = record.get("id")
|
|
90
|
+
if isinstance(ident, str):
|
|
91
|
+
# Rust dedup only removes byte-identical lines, so a repeated id
|
|
92
|
+
# here means two nodes share an id with different content.
|
|
93
|
+
if ident in scan.ids:
|
|
94
|
+
scan.duplicate_ids[ident] += 1
|
|
95
|
+
scan.ids.add(ident)
|
|
96
|
+
return scan
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def study_kgx(nodes_path: Path, edges_path: Path, *, example_limit: int = 10) -> list[StudyViolation]:
|
|
100
|
+
"""Assert over the final KGX NDJSON files, in the spirit of studyKGtsvs.pl.
|
|
101
|
+
|
|
102
|
+
Streams both files once each and checks: duplicate node ids, nodes referenced
|
|
103
|
+
by edges but never declared (``undeclared``), declared nodes participating in
|
|
104
|
+
no edge (``isolated``), empty/malformed lines, and string values carrying
|
|
105
|
+
leading/trailing whitespace. Every check is an assertion -- the caller decides
|
|
106
|
+
whether violations fail the build.
|
|
107
|
+
|
|
108
|
+
Args:
|
|
109
|
+
nodes_path: Path to ``<name>_<version>.nodes.ndjson``.
|
|
110
|
+
edges_path: Path to ``<name>_<version>.edges.ndjson``.
|
|
111
|
+
example_limit: Maximum number of examples retained per violation.
|
|
112
|
+
|
|
113
|
+
Returns:
|
|
114
|
+
A list of :class:`StudyViolation`; empty when every assertion passes.
|
|
115
|
+
"""
|
|
116
|
+
nodes: _FileScan = _scan_ndjson(nodes_path, edge=False)
|
|
117
|
+
edges: _FileScan = _scan_ndjson(edges_path, edge=True)
|
|
118
|
+
|
|
119
|
+
violations: list[StudyViolation] = []
|
|
120
|
+
for label, scan in (("nodes", nodes), ("edges", edges)):
|
|
121
|
+
if scan.missing:
|
|
122
|
+
violations.append(StudyViolation("file-missing", label, 1, [str(scan.path)]))
|
|
123
|
+
continue
|
|
124
|
+
if scan.malformed:
|
|
125
|
+
violations.append(StudyViolation("malformed-lines", label, scan.malformed))
|
|
126
|
+
if scan.whitespace:
|
|
127
|
+
examples: list[str] = [f"{field_name} ({n})" for field_name, n in scan.whitespace.most_common(example_limit)]
|
|
128
|
+
violations.append(StudyViolation("whitespace-values", label, sum(scan.whitespace.values()), examples))
|
|
129
|
+
if nodes.duplicate_ids:
|
|
130
|
+
examples = [ident for ident, _ in nodes.duplicate_ids.most_common(example_limit)]
|
|
131
|
+
violations.append(StudyViolation("duplicate-node-ids", "nodes", len(nodes.duplicate_ids), examples))
|
|
132
|
+
if not nodes.missing and not edges.missing:
|
|
133
|
+
undeclared: list[str] = sorted(edges.ids - nodes.ids)
|
|
134
|
+
if undeclared:
|
|
135
|
+
violations.append(StudyViolation("undeclared-nodes", "edges", len(undeclared), undeclared[:example_limit]))
|
|
136
|
+
isolated: list[str] = sorted(nodes.ids - edges.ids)
|
|
137
|
+
if isolated:
|
|
138
|
+
violations.append(StudyViolation("isolated-nodes", "nodes", len(isolated), isolated[:example_limit]))
|
|
139
|
+
return violations
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def format_violations(violations: list[StudyViolation]) -> str:
|
|
143
|
+
"""Render violations as a human-readable, one-line-per-assertion summary.
|
|
144
|
+
|
|
145
|
+
Args:
|
|
146
|
+
violations: The failed assertions returned by :func:`study_kgx`.
|
|
147
|
+
|
|
148
|
+
Returns:
|
|
149
|
+
Newline-joined summary lines, e.g.
|
|
150
|
+
``nodes: 3 duplicate node ids (e.g. HGNC:5, HGNC:6)``.
|
|
151
|
+
"""
|
|
152
|
+
lines: list[str] = []
|
|
153
|
+
for violation in violations:
|
|
154
|
+
line: str = f"{violation.label}: {violation.count} {_MESSAGES[violation.check]}"
|
|
155
|
+
if violation.examples:
|
|
156
|
+
line += f" (e.g. {', '.join(violation.examples)})"
|
|
157
|
+
lines.append(line)
|
|
158
|
+
return "\n".join(lines)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|