tablassert 10.0.0__tar.gz → 10.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. {tablassert-10.0.0 → tablassert-10.1.0}/PKG-INFO +1 -1
  2. {tablassert-10.0.0 → tablassert-10.1.0}/pyproject.toml +1 -1
  3. {tablassert-10.0.0 → tablassert-10.1.0}/src/tablassert/cli.py +36 -5
  4. {tablassert-10.0.0 → tablassert-10.1.0}/src/tablassert/lib.py +17 -7
  5. {tablassert-10.0.0 → tablassert-10.1.0}/src/tablassert/models.py +11 -4
  6. tablassert-10.1.0/src/tablassert/study.py +158 -0
  7. {tablassert-10.0.0 → tablassert-10.1.0}/LICENSE +0 -0
  8. {tablassert-10.0.0 → tablassert-10.1.0}/README.md +0 -0
  9. {tablassert-10.0.0 → tablassert-10.1.0}/rust/Cargo.lock +0 -0
  10. {tablassert-10.0.0 → tablassert-10.1.0}/rust/Cargo.toml +0 -0
  11. {tablassert-10.0.0 → tablassert-10.1.0}/rust/examples/count_tables.rs +0 -0
  12. {tablassert-10.0.0 → tablassert-10.1.0}/rust/src/fullmap.rs +0 -0
  13. {tablassert-10.0.0 → tablassert-10.1.0}/rust/src/json.rs +0 -0
  14. {tablassert-10.0.0 → tablassert-10.1.0}/rust/src/lib.rs +0 -0
  15. {tablassert-10.0.0 → tablassert-10.1.0}/rust/src/ndjson.rs +0 -0
  16. {tablassert-10.0.0 → tablassert-10.1.0}/rust/src/uuid.rs +0 -0
  17. {tablassert-10.0.0 → tablassert-10.1.0}/rust/tests/build_golden.rs +0 -0
  18. {tablassert-10.0.0 → tablassert-10.1.0}/src/tablassert/__init__.py +0 -0
  19. {tablassert-10.0.0 → tablassert-10.1.0}/src/tablassert/_lazy.py +0 -0
  20. {tablassert-10.0.0 → tablassert-10.1.0}/src/tablassert/agent.py +0 -0
  21. {tablassert-10.0.0 → tablassert-10.1.0}/src/tablassert/biolink.py +0 -0
  22. {tablassert-10.0.0 → tablassert-10.1.0}/src/tablassert/coerce.py +0 -0
  23. {tablassert-10.0.0 → tablassert-10.1.0}/src/tablassert/enums.py +0 -0
  24. {tablassert-10.0.0 → tablassert-10.1.0}/src/tablassert/errors.py +0 -0
  25. {tablassert-10.0.0 → tablassert-10.1.0}/src/tablassert/extras.py +0 -0
  26. {tablassert-10.0.0 → tablassert-10.1.0}/src/tablassert/fullmap.py +0 -0
  27. {tablassert-10.0.0 → tablassert-10.1.0}/src/tablassert/graph_registry.py +0 -0
  28. {tablassert-10.0.0 → tablassert-10.1.0}/src/tablassert/ingests.py +0 -0
  29. {tablassert-10.0.0 → tablassert-10.1.0}/src/tablassert/log.py +0 -0
  30. {tablassert-10.0.0 → tablassert-10.1.0}/src/tablassert/nlp.py +0 -0
  31. {tablassert-10.0.0 → tablassert-10.1.0}/src/tablassert/progress.py +0 -0
  32. {tablassert-10.0.0 → tablassert-10.1.0}/src/tablassert/qc.py +0 -0
  33. {tablassert-10.0.0 → tablassert-10.1.0}/src/tablassert/rig.py +0 -0
  34. {tablassert-10.0.0 → tablassert-10.1.0}/src/tablassert/rs.pyi +0 -0
  35. {tablassert-10.0.0 → tablassert-10.1.0}/src/tablassert/utils.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: tablassert
3
- Version: 10.0.0
3
+ Version: 10.1.0
4
4
  Classifier: License :: OSI Approved :: Apache Software License
5
5
  Classifier: Development Status :: 5 - Production/Stable
6
6
  Classifier: Intended Audience :: Science/Research
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "tablassert"
3
- version = "10.0.0"
3
+ version = "10.1.0"
4
4
  description = "Extract knowledge assertions from tabular data into NCATS Translator-compliant KGX NDJSON — declaratively, with entity resolution and quality control built in."
5
5
  authors = [
6
6
  { name = "Skye Lane Goetz", email = "sgoetz@isbscience.org" }
@@ -121,13 +121,15 @@ def build_pipeline(
121
121
  """Build a knowledge graph from a YAML configuration file.
122
122
 
123
123
  Runs the six-stage build pipeline: load tables → extract sections → build
124
- Tcodes → collect instructions → build subgraphs → compile graph.
124
+ Tcodes → collect instructions → build subgraphs → compile graph. With ``qc``
125
+ enabled a seventh stage studies the final NDJSON files.
125
126
 
126
127
  Args:
127
128
  configuration_file: Path to the graph YAML file.
128
129
  progress: Pipeline progress reporter.
129
130
  release: When ``True``, emit release-mode artifacts.
130
- qc: When ``True``, run quality-control audits on each section.
131
+ qc: When ``True``, run quality-control audits on each section and assert
132
+ over the final NDJSON files (failing the build on any violation).
131
133
  log: When ``True``, enable per-section verbose logging.
132
134
  head: When ``True``, preview a random sample of up to 5 rows per section (fast schema/shape check).
133
135
 
@@ -218,9 +220,35 @@ def build_pipeline(
218
220
  subgraphs, g.name, g.version, g.description, g.contributions, g.ui_explanation, g.tables, g.infores, on_phase=sub_step, on_subgraph=advance
219
221
  )
220
222
 
223
+ # Stage 7/7 (only with --qc): assert over the final NDJSON files.
224
+ if qc:
225
+ progress.stage("Studying Graph")
226
+ study_final_ndjson(g.name, g.version)
227
+
221
228
  logger.info("Built graph {name} v{version}: {n} sections", name=g.name, version=g.version, n=n)
222
229
 
223
230
 
231
+ def study_final_ndjson(name: str, version: str) -> None:
232
+ """Run study assertions over a build's final NDJSON files (the ``--qc`` stage 7).
233
+
234
+ Args:
235
+ name: Graph name, used to locate ``./<name>_<version>.nodes.ndjson``.
236
+ version: Graph version, used to locate ``./<name>_<version>.edges.ndjson``.
237
+
238
+ Raises:
239
+ SystemExit: With status 1 when any study assertion is violated.
240
+ """
241
+ from tablassert.study import format_violations, study_kgx
242
+
243
+ violations = study_kgx(Path(f"./{name}_{version}.nodes.ndjson"), Path(f"./{name}_{version}.edges.ndjson"))
244
+ if violations:
245
+ summary: str = format_violations(violations)
246
+ print(summary, file=sys.stderr)
247
+ logger.warning("study assertions failed on final NDJSON:\n{summary}", summary=summary)
248
+ raise SystemExit(1)
249
+ logger.info("study assertions passed on final NDJSON")
250
+
251
+
224
252
  def validate_pipeline(table_configuration_file: Path, progress: PipelineProgress) -> None:
225
253
  """Validate section syntax from a YAML configuration file.
226
254
 
@@ -599,11 +627,14 @@ def build_kg(
599
627
 
600
628
  ``--qc`` requires the ``[qc]`` extra (``pip install "tablassert[qc]"``); it is
601
629
  checked before the build starts, because the audit stage runs LAST and a missing
602
- extra would otherwise surface only after entity resolution has finished.
630
+ extra would otherwise surface only after entity resolution has finished. It also
631
+ runs a final study stage that asserts over the emitted NDJSON -- no duplicate node
632
+ ids, no undeclared or isolated nodes, no malformed lines or stray whitespace --
633
+ and fails the build (non-zero exit) when any assertion is violated.
603
634
  """
604
635
  if qc:
605
636
  extras.require("qc", required_by="--qc")
606
- run(6, build_pipeline, graph_configuration_file, release=release, qc=qc, log=log, head=head)
637
+ run(7 if qc else 6, build_pipeline, graph_configuration_file, release=release, qc=qc, log=log, head=head)
607
638
 
608
639
 
609
640
  @APP.command(name="validate")
@@ -950,7 +981,7 @@ def _prebuilt_fullmap_urls(babel_version: str) -> tuple[str, str]:
950
981
 
951
982
  RENCI publishes a prebuilt ``fullmap.tar.zst`` (and a ``sha256sum.txt``) under
952
983
  ``{BABEL_BASE}/{babel_version}/fullmap/{tablassert_version}/``, where the version
953
- directory is the INSTALLED Tablassert package version (e.g. ``10.0.0``) — resolved from
984
+ directory is the INSTALLED Tablassert package version (e.g. ``10.1.0``) — resolved from
954
985
  installed-package metadata, never hardcoded, so a new release looks itself up.
955
986
 
956
987
  Args:
@@ -262,8 +262,10 @@ def prune_to_class(lf: pl.LazyFrame) -> pl.LazyFrame:
262
262
 
263
263
  Categories vary per row within a section, so this masks per row rather than
264
264
  dropping columns: values are nulled where the row's class rejects them, and the
265
- Rust null-stripper then removes the key entirely. Scalars are wrapped where the
266
- class declares the slot multivalued.
265
+ Rust null-stripper then removes the key entirely. Scalars are wrapped into
266
+ one-element lists when the slot is multivalued on every class that declares it
267
+ (a column has one dtype, so per-row wrapping is impossible; a slot Biolink
268
+ declares scalar on some classes and multivalued on others stays scalar).
267
269
 
268
270
  Args:
269
271
  lf: Edges LazyFrame carrying a resolved ``category`` column.
@@ -306,10 +308,18 @@ def prune_to_class(lf: pl.LazyFrame) -> pl.LazyFrame:
306
308
  # Preserve what the class refuses rather than deleting it outright; the
307
309
  # value is real evidence, it just has no slot on this association class.
308
310
  rescued.append(pl.when(ok | text.is_null()).then(None).otherwise(pl.concat_str([pl.lit(f"{col}="), text])))
309
- # Biolink makes the same slot multivalued on some classes and scalar on others.
310
- listed: dict[str, bool] = {cat: is_multivalued(association_class(cat), col) for cat in categories}
311
- if any(listed.values()) and not isinstance(schema[col], pl.List):
312
- keep = pl.when(first.replace_strict(listed, default=False)).then(pl.concat_list(keep)).otherwise(keep)
311
+ # A column has one dtype, so the multivalued wrap must be uniform across rows:
312
+ # wrap every value into a one-element list when every class declaring the slot
313
+ # types it multivalued. `is_multivalued` is False for classes that do not declare
314
+ # the slot at all, so the scan is restricted to declaring classes -- including
315
+ # them is what produced a spuriously mixed per-row wrap that died in
316
+ # strict_cast at collect. Rows whose class rejects the slot are already null and
317
+ # stay null. A hypothetically mixed slot keeps its scalar rather than crashing.
318
+ declaring: list[str] = [cat for cat in categories if accepts[cat]]
319
+ listed: dict[str, bool] = {cat: is_multivalued(association_class(cat), col) for cat in declaring}
320
+ if listed and all(listed.values()) and not isinstance(schema[col], pl.List):
321
+ # concat_list maps null -> [null]; the when preserves real nulls instead.
322
+ keep = pl.when(keep.is_null()).then(None).otherwise(pl.concat_list(keep))
313
323
  updates.append(keep.alias(col))
314
324
  if rescued:
315
325
  updates.append(pl.concat_list(rescued).list.drop_nulls().alias(PRUNED_COLUMN))
@@ -632,7 +642,7 @@ def split_list(lf: pl.LazyFrame, col: str, delimiter: str) -> pl.LazyFrame:
632
642
 
633
643
  A column encoding is scalar by construction, so a multivalued Biolink slot such as
634
644
  ``has_evidence`` fed from an aggregated cell would otherwise be emitted as a single
635
- joined string -- and ``mask_illegal_edge_fields`` wraps that scalar into a
645
+ joined string -- and ``prune_to_class`` wraps that scalar into a
636
646
  one-element list, so the value survives Biolink validation while consumers iterate a
637
647
  single ``"a|b|c"`` blob instead of three ids.
638
648
 
@@ -19,6 +19,7 @@ from tablassert.biolink import (
19
19
  Predicates,
20
20
  Qualifiers,
21
21
  )
22
+ from tablassert.coerce import effect_size_target, effect_type_target, pvalue_target, study_size_target
22
23
  from tablassert.enums import Comparisons, EncodingMethods, Files, FillMethods, Functions, Repositories, Tokens
23
24
  from tablassert.errors import BiolinkRelocationWarning, TablassertErrorCodes, TablassertValidationError
24
25
 
@@ -514,18 +515,24 @@ class Annotation(Encoding):
514
515
  # is the real problem -- an author asking for `supporting_study_size` has no way to discover
515
516
  # that Biolink attaches it to no class and the pipeline rerouted it.
516
517
  name: str = str(self.annotation)
517
- if name in UNSATISFIABLE_EDGE_FIELDS:
518
+ # Judge the coerced target, not the raw alias: the clean-phase column coercions rename
519
+ # statistical aliases to their canonical slot before any relocation runs, so
520
+ # `adjusted p value` reaches the edge as `adjusted_p_value` and warning on the alias
521
+ # is a false positive.
522
+ target: str = pvalue_target(name) or study_size_target(name) or effect_size_target(name) or effect_type_target(name) or name
523
+ shown: str = f"`{name}` (coerced to `{target}`)" if target != name else f"`{name}`"
524
+ if target in UNSATISFIABLE_EDGE_FIELDS:
518
525
  warnings.warn(
519
- f"`{name}` is declared in biolink-model {BIOLINK_VERSION} but attached to no association class, "
526
+ f"{shown} is declared in biolink-model {BIOLINK_VERSION} but attached to no association class, "
520
527
  "so it cannot be emitted on an edge; its value is routed onto the inlined supporting study "
521
528
  "instead. Use a slot a Biolink association declares (e.g. `p_value`, `adjusted_p_value`) if you "
522
529
  "need it on the edge itself.",
523
530
  BiolinkRelocationWarning,
524
531
  stacklevel=2,
525
532
  )
526
- elif name not in ALLOWED_EDGE_FIELDS:
533
+ elif target not in ALLOWED_EDGE_FIELDS:
527
534
  warnings.warn(
528
- f"`{name}` is not a Biolink association slot, so it is folded into `supporting_text` as a "
535
+ f"{shown} is not a Biolink association slot, so it is folded into `supporting_text` as a "
529
536
  f'"{name}: <value>" string rather than emitted as its own edge field.',
530
537
  BiolinkRelocationWarning,
531
538
  stacklevel=2,
@@ -0,0 +1,158 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ from collections import Counter
5
+ from dataclasses import dataclass, field
6
+ from pathlib import Path
7
+
8
+ from tablassert.log import cat
9
+
10
+ logger = cat("QC")
11
+
12
+ # Human-readable phrasing for each assertion, used by format_violations.
13
+ _MESSAGES: dict[str, str] = {
14
+ "file-missing": "file not found",
15
+ "malformed-lines": "empty or malformed JSON lines",
16
+ "whitespace-values": "values with leading/trailing whitespace",
17
+ "duplicate-node-ids": "duplicate node ids",
18
+ "undeclared-nodes": "nodes referenced by edges but not declared in the nodes file",
19
+ "isolated-nodes": "declared nodes participating in no edge",
20
+ }
21
+
22
+
23
+ @dataclass
24
+ class StudyViolation:
25
+ """One failed study assertion over the final KGX NDJSON files.
26
+
27
+ Attributes:
28
+ check: Assertion key into ``_MESSAGES`` (e.g. ``"duplicate-node-ids"``).
29
+ label: Which file the violation belongs to (``"nodes"`` or ``"edges"``).
30
+ count: Exact number of offending records/values.
31
+ examples: Capped list of example offenders for the stderr summary.
32
+ """
33
+
34
+ check: str
35
+ label: str
36
+ count: int
37
+ examples: list[str] = field(default_factory=list)
38
+
39
+
40
+ @dataclass
41
+ class _FileScan:
42
+ """Accumulated facts from one streamed pass over an NDJSON file."""
43
+
44
+ ids: set[str]
45
+ duplicate_ids: Counter[str]
46
+ whitespace: Counter[str]
47
+ malformed: int
48
+ missing: bool
49
+ path: Path
50
+
51
+
52
+ def _scan_ndjson(path: Path, *, edge: bool) -> _FileScan:
53
+ """Stream one NDJSON file, collecting the facts the study assertions need.
54
+
55
+ Args:
56
+ path: Path to a ``.nodes.ndjson`` or ``.edges.ndjson`` file.
57
+ edge: ``True`` to collect referenced ids from ``subject``/``object``;
58
+ ``False`` to collect declared node ``id``s and track duplicates.
59
+
60
+ Returns:
61
+ A :class:`_FileScan`; ``missing`` is set (and nothing else) when the
62
+ file does not exist, so a typo'd path can never read as a clean pass.
63
+ """
64
+ scan: _FileScan = _FileScan(set(), Counter(), Counter(), 0, not path.is_file(), path)
65
+ if scan.missing:
66
+ return scan
67
+ with path.open(encoding="utf-8") as handle:
68
+ for line in handle:
69
+ if not line.strip():
70
+ scan.malformed += 1
71
+ continue
72
+ try:
73
+ record: object = json.loads(line)
74
+ except json.JSONDecodeError:
75
+ scan.malformed += 1
76
+ continue
77
+ if not isinstance(record, dict):
78
+ scan.malformed += 1
79
+ continue
80
+ for key, value in record.items():
81
+ if isinstance(value, str) and value != value.strip():
82
+ scan.whitespace[key] += 1
83
+ if edge:
84
+ for role in ("subject", "object"):
85
+ ident: object = record.get(role)
86
+ if isinstance(ident, str):
87
+ scan.ids.add(ident)
88
+ else:
89
+ ident = record.get("id")
90
+ if isinstance(ident, str):
91
+ # Rust dedup only removes byte-identical lines, so a repeated id
92
+ # here means two nodes share an id with different content.
93
+ if ident in scan.ids:
94
+ scan.duplicate_ids[ident] += 1
95
+ scan.ids.add(ident)
96
+ return scan
97
+
98
+
99
+ def study_kgx(nodes_path: Path, edges_path: Path, *, example_limit: int = 10) -> list[StudyViolation]:
100
+ """Assert over the final KGX NDJSON files, in the spirit of studyKGtsvs.pl.
101
+
102
+ Streams both files once each and checks: duplicate node ids, nodes referenced
103
+ by edges but never declared (``undeclared``), declared nodes participating in
104
+ no edge (``isolated``), empty/malformed lines, and string values carrying
105
+ leading/trailing whitespace. Every check is an assertion -- the caller decides
106
+ whether violations fail the build.
107
+
108
+ Args:
109
+ nodes_path: Path to ``<name>_<version>.nodes.ndjson``.
110
+ edges_path: Path to ``<name>_<version>.edges.ndjson``.
111
+ example_limit: Maximum number of examples retained per violation.
112
+
113
+ Returns:
114
+ A list of :class:`StudyViolation`; empty when every assertion passes.
115
+ """
116
+ nodes: _FileScan = _scan_ndjson(nodes_path, edge=False)
117
+ edges: _FileScan = _scan_ndjson(edges_path, edge=True)
118
+
119
+ violations: list[StudyViolation] = []
120
+ for label, scan in (("nodes", nodes), ("edges", edges)):
121
+ if scan.missing:
122
+ violations.append(StudyViolation("file-missing", label, 1, [str(scan.path)]))
123
+ continue
124
+ if scan.malformed:
125
+ violations.append(StudyViolation("malformed-lines", label, scan.malformed))
126
+ if scan.whitespace:
127
+ examples: list[str] = [f"{field_name} ({n})" for field_name, n in scan.whitespace.most_common(example_limit)]
128
+ violations.append(StudyViolation("whitespace-values", label, sum(scan.whitespace.values()), examples))
129
+ if nodes.duplicate_ids:
130
+ examples = [ident for ident, _ in nodes.duplicate_ids.most_common(example_limit)]
131
+ violations.append(StudyViolation("duplicate-node-ids", "nodes", len(nodes.duplicate_ids), examples))
132
+ if not nodes.missing and not edges.missing:
133
+ undeclared: list[str] = sorted(edges.ids - nodes.ids)
134
+ if undeclared:
135
+ violations.append(StudyViolation("undeclared-nodes", "edges", len(undeclared), undeclared[:example_limit]))
136
+ isolated: list[str] = sorted(nodes.ids - edges.ids)
137
+ if isolated:
138
+ violations.append(StudyViolation("isolated-nodes", "nodes", len(isolated), isolated[:example_limit]))
139
+ return violations
140
+
141
+
142
+ def format_violations(violations: list[StudyViolation]) -> str:
143
+ """Render violations as a human-readable, one-line-per-assertion summary.
144
+
145
+ Args:
146
+ violations: The failed assertions returned by :func:`study_kgx`.
147
+
148
+ Returns:
149
+ Newline-joined summary lines, e.g.
150
+ ``nodes: 3 duplicate node ids (e.g. HGNC:5, HGNC:6)``.
151
+ """
152
+ lines: list[str] = []
153
+ for violation in violations:
154
+ line: str = f"{violation.label}: {violation.count} {_MESSAGES[violation.check]}"
155
+ if violation.examples:
156
+ line += f" (e.g. {', '.join(violation.examples)})"
157
+ lines.append(line)
158
+ return "\n".join(lines)
File without changes
File without changes
File without changes
File without changes
File without changes