tablassert 9.1.0__tar.gz → 10.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. {tablassert-9.1.0 → tablassert-10.1.0}/PKG-INFO +1 -1
  2. {tablassert-9.1.0 → tablassert-10.1.0}/pyproject.toml +1 -1
  3. {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/agent.py +5 -0
  4. {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/cli.py +36 -5
  5. {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/enums.py +0 -1
  6. {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/errors.py +2 -3
  7. {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/fullmap.py +18 -4
  8. {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/lib.py +42 -13
  9. {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/models.py +65 -80
  10. tablassert-10.1.0/src/tablassert/study.py +158 -0
  11. {tablassert-9.1.0 → tablassert-10.1.0}/LICENSE +0 -0
  12. {tablassert-9.1.0 → tablassert-10.1.0}/README.md +0 -0
  13. {tablassert-9.1.0 → tablassert-10.1.0}/rust/Cargo.lock +0 -0
  14. {tablassert-9.1.0 → tablassert-10.1.0}/rust/Cargo.toml +0 -0
  15. {tablassert-9.1.0 → tablassert-10.1.0}/rust/examples/count_tables.rs +0 -0
  16. {tablassert-9.1.0 → tablassert-10.1.0}/rust/src/fullmap.rs +0 -0
  17. {tablassert-9.1.0 → tablassert-10.1.0}/rust/src/json.rs +0 -0
  18. {tablassert-9.1.0 → tablassert-10.1.0}/rust/src/lib.rs +0 -0
  19. {tablassert-9.1.0 → tablassert-10.1.0}/rust/src/ndjson.rs +0 -0
  20. {tablassert-9.1.0 → tablassert-10.1.0}/rust/src/uuid.rs +0 -0
  21. {tablassert-9.1.0 → tablassert-10.1.0}/rust/tests/build_golden.rs +0 -0
  22. {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/__init__.py +0 -0
  23. {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/_lazy.py +0 -0
  24. {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/biolink.py +0 -0
  25. {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/coerce.py +0 -0
  26. {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/extras.py +0 -0
  27. {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/graph_registry.py +0 -0
  28. {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/ingests.py +0 -0
  29. {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/log.py +0 -0
  30. {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/nlp.py +0 -0
  31. {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/progress.py +0 -0
  32. {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/qc.py +0 -0
  33. {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/rig.py +0 -0
  34. {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/rs.pyi +0 -0
  35. {tablassert-9.1.0 → tablassert-10.1.0}/src/tablassert/utils.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: tablassert
3
- Version: 9.1.0
3
+ Version: 10.1.0
4
4
  Classifier: License :: OSI Approved :: Apache Software License
5
5
  Classifier: Development Status :: 5 - Production/Stable
6
6
  Classifier: Intended Audience :: Science/Research
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "tablassert"
3
- version = "9.1.0"
3
+ version = "10.1.0"
4
4
  description = "Extract knowledge assertions from tabular data into NCATS Translator-compliant KGX NDJSON — declaratively, with entity resolution and quality control built in."
5
5
  authors = [
6
6
  { name = "Skye Lane Goetz", email = "sgoetz@isbscience.org" }
@@ -2172,6 +2172,11 @@ qualifier and evidence slot the specific class declared. build_and_audit reports
2172
2172
  description rather than emitted on the edge. `q_value`, `fold_change`, `z_score`, `beta` and
2173
2173
  similar are not association slots at all and are folded into `supporting_text`. Prefer
2174
2174
  `p_value`, `adjusted_p_value`, `effect_size`, `effect_type`, `has_evidence`.
2175
+ - MULTIVALUED slots (`has_evidence` and friends) take a real JSON array, never a joined string:
2176
+ declare the annotation `{method: column, encoding: <letter>, split_by: "|"}` so each cell's
2177
+ delimited text splits into its own per-row array. `split_by` is the ONLY multivalued encoding
2178
+ — there is no literal-list method, and a scalar bound for a multivalued slot ships to consumers
2179
+ as one unusable "a|b|c" blob.
2175
2180
  - `effect_size` / `effect_type` are deliberate Tablassert extras pending biolink-model#1774 and
2176
2181
  are EXEMPT from the validity score: a `biolink_valid_pct` below 1.0 is never caused by them.
2177
2182
  - QUALIFIERS: enum-ranged qualifiers take a literal TOKEN, never a CURIE
@@ -121,13 +121,15 @@ def build_pipeline(
121
121
  """Build a knowledge graph from a YAML configuration file.
122
122
 
123
123
  Runs the six-stage build pipeline: load tables → extract sections → build
124
- Tcodes → collect instructions → build subgraphs → compile graph.
124
+ Tcodes → collect instructions → build subgraphs → compile graph. With ``qc``
125
+ enabled a seventh stage studies the final NDJSON files.
125
126
 
126
127
  Args:
127
128
  configuration_file: Path to the graph YAML file.
128
129
  progress: Pipeline progress reporter.
129
130
  release: When ``True``, emit release-mode artifacts.
130
- qc: When ``True``, run quality-control audits on each section.
131
+ qc: When ``True``, run quality-control audits on each section and assert
132
+ over the final NDJSON files (failing the build on any violation).
131
133
  log: When ``True``, enable per-section verbose logging.
132
134
  head: When ``True``, preview a random sample of up to 5 rows per section (fast schema/shape check).
133
135
 
@@ -218,9 +220,35 @@ def build_pipeline(
218
220
  subgraphs, g.name, g.version, g.description, g.contributions, g.ui_explanation, g.tables, g.infores, on_phase=sub_step, on_subgraph=advance
219
221
  )
220
222
 
223
+ # Stage 7/7 (only with --qc): assert over the final NDJSON files.
224
+ if qc:
225
+ progress.stage("Studying Graph")
226
+ study_final_ndjson(g.name, g.version)
227
+
221
228
  logger.info("Built graph {name} v{version}: {n} sections", name=g.name, version=g.version, n=n)
222
229
 
223
230
 
231
+ def study_final_ndjson(name: str, version: str) -> None:
232
+ """Run study assertions over a build's final NDJSON files (the ``--qc`` stage 7).
233
+
234
+ Args:
235
+ name: Graph name, used to locate ``./<name>_<version>.nodes.ndjson``.
236
+ version: Graph version, used to locate ``./<name>_<version>.edges.ndjson``.
237
+
238
+ Raises:
239
+ SystemExit: With status 1 when any study assertion is violated.
240
+ """
241
+ from tablassert.study import format_violations, study_kgx
242
+
243
+ violations = study_kgx(Path(f"./{name}_{version}.nodes.ndjson"), Path(f"./{name}_{version}.edges.ndjson"))
244
+ if violations:
245
+ summary: str = format_violations(violations)
246
+ print(summary, file=sys.stderr)
247
+ logger.warning("study assertions failed on final NDJSON:\n{summary}", summary=summary)
248
+ raise SystemExit(1)
249
+ logger.info("study assertions passed on final NDJSON")
250
+
251
+
224
252
  def validate_pipeline(table_configuration_file: Path, progress: PipelineProgress) -> None:
225
253
  """Validate section syntax from a YAML configuration file.
226
254
 
@@ -599,11 +627,14 @@ def build_kg(
599
627
 
600
628
  ``--qc`` requires the ``[qc]`` extra (``pip install "tablassert[qc]"``); it is
601
629
  checked before the build starts, because the audit stage runs LAST and a missing
602
- extra would otherwise surface only after entity resolution has finished.
630
+ extra would otherwise surface only after entity resolution has finished. It also
631
+ runs a final study stage that asserts over the emitted NDJSON -- no duplicate node
632
+ ids, no undeclared or isolated nodes, no malformed lines or stray whitespace --
633
+ and fails the build (non-zero exit) when any assertion is violated.
603
634
  """
604
635
  if qc:
605
636
  extras.require("qc", required_by="--qc")
606
- run(6, build_pipeline, graph_configuration_file, release=release, qc=qc, log=log, head=head)
637
+ run(7 if qc else 6, build_pipeline, graph_configuration_file, release=release, qc=qc, log=log, head=head)
607
638
 
608
639
 
609
640
  @APP.command(name="validate")
@@ -950,7 +981,7 @@ def _prebuilt_fullmap_urls(babel_version: str) -> tuple[str, str]:
950
981
 
951
982
  RENCI publishes a prebuilt ``fullmap.tar.zst`` (and a ``sha256sum.txt``) under
952
983
  ``{BABEL_BASE}/{babel_version}/fullmap/{tablassert_version}/``, where the version
953
- directory is the INSTALLED Tablassert package version (e.g. ``9.1.0``) — resolved from
984
+ directory is the INSTALLED Tablassert package version (e.g. ``10.1.0``) — resolved from
954
985
  installed-package metadata, never hardcoded, so a new release looks itself up.
955
986
 
956
987
  Args:
@@ -57,7 +57,6 @@ class Files(str, Enum):
57
57
  class EncodingMethods(str, Enum):
58
58
  VALUE = "value"
59
59
  COLUMN = "column"
60
- LIST = "list"
61
60
 
62
61
 
63
62
  class FillMethods(str, Enum):
@@ -25,14 +25,13 @@ TablassertErrorCodes = Literal[
25
25
  "provenance-bad-pmc-id",
26
26
  "provenance-missing-publication",
27
27
  "provenance-publication-and-override",
28
- "encoding-list-requires-list",
29
- "encoding-list-incompatible-ops",
30
- "encoding-list-annotation-only",
28
+ "encoding-list-method-removed",
31
29
  "annotation-split-by-requires-column",
32
30
  "annotation-split-by-empty",
33
31
  "qualifier-auto-derived",
34
32
  "qualifier-bad-value",
35
33
  "qualifier-unsatisfiable",
34
+ "qualifier-nullable-literal",
36
35
  ]
37
36
 
38
37
 
@@ -473,7 +473,7 @@ def _coalesce_expr(col: str, suffix: str, base: str, prefix: str, cast_str: bool
473
473
  return pl.when(pl.col(base).is_not_null()).then(l1).otherwise(l2).alias(col + suffix)
474
474
 
475
475
 
476
- def join_matches(lf: pl.LazyFrame, col: str, matches: pl.DataFrame, tag: str = "_two") -> pl.LazyFrame:
476
+ def join_matches(lf: pl.LazyFrame, col: str, matches: pl.DataFrame, tag: str = "_two", drop_unresolved: bool = True) -> pl.LazyFrame:
477
477
  """Join ranked fullmap matches back into ``lf`` for one column.
478
478
 
479
479
  Coalesces level-one and level-two hits per row (level one wins when
@@ -486,10 +486,15 @@ def join_matches(lf: pl.LazyFrame, col: str, matches: pl.DataFrame, tag: str = "
486
486
  col: Column being resolved.
487
487
  matches: Ranked matches for this column from ``filter_and_rank``.
488
488
  tag: Suffix used to derive the level-two column name.
489
+ drop_unresolved: When True (default) rows whose ``col`` did not match are
490
+ dropped — the behavior subject/object require. When False the row is
491
+ kept and the column (plus its derived ``<col>_*`` columns) stays null;
492
+ used by ``nullable`` qualifiers so a blank/unresolvable cell keeps the
493
+ edge and the null-stripper omits the qualifier key.
489
494
 
490
495
  Returns:
491
496
  New LazyFrame with resolved columns; rows whose ``col`` did not match
492
- are dropped.
497
+ are dropped unless ``drop_unresolved`` is False.
493
498
 
494
499
  Notes:
495
500
  Split out of ``resolve`` so ``resolve_batch`` can apply per-column
@@ -511,7 +516,10 @@ def join_matches(lf: pl.LazyFrame, col: str, matches: pl.DataFrame, tag: str = "
511
516
  result = result.select(pl.exclude(r"^(CURIE|PREFERRED_NAME|CATEGORY_NAME|TAXON_ID|SOURCE_NAME|SOURCE_VERSION|NLP_LEVEL|PR|FREQUENCY)(_l2)?$"))
512
517
  result = result.select(pl.exclude(col + tag))
513
518
  result = result.with_columns(pl.col(f"{col}_taxon").replace("NCBITaxon:0", None))
514
- result = result.filter(pl.col(col).is_not_null())
519
+ if drop_unresolved:
520
+ # Subject/object (and strict qualifiers) drop rows that failed resolution; a
521
+ # nullable qualifier keeps the edge and leaves the column null for the null-stripper.
522
+ result = result.filter(pl.col(col).is_not_null())
515
523
 
516
524
  return result.lazy()
517
525
 
@@ -525,6 +533,12 @@ class ResolveSpec(NamedTuple):
525
533
  avoid: list[Categories] | None = None
526
534
  exclude_prefixes: list[str] | None = None
527
535
  exclude_regex: list[str] | None = None
536
+ nullable: bool = False
537
+ """When True, an unresolved/blank cell keeps its row instead of dropping the edge.
538
+
539
+ Set only for ``nullable`` qualifiers; subject/object always resolve strict so an
540
+ edge with a missing node is dropped, never emitted.
541
+ """
528
542
 
529
543
 
530
544
  def resolve_batch(
@@ -585,7 +599,7 @@ def resolve_batch(
585
599
  )
586
600
  if log:
587
601
  log_unmatched(spec.col, terms_by_col[spec.col], matches, section_hash, config_file)
588
- result = join_matches(result, spec.col, matches, tag)
602
+ result = join_matches(result, spec.col, matches, tag, drop_unresolved=not spec.nullable)
589
603
 
590
604
  return result
591
605
 
@@ -262,8 +262,10 @@ def prune_to_class(lf: pl.LazyFrame) -> pl.LazyFrame:
262
262
 
263
263
  Categories vary per row within a section, so this masks per row rather than
264
264
  dropping columns: values are nulled where the row's class rejects them, and the
265
- Rust null-stripper then removes the key entirely. Scalars are wrapped where the
266
- class declares the slot multivalued.
265
+ Rust null-stripper then removes the key entirely. Scalars are wrapped into
266
+ one-element lists when the slot is multivalued on every class that declares it
267
+ (a column has one dtype, so per-row wrapping is impossible; a slot Biolink
268
+ declares scalar on some classes and multivalued on others stays scalar).
267
269
 
268
270
  Args:
269
271
  lf: Edges LazyFrame carrying a resolved ``category`` column.
@@ -306,10 +308,18 @@ def prune_to_class(lf: pl.LazyFrame) -> pl.LazyFrame:
306
308
  # Preserve what the class refuses rather than deleting it outright; the
307
309
  # value is real evidence, it just has no slot on this association class.
308
310
  rescued.append(pl.when(ok | text.is_null()).then(None).otherwise(pl.concat_str([pl.lit(f"{col}="), text])))
309
- # Biolink makes the same slot multivalued on some classes and scalar on others.
310
- listed: dict[str, bool] = {cat: is_multivalued(association_class(cat), col) for cat in categories}
311
- if any(listed.values()) and not isinstance(schema[col], pl.List):
312
- keep = pl.when(first.replace_strict(listed, default=False)).then(pl.concat_list(keep)).otherwise(keep)
311
+ # A column has one dtype, so the multivalued wrap must be uniform across rows:
312
+ # wrap every value into a one-element list when every class declaring the slot
313
+ # types it multivalued. `is_multivalued` is False for classes that do not declare
314
+ # the slot at all, so the scan is restricted to declaring classes -- including
315
+ # them is what produced a spuriously mixed per-row wrap that died in
316
+ # strict_cast at collect. Rows whose class rejects the slot are already null and
317
+ # stay null. A hypothetically mixed slot keeps its scalar rather than crashing.
318
+ declaring: list[str] = [cat for cat in categories if accepts[cat]]
319
+ listed: dict[str, bool] = {cat: is_multivalued(association_class(cat), col) for cat in declaring}
320
+ if listed and all(listed.values()) and not isinstance(schema[col], pl.List):
321
+ # concat_list maps null -> [null]; the when preserves real nulls instead.
322
+ keep = pl.when(keep.is_null()).then(None).otherwise(pl.concat_list(keep))
313
323
  updates.append(keep.alias(col))
314
324
  if rescued:
315
325
  updates.append(pl.concat_list(rescued).list.drop_nulls().alias(PRUNED_COLUMN))
@@ -632,13 +642,12 @@ def split_list(lf: pl.LazyFrame, col: str, delimiter: str) -> pl.LazyFrame:
632
642
 
633
643
  A column encoding is scalar by construction, so a multivalued Biolink slot such as
634
644
  ``has_evidence`` fed from an aggregated cell would otherwise be emitted as a single
635
- joined string -- and ``mask_illegal_edge_fields`` wraps that scalar into a
645
+ joined string -- and ``prune_to_class`` wraps that scalar into a
636
646
  one-element list, so the value survives Biolink validation while consumers iterate a
637
647
  single ``"a|b|c"`` blob instead of three ids.
638
648
 
639
- Same split as ``explode``, minus the fan-out: this is the per-row counterpart of
640
- ``method: list`` (the literal form covers a fixed array known at config time, this
641
- covers an array that differs on every row).
649
+ Same split as ``explode``, minus the fan-out: this is the one multivalued
650
+ encoding (``split_by``), turning each row's cell into its own JSON array.
642
651
 
643
652
  Args:
644
653
  lf: Source LazyFrame.
@@ -930,7 +939,7 @@ class Tcode(Section):
930
939
  containing ``None`` placeholders) ready for ``clean`` to filter.
931
940
  """
932
941
  return [
933
- (value, (col, x.encoding)) if x.method in (EncodingMethods.VALUE, EncodingMethods.LIST) else None,
942
+ (value, (col, x.encoding)) if x.method == EncodingMethods.VALUE else None,
934
943
  (column, (col, idxname(x.encoding))) if x.method == EncodingMethods.COLUMN else None,
935
944
  (column, (f"original_{col}", col)) if table_literal else None,
936
945
  (fill, (col, x.fill)) if x.fill else None,
@@ -1034,8 +1043,20 @@ class Tcode(Section):
1034
1043
  *[(x, x.qualifier) for x in qualifiers if x.resolved],
1035
1044
  ]
1036
1045
  literals: list[Qualifier] = [x for x in qualifiers if not x.resolved]
1046
+ # A nullable qualifier keeps its edge when the cell is blank or unresolvable (the
1047
+ # column stays null and the null-stripper omits the key); subject/object and
1048
+ # strict qualifiers drop the row as before.
1037
1049
  specs: list[ResolveSpec] = [
1038
- ResolveSpec(col, str(x.taxon) if x.taxon else None, x.prioritize, x.avoid, x.exclude_prefixes, x.exclude_regex) for x, col in node_columns
1050
+ ResolveSpec(
1051
+ col,
1052
+ str(x.taxon) if x.taxon else None,
1053
+ x.prioritize,
1054
+ x.avoid,
1055
+ x.exclude_prefixes,
1056
+ x.exclude_regex,
1057
+ x.nullable if isinstance(x, Qualifier) else False,
1058
+ )
1059
+ for x, col in node_columns
1039
1060
  ]
1040
1061
  return [
1041
1062
  [self.node_prep(x, col) for x, col in node_columns],
@@ -1043,7 +1064,15 @@ class Tcode(Section):
1043
1064
  # which exist to feed entity resolution these columns never undergo.
1044
1065
  [self.encoding(x, x.qualifier) for x in literals],
1045
1066
  (resolve_batch, (specs, db, self.log, self.store.stem, self.config.name, True)),
1046
- [(fullmap_audit, (col, self.store.stem, self.config.name, "passed", True)) for _, col in node_columns] if self.qc else None,
1067
+ # QC audits only the strict columns: a nullable qualifier's nulls are expected
1068
+ # (blank cell / no match), not resolution errors for the audit to delete.
1069
+ [
1070
+ (fullmap_audit, (col, self.store.stem, self.config.name, "passed", True))
1071
+ for x, col in node_columns
1072
+ if not (isinstance(x, Qualifier) and x.nullable)
1073
+ ]
1074
+ if self.qc
1075
+ else None,
1047
1076
  ]
1048
1077
 
1049
1078
  def _provenance_ops(self: Self) -> list[Any]:
@@ -19,6 +19,7 @@ from tablassert.biolink import (
19
19
  Predicates,
20
20
  Qualifiers,
21
21
  )
22
+ from tablassert.coerce import effect_size_target, effect_type_target, pvalue_target, study_size_target
22
23
  from tablassert.enums import Comparisons, EncodingMethods, Files, FillMethods, Functions, Repositories, Tokens
23
24
  from tablassert.errors import BiolinkRelocationWarning, TablassertErrorCodes, TablassertValidationError
24
25
 
@@ -164,67 +165,40 @@ class Math(TablaBase):
164
165
  class Encoding(TablaBase):
165
166
  method: EncodingMethods = Field(
166
167
  EncodingMethods.VALUE,
167
- description="Interpret `encoding` as a literal value, a list of literal values, or source column letters.",
168
- examples=[EncodingMethods.VALUE, EncodingMethods.COLUMN, EncodingMethods.LIST],
169
- )
170
- encoding: str | int | float | list[str | int | float] = Field(
171
- ...,
172
- description="Literal value, list of literal values (with `method: list`), or source column letters.",
173
- examples=["A", "BRCA1", 1.0, ["EFO:0001", "EFO:0002"]],
168
+ description="Interpret `encoding` as a literal value or source column letters.",
169
+ examples=[EncodingMethods.VALUE, EncodingMethods.COLUMN],
174
170
  )
171
+ encoding: str | int | float = Field(..., description="Literal value or source column letters.", examples=["A", "BRCA1", 1.0])
172
+
173
+ @model_validator(mode="before")
174
+ @classmethod
175
+ def reject_removed_list_method(cls, data: Any) -> Any:
176
+ """Fail configs still declaring the removed ``method: list`` with a migration pointer.
177
+
178
+ ``method: list`` (a literal list emitted as one fixed JSON array on every row)
179
+ was removed: ``split_by`` on a ``method: column`` annotation is the one
180
+ multivalued encoding now, and it covers the per-row case the literal never
181
+ could. A bare pydantic enum error would only say the value is invalid, so
182
+ this hook turns the stale config into the actionable coded error the
183
+ migration needs.
184
+ """
185
+ if isinstance(data, dict) and data.get("method") == "list":
186
+ raise TablassertValidationError(
187
+ "`method: list` was removed; for a multivalued annotation use `method: column` with `split_by` "
188
+ "to split each cell's delimited text into a JSON array (subject/object/qualifier nodes are single entities).",
189
+ code="encoding-list-method-removed",
190
+ )
191
+ return data
175
192
 
176
193
  @model_validator(mode="after")
177
194
  def excel_style_columns(self: Self) -> Self:
178
- # A list encoding is only valid under `method: list` (checked by `list_method_consistency`);
179
- # skip the Excel-letter check here so that case reports the clearer list error.
180
- if self.method == EncodingMethods.COLUMN and not isinstance(self.encoding, list):
195
+ if self.method == EncodingMethods.COLUMN:
181
196
  x = self.encoding
182
197
  if not re.search(r"^[A-Z]{1,3}$", str(x)):
183
198
  raise TablassertValidationError(f"`encoding` must be an Excel-style column name (A-ZZ), got {x!r}.", code="encoding-bad-excel-column")
184
199
 
185
200
  return self
186
201
 
187
- @model_validator(mode="after")
188
- def list_method_consistency(self: Self) -> Self:
189
- """Enforce that ``method: list`` carries a literal list and no scalar string ops.
190
-
191
- ``method: list`` is the multivalued counterpart of ``method: value``: the
192
- ``encoding`` is a literal list emitted as a real JSON array (for multivalued
193
- Biolink slots such as ``has_evidence``). The scalar string ops
194
- (``regex``/``remove``/``prefix``/``suffix``/``transformations``/``fill``/``explode_by``)
195
- operate on a single string per row and would mangle a list column, so they are
196
- rejected here — encode the final values directly instead.
197
- """
198
- is_list: bool = isinstance(self.encoding, list)
199
- if self.method == EncodingMethods.LIST:
200
- if not is_list:
201
- raise TablassertValidationError("`method: list` requires `encoding` to be a list of values.", code="encoding-list-requires-list")
202
- scalar_ops: list[str] = []
203
- if self.regex:
204
- scalar_ops.append("regex")
205
- if self.fill is not None:
206
- scalar_ops.append("fill")
207
- if self.explode_by is not None:
208
- scalar_ops.append("explode_by")
209
- if self.remove:
210
- scalar_ops.append("remove")
211
- if self.prefix:
212
- scalar_ops.append("prefix")
213
- if self.suffix:
214
- scalar_ops.append("suffix")
215
- if self.transformations:
216
- scalar_ops.append("transformations")
217
- if scalar_ops:
218
- raise TablassertValidationError(
219
- f"`method: list` is a literal list and is incompatible with the scalar string ops "
220
- f"({', '.join(scalar_ops)}); apply them upstream or encode the final values directly.",
221
- code="encoding-list-incompatible-ops",
222
- )
223
- elif is_list:
224
- raise TablassertValidationError("A list `encoding` requires `method: list`.", code="encoding-list-requires-list")
225
-
226
- return self
227
-
228
202
  regex: list[Regex] | None = Field(
229
203
  None,
230
204
  description="Ordered regex replacements applied to encoded text.",
@@ -306,23 +280,6 @@ class NodeEncoding(Encoding):
306
280
 
307
281
  return exclude_regex
308
282
 
309
- @model_validator(mode="after")
310
- def reject_list_method(self: Self) -> Self:
311
- """Reject ``method: list`` on node encodings (subject/object/qualifiers).
312
-
313
- ``method: list`` is the multivalued counterpart of ``method: value`` and only
314
- makes sense on an annotation (a multivalued Biolink slot). A subject/object/
315
- qualifier is a single entity: a list node column crashes resolution deep in the
316
- pipeline (a polars list-to-string cast) instead of failing at config time, and an
317
- enum-ranged qualifier would silently emit a list where Biolink expects one token.
318
- """
319
- if self.method == EncodingMethods.LIST:
320
- raise TablassertValidationError(
321
- "`method: list` is only valid on annotations (multivalued Biolink slots); subject/object/qualifier nodes are single entities.",
322
- code="encoding-list-annotation-only",
323
- )
324
- return self
325
-
326
283
 
327
284
  class Qualifier(NodeEncoding):
328
285
  qualifier: Qualifiers = Field(
@@ -330,6 +287,15 @@ class Qualifier(NodeEncoding):
330
287
  description="Qualifier predicate key used as the output qualifier column.",
331
288
  examples=[Qualifiers.OBJECT_DIRECTION_QUALIFIER, Qualifiers.SUBJECT_CONTEXT_QUALIFIER],
332
289
  )
290
+ nullable: bool = Field(
291
+ False,
292
+ description=(
293
+ "When True, a blank or unresolvable ``method: column`` cell keeps the edge and omits the "
294
+ "qualifier for that row (the column stays null and the null-stripper drops the key); when "
295
+ "False (default) such a row is dropped, exactly like an unresolved subject/object. Only "
296
+ "meaningful for ``method: column`` — a literal qualifier can never be null."
297
+ ),
298
+ )
333
299
 
334
300
  @property
335
301
  def vocabulary(self: Self) -> frozenset[str] | None:
@@ -400,6 +366,23 @@ class Qualifier(NodeEncoding):
400
366
  )
401
367
  return self
402
368
 
369
+ @model_validator(mode="after")
370
+ def reject_nullable_literal_qualifiers(self: Self) -> Self:
371
+ """Reject ``nullable: true`` on literal qualifiers.
372
+
373
+ ``nullable`` only has meaning for a ``method: column`` qualifier: a blank or
374
+ unresolved cell keeps the edge and the qualifier is omitted for that row. A
375
+ ``method: value`` qualifier is a config-time constant that can never be blank,
376
+ so ``nullable`` would be dead config that misleads the reader. Fail loudly at
377
+ config time instead (the removed ``method: list`` is already rejected upstream
378
+ by :meth:`Encoding.reject_removed_list_method`).
379
+ """
380
+ if self.nullable and self.method != EncodingMethods.COLUMN:
381
+ raise TablassertValidationError(
382
+ "`nullable` only applies to `method: column` qualifiers; a literal qualifier can never be null.", code="qualifier-nullable-literal"
383
+ )
384
+ return self
385
+
403
386
 
404
387
  class Statement(TablaBase):
405
388
  subject: NodeEncoding = Field(..., description="Subject node encoding and mapping configuration.")
@@ -508,19 +491,15 @@ class Annotation(Encoding):
508
491
  def split_by_requires_a_column(self) -> Self:
509
492
  """Enforce that ``split_by`` carries a real separator and a ``method: column`` encoding.
510
493
 
511
- ``split_by`` is the per-row counterpart of ``method: list``: it turns each cell's
512
- own delimited text into a real JSON array, which is the one multivalued shape a
513
- literal cannot express (a list ``encoding`` is fixed at config time, so it emits
514
- the same array on every row). A ``value``/``list`` encoding therefore declares its
515
- members directly rather than round-tripping them through a separator.
494
+ ``split_by`` is the one multivalued encoding: it turns each cell's own
495
+ delimited text into a real JSON array, per row. A ``value`` encoding is a
496
+ scalar literal with no per-row text to split, so it rejects ``split_by``.
516
497
  """
517
498
  if self.split_by is None:
518
499
  return self
519
500
  if self.method != EncodingMethods.COLUMN:
520
501
  raise TablassertValidationError(
521
- "`split_by` splits a column's per-row text and requires `method: column`; "
522
- "declare a literal multivalued annotation with `method: list` instead.",
523
- code="annotation-split-by-requires-column",
502
+ "`split_by` splits a column's per-row text and requires `method: column`.", code="annotation-split-by-requires-column"
524
503
  )
525
504
  if not self.split_by:
526
505
  # An empty separator splits into individual characters -- exactly the
@@ -536,18 +515,24 @@ class Annotation(Encoding):
536
515
  # is the real problem -- an author asking for `supporting_study_size` has no way to discover
537
516
  # that Biolink attaches it to no class and the pipeline rerouted it.
538
517
  name: str = str(self.annotation)
539
- if name in UNSATISFIABLE_EDGE_FIELDS:
518
+ # Judge the coerced target, not the raw alias: the clean-phase column coercions rename
519
+ # statistical aliases to their canonical slot before any relocation runs, so
520
+ # `adjusted p value` reaches the edge as `adjusted_p_value` and warning on the alias
521
+ # is a false positive.
522
+ target: str = pvalue_target(name) or study_size_target(name) or effect_size_target(name) or effect_type_target(name) or name
523
+ shown: str = f"`{name}` (coerced to `{target}`)" if target != name else f"`{name}`"
524
+ if target in UNSATISFIABLE_EDGE_FIELDS:
540
525
  warnings.warn(
541
- f"`{name}` is declared in biolink-model {BIOLINK_VERSION} but attached to no association class, "
526
+ f"{shown} is declared in biolink-model {BIOLINK_VERSION} but attached to no association class, "
542
527
  "so it cannot be emitted on an edge; its value is routed onto the inlined supporting study "
543
528
  "instead. Use a slot a Biolink association declares (e.g. `p_value`, `adjusted_p_value`) if you "
544
529
  "need it on the edge itself.",
545
530
  BiolinkRelocationWarning,
546
531
  stacklevel=2,
547
532
  )
548
- elif name not in ALLOWED_EDGE_FIELDS:
533
+ elif target not in ALLOWED_EDGE_FIELDS:
549
534
  warnings.warn(
550
- f"`{name}` is not a Biolink association slot, so it is folded into `supporting_text` as a "
535
+ f"{shown} is not a Biolink association slot, so it is folded into `supporting_text` as a "
551
536
  f'"{name}: <value>" string rather than emitted as its own edge field.',
552
537
  BiolinkRelocationWarning,
553
538
  stacklevel=2,
@@ -0,0 +1,158 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ from collections import Counter
5
+ from dataclasses import dataclass, field
6
+ from pathlib import Path
7
+
8
+ from tablassert.log import cat
9
+
10
+ logger = cat("QC")
11
+
12
+ # Human-readable phrasing for each assertion, used by format_violations.
13
+ _MESSAGES: dict[str, str] = {
14
+ "file-missing": "file not found",
15
+ "malformed-lines": "empty or malformed JSON lines",
16
+ "whitespace-values": "values with leading/trailing whitespace",
17
+ "duplicate-node-ids": "duplicate node ids",
18
+ "undeclared-nodes": "nodes referenced by edges but not declared in the nodes file",
19
+ "isolated-nodes": "declared nodes participating in no edge",
20
+ }
21
+
22
+
23
+ @dataclass
24
+ class StudyViolation:
25
+ """One failed study assertion over the final KGX NDJSON files.
26
+
27
+ Attributes:
28
+ check: Assertion key into ``_MESSAGES`` (e.g. ``"duplicate-node-ids"``).
29
+ label: Which file the violation belongs to (``"nodes"`` or ``"edges"``).
30
+ count: Exact number of offending records/values.
31
+ examples: Capped list of example offenders for the stderr summary.
32
+ """
33
+
34
+ check: str
35
+ label: str
36
+ count: int
37
+ examples: list[str] = field(default_factory=list)
38
+
39
+
40
+ @dataclass
41
+ class _FileScan:
42
+ """Accumulated facts from one streamed pass over an NDJSON file."""
43
+
44
+ ids: set[str]
45
+ duplicate_ids: Counter[str]
46
+ whitespace: Counter[str]
47
+ malformed: int
48
+ missing: bool
49
+ path: Path
50
+
51
+
52
+ def _scan_ndjson(path: Path, *, edge: bool) -> _FileScan:
53
+ """Stream one NDJSON file, collecting the facts the study assertions need.
54
+
55
+ Args:
56
+ path: Path to a ``.nodes.ndjson`` or ``.edges.ndjson`` file.
57
+ edge: ``True`` to collect referenced ids from ``subject``/``object``;
58
+ ``False`` to collect declared node ``id``s and track duplicates.
59
+
60
+ Returns:
61
+ A :class:`_FileScan`; ``missing`` is set (and nothing else) when the
62
+ file does not exist, so a typo'd path can never read as a clean pass.
63
+ """
64
+ scan: _FileScan = _FileScan(set(), Counter(), Counter(), 0, not path.is_file(), path)
65
+ if scan.missing:
66
+ return scan
67
+ with path.open(encoding="utf-8") as handle:
68
+ for line in handle:
69
+ if not line.strip():
70
+ scan.malformed += 1
71
+ continue
72
+ try:
73
+ record: object = json.loads(line)
74
+ except json.JSONDecodeError:
75
+ scan.malformed += 1
76
+ continue
77
+ if not isinstance(record, dict):
78
+ scan.malformed += 1
79
+ continue
80
+ for key, value in record.items():
81
+ if isinstance(value, str) and value != value.strip():
82
+ scan.whitespace[key] += 1
83
+ if edge:
84
+ for role in ("subject", "object"):
85
+ ident: object = record.get(role)
86
+ if isinstance(ident, str):
87
+ scan.ids.add(ident)
88
+ else:
89
+ ident = record.get("id")
90
+ if isinstance(ident, str):
91
+ # Rust dedup only removes byte-identical lines, so a repeated id
92
+ # here means two nodes share an id with different content.
93
+ if ident in scan.ids:
94
+ scan.duplicate_ids[ident] += 1
95
+ scan.ids.add(ident)
96
+ return scan
97
+
98
+
99
+ def study_kgx(nodes_path: Path, edges_path: Path, *, example_limit: int = 10) -> list[StudyViolation]:
100
+ """Assert over the final KGX NDJSON files, in the spirit of studyKGtsvs.pl.
101
+
102
+ Streams both files once each and checks: duplicate node ids, nodes referenced
103
+ by edges but never declared (``undeclared``), declared nodes participating in
104
+ no edge (``isolated``), empty/malformed lines, and string values carrying
105
+ leading/trailing whitespace. Every check is an assertion -- the caller decides
106
+ whether violations fail the build.
107
+
108
+ Args:
109
+ nodes_path: Path to ``<name>_<version>.nodes.ndjson``.
110
+ edges_path: Path to ``<name>_<version>.edges.ndjson``.
111
+ example_limit: Maximum number of examples retained per violation.
112
+
113
+ Returns:
114
+ A list of :class:`StudyViolation`; empty when every assertion passes.
115
+ """
116
+ nodes: _FileScan = _scan_ndjson(nodes_path, edge=False)
117
+ edges: _FileScan = _scan_ndjson(edges_path, edge=True)
118
+
119
+ violations: list[StudyViolation] = []
120
+ for label, scan in (("nodes", nodes), ("edges", edges)):
121
+ if scan.missing:
122
+ violations.append(StudyViolation("file-missing", label, 1, [str(scan.path)]))
123
+ continue
124
+ if scan.malformed:
125
+ violations.append(StudyViolation("malformed-lines", label, scan.malformed))
126
+ if scan.whitespace:
127
+ examples: list[str] = [f"{field_name} ({n})" for field_name, n in scan.whitespace.most_common(example_limit)]
128
+ violations.append(StudyViolation("whitespace-values", label, sum(scan.whitespace.values()), examples))
129
+ if nodes.duplicate_ids:
130
+ examples = [ident for ident, _ in nodes.duplicate_ids.most_common(example_limit)]
131
+ violations.append(StudyViolation("duplicate-node-ids", "nodes", len(nodes.duplicate_ids), examples))
132
+ if not nodes.missing and not edges.missing:
133
+ undeclared: list[str] = sorted(edges.ids - nodes.ids)
134
+ if undeclared:
135
+ violations.append(StudyViolation("undeclared-nodes", "edges", len(undeclared), undeclared[:example_limit]))
136
+ isolated: list[str] = sorted(nodes.ids - edges.ids)
137
+ if isolated:
138
+ violations.append(StudyViolation("isolated-nodes", "nodes", len(isolated), isolated[:example_limit]))
139
+ return violations
140
+
141
+
142
+ def format_violations(violations: list[StudyViolation]) -> str:
143
+ """Render violations as a human-readable, one-line-per-assertion summary.
144
+
145
+ Args:
146
+ violations: The failed assertions returned by :func:`study_kgx`.
147
+
148
+ Returns:
149
+ Newline-joined summary lines, e.g.
150
+ ``nodes: 3 duplicate node ids (e.g. HGNC:5, HGNC:6)``.
151
+ """
152
+ lines: list[str] = []
153
+ for violation in violations:
154
+ line: str = f"{violation.label}: {violation.count} {_MESSAGES[violation.check]}"
155
+ if violation.examples:
156
+ line += f" (e.g. {', '.join(violation.examples)})"
157
+ lines.append(line)
158
+ return "\n".join(lines)
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes