tablassert 9.1.0__tar.gz → 10.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. {tablassert-9.1.0 → tablassert-10.0.0}/PKG-INFO +1 -1
  2. {tablassert-9.1.0 → tablassert-10.0.0}/pyproject.toml +1 -1
  3. {tablassert-9.1.0 → tablassert-10.0.0}/src/tablassert/agent.py +5 -0
  4. {tablassert-9.1.0 → tablassert-10.0.0}/src/tablassert/cli.py +1 -1
  5. {tablassert-9.1.0 → tablassert-10.0.0}/src/tablassert/enums.py +0 -1
  6. {tablassert-9.1.0 → tablassert-10.0.0}/src/tablassert/errors.py +2 -3
  7. {tablassert-9.1.0 → tablassert-10.0.0}/src/tablassert/fullmap.py +18 -4
  8. {tablassert-9.1.0 → tablassert-10.0.0}/src/tablassert/lib.py +25 -6
  9. {tablassert-9.1.0 → tablassert-10.0.0}/src/tablassert/models.py +54 -76
  10. {tablassert-9.1.0 → tablassert-10.0.0}/LICENSE +0 -0
  11. {tablassert-9.1.0 → tablassert-10.0.0}/README.md +0 -0
  12. {tablassert-9.1.0 → tablassert-10.0.0}/rust/Cargo.lock +0 -0
  13. {tablassert-9.1.0 → tablassert-10.0.0}/rust/Cargo.toml +0 -0
  14. {tablassert-9.1.0 → tablassert-10.0.0}/rust/examples/count_tables.rs +0 -0
  15. {tablassert-9.1.0 → tablassert-10.0.0}/rust/src/fullmap.rs +0 -0
  16. {tablassert-9.1.0 → tablassert-10.0.0}/rust/src/json.rs +0 -0
  17. {tablassert-9.1.0 → tablassert-10.0.0}/rust/src/lib.rs +0 -0
  18. {tablassert-9.1.0 → tablassert-10.0.0}/rust/src/ndjson.rs +0 -0
  19. {tablassert-9.1.0 → tablassert-10.0.0}/rust/src/uuid.rs +0 -0
  20. {tablassert-9.1.0 → tablassert-10.0.0}/rust/tests/build_golden.rs +0 -0
  21. {tablassert-9.1.0 → tablassert-10.0.0}/src/tablassert/__init__.py +0 -0
  22. {tablassert-9.1.0 → tablassert-10.0.0}/src/tablassert/_lazy.py +0 -0
  23. {tablassert-9.1.0 → tablassert-10.0.0}/src/tablassert/biolink.py +0 -0
  24. {tablassert-9.1.0 → tablassert-10.0.0}/src/tablassert/coerce.py +0 -0
  25. {tablassert-9.1.0 → tablassert-10.0.0}/src/tablassert/extras.py +0 -0
  26. {tablassert-9.1.0 → tablassert-10.0.0}/src/tablassert/graph_registry.py +0 -0
  27. {tablassert-9.1.0 → tablassert-10.0.0}/src/tablassert/ingests.py +0 -0
  28. {tablassert-9.1.0 → tablassert-10.0.0}/src/tablassert/log.py +0 -0
  29. {tablassert-9.1.0 → tablassert-10.0.0}/src/tablassert/nlp.py +0 -0
  30. {tablassert-9.1.0 → tablassert-10.0.0}/src/tablassert/progress.py +0 -0
  31. {tablassert-9.1.0 → tablassert-10.0.0}/src/tablassert/qc.py +0 -0
  32. {tablassert-9.1.0 → tablassert-10.0.0}/src/tablassert/rig.py +0 -0
  33. {tablassert-9.1.0 → tablassert-10.0.0}/src/tablassert/rs.pyi +0 -0
  34. {tablassert-9.1.0 → tablassert-10.0.0}/src/tablassert/utils.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: tablassert
3
- Version: 9.1.0
3
+ Version: 10.0.0
4
4
  Classifier: License :: OSI Approved :: Apache Software License
5
5
  Classifier: Development Status :: 5 - Production/Stable
6
6
  Classifier: Intended Audience :: Science/Research
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "tablassert"
3
- version = "9.1.0"
3
+ version = "10.0.0"
4
4
  description = "Extract knowledge assertions from tabular data into NCATS Translator-compliant KGX NDJSON — declaratively, with entity resolution and quality control built in."
5
5
  authors = [
6
6
  { name = "Skye Lane Goetz", email = "sgoetz@isbscience.org" }
@@ -2172,6 +2172,11 @@ qualifier and evidence slot the specific class declared. build_and_audit reports
2172
2172
  description rather than emitted on the edge. `q_value`, `fold_change`, `z_score`, `beta` and
2173
2173
  similar are not association slots at all and are folded into `supporting_text`. Prefer
2174
2174
  `p_value`, `adjusted_p_value`, `effect_size`, `effect_type`, `has_evidence`.
2175
+ - MULTIVALUED slots (`has_evidence` and friends) take a real JSON array, never a joined string:
2176
+ declare the annotation `{method: column, encoding: <letter>, split_by: "|"}` so each cell's
2177
+ delimited text splits into its own per-row array. `split_by` is the ONLY multivalued encoding
2178
+ — there is no literal-list method, and a scalar bound for a multivalued slot ships to consumers
2179
+ as one unusable "a|b|c" blob.
2175
2180
  - `effect_size` / `effect_type` are deliberate Tablassert extras pending biolink-model#1774 and
2176
2181
  are EXEMPT from the validity score: a `biolink_valid_pct` below 1.0 is never caused by them.
2177
2182
  - QUALIFIERS: enum-ranged qualifiers take a literal TOKEN, never a CURIE
@@ -950,7 +950,7 @@ def _prebuilt_fullmap_urls(babel_version: str) -> tuple[str, str]:
950
950
 
951
951
  RENCI publishes a prebuilt ``fullmap.tar.zst`` (and a ``sha256sum.txt``) under
952
952
  ``{BABEL_BASE}/{babel_version}/fullmap/{tablassert_version}/``, where the version
953
- directory is the INSTALLED Tablassert package version (e.g. ``9.1.0``) — resolved from
953
+ directory is the INSTALLED Tablassert package version (e.g. ``10.0.0``) — resolved from
954
954
  installed-package metadata, never hardcoded, so a new release looks itself up.
955
955
 
956
956
  Args:
@@ -57,7 +57,6 @@ class Files(str, Enum):
57
57
  class EncodingMethods(str, Enum):
58
58
  VALUE = "value"
59
59
  COLUMN = "column"
60
- LIST = "list"
61
60
 
62
61
 
63
62
  class FillMethods(str, Enum):
@@ -25,14 +25,13 @@ TablassertErrorCodes = Literal[
25
25
  "provenance-bad-pmc-id",
26
26
  "provenance-missing-publication",
27
27
  "provenance-publication-and-override",
28
- "encoding-list-requires-list",
29
- "encoding-list-incompatible-ops",
30
- "encoding-list-annotation-only",
28
+ "encoding-list-method-removed",
31
29
  "annotation-split-by-requires-column",
32
30
  "annotation-split-by-empty",
33
31
  "qualifier-auto-derived",
34
32
  "qualifier-bad-value",
35
33
  "qualifier-unsatisfiable",
34
+ "qualifier-nullable-literal",
36
35
  ]
37
36
 
38
37
 
@@ -473,7 +473,7 @@ def _coalesce_expr(col: str, suffix: str, base: str, prefix: str, cast_str: bool
473
473
  return pl.when(pl.col(base).is_not_null()).then(l1).otherwise(l2).alias(col + suffix)
474
474
 
475
475
 
476
- def join_matches(lf: pl.LazyFrame, col: str, matches: pl.DataFrame, tag: str = "_two") -> pl.LazyFrame:
476
+ def join_matches(lf: pl.LazyFrame, col: str, matches: pl.DataFrame, tag: str = "_two", drop_unresolved: bool = True) -> pl.LazyFrame:
477
477
  """Join ranked fullmap matches back into ``lf`` for one column.
478
478
 
479
479
  Coalesces level-one and level-two hits per row (level one wins when
@@ -486,10 +486,15 @@ def join_matches(lf: pl.LazyFrame, col: str, matches: pl.DataFrame, tag: str = "
486
486
  col: Column being resolved.
487
487
  matches: Ranked matches for this column from ``filter_and_rank``.
488
488
  tag: Suffix used to derive the level-two column name.
489
+ drop_unresolved: When True (default) rows whose ``col`` did not match are
490
+ dropped — the behavior subject/object require. When False the row is
491
+ kept and the column (plus its derived ``<col>_*`` columns) stays null;
492
+ used by ``nullable`` qualifiers so a blank/unresolvable cell keeps the
493
+ edge and the null-stripper omits the qualifier key.
489
494
 
490
495
  Returns:
491
496
  New LazyFrame with resolved columns; rows whose ``col`` did not match
492
- are dropped.
497
+ are dropped unless ``drop_unresolved`` is False.
493
498
 
494
499
  Notes:
495
500
  Split out of ``resolve`` so ``resolve_batch`` can apply per-column
@@ -511,7 +516,10 @@ def join_matches(lf: pl.LazyFrame, col: str, matches: pl.DataFrame, tag: str = "
511
516
  result = result.select(pl.exclude(r"^(CURIE|PREFERRED_NAME|CATEGORY_NAME|TAXON_ID|SOURCE_NAME|SOURCE_VERSION|NLP_LEVEL|PR|FREQUENCY)(_l2)?$"))
512
517
  result = result.select(pl.exclude(col + tag))
513
518
  result = result.with_columns(pl.col(f"{col}_taxon").replace("NCBITaxon:0", None))
514
- result = result.filter(pl.col(col).is_not_null())
519
+ if drop_unresolved:
520
+ # Subject/object (and strict qualifiers) drop rows that failed resolution; a
521
+ # nullable qualifier keeps the edge and leaves the column null for the null-stripper.
522
+ result = result.filter(pl.col(col).is_not_null())
515
523
 
516
524
  return result.lazy()
517
525
 
@@ -525,6 +533,12 @@ class ResolveSpec(NamedTuple):
525
533
  avoid: list[Categories] | None = None
526
534
  exclude_prefixes: list[str] | None = None
527
535
  exclude_regex: list[str] | None = None
536
+ nullable: bool = False
537
+ """When True, an unresolved/blank cell keeps its row instead of dropping the edge.
538
+
539
+ Set only for ``nullable`` qualifiers; subject/object always resolve strict so an
540
+ edge with a missing node is dropped, never emitted.
541
+ """
528
542
 
529
543
 
530
544
  def resolve_batch(
@@ -585,7 +599,7 @@ def resolve_batch(
585
599
  )
586
600
  if log:
587
601
  log_unmatched(spec.col, terms_by_col[spec.col], matches, section_hash, config_file)
588
- result = join_matches(result, spec.col, matches, tag)
602
+ result = join_matches(result, spec.col, matches, tag, drop_unresolved=not spec.nullable)
589
603
 
590
604
  return result
591
605
 
@@ -636,9 +636,8 @@ def split_list(lf: pl.LazyFrame, col: str, delimiter: str) -> pl.LazyFrame:
636
636
  one-element list, so the value survives Biolink validation while consumers iterate a
637
637
  single ``"a|b|c"`` blob instead of three ids.
638
638
 
639
- Same split as ``explode``, minus the fan-out: this is the per-row counterpart of
640
- ``method: list`` (the literal form covers a fixed array known at config time, this
641
- covers an array that differs on every row).
639
+ Same split as ``explode``, minus the fan-out: this is the one multivalued
640
+ encoding (``split_by``), turning each row's cell into its own JSON array.
642
641
 
643
642
  Args:
644
643
  lf: Source LazyFrame.
@@ -930,7 +929,7 @@ class Tcode(Section):
930
929
  containing ``None`` placeholders) ready for ``clean`` to filter.
931
930
  """
932
931
  return [
933
- (value, (col, x.encoding)) if x.method in (EncodingMethods.VALUE, EncodingMethods.LIST) else None,
932
+ (value, (col, x.encoding)) if x.method == EncodingMethods.VALUE else None,
934
933
  (column, (col, idxname(x.encoding))) if x.method == EncodingMethods.COLUMN else None,
935
934
  (column, (f"original_{col}", col)) if table_literal else None,
936
935
  (fill, (col, x.fill)) if x.fill else None,
@@ -1034,8 +1033,20 @@ class Tcode(Section):
1034
1033
  *[(x, x.qualifier) for x in qualifiers if x.resolved],
1035
1034
  ]
1036
1035
  literals: list[Qualifier] = [x for x in qualifiers if not x.resolved]
1036
+ # A nullable qualifier keeps its edge when the cell is blank or unresolvable (the
1037
+ # column stays null and the null-stripper omits the key); subject/object and
1038
+ # strict qualifiers drop the row as before.
1037
1039
  specs: list[ResolveSpec] = [
1038
- ResolveSpec(col, str(x.taxon) if x.taxon else None, x.prioritize, x.avoid, x.exclude_prefixes, x.exclude_regex) for x, col in node_columns
1040
+ ResolveSpec(
1041
+ col,
1042
+ str(x.taxon) if x.taxon else None,
1043
+ x.prioritize,
1044
+ x.avoid,
1045
+ x.exclude_prefixes,
1046
+ x.exclude_regex,
1047
+ x.nullable if isinstance(x, Qualifier) else False,
1048
+ )
1049
+ for x, col in node_columns
1039
1050
  ]
1040
1051
  return [
1041
1052
  [self.node_prep(x, col) for x, col in node_columns],
@@ -1043,7 +1054,15 @@ class Tcode(Section):
1043
1054
  # which exist to feed entity resolution these columns never undergo.
1044
1055
  [self.encoding(x, x.qualifier) for x in literals],
1045
1056
  (resolve_batch, (specs, db, self.log, self.store.stem, self.config.name, True)),
1046
- [(fullmap_audit, (col, self.store.stem, self.config.name, "passed", True)) for _, col in node_columns] if self.qc else None,
1057
+ # QC audits only the strict columns: a nullable qualifier's nulls are expected
1058
+ # (blank cell / no match), not resolution errors for the audit to delete.
1059
+ [
1060
+ (fullmap_audit, (col, self.store.stem, self.config.name, "passed", True))
1061
+ for x, col in node_columns
1062
+ if not (isinstance(x, Qualifier) and x.nullable)
1063
+ ]
1064
+ if self.qc
1065
+ else None,
1047
1066
  ]
1048
1067
 
1049
1068
  def _provenance_ops(self: Self) -> list[Any]:
@@ -164,67 +164,40 @@ class Math(TablaBase):
164
164
  class Encoding(TablaBase):
165
165
  method: EncodingMethods = Field(
166
166
  EncodingMethods.VALUE,
167
- description="Interpret `encoding` as a literal value, a list of literal values, or source column letters.",
168
- examples=[EncodingMethods.VALUE, EncodingMethods.COLUMN, EncodingMethods.LIST],
169
- )
170
- encoding: str | int | float | list[str | int | float] = Field(
171
- ...,
172
- description="Literal value, list of literal values (with `method: list`), or source column letters.",
173
- examples=["A", "BRCA1", 1.0, ["EFO:0001", "EFO:0002"]],
167
+ description="Interpret `encoding` as a literal value or source column letters.",
168
+ examples=[EncodingMethods.VALUE, EncodingMethods.COLUMN],
174
169
  )
170
+ encoding: str | int | float = Field(..., description="Literal value or source column letters.", examples=["A", "BRCA1", 1.0])
171
+
172
+ @model_validator(mode="before")
173
+ @classmethod
174
+ def reject_removed_list_method(cls, data: Any) -> Any:
175
+ """Fail configs still declaring the removed ``method: list`` with a migration pointer.
176
+
177
+ ``method: list`` (a literal list emitted as one fixed JSON array on every row)
178
+ was removed: ``split_by`` on a ``method: column`` annotation is the one
179
+ multivalued encoding now, and it covers the per-row case the literal never
180
+ could. A bare pydantic enum error would only say the value is invalid, so
181
+ this hook turns the stale config into the actionable coded error the
182
+ migration needs.
183
+ """
184
+ if isinstance(data, dict) and data.get("method") == "list":
185
+ raise TablassertValidationError(
186
+ "`method: list` was removed; for a multivalued annotation use `method: column` with `split_by` "
187
+ "to split each cell's delimited text into a JSON array (subject/object/qualifier nodes are single entities).",
188
+ code="encoding-list-method-removed",
189
+ )
190
+ return data
175
191
 
176
192
  @model_validator(mode="after")
177
193
  def excel_style_columns(self: Self) -> Self:
178
- # A list encoding is only valid under `method: list` (checked by `list_method_consistency`);
179
- # skip the Excel-letter check here so that case reports the clearer list error.
180
- if self.method == EncodingMethods.COLUMN and not isinstance(self.encoding, list):
194
+ if self.method == EncodingMethods.COLUMN:
181
195
  x = self.encoding
182
196
  if not re.search(r"^[A-Z]{1,3}$", str(x)):
183
197
  raise TablassertValidationError(f"`encoding` must be an Excel-style column name (A-ZZ), got {x!r}.", code="encoding-bad-excel-column")
184
198
 
185
199
  return self
186
200
 
187
- @model_validator(mode="after")
188
- def list_method_consistency(self: Self) -> Self:
189
- """Enforce that ``method: list`` carries a literal list and no scalar string ops.
190
-
191
- ``method: list`` is the multivalued counterpart of ``method: value``: the
192
- ``encoding`` is a literal list emitted as a real JSON array (for multivalued
193
- Biolink slots such as ``has_evidence``). The scalar string ops
194
- (``regex``/``remove``/``prefix``/``suffix``/``transformations``/``fill``/``explode_by``)
195
- operate on a single string per row and would mangle a list column, so they are
196
- rejected here — encode the final values directly instead.
197
- """
198
- is_list: bool = isinstance(self.encoding, list)
199
- if self.method == EncodingMethods.LIST:
200
- if not is_list:
201
- raise TablassertValidationError("`method: list` requires `encoding` to be a list of values.", code="encoding-list-requires-list")
202
- scalar_ops: list[str] = []
203
- if self.regex:
204
- scalar_ops.append("regex")
205
- if self.fill is not None:
206
- scalar_ops.append("fill")
207
- if self.explode_by is not None:
208
- scalar_ops.append("explode_by")
209
- if self.remove:
210
- scalar_ops.append("remove")
211
- if self.prefix:
212
- scalar_ops.append("prefix")
213
- if self.suffix:
214
- scalar_ops.append("suffix")
215
- if self.transformations:
216
- scalar_ops.append("transformations")
217
- if scalar_ops:
218
- raise TablassertValidationError(
219
- f"`method: list` is a literal list and is incompatible with the scalar string ops "
220
- f"({', '.join(scalar_ops)}); apply them upstream or encode the final values directly.",
221
- code="encoding-list-incompatible-ops",
222
- )
223
- elif is_list:
224
- raise TablassertValidationError("A list `encoding` requires `method: list`.", code="encoding-list-requires-list")
225
-
226
- return self
227
-
228
201
  regex: list[Regex] | None = Field(
229
202
  None,
230
203
  description="Ordered regex replacements applied to encoded text.",
@@ -306,23 +279,6 @@ class NodeEncoding(Encoding):
306
279
 
307
280
  return exclude_regex
308
281
 
309
- @model_validator(mode="after")
310
- def reject_list_method(self: Self) -> Self:
311
- """Reject ``method: list`` on node encodings (subject/object/qualifiers).
312
-
313
- ``method: list`` is the multivalued counterpart of ``method: value`` and only
314
- makes sense on an annotation (a multivalued Biolink slot). A subject/object/
315
- qualifier is a single entity: a list node column crashes resolution deep in the
316
- pipeline (a polars list-to-string cast) instead of failing at config time, and an
317
- enum-ranged qualifier would silently emit a list where Biolink expects one token.
318
- """
319
- if self.method == EncodingMethods.LIST:
320
- raise TablassertValidationError(
321
- "`method: list` is only valid on annotations (multivalued Biolink slots); subject/object/qualifier nodes are single entities.",
322
- code="encoding-list-annotation-only",
323
- )
324
- return self
325
-
326
282
 
327
283
  class Qualifier(NodeEncoding):
328
284
  qualifier: Qualifiers = Field(
@@ -330,6 +286,15 @@ class Qualifier(NodeEncoding):
330
286
  description="Qualifier predicate key used as the output qualifier column.",
331
287
  examples=[Qualifiers.OBJECT_DIRECTION_QUALIFIER, Qualifiers.SUBJECT_CONTEXT_QUALIFIER],
332
288
  )
289
+ nullable: bool = Field(
290
+ False,
291
+ description=(
292
+ "When True, a blank or unresolvable ``method: column`` cell keeps the edge and omits the "
293
+ "qualifier for that row (the column stays null and the null-stripper drops the key); when "
294
+ "False (default) such a row is dropped, exactly like an unresolved subject/object. Only "
295
+ "meaningful for ``method: column`` — a literal qualifier can never be null."
296
+ ),
297
+ )
333
298
 
334
299
  @property
335
300
  def vocabulary(self: Self) -> frozenset[str] | None:
@@ -400,6 +365,23 @@ class Qualifier(NodeEncoding):
400
365
  )
401
366
  return self
402
367
 
368
+ @model_validator(mode="after")
369
+ def reject_nullable_literal_qualifiers(self: Self) -> Self:
370
+ """Reject ``nullable: true`` on literal qualifiers.
371
+
372
+ ``nullable`` only has meaning for a ``method: column`` qualifier: a blank or
373
+ unresolved cell keeps the edge and the qualifier is omitted for that row. A
374
+ ``method: value`` qualifier is a config-time constant that can never be blank,
375
+ so ``nullable`` would be dead config that misleads the reader. Fail loudly at
376
+ config time instead (the removed ``method: list`` is already rejected upstream
377
+ by :meth:`Encoding.reject_removed_list_method`).
378
+ """
379
+ if self.nullable and self.method != EncodingMethods.COLUMN:
380
+ raise TablassertValidationError(
381
+ "`nullable` only applies to `method: column` qualifiers; a literal qualifier can never be null.", code="qualifier-nullable-literal"
382
+ )
383
+ return self
384
+
403
385
 
404
386
  class Statement(TablaBase):
405
387
  subject: NodeEncoding = Field(..., description="Subject node encoding and mapping configuration.")
@@ -508,19 +490,15 @@ class Annotation(Encoding):
508
490
  def split_by_requires_a_column(self) -> Self:
509
491
  """Enforce that ``split_by`` carries a real separator and a ``method: column`` encoding.
510
492
 
511
- ``split_by`` is the per-row counterpart of ``method: list``: it turns each cell's
512
- own delimited text into a real JSON array, which is the one multivalued shape a
513
- literal cannot express (a list ``encoding`` is fixed at config time, so it emits
514
- the same array on every row). A ``value``/``list`` encoding therefore declares its
515
- members directly rather than round-tripping them through a separator.
493
+ ``split_by`` is the one multivalued encoding: it turns each cell's own
494
+ delimited text into a real JSON array, per row. A ``value`` encoding is a
495
+ scalar literal with no per-row text to split, so it rejects ``split_by``.
516
496
  """
517
497
  if self.split_by is None:
518
498
  return self
519
499
  if self.method != EncodingMethods.COLUMN:
520
500
  raise TablassertValidationError(
521
- "`split_by` splits a column's per-row text and requires `method: column`; "
522
- "declare a literal multivalued annotation with `method: list` instead.",
523
- code="annotation-split-by-requires-column",
501
+ "`split_by` splits a column's per-row text and requires `method: column`.", code="annotation-split-by-requires-column"
524
502
  )
525
503
  if not self.split_by:
526
504
  # An empty separator splits into individual characters -- exactly the
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes