tablassert 12.0.0__tar.gz → 12.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tablassert-12.0.0 → tablassert-12.1.0}/PKG-INFO +1 -1
- {tablassert-12.0.0 → tablassert-12.1.0}/pyproject.toml +1 -1
- {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/agent.py +5 -4
- {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/biolink.py +7 -5
- {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/errors.py +1 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/lib.py +55 -26
- {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/models.py +19 -1
- {tablassert-12.0.0 → tablassert-12.1.0}/LICENSE +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/README.md +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/rust/Cargo.lock +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/rust/Cargo.toml +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/rust/examples/count_tables.rs +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/rust/src/fullmap.rs +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/rust/src/json.rs +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/rust/src/lib.rs +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/rust/src/ndjson.rs +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/rust/src/uuid.rs +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/rust/tests/build_golden.rs +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/rust/tests/common/mod.rs +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/rust/tests/extract_prebuilt.rs +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/__init__.py +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/_lazy.py +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/cli.py +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/coerce.py +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/enums.py +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/extras.py +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/fullmap.py +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/graph_target.py +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/ingests.py +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/log.py +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/nlp.py +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/progress.py +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/qc.py +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/rig.py +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/rs.pyi +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/study.py +0 -0
- {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/utils.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "tablassert"
|
|
3
|
-
version = "12.
|
|
3
|
+
version = "12.1.0"
|
|
4
4
|
description = "Extract knowledge assertions from tabular data into NCATS Translator-compliant KGX NDJSON — declaratively, with entity resolution and quality control built in."
|
|
5
5
|
authors = [
|
|
6
6
|
{ name = "Skye Lane Goetz", email = "sgoetz@isbscience.org" }
|
|
@@ -2288,10 +2288,11 @@ qualifier and evidence slot the specific class declared. build_and_audit reports
|
|
|
2288
2288
|
application numbers, `approval_ids` is a deliberate translator-ingest pass-through: keep
|
|
2289
2289
|
the pipe-joined value as a scalar and do not add `split_by`.
|
|
2290
2290
|
- MULTIVALUED slots (`has_evidence` and friends) take a real JSON array, never a joined string:
|
|
2291
|
-
|
|
2292
|
-
|
|
2293
|
-
—
|
|
2294
|
-
|
|
2291
|
+
`split_by` is the ONLY multivalued encoding — there is no literal-list method. INSPECT the
|
|
2292
|
+
column's cells first (read_table shows them); the separator they ACTUALLY use — `|`, `,`, or
|
|
2293
|
+
`;` — is the one you declare: `{method: column, encoding: <letter>, split_by: "<separator>"}`.
|
|
2294
|
+
A SINGLE-value cell gets NO `split_by`: its scalar wraps into a one-element array, the correct
|
|
2295
|
+
shape. Cells that DO join multiple values but OMIT `split_by` ship as one unusable joined blob.
|
|
2295
2296
|
- `effect_size` / `effect_type` are deliberate Tablassert extras pending biolink-model#1774;
|
|
2296
2297
|
`approval_ids` is a deliberate translator-ingest pass-through extra. All three are EXEMPT from
|
|
2297
2298
|
the validity score: a `biolink_valid_pct` below 1.0 is never caused by these intentional fields.
|
|
@@ -695,11 +695,13 @@ def _scalar_types(annotation: Any) -> set[type]:
|
|
|
695
695
|
def numeric_slot_kind(field: str) -> str | None:
|
|
696
696
|
"""Return ``"int"`` / ``"float"`` when a Biolink association slot has a numeric range.
|
|
697
697
|
|
|
698
|
-
Tablassert stringifies its numeric annotation columns for notation control
|
|
699
|
-
|
|
700
|
-
``
|
|
701
|
-
|
|
702
|
-
|
|
698
|
+
Tablassert stringifies its numeric annotation columns for notation control (p-value
|
|
699
|
+
columns are ALWAYS scientific-notation strings, which Pydantic's lax validation
|
|
700
|
+
coerces back for the ``float``-typed ``p_value`` / ``adjusted_p_value`` slots), but
|
|
701
|
+
a non-p-value column that the installed model types ``int`` (``supporting_study_size``
|
|
702
|
+
once ``biolink/biolink-model#1770`` lands) or ``float`` must be emitted as a real
|
|
703
|
+
JSON number. Derived from the installed model so the answer tracks whatever version
|
|
704
|
+
is pinned.
|
|
703
705
|
|
|
704
706
|
Args:
|
|
705
707
|
field: Edge column name.
|
|
@@ -23,6 +23,7 @@ TablassertErrorCodes = Literal[
|
|
|
23
23
|
"graph-bad-infores",
|
|
24
24
|
"override-bad-publication",
|
|
25
25
|
"override-bad-upstream-infores",
|
|
26
|
+
"override-bad-upstream-urls",
|
|
26
27
|
"provenance-bad-pmc-id",
|
|
27
28
|
"provenance-missing-publication",
|
|
28
29
|
"provenance-publication-and-override",
|
|
@@ -469,7 +469,9 @@ def inline_supporting_study(lf: pl.LazyFrame, study_id: str, sheet: str | None,
|
|
|
469
469
|
return out.drop(drop)
|
|
470
470
|
|
|
471
471
|
|
|
472
|
-
def retrieval_sources(
|
|
472
|
+
def retrieval_sources(
|
|
473
|
+
lf: pl.LazyFrame, primary: str, upstream: list[str], urls: list[str], upstream_urls: dict[str, list[str]] | None = None
|
|
474
|
+
) -> pl.LazyFrame:
|
|
473
475
|
"""Add the Biolink ``sources`` retrieval-provenance column.
|
|
474
476
|
|
|
475
477
|
``upstream_resource_ids`` and ``source_record_urls`` have ``domain: retrieval
|
|
@@ -478,21 +480,30 @@ def retrieval_sources(lf: pl.LazyFrame, primary: str, upstream: list[str], urls:
|
|
|
478
480
|
validation with ``extra_forbidden``.
|
|
479
481
|
|
|
480
482
|
Mirrors ``build_association_knowledge_sources()`` from
|
|
481
|
-
``translator-ingests/util/biolink.py``: the primary knowledge source
|
|
482
|
-
|
|
483
|
-
|
|
483
|
+
``translator-ingests/util/biolink.py``: the primary knowledge source lists the
|
|
484
|
+
upstream resources, and each upstream resource additionally appears as its own
|
|
485
|
+
``supporting_data_source`` entry. By default the primary entry carries the
|
|
486
|
+
section's source record URLs; when ``upstream_urls`` is given, URL placement is
|
|
487
|
+
fully determined by that mapping instead -- each listed upstream entry carries
|
|
488
|
+
its own ``source_record_urls`` and the primary entry emits none (the primary is
|
|
489
|
+
the transforming resource, not a downloadable record).
|
|
484
490
|
|
|
485
491
|
Args:
|
|
486
492
|
lf: Source LazyFrame.
|
|
487
493
|
primary: Infores CURIE of the primary knowledge source.
|
|
488
494
|
upstream: Infores CURIEs of upstream/supporting data sources.
|
|
489
|
-
urls: Source record URLs for the primary entry.
|
|
495
|
+
urls: Source record URLs for the primary entry (ignored when ``upstream_urls`` is set).
|
|
496
|
+
upstream_urls: Optional per-upstream source record URLs keyed by infores CURIE.
|
|
490
497
|
|
|
491
498
|
Returns:
|
|
492
499
|
LazyFrame with a ``sources`` ``list[struct]`` column appended.
|
|
493
500
|
"""
|
|
494
|
-
|
|
495
|
-
|
|
501
|
+
if upstream_urls is not None:
|
|
502
|
+
entries: list[pl.Expr] = [_retrieval_source(primary, "primary_knowledge_source", upstream)]
|
|
503
|
+
entries.extend(_retrieval_source(x, "supporting_data_source", urls=upstream_urls.get(x)) for x in upstream)
|
|
504
|
+
else:
|
|
505
|
+
entries = [_retrieval_source(primary, "primary_knowledge_source", upstream, urls)]
|
|
506
|
+
entries.extend(_retrieval_source(x, "supporting_data_source") for x in upstream)
|
|
496
507
|
return lf.with_columns(pl.concat_list(entries).alias("sources"))
|
|
497
508
|
|
|
498
509
|
|
|
@@ -592,15 +603,19 @@ def clean_numeric(lf: pl.LazyFrame) -> pl.LazyFrame:
|
|
|
592
603
|
def format_numeric(lf: pl.LazyFrame) -> pl.LazyFrame:
|
|
593
604
|
"""Normalize numeric annotation columns for output.
|
|
594
605
|
|
|
595
|
-
|
|
596
|
-
|
|
597
|
-
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
|
|
603
|
-
(``
|
|
606
|
+
P-value columns (any name containing ``p_value``) are always emitted as
|
|
607
|
+
controlled scientific-notation strings (``{:.4e}``, e.g. ``"1.0000e-03"``):
|
|
608
|
+
notation is part of the output contract (see the tutorial's edge example), and
|
|
609
|
+
shortest-repr JSON numbers would render the same values as ``0.0001`` / ``0.05``.
|
|
610
|
+
Biolink types ``p_value`` / ``adjusted_p_value`` as ``float``, but the validation
|
|
611
|
+
here and downstream runs in Pydantic's lax mode, which coerces the numeric string
|
|
612
|
+
back -- so the notation control costs no KGX validity.
|
|
613
|
+
|
|
614
|
+
Remaining numeric columns (``effect_size`` / ``supporting_study_size``) that a
|
|
615
|
+
future biolink model types ``int`` / ``float`` (e.g. once
|
|
616
|
+
``biolink/biolink-model#1770`` / ``#1774`` land) are emitted as real JSON numbers;
|
|
617
|
+
today's untyped ones keep the controlled decimal general string notation
|
|
618
|
+
(``{:.4g}``) because they end up in human-readable text (the inlined
|
|
604
619
|
``StudyResult`` description or ``supporting_text``). Null values stay null.
|
|
605
620
|
|
|
606
621
|
Args:
|
|
@@ -617,15 +632,20 @@ def format_numeric(lf: pl.LazyFrame) -> pl.LazyFrame:
|
|
|
617
632
|
# Collection point: batch formatting for notation control.
|
|
618
633
|
df: pl.DataFrame = lf.collect()
|
|
619
634
|
for c in numeric_columns(df.columns):
|
|
620
|
-
kind: str | None = numeric_slot_kind(c)
|
|
621
|
-
if kind == "float":
|
|
622
|
-
df = df.with_columns(pl.col(c).cast(pl.Float64, strict=False).alias(c))
|
|
623
|
-
continue
|
|
624
|
-
if kind == "int":
|
|
625
|
-
df = df.with_columns(pl.col(c).cast(pl.Float64, strict=False).round().cast(pl.Int64, strict=False).alias(c))
|
|
626
|
-
continue
|
|
627
635
|
df = df.with_columns(pl.col(c).cast(pl.Float64, strict=False).alias(c))
|
|
628
|
-
|
|
636
|
+
if "p_value" in c.lower():
|
|
637
|
+
# Scientific notation is the p-value output contract; the branch ordering is
|
|
638
|
+
# the guard, so the numeric-slot short-circuits below can never fire for
|
|
639
|
+
# these columns even though the model types them ``float``.
|
|
640
|
+
fmt: str = "{:.4e}"
|
|
641
|
+
else:
|
|
642
|
+
kind: str | None = numeric_slot_kind(c)
|
|
643
|
+
if kind == "float":
|
|
644
|
+
continue
|
|
645
|
+
if kind == "int":
|
|
646
|
+
df = df.with_columns(pl.col(c).round().cast(pl.Int64, strict=False).alias(c))
|
|
647
|
+
continue
|
|
648
|
+
fmt = "{:.4g}"
|
|
629
649
|
formatted: list[str | None] = [None if v is None else fmt.format(v) for v in df[c].to_list()]
|
|
630
650
|
df = df.with_columns(pl.Series(c, formatted))
|
|
631
651
|
return df.lazy()
|
|
@@ -1132,6 +1152,11 @@ class Tcode(Section):
|
|
|
1132
1152
|
# name, so a RIG and its edges can never disagree about the source identity.
|
|
1133
1153
|
primary_knowledge_source: str | None = self.infores
|
|
1134
1154
|
upstream_ids = override.upstream_resource_ids if override else upstream_resource_ids(self.provenance.repo)
|
|
1155
|
+
upstream_urls: dict[str, list[str]] | None = (
|
|
1156
|
+
{key: [str(u) for u in urls] for key, urls in override.upstream_source_record_urls.items()}
|
|
1157
|
+
if override and override.upstream_source_record_urls is not None
|
|
1158
|
+
else None
|
|
1159
|
+
)
|
|
1135
1160
|
knowledge_level = override.knowledge_level if override else self.provenance.knowledge_level
|
|
1136
1161
|
agent_type = override.agent_type if override else self.provenance.agent_type
|
|
1137
1162
|
publication_values = override.publications if override else [publication_curie(self.provenance.repo, self.provenance.publication or "")]
|
|
@@ -1151,8 +1176,12 @@ class Tcode(Section):
|
|
|
1151
1176
|
(value, ("agent_type", agent_type)),
|
|
1152
1177
|
# Retrieval provenance lives only in the nested `sources` list (Biolink
|
|
1153
1178
|
# RetrievalSource); current translator-ingests emits no flat
|
|
1154
|
-
# `primary_knowledge_source` scalar, so neither do we.
|
|
1155
|
-
|
|
1179
|
+
# `primary_knowledge_source` scalar, so neither do we. A per-upstream URL
|
|
1180
|
+
# mapping (override.upstream_source_record_urls) re-homes the record URLs
|
|
1181
|
+
# from the primary entry onto the matching supporting entries.
|
|
1182
|
+
(retrieval_sources, (primary_knowledge_source, upstream_ids, [str(u) for u in self.source.url], upstream_urls))
|
|
1183
|
+
if primary_knowledge_source
|
|
1184
|
+
else None,
|
|
1156
1185
|
(publications, (publication_values,)) if publication_values else None,
|
|
1157
1186
|
# Prune first so class-rejected values are handed to the study rather than lost.
|
|
1158
1187
|
(prune_to_class, ()),
|
|
@@ -119,7 +119,7 @@ class BaseSource(TablaBase):
|
|
|
119
119
|
url: list[HttpUrl] = Field(
|
|
120
120
|
...,
|
|
121
121
|
min_length=1,
|
|
122
|
-
description="One or more remote source URL(s) recorded as provenance; emitted in the edge `sources` list under the primary entry's `source_record_urls` list and in the RIG. Format-validated only; not fetched.",
|
|
122
|
+
description="One or more remote source URL(s) recorded as provenance; emitted in the edge `sources` list under the primary entry's `source_record_urls` list and in the RIG. When `provenance.override.upstream_source_record_urls` is set, these URLs serve the RIG only and the per-upstream mapping determines edge placement instead. Format-validated only; not fetched.",
|
|
123
123
|
)
|
|
124
124
|
|
|
125
125
|
rows: list[NonNegativeInt] | None = Field(None, description="Zero-based row indices kept after any row_slice crop.", examples=[[0, 2, 5]])
|
|
@@ -464,6 +464,11 @@ class ManualProvenance(TablaBase):
|
|
|
464
464
|
description="Manual upstream source infores CURIEs emitted instead of the repo-derived source map; the sanctioned place for manual infores.",
|
|
465
465
|
examples=[["infores:my-upstream"]],
|
|
466
466
|
)
|
|
467
|
+
upstream_source_record_urls: dict[str, list[HttpUrl]] | None = Field(
|
|
468
|
+
None,
|
|
469
|
+
description="Per-upstream source record URLs keyed by infores CURIE; every key must appear in `upstream_resource_ids`. When set, the section's `source.url` values are NOT emitted on the primary `sources` entry (RIG use only) — each listed upstream supporting entry carries its own `source_record_urls` instead.",
|
|
470
|
+
examples=[{"infores:my-upstream": ["https://example.org/dataset"]}],
|
|
471
|
+
)
|
|
467
472
|
publications: list[str] | None = Field(
|
|
468
473
|
None,
|
|
469
474
|
description="Publication CURIEs emitted verbatim; currently PMCID CURIEs are required for manual provenance.",
|
|
@@ -493,6 +498,19 @@ class ManualProvenance(TablaBase):
|
|
|
493
498
|
)
|
|
494
499
|
return values
|
|
495
500
|
|
|
501
|
+
@model_validator(mode="after")
|
|
502
|
+
def upstream_urls_match_resource_ids(self: Self) -> Self:
|
|
503
|
+
if self.upstream_source_record_urls is None:
|
|
504
|
+
return self
|
|
505
|
+
for key in self.upstream_source_record_urls:
|
|
506
|
+
validate_infores_curie(key, "override-bad-upstream-urls")
|
|
507
|
+
unknown = sorted(set(self.upstream_source_record_urls) - set(self.upstream_resource_ids))
|
|
508
|
+
if unknown:
|
|
509
|
+
raise TablassertValidationError(
|
|
510
|
+
f"`upstream_source_record_urls` keys must appear in `upstream_resource_ids`, got {unknown}.", code="override-bad-upstream-urls"
|
|
511
|
+
)
|
|
512
|
+
return self
|
|
513
|
+
|
|
496
514
|
|
|
497
515
|
class Provenance(TablaBase):
|
|
498
516
|
repo: Repositories = Field(Repositories.PUBMED_CENTRAL, description="Publication identifier namespace prefix.")
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|