tablassert 12.0.0__tar.gz → 12.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. {tablassert-12.0.0 → tablassert-12.1.0}/PKG-INFO +1 -1
  2. {tablassert-12.0.0 → tablassert-12.1.0}/pyproject.toml +1 -1
  3. {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/agent.py +5 -4
  4. {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/biolink.py +7 -5
  5. {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/errors.py +1 -0
  6. {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/lib.py +55 -26
  7. {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/models.py +19 -1
  8. {tablassert-12.0.0 → tablassert-12.1.0}/LICENSE +0 -0
  9. {tablassert-12.0.0 → tablassert-12.1.0}/README.md +0 -0
  10. {tablassert-12.0.0 → tablassert-12.1.0}/rust/Cargo.lock +0 -0
  11. {tablassert-12.0.0 → tablassert-12.1.0}/rust/Cargo.toml +0 -0
  12. {tablassert-12.0.0 → tablassert-12.1.0}/rust/examples/count_tables.rs +0 -0
  13. {tablassert-12.0.0 → tablassert-12.1.0}/rust/src/fullmap.rs +0 -0
  14. {tablassert-12.0.0 → tablassert-12.1.0}/rust/src/json.rs +0 -0
  15. {tablassert-12.0.0 → tablassert-12.1.0}/rust/src/lib.rs +0 -0
  16. {tablassert-12.0.0 → tablassert-12.1.0}/rust/src/ndjson.rs +0 -0
  17. {tablassert-12.0.0 → tablassert-12.1.0}/rust/src/uuid.rs +0 -0
  18. {tablassert-12.0.0 → tablassert-12.1.0}/rust/tests/build_golden.rs +0 -0
  19. {tablassert-12.0.0 → tablassert-12.1.0}/rust/tests/common/mod.rs +0 -0
  20. {tablassert-12.0.0 → tablassert-12.1.0}/rust/tests/extract_prebuilt.rs +0 -0
  21. {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/__init__.py +0 -0
  22. {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/_lazy.py +0 -0
  23. {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/cli.py +0 -0
  24. {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/coerce.py +0 -0
  25. {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/enums.py +0 -0
  26. {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/extras.py +0 -0
  27. {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/fullmap.py +0 -0
  28. {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/graph_target.py +0 -0
  29. {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/ingests.py +0 -0
  30. {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/log.py +0 -0
  31. {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/nlp.py +0 -0
  32. {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/progress.py +0 -0
  33. {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/qc.py +0 -0
  34. {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/rig.py +0 -0
  35. {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/rs.pyi +0 -0
  36. {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/study.py +0 -0
  37. {tablassert-12.0.0 → tablassert-12.1.0}/src/tablassert/utils.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: tablassert
3
- Version: 12.0.0
3
+ Version: 12.1.0
4
4
  Classifier: License :: OSI Approved :: Apache Software License
5
5
  Classifier: Development Status :: 5 - Production/Stable
6
6
  Classifier: Intended Audience :: Science/Research
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "tablassert"
3
- version = "12.0.0"
3
+ version = "12.1.0"
4
4
  description = "Extract knowledge assertions from tabular data into NCATS Translator-compliant KGX NDJSON — declaratively, with entity resolution and quality control built in."
5
5
  authors = [
6
6
  { name = "Skye Lane Goetz", email = "sgoetz@isbscience.org" }
@@ -2288,10 +2288,11 @@ qualifier and evidence slot the specific class declared. build_and_audit reports
2288
2288
  application numbers, `approval_ids` is a deliberate translator-ingest pass-through: keep
2289
2289
  the pipe-joined value as a scalar and do not add `split_by`.
2290
2290
  - MULTIVALUED slots (`has_evidence` and friends) take a real JSON array, never a joined string:
2291
- declare the annotation `{method: column, encoding: <letter>, split_by: "|"}` so each cell's
2292
- delimited text splits into its own per-row array. `split_by` is the ONLY multivalued encoding
2293
- — there is no literal-list method, and a scalar bound for a multivalued slot ships to consumers
2294
- as one unusable "a|b|c" blob.
2291
+ `split_by` is the ONLY multivalued encoding — there is no literal-list method. INSPECT the
2292
+ column's cells first (read_table shows them); the separator they ACTUALLY use — `|`, `,`, or
2293
+ `;` — is the one you declare: `{method: column, encoding: <letter>, split_by: "<separator>"}`.
2294
+ A SINGLE-value cell gets NO `split_by`: its scalar wraps into a one-element array, the correct
2295
+ shape. Cells that DO join multiple values but OMIT `split_by` ship as one unusable joined blob.
2295
2296
  - `effect_size` / `effect_type` are deliberate Tablassert extras pending biolink-model#1774;
2296
2297
  `approval_ids` is a deliberate translator-ingest pass-through extra. All three are EXEMPT from
2297
2298
  the validity score: a `biolink_valid_pct` below 1.0 is never caused by these intentional fields.
@@ -695,11 +695,13 @@ def _scalar_types(annotation: Any) -> set[type]:
695
695
  def numeric_slot_kind(field: str) -> str | None:
696
696
  """Return ``"int"`` / ``"float"`` when a Biolink association slot has a numeric range.
697
697
 
698
- Tablassert stringifies its numeric annotation columns for notation control, but
699
- Biolink types ``p_value`` and ``adjusted_p_value`` as ``float`` and (with
700
- ``biolink/biolink-model#1770``) ``supporting_study_size`` as ``int``. Those must be
701
- emitted as real JSON numbers. Derived from the installed model so the answer
702
- tracks whatever version is pinned.
698
+ Tablassert stringifies its numeric annotation columns for notation control (p-value
699
+ columns are ALWAYS scientific-notation strings, which Pydantic's lax validation
700
+ coerces back for the ``float``-typed ``p_value`` / ``adjusted_p_value`` slots), but
701
+ a non-p-value column that the installed model types ``int`` (``supporting_study_size``
702
+ once ``biolink/biolink-model#1770`` lands) or ``float`` must be emitted as a real
703
+ JSON number. Derived from the installed model so the answer tracks whatever version
704
+ is pinned.
703
705
 
704
706
  Args:
705
707
  field: Edge column name.
@@ -23,6 +23,7 @@ TablassertErrorCodes = Literal[
23
23
  "graph-bad-infores",
24
24
  "override-bad-publication",
25
25
  "override-bad-upstream-infores",
26
+ "override-bad-upstream-urls",
26
27
  "provenance-bad-pmc-id",
27
28
  "provenance-missing-publication",
28
29
  "provenance-publication-and-override",
@@ -469,7 +469,9 @@ def inline_supporting_study(lf: pl.LazyFrame, study_id: str, sheet: str | None,
469
469
  return out.drop(drop)
470
470
 
471
471
 
472
- def retrieval_sources(lf: pl.LazyFrame, primary: str, upstream: list[str], urls: list[str]) -> pl.LazyFrame:
472
+ def retrieval_sources(
473
+ lf: pl.LazyFrame, primary: str, upstream: list[str], urls: list[str], upstream_urls: dict[str, list[str]] | None = None
474
+ ) -> pl.LazyFrame:
473
475
  """Add the Biolink ``sources`` retrieval-provenance column.
474
476
 
475
477
  ``upstream_resource_ids`` and ``source_record_urls`` have ``domain: retrieval
@@ -478,21 +480,30 @@ def retrieval_sources(lf: pl.LazyFrame, primary: str, upstream: list[str], urls:
478
480
  validation with ``extra_forbidden``.
479
481
 
480
482
  Mirrors ``build_association_knowledge_sources()`` from
481
- ``translator-ingests/util/biolink.py``: the primary knowledge source carries the
482
- source record URLs and lists the upstream resources, and each upstream resource
483
- additionally appears as its own ``supporting_data_source`` entry.
483
+ ``translator-ingests/util/biolink.py``: the primary knowledge source lists the
484
+ upstream resources, and each upstream resource additionally appears as its own
485
+ ``supporting_data_source`` entry. By default the primary entry carries the
486
+ section's source record URLs; when ``upstream_urls`` is given, URL placement is
487
+ fully determined by that mapping instead -- each listed upstream entry carries
488
+ its own ``source_record_urls`` and the primary entry emits none (the primary is
489
+ the transforming resource, not a downloadable record).
484
490
 
485
491
  Args:
486
492
  lf: Source LazyFrame.
487
493
  primary: Infores CURIE of the primary knowledge source.
488
494
  upstream: Infores CURIEs of upstream/supporting data sources.
489
- urls: Source record URLs for the primary entry.
495
+ urls: Source record URLs for the primary entry (ignored when ``upstream_urls`` is set).
496
+ upstream_urls: Optional per-upstream source record URLs keyed by infores CURIE.
490
497
 
491
498
  Returns:
492
499
  LazyFrame with a ``sources`` ``list[struct]`` column appended.
493
500
  """
494
- entries: list[pl.Expr] = [_retrieval_source(primary, "primary_knowledge_source", upstream, urls)]
495
- entries.extend(_retrieval_source(x, "supporting_data_source") for x in upstream)
501
+ if upstream_urls is not None:
502
+ entries: list[pl.Expr] = [_retrieval_source(primary, "primary_knowledge_source", upstream)]
503
+ entries.extend(_retrieval_source(x, "supporting_data_source", urls=upstream_urls.get(x)) for x in upstream)
504
+ else:
505
+ entries = [_retrieval_source(primary, "primary_knowledge_source", upstream, urls)]
506
+ entries.extend(_retrieval_source(x, "supporting_data_source") for x in upstream)
496
507
  return lf.with_columns(pl.concat_list(entries).alias("sources"))
497
508
 
498
509
 
@@ -592,15 +603,19 @@ def clean_numeric(lf: pl.LazyFrame) -> pl.LazyFrame:
592
603
  def format_numeric(lf: pl.LazyFrame) -> pl.LazyFrame:
593
604
  """Normalize numeric annotation columns for output.
594
605
 
595
- Columns that map to a numeric Biolink slot are emitted as real JSON numbers:
596
- ``p_value`` and ``adjusted_p_value`` are typed ``float`` in the model (and
597
- ``supporting_study_size`` ``integer`` once ``biolink/biolink-model#1770`` lands),
598
- so writing ``"6.5200e-06"`` produces a file that strict consumers reject even
599
- though Pydantic's lax mode happens to coerce it.
600
-
601
- Columns with no numeric Biolink slot keep the controlled string notation --
602
- p-value-like names use scientific (``{:.4e}``), others decimal general
603
- (``{:.4g}``) -- because they end up in human-readable text (the inlined
606
+ P-value columns (any name containing ``p_value``) are always emitted as
607
+ controlled scientific-notation strings (``{:.4e}``, e.g. ``"1.0000e-03"``):
608
+ notation is part of the output contract (see the tutorial's edge example), and
609
+ shortest-repr JSON numbers would render the same values as ``0.0001`` / ``0.05``.
610
+ Biolink types ``p_value`` / ``adjusted_p_value`` as ``float``, but the validation
611
+ here and downstream runs in Pydantic's lax mode, which coerces the numeric string
612
+ back -- so the notation control costs no KGX validity.
613
+
614
+ Remaining numeric columns (``effect_size`` / ``supporting_study_size``) that a
615
+ future biolink model types ``int`` / ``float`` (e.g. once
616
+ ``biolink/biolink-model#1770`` / ``#1774`` land) are emitted as real JSON numbers;
617
+ today's untyped ones keep the controlled decimal general string notation
618
+ (``{:.4g}``) because they end up in human-readable text (the inlined
604
619
  ``StudyResult`` description or ``supporting_text``). Null values stay null.
605
620
 
606
621
  Args:
@@ -617,15 +632,20 @@ def format_numeric(lf: pl.LazyFrame) -> pl.LazyFrame:
617
632
  # Collection point: batch formatting for notation control.
618
633
  df: pl.DataFrame = lf.collect()
619
634
  for c in numeric_columns(df.columns):
620
- kind: str | None = numeric_slot_kind(c)
621
- if kind == "float":
622
- df = df.with_columns(pl.col(c).cast(pl.Float64, strict=False).alias(c))
623
- continue
624
- if kind == "int":
625
- df = df.with_columns(pl.col(c).cast(pl.Float64, strict=False).round().cast(pl.Int64, strict=False).alias(c))
626
- continue
627
635
  df = df.with_columns(pl.col(c).cast(pl.Float64, strict=False).alias(c))
628
- fmt: str = "{:.4e}" if "p_value" in c.lower() else "{:.4g}"
636
+ if "p_value" in c.lower():
637
+ # Scientific notation is the p-value output contract; the branch ordering is
638
+ # the guard, so the numeric-slot short-circuits below can never fire for
639
+ # these columns even though the model types them ``float``.
640
+ fmt: str = "{:.4e}"
641
+ else:
642
+ kind: str | None = numeric_slot_kind(c)
643
+ if kind == "float":
644
+ continue
645
+ if kind == "int":
646
+ df = df.with_columns(pl.col(c).round().cast(pl.Int64, strict=False).alias(c))
647
+ continue
648
+ fmt = "{:.4g}"
629
649
  formatted: list[str | None] = [None if v is None else fmt.format(v) for v in df[c].to_list()]
630
650
  df = df.with_columns(pl.Series(c, formatted))
631
651
  return df.lazy()
@@ -1132,6 +1152,11 @@ class Tcode(Section):
1132
1152
  # name, so a RIG and its edges can never disagree about the source identity.
1133
1153
  primary_knowledge_source: str | None = self.infores
1134
1154
  upstream_ids = override.upstream_resource_ids if override else upstream_resource_ids(self.provenance.repo)
1155
+ upstream_urls: dict[str, list[str]] | None = (
1156
+ {key: [str(u) for u in urls] for key, urls in override.upstream_source_record_urls.items()}
1157
+ if override and override.upstream_source_record_urls is not None
1158
+ else None
1159
+ )
1135
1160
  knowledge_level = override.knowledge_level if override else self.provenance.knowledge_level
1136
1161
  agent_type = override.agent_type if override else self.provenance.agent_type
1137
1162
  publication_values = override.publications if override else [publication_curie(self.provenance.repo, self.provenance.publication or "")]
@@ -1151,8 +1176,12 @@ class Tcode(Section):
1151
1176
  (value, ("agent_type", agent_type)),
1152
1177
  # Retrieval provenance lives only in the nested `sources` list (Biolink
1153
1178
  # RetrievalSource); current translator-ingests emits no flat
1154
- # `primary_knowledge_source` scalar, so neither do we.
1155
- (retrieval_sources, (primary_knowledge_source, upstream_ids, [str(u) for u in self.source.url])) if primary_knowledge_source else None,
1179
+ # `primary_knowledge_source` scalar, so neither do we. A per-upstream URL
1180
+ # mapping (override.upstream_source_record_urls) re-homes the record URLs
1181
+ # from the primary entry onto the matching supporting entries.
1182
+ (retrieval_sources, (primary_knowledge_source, upstream_ids, [str(u) for u in self.source.url], upstream_urls))
1183
+ if primary_knowledge_source
1184
+ else None,
1156
1185
  (publications, (publication_values,)) if publication_values else None,
1157
1186
  # Prune first so class-rejected values are handed to the study rather than lost.
1158
1187
  (prune_to_class, ()),
@@ -119,7 +119,7 @@ class BaseSource(TablaBase):
119
119
  url: list[HttpUrl] = Field(
120
120
  ...,
121
121
  min_length=1,
122
- description="One or more remote source URL(s) recorded as provenance; emitted in the edge `sources` list under the primary entry's `source_record_urls` list and in the RIG. Format-validated only; not fetched.",
122
+ description="One or more remote source URL(s) recorded as provenance; emitted in the edge `sources` list under the primary entry's `source_record_urls` list and in the RIG. When `provenance.override.upstream_source_record_urls` is set, these URLs serve the RIG only and the per-upstream mapping determines edge placement instead. Format-validated only; not fetched.",
123
123
  )
124
124
 
125
125
  rows: list[NonNegativeInt] | None = Field(None, description="Zero-based row indices kept after any row_slice crop.", examples=[[0, 2, 5]])
@@ -464,6 +464,11 @@ class ManualProvenance(TablaBase):
464
464
  description="Manual upstream source infores CURIEs emitted instead of the repo-derived source map; the sanctioned place for manual infores.",
465
465
  examples=[["infores:my-upstream"]],
466
466
  )
467
+ upstream_source_record_urls: dict[str, list[HttpUrl]] | None = Field(
468
+ None,
469
+ description="Per-upstream source record URLs keyed by infores CURIE; every key must appear in `upstream_resource_ids`. When set, the section's `source.url` values are NOT emitted on the primary `sources` entry (RIG use only) — each listed upstream supporting entry carries its own `source_record_urls` instead.",
470
+ examples=[{"infores:my-upstream": ["https://example.org/dataset"]}],
471
+ )
467
472
  publications: list[str] | None = Field(
468
473
  None,
469
474
  description="Publication CURIEs emitted verbatim; currently PMCID CURIEs are required for manual provenance.",
@@ -493,6 +498,19 @@ class ManualProvenance(TablaBase):
493
498
  )
494
499
  return values
495
500
 
501
+ @model_validator(mode="after")
502
+ def upstream_urls_match_resource_ids(self: Self) -> Self:
503
+ if self.upstream_source_record_urls is None:
504
+ return self
505
+ for key in self.upstream_source_record_urls:
506
+ validate_infores_curie(key, "override-bad-upstream-urls")
507
+ unknown = sorted(set(self.upstream_source_record_urls) - set(self.upstream_resource_ids))
508
+ if unknown:
509
+ raise TablassertValidationError(
510
+ f"`upstream_source_record_urls` keys must appear in `upstream_resource_ids`, got {unknown}.", code="override-bad-upstream-urls"
511
+ )
512
+ return self
513
+
496
514
 
497
515
  class Provenance(TablaBase):
498
516
  repo: Repositories = Field(Repositories.PUBMED_CENTRAL, description="Publication identifier namespace prefix.")
File without changes
File without changes
File without changes
File without changes
File without changes