tablassert 13.0.0__tar.gz → 14.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tablassert-13.0.0 → tablassert-14.0.0}/PKG-INFO +1 -1
- {tablassert-13.0.0 → tablassert-14.0.0}/pyproject.toml +1 -1
- {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/biolink.py +35 -1
- {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/coerce.py +95 -13
- {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/errors.py +1 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/lib.py +83 -11
- {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/models.py +105 -2
- {tablassert-13.0.0 → tablassert-14.0.0}/LICENSE +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/README.md +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/rust/Cargo.lock +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/rust/Cargo.toml +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/rust/examples/count_tables.rs +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/rust/src/fullmap.rs +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/rust/src/json.rs +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/rust/src/lib.rs +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/rust/src/ndjson.rs +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/rust/src/uuid.rs +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/rust/tests/build_golden.rs +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/rust/tests/common/mod.rs +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/rust/tests/extract_prebuilt.rs +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/__init__.py +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/_lazy.py +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/agent.py +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/cli.py +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/enums.py +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/extras.py +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/fullmap.py +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/graph_target.py +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/ingests.py +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/log.py +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/nlp.py +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/progress.py +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/qc.py +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/rig.py +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/rs.pyi +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/study.py +0 -0
- {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/utils.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "tablassert"
|
|
3
|
-
version = "
|
|
3
|
+
version = "14.0.0"
|
|
4
4
|
description = "Extract knowledge assertions from tabular data into NCATS Translator-compliant KGX NDJSON — declaratively, with entity resolution and quality control built in."
|
|
5
5
|
authors = [
|
|
6
6
|
{ name = "Skye Lane Goetz", email = "sgoetz@isbscience.org" }
|
|
@@ -643,6 +643,40 @@ def is_multivalued(cls: type[Any], field: str) -> bool:
|
|
|
643
643
|
return any(get_origin(arg) is list for arg in get_args(annotation))
|
|
644
644
|
|
|
645
645
|
|
|
646
|
+
@cache
|
|
647
|
+
def _retrieval_source_id_required() -> bool:
|
|
648
|
+
"""Whether the installed model still requires the inherited source ``id`` field."""
|
|
649
|
+
retrieval_source: Any = getattr(_bm, "RetrievalSource", None)
|
|
650
|
+
id_field: Any = getattr(retrieval_source, "model_fields", {}).get("id")
|
|
651
|
+
return id_field is not None and id_field.is_required()
|
|
652
|
+
|
|
653
|
+
|
|
654
|
+
def _validation_record(record: dict[str, Any], *, edge: bool) -> dict[str, Any]:
|
|
655
|
+
"""Add only in-memory compatibility aliases needed by the installed Biolink model.
|
|
656
|
+
|
|
657
|
+
``RetrievalSource`` in the currently pinned model still requires the inherited
|
|
658
|
+
``Entity.id`` even though ``resource_id`` is the canonical provenance identifier
|
|
659
|
+
emitted by Tablassert. The alias is used solely for Pydantic validation; it never
|
|
660
|
+
changes the decoded KGX record or the files written by the pipeline.
|
|
661
|
+
"""
|
|
662
|
+
if not edge:
|
|
663
|
+
return record
|
|
664
|
+
if not _retrieval_source_id_required():
|
|
665
|
+
return record
|
|
666
|
+
sources: Any = record.get("sources")
|
|
667
|
+
if not isinstance(sources, list):
|
|
668
|
+
return record
|
|
669
|
+
normalized: list[Any] = []
|
|
670
|
+
changed: bool = False
|
|
671
|
+
for source in sources:
|
|
672
|
+
if isinstance(source, dict) and "id" not in source and "resource_id" in source:
|
|
673
|
+
normalized.append({**source, "id": source["resource_id"]})
|
|
674
|
+
changed = True
|
|
675
|
+
else:
|
|
676
|
+
normalized.append(source)
|
|
677
|
+
return {**record, "sources": normalized} if changed else record
|
|
678
|
+
|
|
679
|
+
|
|
646
680
|
def validate_record(record: dict[str, Any], *, edge: bool) -> list[str]:
|
|
647
681
|
"""Validate one KGX record against the Biolink class named by its ``category``.
|
|
648
682
|
|
|
@@ -659,7 +693,7 @@ def validate_record(record: dict[str, Any], *, edge: bool) -> list[str]:
|
|
|
659
693
|
category: str = categories[0] if isinstance(categories, list) and categories else str(categories or "")
|
|
660
694
|
cls: type[Any] = association_class(category) if edge else node_class(category)
|
|
661
695
|
try:
|
|
662
|
-
cls(**record)
|
|
696
|
+
cls(**_validation_record(record, edge=edge))
|
|
663
697
|
except ValidationError as error:
|
|
664
698
|
return [f"{'.'.join(str(part) for part in item['loc']) or '?'}: {item['type']}" for item in error.errors()]
|
|
665
699
|
return []
|
|
@@ -48,8 +48,6 @@ def sig(lf: pl.LazyFrame, col: str = "p_value", out: str = "statistical_signific
|
|
|
48
48
|
null p-value → null qualifier. The qualifier may only be set when
|
|
49
49
|
``p_value``/``adjusted_p_value`` is populated (Biolink class rule).
|
|
50
50
|
"""
|
|
51
|
-
from rapidfuzz import fuzz
|
|
52
|
-
|
|
53
51
|
names: list[str] = lf.collect_schema().names()
|
|
54
52
|
# Same rigorous classification as ``coerce_pvalue_columns``: a column is a significance
|
|
55
53
|
# source only when ``pvalue_target`` accepts it — NOT by naive substring — so non-p-value
|
|
@@ -66,11 +64,14 @@ def sig(lf: pl.LazyFrame, col: str = "p_value", out: str = "statistical_signific
|
|
|
66
64
|
# bucket is present. Raw p-value is the canonical significance source; adjusted is the fallback.
|
|
67
65
|
preferred: str = col if col in buckets else next(iter(buckets))
|
|
68
66
|
candidates: list[str] = buckets[preferred]
|
|
69
|
-
|
|
70
|
-
#
|
|
71
|
-
|
|
72
|
-
chosen: str = preferred if preferred in candidates else max(candidates, key=lambda c: fuzz.ratio(c, reference))
|
|
67
|
+
# Same selection rule as ``coerce_pvalue_columns``: canonical wins, then a
|
|
68
|
+
# raw candidate beats a -log10 alias, then fuzzy ranking among what is left.
|
|
69
|
+
chosen: str = _best_candidate(candidates, preferred, preferred)
|
|
73
70
|
expr: pl.Expr = pl.col(chosen).cast(pl.Float64, strict=False)
|
|
71
|
+
# A -log10(p) score column must be un-logged before banding, or the bands
|
|
72
|
+
# invert (a score of 8 means p = 1e-8, not p = 8.0 -> not_significant).
|
|
73
|
+
if is_neglog10_column(chosen):
|
|
74
|
+
expr = _unlog10(expr)
|
|
74
75
|
band: pl.Expr = (
|
|
75
76
|
pl.when(expr.is_null())
|
|
76
77
|
.then(pl.lit(None, dtype=pl.String))
|
|
@@ -173,6 +174,54 @@ SIGNIFICANCE_FLAG_PATTERN: re.Pattern[str] = re.compile(
|
|
|
173
174
|
""",
|
|
174
175
|
re.IGNORECASE | re.VERBOSE,
|
|
175
176
|
)
|
|
177
|
+
# A -log10(p/q) score column: an explicit negation marker (negative / negated /
|
|
178
|
+
# neg / -) before a log/log10 p-or-q token ("negative log p value" — the
|
|
179
|
+
# mokg-v12 HOYER1 spelling, "negative log10 p value", "-log10(p)",
|
|
180
|
+
# "neg log10 q value"). The trailing token must be a *complete* p/q-value
|
|
181
|
+
# token (or a bare delimited P), so the [pq] cannot swallow the first letter
|
|
182
|
+
# of an unrelated word ("negative log protein" stays out). A plain
|
|
183
|
+
# "log10 p value" without a negation marker does NOT match: the sign
|
|
184
|
+
# convention is ambiguous there, so those columns keep riding verbatim rather
|
|
185
|
+
# than being un-logged on a guess.
|
|
186
|
+
NEGLOG10_PVALUE_PATTERN: re.Pattern[str] = re.compile(
|
|
187
|
+
rf"""
|
|
188
|
+
(?<![A-Za-z0-9])
|
|
189
|
+
(?: negative | negated | neg | - )
|
|
190
|
+
[\s_.\-()]*
|
|
191
|
+
log (?: 10 )?
|
|
192
|
+
[\s_.\-()]*
|
|
193
|
+
(?:
|
|
194
|
+
[pq] {_SEP} {_VALUE} {_NUMQUAL} # complete p/q-value token ("log p value", "log q-value", "log pvalue")
|
|
195
|
+
| p (?! [A-Za-z0-9]) # bare P standing alone ("-log10(p)", "-log10 P")
|
|
196
|
+
)
|
|
197
|
+
""",
|
|
198
|
+
re.IGNORECASE | re.VERBOSE,
|
|
199
|
+
)
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def is_neglog10_column(name: str) -> bool:
|
|
203
|
+
"""Return True when a column reports a -log10(p/q) score instead of the raw p/q value.
|
|
204
|
+
|
|
205
|
+
Args:
|
|
206
|
+
name: Raw source column name.
|
|
207
|
+
|
|
208
|
+
Returns:
|
|
209
|
+
True when the name carries an explicit negation marker before a
|
|
210
|
+
log/log10 p-or-q token, else False. Plain ``log``/``log10`` spellings
|
|
211
|
+
without a marker are deliberately excluded (sign convention
|
|
212
|
+
ambiguous), as is everything without a log token at all.
|
|
213
|
+
"""
|
|
214
|
+
return bool(NEGLOG10_PVALUE_PATTERN.search(name))
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def _unlog10(expr: pl.Expr) -> pl.Expr:
|
|
218
|
+
"""Convert a -log10 score expression back to its original scale (10**-x).
|
|
219
|
+
|
|
220
|
+
Float64 underflow floors extreme scores at ``0.0`` — indistinguishable
|
|
221
|
+
from ``p ~ 0`` in practice, and the significance band is identical either
|
|
222
|
+
way. Nulls stay null.
|
|
223
|
+
"""
|
|
224
|
+
return pl.lit(10.0) ** (-expr)
|
|
176
225
|
|
|
177
226
|
|
|
178
227
|
def pvalue_target(name: str) -> str | None:
|
|
@@ -215,10 +264,37 @@ def pvalue_target(name: str) -> str | None:
|
|
|
215
264
|
return "adjusted_p_value" if is_adjusted else "p_value"
|
|
216
265
|
|
|
217
266
|
|
|
267
|
+
def _best_candidate(candidates: list[str], target: str, chosen: str) -> str:
|
|
268
|
+
"""Pick the column a numeric p/q-value slot should read from.
|
|
269
|
+
|
|
270
|
+
A raw (non-neglog) candidate is always preferred over a -log10 alias of
|
|
271
|
+
the same statistic: the raw column holds the p/q value itself, while the
|
|
272
|
+
alias needs un-logging. Fuzzy ranking only runs when no raw candidate
|
|
273
|
+
exists (or only neglog candidates do), so a short raw name ("P", "FDR")
|
|
274
|
+
can no longer lose to a long -log10 alias it would then be un-logged over.
|
|
275
|
+
|
|
276
|
+
Args:
|
|
277
|
+
candidates: Column names bucketed onto ``target`` by ``pvalue_target``.
|
|
278
|
+
target: Canonical slot name (``p_value`` / ``adjusted_p_value``).
|
|
279
|
+
chosen: The canonical name when it is itself among the candidates
|
|
280
|
+
(the existing canonical-wins rule), else a sentinel that is not.
|
|
281
|
+
|
|
282
|
+
Returns:
|
|
283
|
+
The winning column name.
|
|
284
|
+
"""
|
|
285
|
+
from rapidfuzz import fuzz
|
|
286
|
+
|
|
287
|
+
pool: list[str] = [c for c in candidates if not is_neglog10_column(c)] or candidates
|
|
288
|
+
return chosen if chosen in pool else max(pool, key=lambda c: fuzz.ratio(c, target.replace("_", " ")))
|
|
289
|
+
|
|
290
|
+
|
|
218
291
|
def coerce_pvalue_columns(lf: pl.LazyFrame) -> pl.LazyFrame:
|
|
219
292
|
"""Rename p-value-like columns to Biolink KGX-compliant ``p_value`` / ``adjusted_p_value``.
|
|
220
293
|
|
|
221
294
|
Picks a single best fuzzy match per target when multiple candidates exist.
|
|
295
|
+
A chosen column that reports a -log10(p/q) score (see
|
|
296
|
+
:func:`is_neglog10_column`) is un-logged (``p = 10**-x``) as it is renamed,
|
|
297
|
+
so the slot receives the p-value the model types it as.
|
|
222
298
|
|
|
223
299
|
Args:
|
|
224
300
|
lf: Source LazyFrame.
|
|
@@ -226,9 +302,6 @@ def coerce_pvalue_columns(lf: pl.LazyFrame) -> pl.LazyFrame:
|
|
|
226
302
|
Returns:
|
|
227
303
|
LazyFrame with the chosen columns renamed (no-op if no candidates).
|
|
228
304
|
"""
|
|
229
|
-
# Picks a single best fuzzy match per target when multiple candidates exist.
|
|
230
|
-
from rapidfuzz import fuzz
|
|
231
|
-
|
|
232
305
|
names: list[str] = lf.collect_schema().names()
|
|
233
306
|
buckets: dict[str, list[str]] = {}
|
|
234
307
|
for n in names:
|
|
@@ -237,14 +310,23 @@ def coerce_pvalue_columns(lf: pl.LazyFrame) -> pl.LazyFrame:
|
|
|
237
310
|
buckets.setdefault(target, []).append(n)
|
|
238
311
|
|
|
239
312
|
renames: dict[str, str] = {}
|
|
313
|
+
unlog_targets: list[str] = []
|
|
240
314
|
for target, candidates in buckets.items():
|
|
241
|
-
|
|
242
|
-
#
|
|
243
|
-
chosen: str =
|
|
315
|
+
# An existing canonical column always wins; a raw candidate beats a
|
|
316
|
+
# -log10 alias before fuzzy ranking has any say.
|
|
317
|
+
chosen: str = _best_candidate(candidates, target, target)
|
|
244
318
|
if chosen != target:
|
|
245
319
|
renames[chosen] = target
|
|
320
|
+
# A -log10(p) score must be un-logged when it lands on the numeric slot.
|
|
321
|
+
if is_neglog10_column(chosen):
|
|
322
|
+
unlog_targets.append(target)
|
|
246
323
|
|
|
247
|
-
|
|
324
|
+
if not renames and not unlog_targets:
|
|
325
|
+
return lf
|
|
326
|
+
out: pl.LazyFrame = lf.rename(renames) if renames else lf
|
|
327
|
+
if unlog_targets:
|
|
328
|
+
out = out.with_columns([_unlog10(pl.col(t).cast(pl.Float64, strict=False)).alias(t) for t in unlog_targets])
|
|
329
|
+
return out
|
|
248
330
|
|
|
249
331
|
|
|
250
332
|
# --- Study-size fragments ----------------------------------------------------
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
|
+
import json
|
|
3
4
|
import math
|
|
4
5
|
import operator
|
|
5
6
|
import re
|
|
@@ -37,6 +38,7 @@ from tablassert.coerce import (
|
|
|
37
38
|
coerced_target,
|
|
38
39
|
effect_size_target,
|
|
39
40
|
effect_type_target,
|
|
41
|
+
is_neglog10_column,
|
|
40
42
|
pvalue_target,
|
|
41
43
|
sig,
|
|
42
44
|
study_size_target,
|
|
@@ -44,7 +46,7 @@ from tablassert.coerce import (
|
|
|
44
46
|
from tablassert.enums import EncodingMethods, Files, InformationResources, Repositories, Tokens
|
|
45
47
|
from tablassert.fullmap import ResolveSpec, fullmap_db_path, resolve, resolve_batch
|
|
46
48
|
from tablassert.log import cat
|
|
47
|
-
from tablassert.models import Encoding, NodeEncoding, Qualifier, RIGConfig, Section
|
|
49
|
+
from tablassert.models import EDGE_ID_PLACEHOLDER, Encoding, NodeEncoding, Qualifier, RIGConfig, Section
|
|
48
50
|
from tablassert.nlp import level_one, level_two
|
|
49
51
|
from tablassert.qc import fullmap_audit
|
|
50
52
|
from tablassert.rig import (
|
|
@@ -81,6 +83,7 @@ __all__ = [
|
|
|
81
83
|
"effect_size_target",
|
|
82
84
|
"effect_type_target",
|
|
83
85
|
"infores",
|
|
86
|
+
"is_neglog10_column",
|
|
84
87
|
"normalize_biolink_category",
|
|
85
88
|
"predicate_options",
|
|
86
89
|
"pvalue_target",
|
|
@@ -354,16 +357,12 @@ def value(lf: pl.LazyFrame, col: str, x: object) -> pl.LazyFrame:
|
|
|
354
357
|
def _retrieval_source(resource_id: str, resource_role: str, upstream: list[str] | None = None, urls: list[str] | None = None) -> pl.Expr:
|
|
355
358
|
"""Build one ``RetrievalSource`` struct expression.
|
|
356
359
|
|
|
357
|
-
Every entry declares the same
|
|
360
|
+
Every entry declares the same four fields so that :func:`retrieval_sources` can
|
|
358
361
|
``concat_list`` them into a single ``list[struct]`` column; absent list fields are
|
|
359
362
|
typed nulls, which the Rust null-stripper removes from the emitted JSON.
|
|
360
363
|
"""
|
|
361
364
|
empty: pl.Expr = pl.lit(None, dtype=pl.List(pl.String))
|
|
362
365
|
return pl.struct(
|
|
363
|
-
# `id` duplicates `resource_id`, but RetrievalSource inherits `id` from
|
|
364
|
-
# `entity` and the generated Pydantic classes require it, so omitting it
|
|
365
|
-
# fails KGX validation. Stays until biolink-model #1706/#1731 land.
|
|
366
|
-
pl.lit(resource_id).alias("id"),
|
|
367
366
|
pl.lit(resource_id).alias("resource_id"),
|
|
368
367
|
pl.lit(resource_role).alias("resource_role"),
|
|
369
368
|
(pl.concat_list([pl.lit(x) for x in upstream]) if upstream else empty).alias("upstream_resource_ids"),
|
|
@@ -558,7 +557,12 @@ def inline_supporting_study(lf: pl.LazyFrame, study_id: str, study_name: str | N
|
|
|
558
557
|
|
|
559
558
|
|
|
560
559
|
def retrieval_sources(
|
|
561
|
-
lf: pl.LazyFrame,
|
|
560
|
+
lf: pl.LazyFrame,
|
|
561
|
+
primary: str,
|
|
562
|
+
upstream: list[str],
|
|
563
|
+
urls: list[str],
|
|
564
|
+
upstream_urls: dict[str, list[str]] | None = None,
|
|
565
|
+
explicit: list[dict[str, Any]] | None = None,
|
|
562
566
|
) -> pl.LazyFrame:
|
|
563
567
|
"""Add the Biolink ``sources`` retrieval-provenance column.
|
|
564
568
|
|
|
@@ -576,18 +580,32 @@ def retrieval_sources(
|
|
|
576
580
|
its own ``source_record_urls`` and the primary entry emits none (the primary is
|
|
577
581
|
the transforming resource, not a downloadable record).
|
|
578
582
|
|
|
583
|
+
When ``explicit`` is given, exactly those entries are emitted, in order, and
|
|
584
|
+
``primary``/``upstream``/``urls``/``upstream_urls`` are ignored. Each entry
|
|
585
|
+
template carries ``resource_id``, ``resource_role``, and optional
|
|
586
|
+
``upstream_resource_ids``/``source_record_urls``; ``source_record_urls`` values
|
|
587
|
+
may contain the literal ``{edge_id}`` placeholder, which is NOT resolved here
|
|
588
|
+
(the edge id is only assigned at the final dedup stage) but in a post-dedup
|
|
589
|
+
sweep of the final edges NDJSON.
|
|
590
|
+
|
|
579
591
|
Args:
|
|
580
592
|
lf: Source LazyFrame.
|
|
581
593
|
primary: Infores CURIE of the primary knowledge source.
|
|
582
594
|
upstream: Infores CURIEs of upstream/supporting data sources.
|
|
583
595
|
urls: Source record URLs for the primary entry (ignored when ``upstream_urls`` is set).
|
|
584
596
|
upstream_urls: Optional per-upstream source record URLs keyed by infores CURIE.
|
|
597
|
+
explicit: Optional explicit entry templates emitted verbatim, in order.
|
|
585
598
|
|
|
586
599
|
Returns:
|
|
587
600
|
LazyFrame with a ``sources`` ``list[struct]`` column appended.
|
|
588
601
|
"""
|
|
589
|
-
if
|
|
590
|
-
entries: list[pl.Expr] = [
|
|
602
|
+
if explicit is not None:
|
|
603
|
+
entries: list[pl.Expr] = [
|
|
604
|
+
_retrieval_source(entry["resource_id"], entry["resource_role"], entry.get("upstream_resource_ids"), entry.get("source_record_urls"))
|
|
605
|
+
for entry in explicit
|
|
606
|
+
]
|
|
607
|
+
elif upstream_urls is not None:
|
|
608
|
+
entries = [_retrieval_source(primary, "primary_knowledge_source", upstream)]
|
|
591
609
|
entries.extend(_retrieval_source(x, "supporting_data_source", urls=upstream_urls.get(x)) for x in upstream)
|
|
592
610
|
else:
|
|
593
611
|
entries = [_retrieval_source(primary, "primary_knowledge_source", upstream, urls)]
|
|
@@ -1257,6 +1275,12 @@ class Tcode(Section):
|
|
|
1257
1275
|
if override and override.upstream_source_record_urls is not None
|
|
1258
1276
|
else None
|
|
1259
1277
|
)
|
|
1278
|
+
# An explicit `sources` template replaces the derived primary/upstream
|
|
1279
|
+
# emission entirely; the model already forbids combining it with
|
|
1280
|
+
# `upstream_resource_ids`/`upstream_source_record_urls`.
|
|
1281
|
+
explicit_sources: list[dict[str, Any]] | None = (
|
|
1282
|
+
[entry.model_dump(exclude_none=True) for entry in override.sources] if override and override.sources is not None else None
|
|
1283
|
+
)
|
|
1260
1284
|
knowledge_level = override.knowledge_level if override else self.provenance.knowledge_level
|
|
1261
1285
|
agent_type = override.agent_type if override else self.provenance.agent_type
|
|
1262
1286
|
publication_values = override.publications if override else [publication_curie(self.provenance.repo, self.provenance.publication or "")]
|
|
@@ -1282,8 +1306,9 @@ class Tcode(Section):
|
|
|
1282
1306
|
# RetrievalSource); current translator-ingests emits no flat
|
|
1283
1307
|
# `primary_knowledge_source` scalar, so neither do we. A per-upstream URL
|
|
1284
1308
|
# mapping (override.upstream_source_record_urls) re-homes the record URLs
|
|
1285
|
-
# from the primary entry onto the matching supporting entries
|
|
1286
|
-
|
|
1309
|
+
# from the primary entry onto the matching supporting entries; an explicit
|
|
1310
|
+
# `sources` template (override.sources) replaces the whole derivation.
|
|
1311
|
+
(retrieval_sources, (primary_knowledge_source, upstream_ids, [str(u) for u in self.source.url], upstream_urls, explicit_sources))
|
|
1287
1312
|
if primary_knowledge_source
|
|
1288
1313
|
else None,
|
|
1289
1314
|
(publications, (publication_values,)) if publication_values else None,
|
|
@@ -1524,6 +1549,44 @@ def dedup_stream(p_in: Path, is_edges: bool) -> None:
|
|
|
1524
1549
|
p_in.unlink()
|
|
1525
1550
|
|
|
1526
1551
|
|
|
1552
|
+
def _resolve_edge_id_placeholders(edges_path: Path) -> None:
|
|
1553
|
+
"""Resolve ``{edge_id}`` placeholders in a final edges NDJSON file.
|
|
1554
|
+
|
|
1555
|
+
The edge ``id`` is a deterministic content hash assigned by the Rust deduper
|
|
1556
|
+
after subgraphs are written, so explicit ``override.sources`` record URLs
|
|
1557
|
+
cannot embed it during the polars build: the literal placeholder is what
|
|
1558
|
+
gets hashed, and this post-dedup sweep substitutes each record's own id into
|
|
1559
|
+
every string inside every ``sources[].source_record_urls`` list. The pass is
|
|
1560
|
+
skipped entirely when no line contains the marker (cheap substring precheck,
|
|
1561
|
+
no full parse), leaving the file byte-identical.
|
|
1562
|
+
|
|
1563
|
+
Args:
|
|
1564
|
+
edges_path: Path to the deduplicated ``*.edges.ndjson`` file.
|
|
1565
|
+
|
|
1566
|
+
Returns:
|
|
1567
|
+
``None``; rewrites ``edges_path`` in place via a temp file when any
|
|
1568
|
+
placeholder was resolved.
|
|
1569
|
+
"""
|
|
1570
|
+
tmp_path: Path = edges_path.with_name(edges_path.name + ".placeholder.tmp")
|
|
1571
|
+
resolved: bool = False
|
|
1572
|
+
with edges_path.open("r", encoding="utf-8") as src, tmp_path.open("w", encoding="utf-8") as dst:
|
|
1573
|
+
for line in src:
|
|
1574
|
+
if EDGE_ID_PLACEHOLDER not in line:
|
|
1575
|
+
dst.write(line)
|
|
1576
|
+
continue
|
|
1577
|
+
record: dict[str, Any] = json.loads(line)
|
|
1578
|
+
for source in record.get("sources") or []:
|
|
1579
|
+
urls: list[str] | None = source.get("source_record_urls")
|
|
1580
|
+
if urls:
|
|
1581
|
+
source["source_record_urls"] = [url.replace(EDGE_ID_PLACEHOLDER, record["id"]) for url in urls]
|
|
1582
|
+
dst.write(json.dumps(record, ensure_ascii=False) + "\n")
|
|
1583
|
+
resolved = True
|
|
1584
|
+
if resolved:
|
|
1585
|
+
tmp_path.replace(edges_path)
|
|
1586
|
+
else:
|
|
1587
|
+
tmp_path.unlink()
|
|
1588
|
+
|
|
1589
|
+
|
|
1527
1590
|
def fold_unknown_to_supporting_text(lf: pl.LazyFrame) -> pl.LazyFrame:
|
|
1528
1591
|
"""Fold any non-Biolink edge column into ``supporting_text`` as ``col: value`` strings.
|
|
1529
1592
|
|
|
@@ -1635,6 +1698,12 @@ def _write_ndjson(
|
|
|
1635
1698
|
``compile_graph``; each commented phase boundary below is a hook point for
|
|
1636
1699
|
the US-009 ``on_phase`` progress callback.
|
|
1637
1700
|
|
|
1701
|
+
Dedup assigns each edge ``id`` as the deterministic content hash of the
|
|
1702
|
+
PRE-resolution record -- the literal ``{edge_id}`` placeholder in explicit
|
|
1703
|
+
``override.sources`` record URLs is what gets hashed -- and the placeholder
|
|
1704
|
+
sweep that follows substitutes the assigned id into the final edges file
|
|
1705
|
+
only, keeping ids deterministic.
|
|
1706
|
+
|
|
1638
1707
|
Args:
|
|
1639
1708
|
subnodes: Per-section node LazyFrames from ``_collect_subframes``.
|
|
1640
1709
|
subedges: Per-section edge LazyFrames from ``_collect_subframes``.
|
|
@@ -1669,6 +1738,9 @@ def _write_ndjson(
|
|
|
1669
1738
|
if on_phase is not None:
|
|
1670
1739
|
on_phase("dedup")
|
|
1671
1740
|
dedup_stream(edges_tmp, is_edges=True)
|
|
1741
|
+
# The deduper hashes the record WITH the literal `{edge_id}` placeholder still
|
|
1742
|
+
# in place, so edge ids stay deterministic regardless of this resolution pass.
|
|
1743
|
+
_resolve_edge_id_placeholders(edges_tmp.with_suffix(""))
|
|
1672
1744
|
dedup_stream(nodes_tmp, is_edges=False)
|
|
1673
1745
|
|
|
1674
1746
|
# Phase: rig. Summaries come from the FINAL deduplicated KGX files, and the
|
|
@@ -466,16 +466,99 @@ def validate_infores_curie(value: str, code: TablassertErrorCodes) -> str:
|
|
|
466
466
|
return value
|
|
467
467
|
|
|
468
468
|
|
|
469
|
+
EDGE_ID_PLACEHOLDER: str = "{edge_id}"
|
|
470
|
+
"""Placeholder for the final edge id inside ``override.sources`` record URLs.
|
|
471
|
+
|
|
472
|
+
The edge ``id`` is a deterministic content hash computed during the final dedup
|
|
473
|
+
stage -- after subgraphs are written -- so a per-edge URL cannot embed it during
|
|
474
|
+
the table build. The literal placeholder is emitted as-is and resolved against
|
|
475
|
+
the deduplicated ``*.edges.ndjson`` in a post-dedup sweep.
|
|
476
|
+
"""
|
|
477
|
+
|
|
478
|
+
RESOURCE_ROLES: tuple[str, ...] = ("primary_knowledge_source", "aggregator_knowledge_source", "supporting_data_source")
|
|
479
|
+
"""Valid Biolink ``ResourceRoleEnum`` values for retrieval ``sources`` entries.
|
|
480
|
+
|
|
481
|
+
Kept as literals instead of importing ``ResourceRoleEnum`` from the generated
|
|
482
|
+
``biolink_model`` Pydantic classes so config validation stays cheap; any other
|
|
483
|
+
value fails KGX validation downstream.
|
|
484
|
+
"""
|
|
485
|
+
|
|
486
|
+
|
|
487
|
+
class SourceOverride(TablaBase):
|
|
488
|
+
"""One explicit retrieval-``sources`` entry template for manual provenance.
|
|
489
|
+
|
|
490
|
+
Each entry becomes one Biolink ``RetrievalSource`` struct on the edge's
|
|
491
|
+
``sources`` list, replacing the derived primary/upstream emission entirely.
|
|
492
|
+
"""
|
|
493
|
+
|
|
494
|
+
resource_id: str = Field(description="Infores CURIE of this retrieval source entry.", examples=["infores:my-source"])
|
|
495
|
+
resource_role: str = Field(
|
|
496
|
+
description="Biolink ResourceRoleEnum value for this entry.",
|
|
497
|
+
examples=["primary_knowledge_source", "aggregator_knowledge_source", "supporting_data_source"],
|
|
498
|
+
)
|
|
499
|
+
upstream_resource_ids: list[str] | None = Field(
|
|
500
|
+
None, description="Upstream infores CURIEs carried by this entry.", examples=[["infores:my-upstream"]]
|
|
501
|
+
)
|
|
502
|
+
source_record_urls: list[str] | None = Field(
|
|
503
|
+
None,
|
|
504
|
+
description=f"Source record URLs carried by this entry; `{EDGE_ID_PLACEHOLDER}` is replaced with the final edge id after dedup.",
|
|
505
|
+
examples=[["https://example.org/edge?id={edge_id}"]],
|
|
506
|
+
)
|
|
507
|
+
|
|
508
|
+
@field_validator("resource_id", mode="after")
|
|
509
|
+
@classmethod
|
|
510
|
+
def infores_resource_id(cls, value: str) -> str:
|
|
511
|
+
return validate_infores_curie(value, "override-bad-sources")
|
|
512
|
+
|
|
513
|
+
@field_validator("resource_role", mode="after")
|
|
514
|
+
@classmethod
|
|
515
|
+
def biolink_resource_role(cls, value: str) -> str:
|
|
516
|
+
if value not in RESOURCE_ROLES:
|
|
517
|
+
raise TablassertValidationError(f"`resource_role` must be one of {list(RESOURCE_ROLES)}, got {value!r}.", code="override-bad-sources")
|
|
518
|
+
return value
|
|
519
|
+
|
|
520
|
+
@field_validator("upstream_resource_ids", mode="after")
|
|
521
|
+
@classmethod
|
|
522
|
+
def infores_upstream_resource_ids(cls, values: list[str] | None) -> list[str] | None:
|
|
523
|
+
if values is None:
|
|
524
|
+
return None
|
|
525
|
+
for value in values:
|
|
526
|
+
validate_infores_curie(value, "override-bad-sources")
|
|
527
|
+
return values
|
|
528
|
+
|
|
529
|
+
@field_validator("source_record_urls", mode="after")
|
|
530
|
+
@classmethod
|
|
531
|
+
def url_or_edge_id_template(cls, values: list[str] | None) -> list[str] | None:
|
|
532
|
+
# Typed as plain str (not HttpUrl) so the `{edge_id}` placeholder survives
|
|
533
|
+
# validation; after removing placeholder occurrences the rest must still be
|
|
534
|
+
# an absolute http(s) URL.
|
|
535
|
+
if values is None:
|
|
536
|
+
return None
|
|
537
|
+
for value in values:
|
|
538
|
+
stripped: str = value.replace(EDGE_ID_PLACEHOLDER, "")
|
|
539
|
+
if not stripped.startswith(("https://", "http://")):
|
|
540
|
+
raise TablassertValidationError(
|
|
541
|
+
f"`source_record_urls` entries must be http(s) URLs (optionally containing `{EDGE_ID_PLACEHOLDER}`), got {value!r}.",
|
|
542
|
+
code="override-bad-sources",
|
|
543
|
+
)
|
|
544
|
+
return values
|
|
545
|
+
|
|
546
|
+
|
|
469
547
|
class ManualProvenance(TablaBase):
|
|
470
548
|
"""Manually-specified provenance for non-PMID/PMC source graphs.
|
|
471
549
|
|
|
472
550
|
When present under :class:`Provenance`, these values replace the legacy
|
|
473
551
|
repo/publication-derived provenance while keeping the same KL/AT defaults.
|
|
474
552
|
The primary ``sources`` entry (``resource_role: primary_knowledge_source``)
|
|
475
|
-
|
|
476
|
-
infores CURIEs
|
|
553
|
+
derives from the graph-level ``rig.source_info.infores_id`` unless an
|
|
554
|
+
explicit ``sources`` template is given; manual infores CURIEs otherwise
|
|
555
|
+
belong in ``upstream_resource_ids``.
|
|
477
556
|
"""
|
|
478
557
|
|
|
558
|
+
sources: list[SourceOverride] | None = Field(
|
|
559
|
+
None,
|
|
560
|
+
description="Explicit retrieval-`sources` entry templates replacing the derived primary/upstream emission entirely; mutually exclusive with `upstream_resource_ids` and `upstream_source_record_urls`, which it subsumes.",
|
|
561
|
+
)
|
|
479
562
|
upstream_resource_ids: list[str] = Field(
|
|
480
563
|
default_factory=list,
|
|
481
564
|
description="Manual upstream source infores CURIEs emitted instead of the repo-derived source map; the sanctioned place for manual infores.",
|
|
@@ -515,6 +598,26 @@ class ManualProvenance(TablaBase):
|
|
|
515
598
|
)
|
|
516
599
|
return values
|
|
517
600
|
|
|
601
|
+
@model_validator(mode="after")
|
|
602
|
+
def sources_template_is_coherent(self: Self) -> Self:
|
|
603
|
+
if self.sources is None:
|
|
604
|
+
return self
|
|
605
|
+
if self.upstream_resource_ids or self.upstream_source_record_urls is not None:
|
|
606
|
+
raise TablassertValidationError(
|
|
607
|
+
"`sources` is mutually exclusive with `upstream_resource_ids` and `upstream_source_record_urls`; the explicit template subsumes both.",
|
|
608
|
+
code="override-bad-sources",
|
|
609
|
+
)
|
|
610
|
+
if not self.sources:
|
|
611
|
+
raise TablassertValidationError("`sources` must contain at least one entry when set.", code="override-bad-sources")
|
|
612
|
+
resource_ids: list[str] = [entry.resource_id for entry in self.sources]
|
|
613
|
+
if len(set(resource_ids)) != len(resource_ids):
|
|
614
|
+
raise TablassertValidationError("`sources` entries must have unique `resource_id` values.", code="override-bad-sources")
|
|
615
|
+
if not any(entry.resource_role in ("primary_knowledge_source", "aggregator_knowledge_source") for entry in self.sources):
|
|
616
|
+
raise TablassertValidationError(
|
|
617
|
+
"`sources` must include at least one `primary_knowledge_source` or `aggregator_knowledge_source` entry.", code="override-bad-sources"
|
|
618
|
+
)
|
|
619
|
+
return self
|
|
620
|
+
|
|
518
621
|
@model_validator(mode="after")
|
|
519
622
|
def upstream_urls_match_resource_ids(self: Self) -> Self:
|
|
520
623
|
if self.upstream_source_record_urls is None:
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|