tablassert 13.0.0__tar.gz → 14.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. {tablassert-13.0.0 → tablassert-14.0.0}/PKG-INFO +1 -1
  2. {tablassert-13.0.0 → tablassert-14.0.0}/pyproject.toml +1 -1
  3. {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/biolink.py +35 -1
  4. {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/coerce.py +95 -13
  5. {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/errors.py +1 -0
  6. {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/lib.py +83 -11
  7. {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/models.py +105 -2
  8. {tablassert-13.0.0 → tablassert-14.0.0}/LICENSE +0 -0
  9. {tablassert-13.0.0 → tablassert-14.0.0}/README.md +0 -0
  10. {tablassert-13.0.0 → tablassert-14.0.0}/rust/Cargo.lock +0 -0
  11. {tablassert-13.0.0 → tablassert-14.0.0}/rust/Cargo.toml +0 -0
  12. {tablassert-13.0.0 → tablassert-14.0.0}/rust/examples/count_tables.rs +0 -0
  13. {tablassert-13.0.0 → tablassert-14.0.0}/rust/src/fullmap.rs +0 -0
  14. {tablassert-13.0.0 → tablassert-14.0.0}/rust/src/json.rs +0 -0
  15. {tablassert-13.0.0 → tablassert-14.0.0}/rust/src/lib.rs +0 -0
  16. {tablassert-13.0.0 → tablassert-14.0.0}/rust/src/ndjson.rs +0 -0
  17. {tablassert-13.0.0 → tablassert-14.0.0}/rust/src/uuid.rs +0 -0
  18. {tablassert-13.0.0 → tablassert-14.0.0}/rust/tests/build_golden.rs +0 -0
  19. {tablassert-13.0.0 → tablassert-14.0.0}/rust/tests/common/mod.rs +0 -0
  20. {tablassert-13.0.0 → tablassert-14.0.0}/rust/tests/extract_prebuilt.rs +0 -0
  21. {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/__init__.py +0 -0
  22. {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/_lazy.py +0 -0
  23. {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/agent.py +0 -0
  24. {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/cli.py +0 -0
  25. {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/enums.py +0 -0
  26. {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/extras.py +0 -0
  27. {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/fullmap.py +0 -0
  28. {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/graph_target.py +0 -0
  29. {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/ingests.py +0 -0
  30. {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/log.py +0 -0
  31. {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/nlp.py +0 -0
  32. {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/progress.py +0 -0
  33. {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/qc.py +0 -0
  34. {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/rig.py +0 -0
  35. {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/rs.pyi +0 -0
  36. {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/study.py +0 -0
  37. {tablassert-13.0.0 → tablassert-14.0.0}/src/tablassert/utils.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: tablassert
3
- Version: 13.0.0
3
+ Version: 14.0.0
4
4
  Classifier: License :: OSI Approved :: Apache Software License
5
5
  Classifier: Development Status :: 5 - Production/Stable
6
6
  Classifier: Intended Audience :: Science/Research
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "tablassert"
3
- version = "13.0.0"
3
+ version = "14.0.0"
4
4
  description = "Extract knowledge assertions from tabular data into NCATS Translator-compliant KGX NDJSON — declaratively, with entity resolution and quality control built in."
5
5
  authors = [
6
6
  { name = "Skye Lane Goetz", email = "sgoetz@isbscience.org" }
@@ -643,6 +643,40 @@ def is_multivalued(cls: type[Any], field: str) -> bool:
643
643
  return any(get_origin(arg) is list for arg in get_args(annotation))
644
644
 
645
645
 
646
+ @cache
647
+ def _retrieval_source_id_required() -> bool:
648
+ """Whether the installed model still requires the inherited source ``id`` field."""
649
+ retrieval_source: Any = getattr(_bm, "RetrievalSource", None)
650
+ id_field: Any = getattr(retrieval_source, "model_fields", {}).get("id")
651
+ return id_field is not None and id_field.is_required()
652
+
653
+
654
+ def _validation_record(record: dict[str, Any], *, edge: bool) -> dict[str, Any]:
655
+ """Add only in-memory compatibility aliases needed by the installed Biolink model.
656
+
657
+ ``RetrievalSource`` in the currently pinned model still requires the inherited
658
+ ``Entity.id`` even though ``resource_id`` is the canonical provenance identifier
659
+ emitted by Tablassert. The alias is used solely for Pydantic validation; it never
660
+ changes the decoded KGX record or the files written by the pipeline.
661
+ """
662
+ if not edge:
663
+ return record
664
+ if not _retrieval_source_id_required():
665
+ return record
666
+ sources: Any = record.get("sources")
667
+ if not isinstance(sources, list):
668
+ return record
669
+ normalized: list[Any] = []
670
+ changed: bool = False
671
+ for source in sources:
672
+ if isinstance(source, dict) and "id" not in source and "resource_id" in source:
673
+ normalized.append({**source, "id": source["resource_id"]})
674
+ changed = True
675
+ else:
676
+ normalized.append(source)
677
+ return {**record, "sources": normalized} if changed else record
678
+
679
+
646
680
  def validate_record(record: dict[str, Any], *, edge: bool) -> list[str]:
647
681
  """Validate one KGX record against the Biolink class named by its ``category``.
648
682
 
@@ -659,7 +693,7 @@ def validate_record(record: dict[str, Any], *, edge: bool) -> list[str]:
659
693
  category: str = categories[0] if isinstance(categories, list) and categories else str(categories or "")
660
694
  cls: type[Any] = association_class(category) if edge else node_class(category)
661
695
  try:
662
- cls(**record)
696
+ cls(**_validation_record(record, edge=edge))
663
697
  except ValidationError as error:
664
698
  return [f"{'.'.join(str(part) for part in item['loc']) or '?'}: {item['type']}" for item in error.errors()]
665
699
  return []
@@ -48,8 +48,6 @@ def sig(lf: pl.LazyFrame, col: str = "p_value", out: str = "statistical_signific
48
48
  null p-value → null qualifier. The qualifier may only be set when
49
49
  ``p_value``/``adjusted_p_value`` is populated (Biolink class rule).
50
50
  """
51
- from rapidfuzz import fuzz
52
-
53
51
  names: list[str] = lf.collect_schema().names()
54
52
  # Same rigorous classification as ``coerce_pvalue_columns``: a column is a significance
55
53
  # source only when ``pvalue_target`` accepts it — NOT by naive substring — so non-p-value
@@ -66,11 +64,14 @@ def sig(lf: pl.LazyFrame, col: str = "p_value", out: str = "statistical_signific
66
64
  # bucket is present. Raw p-value is the canonical significance source; adjusted is the fallback.
67
65
  preferred: str = col if col in buckets else next(iter(buckets))
68
66
  candidates: list[str] = buckets[preferred]
69
- reference: str = preferred.replace("_", " ")
70
- # An existing canonical column always wins; fuzzy ranking only picks among aliases
71
- # (same rule as ``coerce_pvalue_columns``).
72
- chosen: str = preferred if preferred in candidates else max(candidates, key=lambda c: fuzz.ratio(c, reference))
67
+ # Same selection rule as ``coerce_pvalue_columns``: canonical wins, then a
68
+ # raw candidate beats a -log10 alias, then fuzzy ranking among what is left.
69
+ chosen: str = _best_candidate(candidates, preferred, preferred)
73
70
  expr: pl.Expr = pl.col(chosen).cast(pl.Float64, strict=False)
71
+ # A -log10(p) score column must be un-logged before banding, or the bands
72
+ # invert (a score of 8 means p = 1e-8, not p = 8.0 -> not_significant).
73
+ if is_neglog10_column(chosen):
74
+ expr = _unlog10(expr)
74
75
  band: pl.Expr = (
75
76
  pl.when(expr.is_null())
76
77
  .then(pl.lit(None, dtype=pl.String))
@@ -173,6 +174,54 @@ SIGNIFICANCE_FLAG_PATTERN: re.Pattern[str] = re.compile(
173
174
  """,
174
175
  re.IGNORECASE | re.VERBOSE,
175
176
  )
177
+ # A -log10(p/q) score column: an explicit negation marker (negative / negated /
178
+ # neg / -) before a log/log10 p-or-q token ("negative log p value" — the
179
+ # mokg-v12 HOYER1 spelling, "negative log10 p value", "-log10(p)",
180
+ # "neg log10 q value"). The trailing token must be a *complete* p/q-value
181
+ # token (or a bare delimited P), so the [pq] cannot swallow the first letter
182
+ # of an unrelated word ("negative log protein" stays out). A plain
183
+ # "log10 p value" without a negation marker does NOT match: the sign
184
+ # convention is ambiguous there, so those columns keep riding verbatim rather
185
+ # than being un-logged on a guess.
186
+ NEGLOG10_PVALUE_PATTERN: re.Pattern[str] = re.compile(
187
+ rf"""
188
+ (?<![A-Za-z0-9])
189
+ (?: negative | negated | neg | - )
190
+ [\s_.\-()]*
191
+ log (?: 10 )?
192
+ [\s_.\-()]*
193
+ (?:
194
+ [pq] {_SEP} {_VALUE} {_NUMQUAL} # complete p/q-value token ("log p value", "log q-value", "log pvalue")
195
+ | p (?! [A-Za-z0-9]) # bare P standing alone ("-log10(p)", "-log10 P")
196
+ )
197
+ """,
198
+ re.IGNORECASE | re.VERBOSE,
199
+ )
200
+
201
+
202
+ def is_neglog10_column(name: str) -> bool:
203
+ """Return True when a column reports a -log10(p/q) score instead of the raw p/q value.
204
+
205
+ Args:
206
+ name: Raw source column name.
207
+
208
+ Returns:
209
+ True when the name carries an explicit negation marker before a
210
+ log/log10 p-or-q token, else False. Plain ``log``/``log10`` spellings
211
+ without a marker are deliberately excluded (sign convention
212
+ ambiguous), as is everything without a log token at all.
213
+ """
214
+ return bool(NEGLOG10_PVALUE_PATTERN.search(name))
215
+
216
+
217
+ def _unlog10(expr: pl.Expr) -> pl.Expr:
218
+ """Convert a -log10 score expression back to its original scale (10**-x).
219
+
220
+ Float64 underflow floors extreme scores at ``0.0`` — indistinguishable
221
+ from ``p ~ 0`` in practice, and the significance band is identical either
222
+ way. Nulls stay null.
223
+ """
224
+ return pl.lit(10.0) ** (-expr)
176
225
 
177
226
 
178
227
  def pvalue_target(name: str) -> str | None:
@@ -215,10 +264,37 @@ def pvalue_target(name: str) -> str | None:
215
264
  return "adjusted_p_value" if is_adjusted else "p_value"
216
265
 
217
266
 
267
+ def _best_candidate(candidates: list[str], target: str, chosen: str) -> str:
268
+ """Pick the column a numeric p/q-value slot should read from.
269
+
270
+ A raw (non-neglog) candidate is always preferred over a -log10 alias of
271
+ the same statistic: the raw column holds the p/q value itself, while the
272
+ alias needs un-logging. Fuzzy ranking only runs when no raw candidate
273
+ exists (or only neglog candidates do), so a short raw name ("P", "FDR")
274
+ can no longer lose to a long -log10 alias it would then be un-logged over.
275
+
276
+ Args:
277
+ candidates: Column names bucketed onto ``target`` by ``pvalue_target``.
278
+ target: Canonical slot name (``p_value`` / ``adjusted_p_value``).
279
+ chosen: The canonical name when it is itself among the candidates
280
+ (the existing canonical-wins rule), else a sentinel that is not.
281
+
282
+ Returns:
283
+ The winning column name.
284
+ """
285
+ from rapidfuzz import fuzz
286
+
287
+ pool: list[str] = [c for c in candidates if not is_neglog10_column(c)] or candidates
288
+ return chosen if chosen in pool else max(pool, key=lambda c: fuzz.ratio(c, target.replace("_", " ")))
289
+
290
+
218
291
  def coerce_pvalue_columns(lf: pl.LazyFrame) -> pl.LazyFrame:
219
292
  """Rename p-value-like columns to Biolink KGX-compliant ``p_value`` / ``adjusted_p_value``.
220
293
 
221
294
  Picks a single best fuzzy match per target when multiple candidates exist.
295
+ A chosen column that reports a -log10(p/q) score (see
296
+ :func:`is_neglog10_column`) is un-logged (``p = 10**-x``) as it is renamed,
297
+ so the slot receives the p-value the model types it as.
222
298
 
223
299
  Args:
224
300
  lf: Source LazyFrame.
@@ -226,9 +302,6 @@ def coerce_pvalue_columns(lf: pl.LazyFrame) -> pl.LazyFrame:
226
302
  Returns:
227
303
  LazyFrame with the chosen columns renamed (no-op if no candidates).
228
304
  """
229
- # Picks a single best fuzzy match per target when multiple candidates exist.
230
- from rapidfuzz import fuzz
231
-
232
305
  names: list[str] = lf.collect_schema().names()
233
306
  buckets: dict[str, list[str]] = {}
234
307
  for n in names:
@@ -237,14 +310,23 @@ def coerce_pvalue_columns(lf: pl.LazyFrame) -> pl.LazyFrame:
237
310
  buckets.setdefault(target, []).append(n)
238
311
 
239
312
  renames: dict[str, str] = {}
313
+ unlog_targets: list[str] = []
240
314
  for target, candidates in buckets.items():
241
- reference: str = target.replace("_", " ")
242
- # An existing canonical column always wins; fuzzy ranking only picks among aliases.
243
- chosen: str = target if target in candidates else max(candidates, key=lambda c: fuzz.ratio(c, reference))
315
+ # An existing canonical column always wins; a raw candidate beats a
316
+ # -log10 alias before fuzzy ranking has any say.
317
+ chosen: str = _best_candidate(candidates, target, target)
244
318
  if chosen != target:
245
319
  renames[chosen] = target
320
+ # A -log10(p) score must be un-logged when it lands on the numeric slot.
321
+ if is_neglog10_column(chosen):
322
+ unlog_targets.append(target)
246
323
 
247
- return lf.rename(renames) if renames else lf
324
+ if not renames and not unlog_targets:
325
+ return lf
326
+ out: pl.LazyFrame = lf.rename(renames) if renames else lf
327
+ if unlog_targets:
328
+ out = out.with_columns([_unlog10(pl.col(t).cast(pl.Float64, strict=False)).alias(t) for t in unlog_targets])
329
+ return out
248
330
 
249
331
 
250
332
  # --- Study-size fragments ----------------------------------------------------
@@ -22,6 +22,7 @@ TablassertErrorCodes = Literal[
22
22
  "encoding-bad-remove-entry",
23
23
  "graph-bad-infores",
24
24
  "override-bad-publication",
25
+ "override-bad-sources",
25
26
  "override-bad-upstream-infores",
26
27
  "override-bad-upstream-urls",
27
28
  "provenance-bad-pmc-id",
@@ -1,5 +1,6 @@
1
1
  from __future__ import annotations
2
2
 
3
+ import json
3
4
  import math
4
5
  import operator
5
6
  import re
@@ -37,6 +38,7 @@ from tablassert.coerce import (
37
38
  coerced_target,
38
39
  effect_size_target,
39
40
  effect_type_target,
41
+ is_neglog10_column,
40
42
  pvalue_target,
41
43
  sig,
42
44
  study_size_target,
@@ -44,7 +46,7 @@ from tablassert.coerce import (
44
46
  from tablassert.enums import EncodingMethods, Files, InformationResources, Repositories, Tokens
45
47
  from tablassert.fullmap import ResolveSpec, fullmap_db_path, resolve, resolve_batch
46
48
  from tablassert.log import cat
47
- from tablassert.models import Encoding, NodeEncoding, Qualifier, RIGConfig, Section
49
+ from tablassert.models import EDGE_ID_PLACEHOLDER, Encoding, NodeEncoding, Qualifier, RIGConfig, Section
48
50
  from tablassert.nlp import level_one, level_two
49
51
  from tablassert.qc import fullmap_audit
50
52
  from tablassert.rig import (
@@ -81,6 +83,7 @@ __all__ = [
81
83
  "effect_size_target",
82
84
  "effect_type_target",
83
85
  "infores",
86
+ "is_neglog10_column",
84
87
  "normalize_biolink_category",
85
88
  "predicate_options",
86
89
  "pvalue_target",
@@ -354,16 +357,12 @@ def value(lf: pl.LazyFrame, col: str, x: object) -> pl.LazyFrame:
354
357
  def _retrieval_source(resource_id: str, resource_role: str, upstream: list[str] | None = None, urls: list[str] | None = None) -> pl.Expr:
355
358
  """Build one ``RetrievalSource`` struct expression.
356
359
 
357
- Every entry declares the same five fields so that :func:`retrieval_sources` can
360
+ Every entry declares the same four fields so that :func:`retrieval_sources` can
358
361
  ``concat_list`` them into a single ``list[struct]`` column; absent list fields are
359
362
  typed nulls, which the Rust null-stripper removes from the emitted JSON.
360
363
  """
361
364
  empty: pl.Expr = pl.lit(None, dtype=pl.List(pl.String))
362
365
  return pl.struct(
363
- # `id` duplicates `resource_id`, but RetrievalSource inherits `id` from
364
- # `entity` and the generated Pydantic classes require it, so omitting it
365
- # fails KGX validation. Stays until biolink-model #1706/#1731 land.
366
- pl.lit(resource_id).alias("id"),
367
366
  pl.lit(resource_id).alias("resource_id"),
368
367
  pl.lit(resource_role).alias("resource_role"),
369
368
  (pl.concat_list([pl.lit(x) for x in upstream]) if upstream else empty).alias("upstream_resource_ids"),
@@ -558,7 +557,12 @@ def inline_supporting_study(lf: pl.LazyFrame, study_id: str, study_name: str | N
558
557
 
559
558
 
560
559
  def retrieval_sources(
561
- lf: pl.LazyFrame, primary: str, upstream: list[str], urls: list[str], upstream_urls: dict[str, list[str]] | None = None
560
+ lf: pl.LazyFrame,
561
+ primary: str,
562
+ upstream: list[str],
563
+ urls: list[str],
564
+ upstream_urls: dict[str, list[str]] | None = None,
565
+ explicit: list[dict[str, Any]] | None = None,
562
566
  ) -> pl.LazyFrame:
563
567
  """Add the Biolink ``sources`` retrieval-provenance column.
564
568
 
@@ -576,18 +580,32 @@ def retrieval_sources(
576
580
  its own ``source_record_urls`` and the primary entry emits none (the primary is
577
581
  the transforming resource, not a downloadable record).
578
582
 
583
+ When ``explicit`` is given, exactly those entries are emitted, in order, and
584
+ ``primary``/``upstream``/``urls``/``upstream_urls`` are ignored. Each entry
585
+ template carries ``resource_id``, ``resource_role``, and optional
586
+ ``upstream_resource_ids``/``source_record_urls``; ``source_record_urls`` values
587
+ may contain the literal ``{edge_id}`` placeholder, which is NOT resolved here
588
+ (the edge id is only assigned at the final dedup stage) but in a post-dedup
589
+ sweep of the final edges NDJSON.
590
+
579
591
  Args:
580
592
  lf: Source LazyFrame.
581
593
  primary: Infores CURIE of the primary knowledge source.
582
594
  upstream: Infores CURIEs of upstream/supporting data sources.
583
595
  urls: Source record URLs for the primary entry (ignored when ``upstream_urls`` is set).
584
596
  upstream_urls: Optional per-upstream source record URLs keyed by infores CURIE.
597
+ explicit: Optional explicit entry templates emitted verbatim, in order.
585
598
 
586
599
  Returns:
587
600
  LazyFrame with a ``sources`` ``list[struct]`` column appended.
588
601
  """
589
- if upstream_urls is not None:
590
- entries: list[pl.Expr] = [_retrieval_source(primary, "primary_knowledge_source", upstream)]
602
+ if explicit is not None:
603
+ entries: list[pl.Expr] = [
604
+ _retrieval_source(entry["resource_id"], entry["resource_role"], entry.get("upstream_resource_ids"), entry.get("source_record_urls"))
605
+ for entry in explicit
606
+ ]
607
+ elif upstream_urls is not None:
608
+ entries = [_retrieval_source(primary, "primary_knowledge_source", upstream)]
591
609
  entries.extend(_retrieval_source(x, "supporting_data_source", urls=upstream_urls.get(x)) for x in upstream)
592
610
  else:
593
611
  entries = [_retrieval_source(primary, "primary_knowledge_source", upstream, urls)]
@@ -1257,6 +1275,12 @@ class Tcode(Section):
1257
1275
  if override and override.upstream_source_record_urls is not None
1258
1276
  else None
1259
1277
  )
1278
+ # An explicit `sources` template replaces the derived primary/upstream
1279
+ # emission entirely; the model already forbids combining it with
1280
+ # `upstream_resource_ids`/`upstream_source_record_urls`.
1281
+ explicit_sources: list[dict[str, Any]] | None = (
1282
+ [entry.model_dump(exclude_none=True) for entry in override.sources] if override and override.sources is not None else None
1283
+ )
1260
1284
  knowledge_level = override.knowledge_level if override else self.provenance.knowledge_level
1261
1285
  agent_type = override.agent_type if override else self.provenance.agent_type
1262
1286
  publication_values = override.publications if override else [publication_curie(self.provenance.repo, self.provenance.publication or "")]
@@ -1282,8 +1306,9 @@ class Tcode(Section):
1282
1306
  # RetrievalSource); current translator-ingests emits no flat
1283
1307
  # `primary_knowledge_source` scalar, so neither do we. A per-upstream URL
1284
1308
  # mapping (override.upstream_source_record_urls) re-homes the record URLs
1285
- # from the primary entry onto the matching supporting entries.
1286
- (retrieval_sources, (primary_knowledge_source, upstream_ids, [str(u) for u in self.source.url], upstream_urls))
1309
+ # from the primary entry onto the matching supporting entries; an explicit
1310
+ # `sources` template (override.sources) replaces the whole derivation.
1311
+ (retrieval_sources, (primary_knowledge_source, upstream_ids, [str(u) for u in self.source.url], upstream_urls, explicit_sources))
1287
1312
  if primary_knowledge_source
1288
1313
  else None,
1289
1314
  (publications, (publication_values,)) if publication_values else None,
@@ -1524,6 +1549,44 @@ def dedup_stream(p_in: Path, is_edges: bool) -> None:
1524
1549
  p_in.unlink()
1525
1550
 
1526
1551
 
1552
+ def _resolve_edge_id_placeholders(edges_path: Path) -> None:
1553
+ """Resolve ``{edge_id}`` placeholders in a final edges NDJSON file.
1554
+
1555
+ The edge ``id`` is a deterministic content hash assigned by the Rust deduper
1556
+ after subgraphs are written, so explicit ``override.sources`` record URLs
1557
+ cannot embed it during the polars build: the literal placeholder is what
1558
+ gets hashed, and this post-dedup sweep substitutes each record's own id into
1559
+ every string inside every ``sources[].source_record_urls`` list. The pass is
1560
+ skipped entirely when no line contains the marker (cheap substring precheck,
1561
+ no full parse), leaving the file byte-identical.
1562
+
1563
+ Args:
1564
+ edges_path: Path to the deduplicated ``*.edges.ndjson`` file.
1565
+
1566
+ Returns:
1567
+ ``None``; rewrites ``edges_path`` in place via a temp file when any
1568
+ placeholder was resolved.
1569
+ """
1570
+ tmp_path: Path = edges_path.with_name(edges_path.name + ".placeholder.tmp")
1571
+ resolved: bool = False
1572
+ with edges_path.open("r", encoding="utf-8") as src, tmp_path.open("w", encoding="utf-8") as dst:
1573
+ for line in src:
1574
+ if EDGE_ID_PLACEHOLDER not in line:
1575
+ dst.write(line)
1576
+ continue
1577
+ record: dict[str, Any] = json.loads(line)
1578
+ for source in record.get("sources") or []:
1579
+ urls: list[str] | None = source.get("source_record_urls")
1580
+ if urls:
1581
+ source["source_record_urls"] = [url.replace(EDGE_ID_PLACEHOLDER, record["id"]) for url in urls]
1582
+ dst.write(json.dumps(record, ensure_ascii=False) + "\n")
1583
+ resolved = True
1584
+ if resolved:
1585
+ tmp_path.replace(edges_path)
1586
+ else:
1587
+ tmp_path.unlink()
1588
+
1589
+
1527
1590
  def fold_unknown_to_supporting_text(lf: pl.LazyFrame) -> pl.LazyFrame:
1528
1591
  """Fold any non-Biolink edge column into ``supporting_text`` as ``col: value`` strings.
1529
1592
 
@@ -1635,6 +1698,12 @@ def _write_ndjson(
1635
1698
  ``compile_graph``; each commented phase boundary below is a hook point for
1636
1699
  the US-009 ``on_phase`` progress callback.
1637
1700
 
1701
+ Dedup assigns each edge ``id`` as the deterministic content hash of the
1702
+ PRE-resolution record -- the literal ``{edge_id}`` placeholder in explicit
1703
+ ``override.sources`` record URLs is what gets hashed -- and the placeholder
1704
+ sweep that follows substitutes the assigned id into the final edges file
1705
+ only, keeping ids deterministic.
1706
+
1638
1707
  Args:
1639
1708
  subnodes: Per-section node LazyFrames from ``_collect_subframes``.
1640
1709
  subedges: Per-section edge LazyFrames from ``_collect_subframes``.
@@ -1669,6 +1738,9 @@ def _write_ndjson(
1669
1738
  if on_phase is not None:
1670
1739
  on_phase("dedup")
1671
1740
  dedup_stream(edges_tmp, is_edges=True)
1741
+ # The deduper hashes the record WITH the literal `{edge_id}` placeholder still
1742
+ # in place, so edge ids stay deterministic regardless of this resolution pass.
1743
+ _resolve_edge_id_placeholders(edges_tmp.with_suffix(""))
1672
1744
  dedup_stream(nodes_tmp, is_edges=False)
1673
1745
 
1674
1746
  # Phase: rig. Summaries come from the FINAL deduplicated KGX files, and the
@@ -466,16 +466,99 @@ def validate_infores_curie(value: str, code: TablassertErrorCodes) -> str:
466
466
  return value
467
467
 
468
468
 
469
+ EDGE_ID_PLACEHOLDER: str = "{edge_id}"
470
+ """Placeholder for the final edge id inside ``override.sources`` record URLs.
471
+
472
+ The edge ``id`` is a deterministic content hash computed during the final dedup
473
+ stage -- after subgraphs are written -- so a per-edge URL cannot embed it during
474
+ the table build. The literal placeholder is emitted as-is and resolved against
475
+ the deduplicated ``*.edges.ndjson`` in a post-dedup sweep.
476
+ """
477
+
478
+ RESOURCE_ROLES: tuple[str, ...] = ("primary_knowledge_source", "aggregator_knowledge_source", "supporting_data_source")
479
+ """Valid Biolink ``ResourceRoleEnum`` values for retrieval ``sources`` entries.
480
+
481
+ Kept as literals instead of importing ``ResourceRoleEnum`` from the generated
482
+ ``biolink_model`` Pydantic classes so config validation stays cheap; any other
483
+ value fails KGX validation downstream.
484
+ """
485
+
486
+
487
+ class SourceOverride(TablaBase):
488
+ """One explicit retrieval-``sources`` entry template for manual provenance.
489
+
490
+ Each entry becomes one Biolink ``RetrievalSource`` struct on the edge's
491
+ ``sources`` list, replacing the derived primary/upstream emission entirely.
492
+ """
493
+
494
+ resource_id: str = Field(description="Infores CURIE of this retrieval source entry.", examples=["infores:my-source"])
495
+ resource_role: str = Field(
496
+ description="Biolink ResourceRoleEnum value for this entry.",
497
+ examples=["primary_knowledge_source", "aggregator_knowledge_source", "supporting_data_source"],
498
+ )
499
+ upstream_resource_ids: list[str] | None = Field(
500
+ None, description="Upstream infores CURIEs carried by this entry.", examples=[["infores:my-upstream"]]
501
+ )
502
+ source_record_urls: list[str] | None = Field(
503
+ None,
504
+ description=f"Source record URLs carried by this entry; `{EDGE_ID_PLACEHOLDER}` is replaced with the final edge id after dedup.",
505
+ examples=[["https://example.org/edge?id={edge_id}"]],
506
+ )
507
+
508
+ @field_validator("resource_id", mode="after")
509
+ @classmethod
510
+ def infores_resource_id(cls, value: str) -> str:
511
+ return validate_infores_curie(value, "override-bad-sources")
512
+
513
+ @field_validator("resource_role", mode="after")
514
+ @classmethod
515
+ def biolink_resource_role(cls, value: str) -> str:
516
+ if value not in RESOURCE_ROLES:
517
+ raise TablassertValidationError(f"`resource_role` must be one of {list(RESOURCE_ROLES)}, got {value!r}.", code="override-bad-sources")
518
+ return value
519
+
520
+ @field_validator("upstream_resource_ids", mode="after")
521
+ @classmethod
522
+ def infores_upstream_resource_ids(cls, values: list[str] | None) -> list[str] | None:
523
+ if values is None:
524
+ return None
525
+ for value in values:
526
+ validate_infores_curie(value, "override-bad-sources")
527
+ return values
528
+
529
+ @field_validator("source_record_urls", mode="after")
530
+ @classmethod
531
+ def url_or_edge_id_template(cls, values: list[str] | None) -> list[str] | None:
532
+ # Typed as plain str (not HttpUrl) so the `{edge_id}` placeholder survives
533
+ # validation; after removing placeholder occurrences the rest must still be
534
+ # an absolute http(s) URL.
535
+ if values is None:
536
+ return None
537
+ for value in values:
538
+ stripped: str = value.replace(EDGE_ID_PLACEHOLDER, "")
539
+ if not stripped.startswith(("https://", "http://")):
540
+ raise TablassertValidationError(
541
+ f"`source_record_urls` entries must be http(s) URLs (optionally containing `{EDGE_ID_PLACEHOLDER}`), got {value!r}.",
542
+ code="override-bad-sources",
543
+ )
544
+ return values
545
+
546
+
469
547
  class ManualProvenance(TablaBase):
470
548
  """Manually-specified provenance for non-PMID/PMC source graphs.
471
549
 
472
550
  When present under :class:`Provenance`, these values replace the legacy
473
551
  repo/publication-derived provenance while keeping the same KL/AT defaults.
474
552
  The primary ``sources`` entry (``resource_role: primary_knowledge_source``)
475
- always derives from the graph-level ``rig.source_info.infores_id``; manual
476
- infores CURIEs belong in ``upstream_resource_ids``.
553
+ derives from the graph-level ``rig.source_info.infores_id`` unless an
554
+ explicit ``sources`` template is given; manual infores CURIEs otherwise
555
+ belong in ``upstream_resource_ids``.
477
556
  """
478
557
 
558
+ sources: list[SourceOverride] | None = Field(
559
+ None,
560
+ description="Explicit retrieval-`sources` entry templates replacing the derived primary/upstream emission entirely; mutually exclusive with `upstream_resource_ids` and `upstream_source_record_urls`, which it subsumes.",
561
+ )
479
562
  upstream_resource_ids: list[str] = Field(
480
563
  default_factory=list,
481
564
  description="Manual upstream source infores CURIEs emitted instead of the repo-derived source map; the sanctioned place for manual infores.",
@@ -515,6 +598,26 @@ class ManualProvenance(TablaBase):
515
598
  )
516
599
  return values
517
600
 
601
+ @model_validator(mode="after")
602
+ def sources_template_is_coherent(self: Self) -> Self:
603
+ if self.sources is None:
604
+ return self
605
+ if self.upstream_resource_ids or self.upstream_source_record_urls is not None:
606
+ raise TablassertValidationError(
607
+ "`sources` is mutually exclusive with `upstream_resource_ids` and `upstream_source_record_urls`; the explicit template subsumes both.",
608
+ code="override-bad-sources",
609
+ )
610
+ if not self.sources:
611
+ raise TablassertValidationError("`sources` must contain at least one entry when set.", code="override-bad-sources")
612
+ resource_ids: list[str] = [entry.resource_id for entry in self.sources]
613
+ if len(set(resource_ids)) != len(resource_ids):
614
+ raise TablassertValidationError("`sources` entries must have unique `resource_id` values.", code="override-bad-sources")
615
+ if not any(entry.resource_role in ("primary_knowledge_source", "aggregator_knowledge_source") for entry in self.sources):
616
+ raise TablassertValidationError(
617
+ "`sources` must include at least one `primary_knowledge_source` or `aggregator_knowledge_source` entry.", code="override-bad-sources"
618
+ )
619
+ return self
620
+
518
621
  @model_validator(mode="after")
519
622
  def upstream_urls_match_resource_ids(self: Self) -> Self:
520
623
  if self.upstream_source_record_urls is None:
File without changes
File without changes
File without changes
File without changes
File without changes