tablassert 10.0.0__tar.gz → 11.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tablassert-10.0.0 → tablassert-11.0.0}/PKG-INFO +19 -3
- {tablassert-10.0.0 → tablassert-11.0.0}/README.md +18 -2
- {tablassert-10.0.0 → tablassert-11.0.0}/pyproject.toml +25 -7
- {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/agent.py +29 -4
- {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/cli.py +55 -9
- {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/coerce.py +20 -0
- tablassert-11.0.0/src/tablassert/enums.py +131 -0
- {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/errors.py +10 -0
- {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/fullmap.py +35 -0
- {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/graph_registry.py +38 -7
- {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/lib.py +66 -73
- {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/models.py +381 -23
- tablassert-11.0.0/src/tablassert/rig.py +686 -0
- tablassert-11.0.0/src/tablassert/study.py +166 -0
- tablassert-10.0.0/src/tablassert/enums.py +0 -69
- tablassert-10.0.0/src/tablassert/rig.py +0 -290
- {tablassert-10.0.0 → tablassert-11.0.0}/LICENSE +0 -0
- {tablassert-10.0.0 → tablassert-11.0.0}/rust/Cargo.lock +0 -0
- {tablassert-10.0.0 → tablassert-11.0.0}/rust/Cargo.toml +0 -0
- {tablassert-10.0.0 → tablassert-11.0.0}/rust/examples/count_tables.rs +0 -0
- {tablassert-10.0.0 → tablassert-11.0.0}/rust/src/fullmap.rs +0 -0
- {tablassert-10.0.0 → tablassert-11.0.0}/rust/src/json.rs +0 -0
- {tablassert-10.0.0 → tablassert-11.0.0}/rust/src/lib.rs +0 -0
- {tablassert-10.0.0 → tablassert-11.0.0}/rust/src/ndjson.rs +0 -0
- {tablassert-10.0.0 → tablassert-11.0.0}/rust/src/uuid.rs +0 -0
- {tablassert-10.0.0 → tablassert-11.0.0}/rust/tests/build_golden.rs +0 -0
- {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/__init__.py +0 -0
- {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/_lazy.py +0 -0
- {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/biolink.py +0 -0
- {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/extras.py +0 -0
- {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/ingests.py +0 -0
- {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/log.py +0 -0
- {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/nlp.py +0 -0
- {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/progress.py +0 -0
- {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/qc.py +0 -0
- {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/rs.pyi +0 -0
- {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/utils.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: tablassert
|
|
3
|
-
Version:
|
|
3
|
+
Version: 11.0.0
|
|
4
4
|
Classifier: License :: OSI Approved :: Apache Software License
|
|
5
5
|
Classifier: Development Status :: 5 - Production/Stable
|
|
6
6
|
Classifier: Intended Audience :: Science/Research
|
|
@@ -99,15 +99,31 @@ template:
|
|
|
99
99
|
- { annotation: supporting_study_size, method: column, encoding: D }
|
|
100
100
|
```
|
|
101
101
|
|
|
102
|
-
Wrap it in a graph config (`graph.yaml`) pointing at your fullmap entity-resolution database
|
|
102
|
+
Wrap it in a graph config (`graph.yaml`) pointing at your fullmap entity-resolution database
|
|
103
|
+
and carrying the required `rig:` metadata for the generated Resource Ingest Guide:
|
|
103
104
|
|
|
104
105
|
```yaml
|
|
105
106
|
name: MY_KG
|
|
106
107
|
version: 1.0.0
|
|
107
|
-
description: Gene–disease associations extracted from tabular sources.
|
|
108
108
|
tables:
|
|
109
109
|
- ./table.yaml
|
|
110
110
|
fullmap: /path/to/fullmap
|
|
111
|
+
rig:
|
|
112
|
+
source_info:
|
|
113
|
+
infores_id: infores:my-kg
|
|
114
|
+
terms_of_use_info:
|
|
115
|
+
terms_of_use_url: https://example.org/terms
|
|
116
|
+
data_access_locations:
|
|
117
|
+
- My source downloads - https://example.org/downloads
|
|
118
|
+
source_status: maintained_regular_updates
|
|
119
|
+
ingest_info:
|
|
120
|
+
utility: Gene-disease associations support Translator disease-mechanism queries.
|
|
121
|
+
scope: Gene-disease associations extracted from tabular sources.
|
|
122
|
+
provenance_info:
|
|
123
|
+
contributions:
|
|
124
|
+
- "Author Name - code author, data modeling"
|
|
125
|
+
artifact_base_url: https://example.org/my-kg
|
|
126
|
+
artifact_base_path: ./published/my-kg
|
|
111
127
|
```
|
|
112
128
|
|
|
113
129
|
Build the knowledge graph:
|
|
@@ -44,15 +44,31 @@ template:
|
|
|
44
44
|
- { annotation: supporting_study_size, method: column, encoding: D }
|
|
45
45
|
```
|
|
46
46
|
|
|
47
|
-
Wrap it in a graph config (`graph.yaml`) pointing at your fullmap entity-resolution database
|
|
47
|
+
Wrap it in a graph config (`graph.yaml`) pointing at your fullmap entity-resolution database
|
|
48
|
+
and carrying the required `rig:` metadata for the generated Resource Ingest Guide:
|
|
48
49
|
|
|
49
50
|
```yaml
|
|
50
51
|
name: MY_KG
|
|
51
52
|
version: 1.0.0
|
|
52
|
-
description: Gene–disease associations extracted from tabular sources.
|
|
53
53
|
tables:
|
|
54
54
|
- ./table.yaml
|
|
55
55
|
fullmap: /path/to/fullmap
|
|
56
|
+
rig:
|
|
57
|
+
source_info:
|
|
58
|
+
infores_id: infores:my-kg
|
|
59
|
+
terms_of_use_info:
|
|
60
|
+
terms_of_use_url: https://example.org/terms
|
|
61
|
+
data_access_locations:
|
|
62
|
+
- My source downloads - https://example.org/downloads
|
|
63
|
+
source_status: maintained_regular_updates
|
|
64
|
+
ingest_info:
|
|
65
|
+
utility: Gene-disease associations support Translator disease-mechanism queries.
|
|
66
|
+
scope: Gene-disease associations extracted from tabular sources.
|
|
67
|
+
provenance_info:
|
|
68
|
+
contributions:
|
|
69
|
+
- "Author Name - code author, data modeling"
|
|
70
|
+
artifact_base_url: https://example.org/my-kg
|
|
71
|
+
artifact_base_path: ./published/my-kg
|
|
56
72
|
```
|
|
57
73
|
|
|
58
74
|
Build the knowledge graph:
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "tablassert"
|
|
3
|
-
version = "
|
|
3
|
+
version = "11.0.0"
|
|
4
4
|
description = "Extract knowledge assertions from tabular data into NCATS Translator-compliant KGX NDJSON — declaratively, with entity resolution and quality control built in."
|
|
5
5
|
authors = [
|
|
6
6
|
{ name = "Skye Lane Goetz", email = "sgoetz@isbscience.org" }
|
|
@@ -96,17 +96,35 @@ optimize = [
|
|
|
96
96
|
]
|
|
97
97
|
|
|
98
98
|
[dependency-groups]
|
|
99
|
-
|
|
100
|
-
"mkdocs>=1.6.1",
|
|
101
|
-
"mkdocs-material>=9.6.0",
|
|
99
|
+
test = [
|
|
102
100
|
"maturin>=1.10,<2.0",
|
|
103
|
-
"pre-commit>=4.5.1",
|
|
104
|
-
"pyright>=1.1.411",
|
|
105
101
|
"pytest>=9.0.2",
|
|
106
102
|
"pytest-cov>=7.1.0",
|
|
107
|
-
"ruff>=0.15.6",
|
|
108
103
|
"pytest-xdist>=3.8.0",
|
|
109
104
|
]
|
|
105
|
+
typecheck = [
|
|
106
|
+
"pyright>=1.1.411",
|
|
107
|
+
]
|
|
108
|
+
docs = [
|
|
109
|
+
"mkdocs>=1.6.1",
|
|
110
|
+
"mkdocs-material>=9.6.0",
|
|
111
|
+
]
|
|
112
|
+
# The EXACT set both CI Python jobs install. Keeping `python-type` and `python-test` on one
|
|
113
|
+
# identical sync is deliberate: `astral-sh/setup-uv` keys its cache on the lockfile, so jobs
|
|
114
|
+
# with different dependency sets race to save a cache the others then restore and miss on.
|
|
115
|
+
# One shared set means one genuinely warm cache. Excludes docs/pre-commit/ruff, which no CI
|
|
116
|
+
# job that syncs needs (lint runs ruff standalone; docs uses `--group docs`).
|
|
117
|
+
ci = [
|
|
118
|
+
{ include-group = "test" },
|
|
119
|
+
{ include-group = "typecheck" },
|
|
120
|
+
]
|
|
121
|
+
# Superset for local development, so `make setup` still installs everything.
|
|
122
|
+
dev = [
|
|
123
|
+
{ include-group = "ci" },
|
|
124
|
+
{ include-group = "docs" },
|
|
125
|
+
"pre-commit>=4.5.1",
|
|
126
|
+
"ruff>=0.15.6",
|
|
127
|
+
]
|
|
110
128
|
|
|
111
129
|
[tool.pytest.ini_options]
|
|
112
130
|
testpaths = ["tests"]
|
|
@@ -1313,12 +1313,34 @@ def build_and_audit(
|
|
|
1313
1313
|
table_cfg: dict[str, object] = data if ("template" in data or "sections" in data) else {"template": data}
|
|
1314
1314
|
|
|
1315
1315
|
(root / "table.yaml").write_text(yaml.safe_dump(table_cfg, sort_keys=False))
|
|
1316
|
+
# The measurement build still emits a RIG (every build does), so it carries an
|
|
1317
|
+
# honest minimal rig block: the agent only mines PMC open-access tables, and the
|
|
1318
|
+
# artifacts live in this throwaway workdir (file:// base = unpublished).
|
|
1319
|
+
resolved_root: str = str(root.resolve())
|
|
1316
1320
|
graph_cfg: dict[str, object] = {
|
|
1317
1321
|
"name": name,
|
|
1318
1322
|
"version": version,
|
|
1319
|
-
"description": f"Agent-built graph for {name}",
|
|
1320
1323
|
"tables": ["table.yaml"], # relative to workdir (the pipelines chdir there)
|
|
1321
1324
|
"fullmap": str(fullmap),
|
|
1325
|
+
"rig": {
|
|
1326
|
+
"source_info": {
|
|
1327
|
+
"infores_id": f"infores:{name.lower().replace('_', '-')}",
|
|
1328
|
+
"name": f"Agent-built measurement graph {name}",
|
|
1329
|
+
"terms_of_use_info": {
|
|
1330
|
+
"terms_of_use_url": "https://pmc.ncbi.nlm.nih.gov/about/copyright/",
|
|
1331
|
+
"terms_of_use_description": "PubMed Central open-access supplementary table; individual article licenses apply.",
|
|
1332
|
+
},
|
|
1333
|
+
"data_access_locations": ["PubMed Central - https://pmc.ncbi.nlm.nih.gov/"],
|
|
1334
|
+
"source_status": "unknown",
|
|
1335
|
+
},
|
|
1336
|
+
"ingest_info": {
|
|
1337
|
+
"utility": f"Transient measurement graph used to score agent-derived table configs for {name}.",
|
|
1338
|
+
"scope": "Associations mined from one PMC supplementary table config under audit.",
|
|
1339
|
+
},
|
|
1340
|
+
"provenance_info": {"contributions": ["Tablassert agent: automated config derivation and measurement build"]},
|
|
1341
|
+
"artifact_base_url": f"file://{resolved_root}",
|
|
1342
|
+
"artifact_base_path": resolved_root,
|
|
1343
|
+
},
|
|
1322
1344
|
}
|
|
1323
1345
|
(root / "graph.yaml").write_text(yaml.safe_dump(graph_cfg, sort_keys=False))
|
|
1324
1346
|
|
|
@@ -2153,9 +2175,12 @@ section choose column-letter encodings for entity columns and literal CURIEs for
|
|
|
2153
2175
|
pick a predicate the subject/object pair actually permits (see BIOLINK MODELING below); add
|
|
2154
2176
|
statistical annotations (p_value / effect_size / effect_type) when that table has them —
|
|
2155
2177
|
method: column for table-provided columns, method: value for a fixed valid value (e.g.
|
|
2156
|
-
effect_type: spearmans_rho when every row is a Spearman correlation).
|
|
2157
|
-
|
|
2158
|
-
|
|
2178
|
+
effect_type: spearmans_rho when every row is a Spearman correlation). effect_size and effect_type
|
|
2179
|
+
are MANDATORY AS A PAIR: either one without the other is a hard validation error that bounces your
|
|
2180
|
+
final answer, so a table with an effect-size column also needs its effect_type (method: value when
|
|
2181
|
+
every row shares one statistic). Alias spellings count — `odds ratio` and the legacy
|
|
2182
|
+
`relationship_strength` both coerce to effect_size. A single-table article is still ONE config with
|
|
2183
|
+
ONE section.
|
|
2159
2184
|
|
|
2160
2185
|
# BIOLINK MODELING (the pipeline enforces these SILENTLY — violating them costs you score)
|
|
2161
2186
|
The build derives each edge's association CLASS from the (subject category, object category)
|
|
@@ -121,13 +121,15 @@ def build_pipeline(
|
|
|
121
121
|
"""Build a knowledge graph from a YAML configuration file.
|
|
122
122
|
|
|
123
123
|
Runs the six-stage build pipeline: load tables → extract sections → build
|
|
124
|
-
Tcodes → collect instructions → build subgraphs → compile graph.
|
|
124
|
+
Tcodes → collect instructions → build subgraphs → compile graph. With ``qc``
|
|
125
|
+
enabled a seventh stage studies the final NDJSON files.
|
|
125
126
|
|
|
126
127
|
Args:
|
|
127
128
|
configuration_file: Path to the graph YAML file.
|
|
128
129
|
progress: Pipeline progress reporter.
|
|
129
130
|
release: When ``True``, emit release-mode artifacts.
|
|
130
|
-
qc: When ``True``, run quality-control audits on each section
|
|
131
|
+
qc: When ``True``, run quality-control audits on each section and assert
|
|
132
|
+
over the final NDJSON files (failing the build on any violation).
|
|
131
133
|
log: When ``True``, enable per-section verbose logging.
|
|
132
134
|
head: When ``True``, preview a random sample of up to 5 rows per section (fast schema/shape check).
|
|
133
135
|
|
|
@@ -166,6 +168,12 @@ def build_pipeline(
|
|
|
166
168
|
advance()
|
|
167
169
|
sections: list[dict[str, Any]] = list(chain.from_iterable(temp))
|
|
168
170
|
n: int = len(sections)
|
|
171
|
+
# Per-section source descriptors for the generated RIG's relevant-file cross-check:
|
|
172
|
+
# each entry records the section's local file name and its validated source URLs.
|
|
173
|
+
section_sources: list[dict[str, Any]] = []
|
|
174
|
+
for s in sections:
|
|
175
|
+
src: dict[str, Any] = s.get("source") or {}
|
|
176
|
+
section_sources.append({"local": str(src.get("local") or ""), "urls": [str(u) for u in src.get("url") or []]})
|
|
169
177
|
|
|
170
178
|
# Stage 3/6: build Tcode.
|
|
171
179
|
progress.stage("Building TCode")
|
|
@@ -179,7 +187,16 @@ def build_pipeline(
|
|
|
179
187
|
try:
|
|
180
188
|
tcode.append(
|
|
181
189
|
Tcode.model_validate(
|
|
182
|
-
{
|
|
190
|
+
{
|
|
191
|
+
**s,
|
|
192
|
+
"store": store,
|
|
193
|
+
"log": log,
|
|
194
|
+
"qc": qc,
|
|
195
|
+
"release": release,
|
|
196
|
+
"head": head,
|
|
197
|
+
"name": g.name,
|
|
198
|
+
"infores": g.rig.source_info.infores_id,
|
|
199
|
+
}
|
|
183
200
|
)
|
|
184
201
|
)
|
|
185
202
|
except pydantic.ValidationError as e:
|
|
@@ -214,13 +231,39 @@ def build_pipeline(
|
|
|
214
231
|
start(f"{g.name} · v{g.version}")
|
|
215
232
|
# on_phase drives the phase tag (scan → normalize → write-nodes → write-edges → dedup → rig);
|
|
216
233
|
# on_subgraph ticks the bar once per subgraph, so the total is len(subgraphs).
|
|
217
|
-
compile_graph(
|
|
218
|
-
|
|
219
|
-
)
|
|
234
|
+
compile_graph(subgraphs, g.name, g.version, g.rig, section_sources, on_phase=sub_step, on_subgraph=advance)
|
|
235
|
+
|
|
236
|
+
# Stage 7/7 (only with --qc): assert over the final NDJSON files.
|
|
237
|
+
if qc:
|
|
238
|
+
progress.stage("Studying Graph")
|
|
239
|
+
study_final_ndjson(g.name, g.version, Path(g.rig.artifact_base_path))
|
|
220
240
|
|
|
221
241
|
logger.info("Built graph {name} v{version}: {n} sections", name=g.name, version=g.version, n=n)
|
|
222
242
|
|
|
223
243
|
|
|
244
|
+
def study_final_ndjson(name: str, version: str, out_dir: Path) -> None:
|
|
245
|
+
"""Run study assertions over a build's final NDJSON files (the ``--qc`` stage 7).
|
|
246
|
+
|
|
247
|
+
Args:
|
|
248
|
+
name: Graph name, used to locate ``<name>_<version>.nodes.ndjson``.
|
|
249
|
+
version: Graph version, used to locate ``<name>_<version>.edges.ndjson``.
|
|
250
|
+
out_dir: Artifact directory the build wrote into
|
|
251
|
+
(``rig.artifact_base_path``).
|
|
252
|
+
|
|
253
|
+
Raises:
|
|
254
|
+
SystemExit: With status 1 when any study assertion is violated.
|
|
255
|
+
"""
|
|
256
|
+
from tablassert.study import format_violations, study_kgx
|
|
257
|
+
|
|
258
|
+
violations = study_kgx(out_dir / f"{name}_{version}.nodes.ndjson", out_dir / f"{name}_{version}.edges.ndjson")
|
|
259
|
+
if violations:
|
|
260
|
+
summary: str = format_violations(violations)
|
|
261
|
+
print(summary, file=sys.stderr)
|
|
262
|
+
logger.warning("study assertions failed on final NDJSON:\n{summary}", summary=summary)
|
|
263
|
+
raise SystemExit(1)
|
|
264
|
+
logger.info("study assertions passed on final NDJSON")
|
|
265
|
+
|
|
266
|
+
|
|
224
267
|
def validate_pipeline(table_configuration_file: Path, progress: PipelineProgress) -> None:
|
|
225
268
|
"""Validate section syntax from a YAML configuration file.
|
|
226
269
|
|
|
@@ -599,11 +642,14 @@ def build_kg(
|
|
|
599
642
|
|
|
600
643
|
``--qc`` requires the ``[qc]`` extra (``pip install "tablassert[qc]"``); it is
|
|
601
644
|
checked before the build starts, because the audit stage runs LAST and a missing
|
|
602
|
-
extra would otherwise surface only after entity resolution has finished.
|
|
645
|
+
extra would otherwise surface only after entity resolution has finished. It also
|
|
646
|
+
runs a final study stage that asserts over the emitted NDJSON -- no duplicate node
|
|
647
|
+
ids, no undeclared or isolated nodes, no malformed lines or stray whitespace --
|
|
648
|
+
and fails the build (non-zero exit) when any assertion is violated.
|
|
603
649
|
"""
|
|
604
650
|
if qc:
|
|
605
651
|
extras.require("qc", required_by="--qc")
|
|
606
|
-
run(6, build_pipeline, graph_configuration_file, release=release, qc=qc, log=log, head=head)
|
|
652
|
+
run(7 if qc else 6, build_pipeline, graph_configuration_file, release=release, qc=qc, log=log, head=head)
|
|
607
653
|
|
|
608
654
|
|
|
609
655
|
@APP.command(name="validate")
|
|
@@ -950,7 +996,7 @@ def _prebuilt_fullmap_urls(babel_version: str) -> tuple[str, str]:
|
|
|
950
996
|
|
|
951
997
|
RENCI publishes a prebuilt ``fullmap.tar.zst`` (and a ``sha256sum.txt``) under
|
|
952
998
|
``{BABEL_BASE}/{babel_version}/fullmap/{tablassert_version}/``, where the version
|
|
953
|
-
directory is the INSTALLED Tablassert package version (e.g. ``10.
|
|
999
|
+
directory is the INSTALLED Tablassert package version (e.g. ``10.1.0``) — resolved from
|
|
954
1000
|
installed-package metadata, never hardcoded, so a new release looks itself up.
|
|
955
1001
|
|
|
956
1002
|
Args:
|
|
@@ -523,6 +523,26 @@ def effect_size_target(name: str) -> str | None:
|
|
|
523
523
|
return None
|
|
524
524
|
|
|
525
525
|
|
|
526
|
+
def coerced_target(name: str) -> str:
|
|
527
|
+
"""Map a column/annotation name to the canonical name the clean phase renames it to.
|
|
528
|
+
|
|
529
|
+
Args:
|
|
530
|
+
name: Raw source column or annotation name.
|
|
531
|
+
|
|
532
|
+
Returns:
|
|
533
|
+
The canonical slot name a ``coerce_*_columns`` op would rename ``name``
|
|
534
|
+
to, or ``name`` unchanged when no coercion claims it.
|
|
535
|
+
|
|
536
|
+
Notes:
|
|
537
|
+
Classifier order mirrors the op order in ``Tcode._source_ops``:
|
|
538
|
+
``coerce_pvalue_columns`` runs first, so a p/q-value alias is claimed
|
|
539
|
+
before the study-size and effect classifiers ever see it. Config-time
|
|
540
|
+
validators judge this target rather than the raw name so they see a
|
|
541
|
+
name exactly as the build will.
|
|
542
|
+
"""
|
|
543
|
+
return pvalue_target(name) or study_size_target(name) or effect_size_target(name) or effect_type_target(name) or name
|
|
544
|
+
|
|
545
|
+
|
|
526
546
|
def coerce_effect_size_columns(lf: pl.LazyFrame) -> pl.LazyFrame:
|
|
527
547
|
"""Rename effect-size-like columns to Biolink KGX-compliant ``effect_size`` (Biolink PR #1774).
|
|
528
548
|
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
"""Tablassert configuration enums (non-Biolink).
|
|
2
|
+
|
|
3
|
+
These enums describe Tablassert's own configuration vocabulary (source kinds,
|
|
4
|
+
comparison operators, encoding/fill methods, contribution labels, and so on).
|
|
5
|
+
|
|
6
|
+
The Biolink Model vocabulary -- ``Categories``, ``Predicates``, ``Qualifiers``,
|
|
7
|
+
``KnowledgeLevels``, ``AgentTypes``, ``EdgeCategories``, and
|
|
8
|
+
``ALLOWED_EDGE_FIELDS`` -- is derived from the ``biolink-model`` package and lives
|
|
9
|
+
in :mod:`tablassert.biolink`. Import those from there.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from enum import Enum
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class Tokens(str, Enum):
|
|
18
|
+
AUTO = "auto"
|
|
19
|
+
VALUES = "values"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class Repositories(str, Enum):
|
|
23
|
+
PUBMED_CENTRAL = "PMC"
|
|
24
|
+
PUBMED = "PMID"
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class InformationResources(str, Enum):
|
|
28
|
+
PUBMED = "infores:pubmed"
|
|
29
|
+
PUBMED_CENTRAL = "infores:pubmed-central"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class Contributions(str, Enum):
|
|
33
|
+
CURATION = "curation"
|
|
34
|
+
VALIDATION = "validation"
|
|
35
|
+
TOOL = "tool"
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class Comparisons(str, Enum):
|
|
39
|
+
GT = "gt"
|
|
40
|
+
GE = "ge"
|
|
41
|
+
LT = "lt"
|
|
42
|
+
LE = "le"
|
|
43
|
+
EQ = "eq"
|
|
44
|
+
NE = "ne"
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class Functions(str, Enum):
|
|
48
|
+
COPYSIGN = "copysign"
|
|
49
|
+
POW = "pow"
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class Files(str, Enum):
|
|
53
|
+
TEXT = "text"
|
|
54
|
+
EXCEL = "excel"
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class EncodingMethods(str, Enum):
|
|
58
|
+
VALUE = "value"
|
|
59
|
+
COLUMN = "column"
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class FillMethods(str, Enum):
|
|
63
|
+
FORWARD = "forward"
|
|
64
|
+
BACKWARD = "backward"
|
|
65
|
+
MIN = "min"
|
|
66
|
+
MAX = "max"
|
|
67
|
+
MEAN = "mean"
|
|
68
|
+
ZERO = "zero"
|
|
69
|
+
ONE = "one"
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
# --- Resource Ingest Guide vocabularies ------------------------------------ #
|
|
73
|
+
# These mirror the enums declared by the released RIG schema
|
|
74
|
+
# (biolink/resource-ingest-guide-schema), so a generated `.RIG.yaml` can only
|
|
75
|
+
# carry values the upstream validator accepts.
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
class SourceStatuses(str, Enum):
|
|
79
|
+
MAINTAINED_REGULAR_UPDATES = "maintained_regular_updates"
|
|
80
|
+
MAINTAINED_AS_NEEDED_UPDATES = "maintained_as_needed_updates"
|
|
81
|
+
NOT_MAINTAINED = "not_maintained"
|
|
82
|
+
UNKNOWN = "unknown"
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
class ProvisionMechanisms(str, Enum):
|
|
86
|
+
FILE_DOWNLOAD = "file_download"
|
|
87
|
+
API_ENDPOINT = "api_endpoint"
|
|
88
|
+
DATABASE_DUMP = "database_dump"
|
|
89
|
+
OTHER = "other"
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
class DataFormats(str, Enum):
|
|
93
|
+
TSV = "tsv"
|
|
94
|
+
XML = "xml"
|
|
95
|
+
CSV = "csv"
|
|
96
|
+
JSON = "json"
|
|
97
|
+
YAML = "yaml"
|
|
98
|
+
OBO = "obo"
|
|
99
|
+
PROTOBUFF = "protobuff"
|
|
100
|
+
KGX = "kgx"
|
|
101
|
+
MYSQL = "mysql"
|
|
102
|
+
POSTGRESQL = "postgresql"
|
|
103
|
+
SQLITE = "sqlite"
|
|
104
|
+
OTHER = "other"
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
class IngestCategories(str, Enum):
|
|
108
|
+
PRIMARY_KNOWLEDGE_PROVIDER = "primary_knowledge_provider"
|
|
109
|
+
AGGREGATION_PROVIDER = "aggregation_provider"
|
|
110
|
+
AGGREGATION_INTERPRETER = "aggregation_interpreter"
|
|
111
|
+
SUPPORTING_DATA_PROVIDER = "supporting_data_provider"
|
|
112
|
+
TRANSLATOR_KNOWLEDGE_CREATOR = "translator_knowledge_creator"
|
|
113
|
+
ONTOLOGY_PROVIDER = "ontology_provider"
|
|
114
|
+
NODE_PROPERTY_ONLY_PROVIDER = "node_property_only_provider"
|
|
115
|
+
OTHER = "other"
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
class ContentCategories(str, Enum):
|
|
119
|
+
EDGE_CONTENT = "edge_content"
|
|
120
|
+
NODE_PROPERTY_CONTENT = "node_property_content"
|
|
121
|
+
EDGE_PROPERTY_CONTENT = "edge_property_content"
|
|
122
|
+
OTHER = "other"
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
class ModelingCategories(str, Enum):
|
|
126
|
+
SPOQ_PATTERN = "spoq_pattern"
|
|
127
|
+
PREDICATES = "predicates"
|
|
128
|
+
QUALIFIERS = "qualifiers"
|
|
129
|
+
EDGE_PROPERTIES = "edge_properties"
|
|
130
|
+
NODE_PROPERTIES = "node_properties"
|
|
131
|
+
OTHER = "other"
|
|
@@ -12,6 +12,7 @@ TablassertErrorCodes = Literal[
|
|
|
12
12
|
"graph-validation-failed",
|
|
13
13
|
"section-validation-failed",
|
|
14
14
|
"babel-download-failed",
|
|
15
|
+
"resolve-bad-specs",
|
|
15
16
|
"config-rows-and-row-slice-conflict",
|
|
16
17
|
"comparison-bad-comparator-type",
|
|
17
18
|
"comparison-nonnumeric-comparator",
|
|
@@ -28,10 +29,19 @@ TablassertErrorCodes = Literal[
|
|
|
28
29
|
"encoding-list-method-removed",
|
|
29
30
|
"annotation-split-by-requires-column",
|
|
30
31
|
"annotation-split-by-empty",
|
|
32
|
+
"annotation-effect-size-without-type",
|
|
33
|
+
"annotation-effect-type-without-size",
|
|
31
34
|
"qualifier-auto-derived",
|
|
32
35
|
"qualifier-bad-value",
|
|
33
36
|
"qualifier-unsatisfiable",
|
|
34
37
|
"qualifier-nullable-literal",
|
|
38
|
+
"qualifier-duplicated",
|
|
39
|
+
"rig-bad-infores",
|
|
40
|
+
"rig-bad-artifact-url",
|
|
41
|
+
"rig-bad-access-location",
|
|
42
|
+
"rig-terms-empty",
|
|
43
|
+
"rig-legacy-keys",
|
|
44
|
+
"rig-validation-failed",
|
|
35
45
|
]
|
|
36
46
|
|
|
37
47
|
|
|
@@ -10,6 +10,7 @@ from typing import TYPE_CHECKING, Any, NamedTuple, cast
|
|
|
10
10
|
from tablassert import rs
|
|
11
11
|
from tablassert._lazy import LazyModule
|
|
12
12
|
from tablassert.biolink import Categories
|
|
13
|
+
from tablassert.errors import TablassertError
|
|
13
14
|
from tablassert.log import cat
|
|
14
15
|
|
|
15
16
|
logger = cat("FULLMAP")
|
|
@@ -575,12 +576,46 @@ def resolve_batch(
|
|
|
575
576
|
|
|
576
577
|
Returns:
|
|
577
578
|
LazyFrame with resolved columns added.
|
|
579
|
+
|
|
580
|
+
Raises:
|
|
581
|
+
TablassertError: ``resolve-bad-specs`` when two specs share a column, or a
|
|
582
|
+
spec's column or its ``<col> + tag`` normalization column is absent from
|
|
583
|
+
the input schema. Checked schema-only (no ``collect``) before any term
|
|
584
|
+
extraction or redb access, so bad specs never surface mid-build as a raw
|
|
585
|
+
polars ``ColumnNotFoundError``.
|
|
578
586
|
"""
|
|
579
587
|
# Each column still gets its own taxon/prioritize/avoid filtering and its own join back into lf;
|
|
580
588
|
# only the redb round trip itself (rs.lookup_fullmap_terms) is pooled across columns.
|
|
581
589
|
if not specs:
|
|
582
590
|
return lf
|
|
583
591
|
|
|
592
|
+
# Fail loudly on malformed specs BEFORE term collection or any redb access. A duplicated
|
|
593
|
+
# spec would otherwise die mid-build in its second join_matches pass (the first pass drops
|
|
594
|
+
# <col> + tag, so the second spec's level-two join hits a raw polars ColumnNotFoundError),
|
|
595
|
+
# and a spec whose normalization columns were never produced would crash distinct().
|
|
596
|
+
seen: set[str] = set()
|
|
597
|
+
duplicates: list[str] = []
|
|
598
|
+
for spec in specs:
|
|
599
|
+
if spec.col in seen and spec.col not in duplicates:
|
|
600
|
+
duplicates.append(spec.col)
|
|
601
|
+
seen.add(spec.col)
|
|
602
|
+
if duplicates:
|
|
603
|
+
raise TablassertError(
|
|
604
|
+
f"resolve_batch received more than one ResolveSpec for column(s) {', '.join(repr(col) for col in duplicates)}; "
|
|
605
|
+
"each node column may be resolved at most once per batch.",
|
|
606
|
+
code="resolve-bad-specs",
|
|
607
|
+
)
|
|
608
|
+
|
|
609
|
+
schema: pl.Schema = lf.collect_schema()
|
|
610
|
+
missing: list[tuple[str, str]] = [(name, spec.col) for spec in specs for name in (spec.col, spec.col + tag) if name not in schema]
|
|
611
|
+
if missing:
|
|
612
|
+
detail: str = ", ".join(f"{name!r} (needed by the spec for {col!r})" for name, col in missing)
|
|
613
|
+
raise TablassertError(
|
|
614
|
+
f"resolve_batch specs reference column(s) absent from the input schema: {detail}. This usually means a node "
|
|
615
|
+
"encoding was declared twice or against the wrong column, so its normalization columns were never produced.",
|
|
616
|
+
code="resolve-bad-specs",
|
|
617
|
+
)
|
|
618
|
+
|
|
584
619
|
terms_by_col: dict[str, pl.LazyFrame] = {spec.col: distinct(lf, spec.col, spec.col + tag) for spec in specs}
|
|
585
620
|
collected_terms: dict[str, pl.DataFrame] = {col: terms.collect() for col, terms in terms_by_col.items()}
|
|
586
621
|
|
|
@@ -36,11 +36,42 @@ TMP_NAME: str = f"{GRAPH_YAML}.tmp"
|
|
|
36
36
|
CORRUPT_PREFIX: str = f"{GRAPH_YAML}.corrupt-"
|
|
37
37
|
GRAPH_NAME: str = "tablassert-agent"
|
|
38
38
|
GRAPH_VERSION: str = "1"
|
|
39
|
-
GRAPH_DESCRIPTION: str = "Aggregate graph of agent-built PMC table configs"
|
|
40
39
|
#: Record statuses whose best config self-registers (both are SUCCESSFUL builds).
|
|
41
40
|
REGISTERED_STATUSES: frozenset[str] = frozenset({"MAPPED", "BUILT_UNMEASURED"})
|
|
42
41
|
|
|
43
42
|
|
|
43
|
+
def _registry_rig(state_dir: Path) -> dict[str, Any]:
|
|
44
|
+
"""The honest ``rig:`` block for the aggregate agent registry graph.
|
|
45
|
+
|
|
46
|
+
Every fact here is mechanical: the agent only mines PubMed Central
|
|
47
|
+
open-access supplementary tables, so source terms/access describe PMC, and
|
|
48
|
+
the artifact bases point at the state directory itself (a ``file://`` base
|
|
49
|
+
is a valid unpublished URI; swap it for a public https base before sending
|
|
50
|
+
the generated RIG anywhere).
|
|
51
|
+
"""
|
|
52
|
+
resolved: str = str(state_dir.resolve())
|
|
53
|
+
return {
|
|
54
|
+
"source_info": {
|
|
55
|
+
"infores_id": "infores:tablassert-agent",
|
|
56
|
+
"name": "PubMed Central open-access supplementary tables",
|
|
57
|
+
"description": "Aggregate of tabular associations mined from PubMed Central open-access supplementary files by the Tablassert agent.",
|
|
58
|
+
"terms_of_use_info": {
|
|
59
|
+
"terms_of_use_url": "https://pmc.ncbi.nlm.nih.gov/about/copyright/",
|
|
60
|
+
"terms_of_use_description": "PubMed Central open-access subset; individual article licenses apply.",
|
|
61
|
+
},
|
|
62
|
+
"data_access_locations": ["PubMed Central - https://pmc.ncbi.nlm.nih.gov/"],
|
|
63
|
+
"source_status": "unknown",
|
|
64
|
+
},
|
|
65
|
+
"ingest_info": {
|
|
66
|
+
"utility": "Aggregates agent-derived tabular knowledge assertions for Translator-style querying.",
|
|
67
|
+
"scope": "All agent-built table configs registered under this state directory.",
|
|
68
|
+
},
|
|
69
|
+
"provenance_info": {"contributions": ["Tablassert agent: automated config derivation and build"]},
|
|
70
|
+
"artifact_base_url": f"file://{resolved}",
|
|
71
|
+
"artifact_base_path": resolved,
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
|
|
44
75
|
@contextlib.contextmanager
|
|
45
76
|
def _registry_lock(state_dir: Path) -> Iterator[None]:
|
|
46
77
|
"""Hold an EXCLUSIVE ``flock`` on ``<state_dir>/graph.yaml.lock`` (created if missing).
|
|
@@ -58,9 +89,9 @@ def _registry_lock(state_dir: Path) -> Iterator[None]:
|
|
|
58
89
|
fcntl.flock(lock_file.fileno(), fcntl.LOCK_UN)
|
|
59
90
|
|
|
60
91
|
|
|
61
|
-
def _fresh_doc() -> dict[str, Any]:
|
|
92
|
+
def _fresh_doc(state_dir: Path) -> dict[str, Any]:
|
|
62
93
|
"""A brand-new registry document (``_apply_fullmap`` fills the first-wins ``fullmap``)."""
|
|
63
|
-
return {"name": GRAPH_NAME, "version": GRAPH_VERSION, "
|
|
94
|
+
return {"name": GRAPH_NAME, "version": GRAPH_VERSION, "tables": [], "rig": _registry_rig(state_dir)}
|
|
64
95
|
|
|
65
96
|
|
|
66
97
|
def _quarantine(state_dir: Path, reason: str) -> Path:
|
|
@@ -85,20 +116,20 @@ def _load_registry(state_dir: Path) -> dict[str, Any]:
|
|
|
85
116
|
"""
|
|
86
117
|
path: Path = state_dir / GRAPH_YAML
|
|
87
118
|
if not path.is_file():
|
|
88
|
-
return _fresh_doc()
|
|
119
|
+
return _fresh_doc(state_dir)
|
|
89
120
|
try:
|
|
90
121
|
data: object = yaml.safe_load(path.read_text(encoding="utf-8"))
|
|
91
122
|
except yaml.YAMLError as exc:
|
|
92
123
|
_quarantine(state_dir, f"YAML parse error: {exc}")
|
|
93
|
-
return _fresh_doc()
|
|
124
|
+
return _fresh_doc(state_dir)
|
|
94
125
|
if not isinstance(data, dict):
|
|
95
126
|
_quarantine(state_dir, f"top level is not a mapping (got {type(data).__name__})")
|
|
96
|
-
return _fresh_doc()
|
|
127
|
+
return _fresh_doc(state_dir)
|
|
97
128
|
try:
|
|
98
129
|
Graph.model_validate(data)
|
|
99
130
|
except pydantic.ValidationError as exc:
|
|
100
131
|
_quarantine(state_dir, f"fails Graph.model_validate ({len(exc.errors())} error(s))")
|
|
101
|
-
return _fresh_doc()
|
|
132
|
+
return _fresh_doc(state_dir)
|
|
102
133
|
return data
|
|
103
134
|
|
|
104
135
|
|