tablassert 10.0.0__tar.gz → 11.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. {tablassert-10.0.0 → tablassert-11.0.0}/PKG-INFO +19 -3
  2. {tablassert-10.0.0 → tablassert-11.0.0}/README.md +18 -2
  3. {tablassert-10.0.0 → tablassert-11.0.0}/pyproject.toml +25 -7
  4. {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/agent.py +29 -4
  5. {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/cli.py +55 -9
  6. {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/coerce.py +20 -0
  7. tablassert-11.0.0/src/tablassert/enums.py +131 -0
  8. {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/errors.py +10 -0
  9. {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/fullmap.py +35 -0
  10. {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/graph_registry.py +38 -7
  11. {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/lib.py +66 -73
  12. {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/models.py +381 -23
  13. tablassert-11.0.0/src/tablassert/rig.py +686 -0
  14. tablassert-11.0.0/src/tablassert/study.py +166 -0
  15. tablassert-10.0.0/src/tablassert/enums.py +0 -69
  16. tablassert-10.0.0/src/tablassert/rig.py +0 -290
  17. {tablassert-10.0.0 → tablassert-11.0.0}/LICENSE +0 -0
  18. {tablassert-10.0.0 → tablassert-11.0.0}/rust/Cargo.lock +0 -0
  19. {tablassert-10.0.0 → tablassert-11.0.0}/rust/Cargo.toml +0 -0
  20. {tablassert-10.0.0 → tablassert-11.0.0}/rust/examples/count_tables.rs +0 -0
  21. {tablassert-10.0.0 → tablassert-11.0.0}/rust/src/fullmap.rs +0 -0
  22. {tablassert-10.0.0 → tablassert-11.0.0}/rust/src/json.rs +0 -0
  23. {tablassert-10.0.0 → tablassert-11.0.0}/rust/src/lib.rs +0 -0
  24. {tablassert-10.0.0 → tablassert-11.0.0}/rust/src/ndjson.rs +0 -0
  25. {tablassert-10.0.0 → tablassert-11.0.0}/rust/src/uuid.rs +0 -0
  26. {tablassert-10.0.0 → tablassert-11.0.0}/rust/tests/build_golden.rs +0 -0
  27. {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/__init__.py +0 -0
  28. {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/_lazy.py +0 -0
  29. {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/biolink.py +0 -0
  30. {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/extras.py +0 -0
  31. {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/ingests.py +0 -0
  32. {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/log.py +0 -0
  33. {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/nlp.py +0 -0
  34. {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/progress.py +0 -0
  35. {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/qc.py +0 -0
  36. {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/rs.pyi +0 -0
  37. {tablassert-10.0.0 → tablassert-11.0.0}/src/tablassert/utils.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: tablassert
3
- Version: 10.0.0
3
+ Version: 11.0.0
4
4
  Classifier: License :: OSI Approved :: Apache Software License
5
5
  Classifier: Development Status :: 5 - Production/Stable
6
6
  Classifier: Intended Audience :: Science/Research
@@ -99,15 +99,31 @@ template:
99
99
  - { annotation: supporting_study_size, method: column, encoding: D }
100
100
  ```
101
101
 
102
- Wrap it in a graph config (`graph.yaml`) pointing at your fullmap entity-resolution database:
102
+ Wrap it in a graph config (`graph.yaml`) pointing at your fullmap entity-resolution database
103
+ and carrying the required `rig:` metadata for the generated Resource Ingest Guide:
103
104
 
104
105
  ```yaml
105
106
  name: MY_KG
106
107
  version: 1.0.0
107
- description: Gene–disease associations extracted from tabular sources.
108
108
  tables:
109
109
  - ./table.yaml
110
110
  fullmap: /path/to/fullmap
111
+ rig:
112
+ source_info:
113
+ infores_id: infores:my-kg
114
+ terms_of_use_info:
115
+ terms_of_use_url: https://example.org/terms
116
+ data_access_locations:
117
+ - My source downloads - https://example.org/downloads
118
+ source_status: maintained_regular_updates
119
+ ingest_info:
120
+ utility: Gene-disease associations support Translator disease-mechanism queries.
121
+ scope: Gene-disease associations extracted from tabular sources.
122
+ provenance_info:
123
+ contributions:
124
+ - "Author Name - code author, data modeling"
125
+ artifact_base_url: https://example.org/my-kg
126
+ artifact_base_path: ./published/my-kg
111
127
  ```
112
128
 
113
129
  Build the knowledge graph:
@@ -44,15 +44,31 @@ template:
44
44
  - { annotation: supporting_study_size, method: column, encoding: D }
45
45
  ```
46
46
 
47
- Wrap it in a graph config (`graph.yaml`) pointing at your fullmap entity-resolution database:
47
+ Wrap it in a graph config (`graph.yaml`) pointing at your fullmap entity-resolution database
48
+ and carrying the required `rig:` metadata for the generated Resource Ingest Guide:
48
49
 
49
50
  ```yaml
50
51
  name: MY_KG
51
52
  version: 1.0.0
52
- description: Gene–disease associations extracted from tabular sources.
53
53
  tables:
54
54
  - ./table.yaml
55
55
  fullmap: /path/to/fullmap
56
+ rig:
57
+ source_info:
58
+ infores_id: infores:my-kg
59
+ terms_of_use_info:
60
+ terms_of_use_url: https://example.org/terms
61
+ data_access_locations:
62
+ - My source downloads - https://example.org/downloads
63
+ source_status: maintained_regular_updates
64
+ ingest_info:
65
+ utility: Gene-disease associations support Translator disease-mechanism queries.
66
+ scope: Gene-disease associations extracted from tabular sources.
67
+ provenance_info:
68
+ contributions:
69
+ - "Author Name - code author, data modeling"
70
+ artifact_base_url: https://example.org/my-kg
71
+ artifact_base_path: ./published/my-kg
56
72
  ```
57
73
 
58
74
  Build the knowledge graph:
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "tablassert"
3
- version = "10.0.0"
3
+ version = "11.0.0"
4
4
  description = "Extract knowledge assertions from tabular data into NCATS Translator-compliant KGX NDJSON — declaratively, with entity resolution and quality control built in."
5
5
  authors = [
6
6
  { name = "Skye Lane Goetz", email = "sgoetz@isbscience.org" }
@@ -96,17 +96,35 @@ optimize = [
96
96
  ]
97
97
 
98
98
  [dependency-groups]
99
- dev = [
100
- "mkdocs>=1.6.1",
101
- "mkdocs-material>=9.6.0",
99
+ test = [
102
100
  "maturin>=1.10,<2.0",
103
- "pre-commit>=4.5.1",
104
- "pyright>=1.1.411",
105
101
  "pytest>=9.0.2",
106
102
  "pytest-cov>=7.1.0",
107
- "ruff>=0.15.6",
108
103
  "pytest-xdist>=3.8.0",
109
104
  ]
105
+ typecheck = [
106
+ "pyright>=1.1.411",
107
+ ]
108
+ docs = [
109
+ "mkdocs>=1.6.1",
110
+ "mkdocs-material>=9.6.0",
111
+ ]
112
+ # The EXACT set both CI Python jobs install. Keeping `python-type` and `python-test` on one
113
+ # identical sync is deliberate: `astral-sh/setup-uv` keys its cache on the lockfile, so jobs
114
+ # with different dependency sets race to save a cache the others then restore and miss on.
115
+ # One shared set means one genuinely warm cache. Excludes docs/pre-commit/ruff, which no CI
116
+ # job that syncs needs (lint runs ruff standalone; docs uses `--group docs`).
117
+ ci = [
118
+ { include-group = "test" },
119
+ { include-group = "typecheck" },
120
+ ]
121
+ # Superset for local development, so `make setup` still installs everything.
122
+ dev = [
123
+ { include-group = "ci" },
124
+ { include-group = "docs" },
125
+ "pre-commit>=4.5.1",
126
+ "ruff>=0.15.6",
127
+ ]
110
128
 
111
129
  [tool.pytest.ini_options]
112
130
  testpaths = ["tests"]
@@ -1313,12 +1313,34 @@ def build_and_audit(
1313
1313
  table_cfg: dict[str, object] = data if ("template" in data or "sections" in data) else {"template": data}
1314
1314
 
1315
1315
  (root / "table.yaml").write_text(yaml.safe_dump(table_cfg, sort_keys=False))
1316
+ # The measurement build still emits a RIG (every build does), so it carries an
1317
+ # honest minimal rig block: the agent only mines PMC open-access tables, and the
1318
+ # artifacts live in this throwaway workdir (file:// base = unpublished).
1319
+ resolved_root: str = str(root.resolve())
1316
1320
  graph_cfg: dict[str, object] = {
1317
1321
  "name": name,
1318
1322
  "version": version,
1319
- "description": f"Agent-built graph for {name}",
1320
1323
  "tables": ["table.yaml"], # relative to workdir (the pipelines chdir there)
1321
1324
  "fullmap": str(fullmap),
1325
+ "rig": {
1326
+ "source_info": {
1327
+ "infores_id": f"infores:{name.lower().replace('_', '-')}",
1328
+ "name": f"Agent-built measurement graph {name}",
1329
+ "terms_of_use_info": {
1330
+ "terms_of_use_url": "https://pmc.ncbi.nlm.nih.gov/about/copyright/",
1331
+ "terms_of_use_description": "PubMed Central open-access supplementary table; individual article licenses apply.",
1332
+ },
1333
+ "data_access_locations": ["PubMed Central - https://pmc.ncbi.nlm.nih.gov/"],
1334
+ "source_status": "unknown",
1335
+ },
1336
+ "ingest_info": {
1337
+ "utility": f"Transient measurement graph used to score agent-derived table configs for {name}.",
1338
+ "scope": "Associations mined from one PMC supplementary table config under audit.",
1339
+ },
1340
+ "provenance_info": {"contributions": ["Tablassert agent: automated config derivation and measurement build"]},
1341
+ "artifact_base_url": f"file://{resolved_root}",
1342
+ "artifact_base_path": resolved_root,
1343
+ },
1322
1344
  }
1323
1345
  (root / "graph.yaml").write_text(yaml.safe_dump(graph_cfg, sort_keys=False))
1324
1346
 
@@ -2153,9 +2175,12 @@ section choose column-letter encodings for entity columns and literal CURIEs for
2153
2175
  pick a predicate the subject/object pair actually permits (see BIOLINK MODELING below); add
2154
2176
  statistical annotations (p_value / effect_size / effect_type) when that table has them —
2155
2177
  method: column for table-provided columns, method: value for a fixed valid value (e.g.
2156
- effect_type: spearmans_rho when every row is a Spearman correlation). Emit effect_type ONLY
2157
- alongside an effect_size annotation: the pipeline nulls an effect_type without a numeric
2158
- effect_size. A single-table article is still ONE config with ONE section.
2178
+ effect_type: spearmans_rho when every row is a Spearman correlation). effect_size and effect_type
2179
+ are MANDATORY AS A PAIR: either one without the other is a hard validation error that bounces your
2180
+ final answer, so a table with an effect-size column also needs its effect_type (method: value when
2181
+ every row shares one statistic). Alias spellings count — `odds ratio` and the legacy
2182
+ `relationship_strength` both coerce to effect_size. A single-table article is still ONE config with
2183
+ ONE section.
2159
2184
 
2160
2185
  # BIOLINK MODELING (the pipeline enforces these SILENTLY — violating them costs you score)
2161
2186
  The build derives each edge's association CLASS from the (subject category, object category)
@@ -121,13 +121,15 @@ def build_pipeline(
121
121
  """Build a knowledge graph from a YAML configuration file.
122
122
 
123
123
  Runs the six-stage build pipeline: load tables → extract sections → build
124
- Tcodes → collect instructions → build subgraphs → compile graph.
124
+ Tcodes → collect instructions → build subgraphs → compile graph. With ``qc``
125
+ enabled a seventh stage studies the final NDJSON files.
125
126
 
126
127
  Args:
127
128
  configuration_file: Path to the graph YAML file.
128
129
  progress: Pipeline progress reporter.
129
130
  release: When ``True``, emit release-mode artifacts.
130
- qc: When ``True``, run quality-control audits on each section.
131
+ qc: When ``True``, run quality-control audits on each section and assert
132
+ over the final NDJSON files (failing the build on any violation).
131
133
  log: When ``True``, enable per-section verbose logging.
132
134
  head: When ``True``, preview a random sample of up to 5 rows per section (fast schema/shape check).
133
135
 
@@ -166,6 +168,12 @@ def build_pipeline(
166
168
  advance()
167
169
  sections: list[dict[str, Any]] = list(chain.from_iterable(temp))
168
170
  n: int = len(sections)
171
+ # Per-section source descriptors for the generated RIG's relevant-file cross-check:
172
+ # each entry records the section's local file name and its validated source URLs.
173
+ section_sources: list[dict[str, Any]] = []
174
+ for s in sections:
175
+ src: dict[str, Any] = s.get("source") or {}
176
+ section_sources.append({"local": str(src.get("local") or ""), "urls": [str(u) for u in src.get("url") or []]})
169
177
 
170
178
  # Stage 3/6: build Tcode.
171
179
  progress.stage("Building TCode")
@@ -179,7 +187,16 @@ def build_pipeline(
179
187
  try:
180
188
  tcode.append(
181
189
  Tcode.model_validate(
182
- {**s, "store": store, "log": log, "qc": qc, "release": release, "head": head, "name": g.name, "infores": g.infores}
190
+ {
191
+ **s,
192
+ "store": store,
193
+ "log": log,
194
+ "qc": qc,
195
+ "release": release,
196
+ "head": head,
197
+ "name": g.name,
198
+ "infores": g.rig.source_info.infores_id,
199
+ }
183
200
  )
184
201
  )
185
202
  except pydantic.ValidationError as e:
@@ -214,13 +231,39 @@ def build_pipeline(
214
231
  start(f"{g.name} · v{g.version}")
215
232
  # on_phase drives the phase tag (scan → normalize → write-nodes → write-edges → dedup → rig);
216
233
  # on_subgraph ticks the bar once per subgraph, so the total is len(subgraphs).
217
- compile_graph(
218
- subgraphs, g.name, g.version, g.description, g.contributions, g.ui_explanation, g.tables, g.infores, on_phase=sub_step, on_subgraph=advance
219
- )
234
+ compile_graph(subgraphs, g.name, g.version, g.rig, section_sources, on_phase=sub_step, on_subgraph=advance)
235
+
236
+ # Stage 7/7 (only with --qc): assert over the final NDJSON files.
237
+ if qc:
238
+ progress.stage("Studying Graph")
239
+ study_final_ndjson(g.name, g.version, Path(g.rig.artifact_base_path))
220
240
 
221
241
  logger.info("Built graph {name} v{version}: {n} sections", name=g.name, version=g.version, n=n)
222
242
 
223
243
 
244
+ def study_final_ndjson(name: str, version: str, out_dir: Path) -> None:
245
+ """Run study assertions over a build's final NDJSON files (the ``--qc`` stage 7).
246
+
247
+ Args:
248
+ name: Graph name, used to locate ``<name>_<version>.nodes.ndjson``.
249
+ version: Graph version, used to locate ``<name>_<version>.edges.ndjson``.
250
+ out_dir: Artifact directory the build wrote into
251
+ (``rig.artifact_base_path``).
252
+
253
+ Raises:
254
+ SystemExit: With status 1 when any study assertion is violated.
255
+ """
256
+ from tablassert.study import format_violations, study_kgx
257
+
258
+ violations = study_kgx(out_dir / f"{name}_{version}.nodes.ndjson", out_dir / f"{name}_{version}.edges.ndjson")
259
+ if violations:
260
+ summary: str = format_violations(violations)
261
+ print(summary, file=sys.stderr)
262
+ logger.warning("study assertions failed on final NDJSON:\n{summary}", summary=summary)
263
+ raise SystemExit(1)
264
+ logger.info("study assertions passed on final NDJSON")
265
+
266
+
224
267
  def validate_pipeline(table_configuration_file: Path, progress: PipelineProgress) -> None:
225
268
  """Validate section syntax from a YAML configuration file.
226
269
 
@@ -599,11 +642,14 @@ def build_kg(
599
642
 
600
643
  ``--qc`` requires the ``[qc]`` extra (``pip install "tablassert[qc]"``); it is
601
644
  checked before the build starts, because the audit stage runs LAST and a missing
602
- extra would otherwise surface only after entity resolution has finished.
645
+ extra would otherwise surface only after entity resolution has finished. It also
646
+ runs a final study stage that asserts over the emitted NDJSON -- no duplicate node
647
+ ids, no undeclared or isolated nodes, no malformed lines or stray whitespace --
648
+ and fails the build (non-zero exit) when any assertion is violated.
603
649
  """
604
650
  if qc:
605
651
  extras.require("qc", required_by="--qc")
606
- run(6, build_pipeline, graph_configuration_file, release=release, qc=qc, log=log, head=head)
652
+ run(7 if qc else 6, build_pipeline, graph_configuration_file, release=release, qc=qc, log=log, head=head)
607
653
 
608
654
 
609
655
  @APP.command(name="validate")
@@ -950,7 +996,7 @@ def _prebuilt_fullmap_urls(babel_version: str) -> tuple[str, str]:
950
996
 
951
997
  RENCI publishes a prebuilt ``fullmap.tar.zst`` (and a ``sha256sum.txt``) under
952
998
  ``{BABEL_BASE}/{babel_version}/fullmap/{tablassert_version}/``, where the version
953
- directory is the INSTALLED Tablassert package version (e.g. ``10.0.0``) — resolved from
999
+ directory is the INSTALLED Tablassert package version (e.g. ``10.1.0``) — resolved from
954
1000
  installed-package metadata, never hardcoded, so a new release looks itself up.
955
1001
 
956
1002
  Args:
@@ -523,6 +523,26 @@ def effect_size_target(name: str) -> str | None:
523
523
  return None
524
524
 
525
525
 
526
+ def coerced_target(name: str) -> str:
527
+ """Map a column/annotation name to the canonical name the clean phase renames it to.
528
+
529
+ Args:
530
+ name: Raw source column or annotation name.
531
+
532
+ Returns:
533
+ The canonical slot name a ``coerce_*_columns`` op would rename ``name``
534
+ to, or ``name`` unchanged when no coercion claims it.
535
+
536
+ Notes:
537
+ Classifier order mirrors the op order in ``Tcode._source_ops``:
538
+ ``coerce_pvalue_columns`` runs first, so a p/q-value alias is claimed
539
+ before the study-size and effect classifiers ever see it. Config-time
540
+ validators judge this target rather than the raw name so they see a
541
+ name exactly as the build will.
542
+ """
543
+ return pvalue_target(name) or study_size_target(name) or effect_size_target(name) or effect_type_target(name) or name
544
+
545
+
526
546
  def coerce_effect_size_columns(lf: pl.LazyFrame) -> pl.LazyFrame:
527
547
  """Rename effect-size-like columns to Biolink KGX-compliant ``effect_size`` (Biolink PR #1774).
528
548
 
@@ -0,0 +1,131 @@
1
+ """Tablassert configuration enums (non-Biolink).
2
+
3
+ These enums describe Tablassert's own configuration vocabulary (source kinds,
4
+ comparison operators, encoding/fill methods, contribution labels, and so on).
5
+
6
+ The Biolink Model vocabulary -- ``Categories``, ``Predicates``, ``Qualifiers``,
7
+ ``KnowledgeLevels``, ``AgentTypes``, ``EdgeCategories``, and
8
+ ``ALLOWED_EDGE_FIELDS`` -- is derived from the ``biolink-model`` package and lives
9
+ in :mod:`tablassert.biolink`. Import those from there.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from enum import Enum
15
+
16
+
17
+ class Tokens(str, Enum):
18
+ AUTO = "auto"
19
+ VALUES = "values"
20
+
21
+
22
+ class Repositories(str, Enum):
23
+ PUBMED_CENTRAL = "PMC"
24
+ PUBMED = "PMID"
25
+
26
+
27
+ class InformationResources(str, Enum):
28
+ PUBMED = "infores:pubmed"
29
+ PUBMED_CENTRAL = "infores:pubmed-central"
30
+
31
+
32
+ class Contributions(str, Enum):
33
+ CURATION = "curation"
34
+ VALIDATION = "validation"
35
+ TOOL = "tool"
36
+
37
+
38
+ class Comparisons(str, Enum):
39
+ GT = "gt"
40
+ GE = "ge"
41
+ LT = "lt"
42
+ LE = "le"
43
+ EQ = "eq"
44
+ NE = "ne"
45
+
46
+
47
+ class Functions(str, Enum):
48
+ COPYSIGN = "copysign"
49
+ POW = "pow"
50
+
51
+
52
+ class Files(str, Enum):
53
+ TEXT = "text"
54
+ EXCEL = "excel"
55
+
56
+
57
+ class EncodingMethods(str, Enum):
58
+ VALUE = "value"
59
+ COLUMN = "column"
60
+
61
+
62
+ class FillMethods(str, Enum):
63
+ FORWARD = "forward"
64
+ BACKWARD = "backward"
65
+ MIN = "min"
66
+ MAX = "max"
67
+ MEAN = "mean"
68
+ ZERO = "zero"
69
+ ONE = "one"
70
+
71
+
72
+ # --- Resource Ingest Guide vocabularies ------------------------------------ #
73
+ # These mirror the enums declared by the released RIG schema
74
+ # (biolink/resource-ingest-guide-schema), so a generated `.RIG.yaml` can only
75
+ # carry values the upstream validator accepts.
76
+
77
+
78
+ class SourceStatuses(str, Enum):
79
+ MAINTAINED_REGULAR_UPDATES = "maintained_regular_updates"
80
+ MAINTAINED_AS_NEEDED_UPDATES = "maintained_as_needed_updates"
81
+ NOT_MAINTAINED = "not_maintained"
82
+ UNKNOWN = "unknown"
83
+
84
+
85
+ class ProvisionMechanisms(str, Enum):
86
+ FILE_DOWNLOAD = "file_download"
87
+ API_ENDPOINT = "api_endpoint"
88
+ DATABASE_DUMP = "database_dump"
89
+ OTHER = "other"
90
+
91
+
92
+ class DataFormats(str, Enum):
93
+ TSV = "tsv"
94
+ XML = "xml"
95
+ CSV = "csv"
96
+ JSON = "json"
97
+ YAML = "yaml"
98
+ OBO = "obo"
99
+ PROTOBUFF = "protobuff"
100
+ KGX = "kgx"
101
+ MYSQL = "mysql"
102
+ POSTGRESQL = "postgresql"
103
+ SQLITE = "sqlite"
104
+ OTHER = "other"
105
+
106
+
107
+ class IngestCategories(str, Enum):
108
+ PRIMARY_KNOWLEDGE_PROVIDER = "primary_knowledge_provider"
109
+ AGGREGATION_PROVIDER = "aggregation_provider"
110
+ AGGREGATION_INTERPRETER = "aggregation_interpreter"
111
+ SUPPORTING_DATA_PROVIDER = "supporting_data_provider"
112
+ TRANSLATOR_KNOWLEDGE_CREATOR = "translator_knowledge_creator"
113
+ ONTOLOGY_PROVIDER = "ontology_provider"
114
+ NODE_PROPERTY_ONLY_PROVIDER = "node_property_only_provider"
115
+ OTHER = "other"
116
+
117
+
118
+ class ContentCategories(str, Enum):
119
+ EDGE_CONTENT = "edge_content"
120
+ NODE_PROPERTY_CONTENT = "node_property_content"
121
+ EDGE_PROPERTY_CONTENT = "edge_property_content"
122
+ OTHER = "other"
123
+
124
+
125
+ class ModelingCategories(str, Enum):
126
+ SPOQ_PATTERN = "spoq_pattern"
127
+ PREDICATES = "predicates"
128
+ QUALIFIERS = "qualifiers"
129
+ EDGE_PROPERTIES = "edge_properties"
130
+ NODE_PROPERTIES = "node_properties"
131
+ OTHER = "other"
@@ -12,6 +12,7 @@ TablassertErrorCodes = Literal[
12
12
  "graph-validation-failed",
13
13
  "section-validation-failed",
14
14
  "babel-download-failed",
15
+ "resolve-bad-specs",
15
16
  "config-rows-and-row-slice-conflict",
16
17
  "comparison-bad-comparator-type",
17
18
  "comparison-nonnumeric-comparator",
@@ -28,10 +29,19 @@ TablassertErrorCodes = Literal[
28
29
  "encoding-list-method-removed",
29
30
  "annotation-split-by-requires-column",
30
31
  "annotation-split-by-empty",
32
+ "annotation-effect-size-without-type",
33
+ "annotation-effect-type-without-size",
31
34
  "qualifier-auto-derived",
32
35
  "qualifier-bad-value",
33
36
  "qualifier-unsatisfiable",
34
37
  "qualifier-nullable-literal",
38
+ "qualifier-duplicated",
39
+ "rig-bad-infores",
40
+ "rig-bad-artifact-url",
41
+ "rig-bad-access-location",
42
+ "rig-terms-empty",
43
+ "rig-legacy-keys",
44
+ "rig-validation-failed",
35
45
  ]
36
46
 
37
47
 
@@ -10,6 +10,7 @@ from typing import TYPE_CHECKING, Any, NamedTuple, cast
10
10
  from tablassert import rs
11
11
  from tablassert._lazy import LazyModule
12
12
  from tablassert.biolink import Categories
13
+ from tablassert.errors import TablassertError
13
14
  from tablassert.log import cat
14
15
 
15
16
  logger = cat("FULLMAP")
@@ -575,12 +576,46 @@ def resolve_batch(
575
576
 
576
577
  Returns:
577
578
  LazyFrame with resolved columns added.
579
+
580
+ Raises:
581
+ TablassertError: ``resolve-bad-specs`` when two specs share a column, or a
582
+ spec's column or its ``<col> + tag`` normalization column is absent from
583
+ the input schema. Checked schema-only (no ``collect``) before any term
584
+ extraction or redb access, so bad specs never surface mid-build as a raw
585
+ polars ``ColumnNotFoundError``.
578
586
  """
579
587
  # Each column still gets its own taxon/prioritize/avoid filtering and its own join back into lf;
580
588
  # only the redb round trip itself (rs.lookup_fullmap_terms) is pooled across columns.
581
589
  if not specs:
582
590
  return lf
583
591
 
592
+ # Fail loudly on malformed specs BEFORE term collection or any redb access. A duplicated
593
+ # spec would otherwise die mid-build in its second join_matches pass (the first pass drops
594
+ # <col> + tag, so the second spec's level-two join hits a raw polars ColumnNotFoundError),
595
+ # and a spec whose normalization columns were never produced would crash distinct().
596
+ seen: set[str] = set()
597
+ duplicates: list[str] = []
598
+ for spec in specs:
599
+ if spec.col in seen and spec.col not in duplicates:
600
+ duplicates.append(spec.col)
601
+ seen.add(spec.col)
602
+ if duplicates:
603
+ raise TablassertError(
604
+ f"resolve_batch received more than one ResolveSpec for column(s) {', '.join(repr(col) for col in duplicates)}; "
605
+ "each node column may be resolved at most once per batch.",
606
+ code="resolve-bad-specs",
607
+ )
608
+
609
+ schema: pl.Schema = lf.collect_schema()
610
+ missing: list[tuple[str, str]] = [(name, spec.col) for spec in specs for name in (spec.col, spec.col + tag) if name not in schema]
611
+ if missing:
612
+ detail: str = ", ".join(f"{name!r} (needed by the spec for {col!r})" for name, col in missing)
613
+ raise TablassertError(
614
+ f"resolve_batch specs reference column(s) absent from the input schema: {detail}. This usually means a node "
615
+ "encoding was declared twice or against the wrong column, so its normalization columns were never produced.",
616
+ code="resolve-bad-specs",
617
+ )
618
+
584
619
  terms_by_col: dict[str, pl.LazyFrame] = {spec.col: distinct(lf, spec.col, spec.col + tag) for spec in specs}
585
620
  collected_terms: dict[str, pl.DataFrame] = {col: terms.collect() for col, terms in terms_by_col.items()}
586
621
 
@@ -36,11 +36,42 @@ TMP_NAME: str = f"{GRAPH_YAML}.tmp"
36
36
  CORRUPT_PREFIX: str = f"{GRAPH_YAML}.corrupt-"
37
37
  GRAPH_NAME: str = "tablassert-agent"
38
38
  GRAPH_VERSION: str = "1"
39
- GRAPH_DESCRIPTION: str = "Aggregate graph of agent-built PMC table configs"
40
39
  #: Record statuses whose best config self-registers (both are SUCCESSFUL builds).
41
40
  REGISTERED_STATUSES: frozenset[str] = frozenset({"MAPPED", "BUILT_UNMEASURED"})
42
41
 
43
42
 
43
+ def _registry_rig(state_dir: Path) -> dict[str, Any]:
44
+ """The honest ``rig:`` block for the aggregate agent registry graph.
45
+
46
+ Every fact here is mechanical: the agent only mines PubMed Central
47
+ open-access supplementary tables, so source terms/access describe PMC, and
48
+ the artifact bases point at the state directory itself (a ``file://`` base
49
+ is a valid unpublished URI; swap it for a public https base before sending
50
+ the generated RIG anywhere).
51
+ """
52
+ resolved: str = str(state_dir.resolve())
53
+ return {
54
+ "source_info": {
55
+ "infores_id": "infores:tablassert-agent",
56
+ "name": "PubMed Central open-access supplementary tables",
57
+ "description": "Aggregate of tabular associations mined from PubMed Central open-access supplementary files by the Tablassert agent.",
58
+ "terms_of_use_info": {
59
+ "terms_of_use_url": "https://pmc.ncbi.nlm.nih.gov/about/copyright/",
60
+ "terms_of_use_description": "PubMed Central open-access subset; individual article licenses apply.",
61
+ },
62
+ "data_access_locations": ["PubMed Central - https://pmc.ncbi.nlm.nih.gov/"],
63
+ "source_status": "unknown",
64
+ },
65
+ "ingest_info": {
66
+ "utility": "Aggregates agent-derived tabular knowledge assertions for Translator-style querying.",
67
+ "scope": "All agent-built table configs registered under this state directory.",
68
+ },
69
+ "provenance_info": {"contributions": ["Tablassert agent: automated config derivation and build"]},
70
+ "artifact_base_url": f"file://{resolved}",
71
+ "artifact_base_path": resolved,
72
+ }
73
+
74
+
44
75
  @contextlib.contextmanager
45
76
  def _registry_lock(state_dir: Path) -> Iterator[None]:
46
77
  """Hold an EXCLUSIVE ``flock`` on ``<state_dir>/graph.yaml.lock`` (created if missing).
@@ -58,9 +89,9 @@ def _registry_lock(state_dir: Path) -> Iterator[None]:
58
89
  fcntl.flock(lock_file.fileno(), fcntl.LOCK_UN)
59
90
 
60
91
 
61
- def _fresh_doc() -> dict[str, Any]:
92
+ def _fresh_doc(state_dir: Path) -> dict[str, Any]:
62
93
  """A brand-new registry document (``_apply_fullmap`` fills the first-wins ``fullmap``)."""
63
- return {"name": GRAPH_NAME, "version": GRAPH_VERSION, "description": GRAPH_DESCRIPTION, "tables": []}
94
+ return {"name": GRAPH_NAME, "version": GRAPH_VERSION, "tables": [], "rig": _registry_rig(state_dir)}
64
95
 
65
96
 
66
97
  def _quarantine(state_dir: Path, reason: str) -> Path:
@@ -85,20 +116,20 @@ def _load_registry(state_dir: Path) -> dict[str, Any]:
85
116
  """
86
117
  path: Path = state_dir / GRAPH_YAML
87
118
  if not path.is_file():
88
- return _fresh_doc()
119
+ return _fresh_doc(state_dir)
89
120
  try:
90
121
  data: object = yaml.safe_load(path.read_text(encoding="utf-8"))
91
122
  except yaml.YAMLError as exc:
92
123
  _quarantine(state_dir, f"YAML parse error: {exc}")
93
- return _fresh_doc()
124
+ return _fresh_doc(state_dir)
94
125
  if not isinstance(data, dict):
95
126
  _quarantine(state_dir, f"top level is not a mapping (got {type(data).__name__})")
96
- return _fresh_doc()
127
+ return _fresh_doc(state_dir)
97
128
  try:
98
129
  Graph.model_validate(data)
99
130
  except pydantic.ValidationError as exc:
100
131
  _quarantine(state_dir, f"fails Graph.model_validate ({len(exc.errors())} error(s))")
101
- return _fresh_doc()
132
+ return _fresh_doc(state_dir)
102
133
  return data
103
134
 
104
135