tablassert 8.0.0__tar.gz → 8.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tablassert-8.0.0 → tablassert-8.0.1}/PKG-INFO +15 -30
- tablassert-8.0.1/README.md +70 -0
- {tablassert-8.0.0 → tablassert-8.0.1}/pyproject.toml +1 -1
- {tablassert-8.0.0 → tablassert-8.0.1}/src/tablassert/agent.py +7 -25
- {tablassert-8.0.0 → tablassert-8.0.1}/src/tablassert/cli.py +5 -15
- {tablassert-8.0.0 → tablassert-8.0.1}/src/tablassert/errors.py +0 -1
- {tablassert-8.0.0 → tablassert-8.0.1}/src/tablassert/lib.py +1 -1
- {tablassert-8.0.0 → tablassert-8.0.1}/src/tablassert/models.py +4 -13
- tablassert-8.0.0/README.md +0 -85
- {tablassert-8.0.0 → tablassert-8.0.1}/LICENSE +0 -0
- {tablassert-8.0.0 → tablassert-8.0.1}/rust/Cargo.lock +0 -0
- {tablassert-8.0.0 → tablassert-8.0.1}/rust/Cargo.toml +0 -0
- {tablassert-8.0.0 → tablassert-8.0.1}/rust/examples/count_tables.rs +0 -0
- {tablassert-8.0.0 → tablassert-8.0.1}/rust/src/fullmap.rs +0 -0
- {tablassert-8.0.0 → tablassert-8.0.1}/rust/src/json.rs +0 -0
- {tablassert-8.0.0 → tablassert-8.0.1}/rust/src/lib.rs +0 -0
- {tablassert-8.0.0 → tablassert-8.0.1}/rust/src/ndjson.rs +0 -0
- {tablassert-8.0.0 → tablassert-8.0.1}/rust/src/uuid.rs +0 -0
- {tablassert-8.0.0 → tablassert-8.0.1}/rust/tests/build_golden.rs +0 -0
- {tablassert-8.0.0 → tablassert-8.0.1}/src/tablassert/__init__.py +0 -0
- {tablassert-8.0.0 → tablassert-8.0.1}/src/tablassert/_lazy.py +0 -0
- {tablassert-8.0.0 → tablassert-8.0.1}/src/tablassert/biolink.py +0 -0
- {tablassert-8.0.0 → tablassert-8.0.1}/src/tablassert/coerce.py +0 -0
- {tablassert-8.0.0 → tablassert-8.0.1}/src/tablassert/enums.py +0 -0
- {tablassert-8.0.0 → tablassert-8.0.1}/src/tablassert/fullmap.py +0 -0
- {tablassert-8.0.0 → tablassert-8.0.1}/src/tablassert/ingests.py +0 -0
- {tablassert-8.0.0 → tablassert-8.0.1}/src/tablassert/log.py +0 -0
- {tablassert-8.0.0 → tablassert-8.0.1}/src/tablassert/nlp.py +0 -0
- {tablassert-8.0.0 → tablassert-8.0.1}/src/tablassert/progress.py +0 -0
- {tablassert-8.0.0 → tablassert-8.0.1}/src/tablassert/qc.py +0 -0
- {tablassert-8.0.0 → tablassert-8.0.1}/src/tablassert/rig.py +0 -0
- {tablassert-8.0.0 → tablassert-8.0.1}/src/tablassert/rs.pyi +0 -0
- {tablassert-8.0.0 → tablassert-8.0.1}/src/tablassert/utils.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: tablassert
|
|
3
|
-
Version: 8.0.
|
|
3
|
+
Version: 8.0.1
|
|
4
4
|
Classifier: License :: OSI Approved :: Apache Software License
|
|
5
5
|
Classifier: Development Status :: 5 - Production/Stable
|
|
6
6
|
Classifier: Intended Audience :: Science/Research
|
|
@@ -71,16 +71,11 @@ tablassert build-kg config.yaml
|
|
|
71
71
|
pip install tablassert
|
|
72
72
|
```
|
|
73
73
|
|
|
74
|
-
The base install
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
```
|
|
80
|
-
|
|
81
|
-
Excel (`.xlsx`) inputs are read through Polars' `calamine` engine and additionally require `python-calamine` (`pip install python-calamine`).
|
|
82
|
-
|
|
83
|
-
QC is opt-in: pass `--qc` to `build-kg` to run the three-stage audit (exact → fuzzy → BioBERT). See the [CLI Reference](https://skyeav.github.io/Tablassert/cli/) for the full flag reference.
|
|
74
|
+
The base install builds knowledge graphs from CSV/TSV/Excel sources. Optional extras (`rt`, `qc`,
|
|
75
|
+
`agent`) add CPU-compatible Polars, the three-stage QC audit, and the autonomous agent — see the
|
|
76
|
+
[Installation guide](https://skyeav.github.io/Tablassert/installation/) for the full matrix. QC is opt-in
|
|
77
|
+
at build time (`build-kg --qc`); see the [CLI Reference](https://skyeav.github.io/Tablassert/cli/) for the
|
|
78
|
+
complete flag reference.
|
|
84
79
|
|
|
85
80
|
## Quick Demo
|
|
86
81
|
|
|
@@ -88,30 +83,20 @@ QC is opt-in: pass `--qc` to `build-kg` to run the three-stage audit (exact →
|
|
|
88
83
|
from pathlib import Path
|
|
89
84
|
from tablassert.lib import resolve_many
|
|
90
85
|
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
col="gene",
|
|
94
|
-
entities=["TP53", "BRCA1", "EGFR"],
|
|
95
|
-
fullmap=Path("/path/to/fullmap"),
|
|
96
|
-
taxon="9606",
|
|
97
|
-
)
|
|
98
|
-
|
|
99
|
-
for row in results:
|
|
100
|
-
print(f"{row['original_gene']} → {row['gene']} ({row['gene_name']})")
|
|
101
|
-
# TP53 → HGNC:11998 (TP53)
|
|
102
|
-
# BRCA1 → HGNC:1100 (BRCA1)
|
|
103
|
-
# EGFR → HGNC:3236 (EGFR)
|
|
86
|
+
results = resolve_many(col="gene", entities=["TP53", "BRCA1"], fullmap=Path("/path/to/fullmap"), taxon="9606")
|
|
87
|
+
# [{"original_gene": "TP53", "gene": "HGNC:11998", "gene_name": "TP53", ...}, ...]
|
|
104
88
|
```
|
|
105
89
|
|
|
106
|
-
Point `resolve_many()` at a fullmap database
|
|
90
|
+
Point `resolve_many()` at a fullmap database to resolve any iterable of entity strings to CURIEs — no
|
|
91
|
+
LazyFrame setup or NLP preprocessing required. See the
|
|
92
|
+
[Batch Resolution API](https://skyeav.github.io/Tablassert/api/lib/) for the full reference; for
|
|
93
|
+
YAML-configured pipeline builds use `tablassert build-kg config.yaml`.
|
|
107
94
|
|
|
108
95
|
## Key Features
|
|
109
96
|
|
|
110
|
-
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
- **KGX Compliance** — NCATS Translator-compatible NDJSON output
|
|
114
|
-
- **Performance** — Lazy evaluation pipelines with Polars and an embedded redb-accelerated entity resolution database
|
|
97
|
+
Declarative YAML configs, built-in entity resolution, optional three-stage QC, and KGX-compliant NDJSON
|
|
98
|
+
output — with lazy Polars pipelines over an embedded redb resolution database. See the
|
|
99
|
+
[documentation](https://skyeav.github.io/Tablassert/) for the full feature overview and use-case gallery.
|
|
115
100
|
|
|
116
101
|
## Developing
|
|
117
102
|
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
# Tablassert
|
|
2
|
+
|
|
3
|
+
[](https://pypi.org/project/tablassert/)
|
|
4
|
+
[](https://pypi.org/project/tablassert/)
|
|
5
|
+
[](https://github.com/SkyeAv/Tablassert/blob/main/LICENSE)
|
|
6
|
+
[](https://skyeav.github.io/Tablassert/)
|
|
7
|
+
|
|
8
|
+
Extract knowledge assertions from tabular data into NCATS Translator-compliant KGX NDJSON — declaratively, with entity resolution built in and optional quality control.
|
|
9
|
+
|
|
10
|
+
```bash
|
|
11
|
+
pip install tablassert
|
|
12
|
+
tablassert build-kg config.yaml
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
**[Full Documentation](https://skyeav.github.io/Tablassert/)** — installation guides, tutorials, configuration reference, and API docs.
|
|
16
|
+
|
|
17
|
+
## Installation
|
|
18
|
+
|
|
19
|
+
```bash
|
|
20
|
+
pip install tablassert
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
The base install builds knowledge graphs from CSV/TSV/Excel sources. Optional extras (`rt`, `qc`,
|
|
24
|
+
`agent`) add CPU-compatible Polars, the three-stage QC audit, and the autonomous agent — see the
|
|
25
|
+
[Installation guide](https://skyeav.github.io/Tablassert/installation/) for the full matrix. QC is opt-in
|
|
26
|
+
at build time (`build-kg --qc`); see the [CLI Reference](https://skyeav.github.io/Tablassert/cli/) for the
|
|
27
|
+
complete flag reference.
|
|
28
|
+
|
|
29
|
+
## Quick Demo
|
|
30
|
+
|
|
31
|
+
```python
|
|
32
|
+
from pathlib import Path
|
|
33
|
+
from tablassert.lib import resolve_many
|
|
34
|
+
|
|
35
|
+
results = resolve_many(col="gene", entities=["TP53", "BRCA1"], fullmap=Path("/path/to/fullmap"), taxon="9606")
|
|
36
|
+
# [{"original_gene": "TP53", "gene": "HGNC:11998", "gene_name": "TP53", ...}, ...]
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
Point `resolve_many()` at a fullmap database to resolve any iterable of entity strings to CURIEs — no
|
|
40
|
+
LazyFrame setup or NLP preprocessing required. See the
|
|
41
|
+
[Batch Resolution API](https://skyeav.github.io/Tablassert/api/lib/) for the full reference; for
|
|
42
|
+
YAML-configured pipeline builds use `tablassert build-kg config.yaml`.
|
|
43
|
+
|
|
44
|
+
## Key Features
|
|
45
|
+
|
|
46
|
+
Declarative YAML configs, built-in entity resolution, optional three-stage QC, and KGX-compliant NDJSON
|
|
47
|
+
output — with lazy Polars pipelines over an embedded redb resolution database. See the
|
|
48
|
+
[documentation](https://skyeav.github.io/Tablassert/) for the full feature overview and use-case gallery.
|
|
49
|
+
|
|
50
|
+
## Developing
|
|
51
|
+
|
|
52
|
+
```bash
|
|
53
|
+
uv sync --group dev --extra qc
|
|
54
|
+
uv run maturin develop --manifest-path rust/Cargo.toml
|
|
55
|
+
make check
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
See **[CONTRIBUTING.md](CONTRIBUTING.md)** for the full development loop, quality gates, and pull request guidelines.
|
|
59
|
+
|
|
60
|
+
## License
|
|
61
|
+
|
|
62
|
+
[Apache License 2.0](LICENSE)
|
|
63
|
+
|
|
64
|
+
## Contributors
|
|
65
|
+
|
|
66
|
+
[Skye Lane Goetz](mailto:sgoetz@isbscience.org) — Institute for Systems Biology
|
|
67
|
+
|
|
68
|
+
[Gwênlyn Glusman](mailto:gglusman@isbscience.org) — Institute for Systems Biology
|
|
69
|
+
|
|
70
|
+
Jared C. Roach — Institute for Systems Biology
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "tablassert"
|
|
3
|
-
version = "8.0.
|
|
3
|
+
version = "8.0.1"
|
|
4
4
|
description = "Extract knowledge assertions from tabular data into NCATS Translator-compliant KGX NDJSON — declaratively, with entity resolution and quality control built in."
|
|
5
5
|
authors = [
|
|
6
6
|
{ name = "Skye Lane Goetz", email = "sgoetz@isbscience.org" }
|
|
@@ -352,8 +352,7 @@ def candidate_tables(files: list[Path]) -> list[Path]:
|
|
|
352
352
|
"""Return EVERY downloaded data-table file, raising ``FileNotFoundError`` when there is none.
|
|
353
353
|
|
|
354
354
|
The supervisor presents all candidates to the agent (which chooses among them and among Excel
|
|
355
|
-
worksheets); the fail-fast guard
|
|
356
|
-
has already run.
|
|
355
|
+
worksheets); the fail-fast guard raises when a fetch yields no data tables.
|
|
357
356
|
"""
|
|
358
357
|
tables: list[Path] = [path for path in files if is_table_file(path.name)]
|
|
359
358
|
if not tables:
|
|
@@ -1598,7 +1597,6 @@ def build_agent(
|
|
|
1598
1597
|
instructions: str = INSTRUCTIONS,
|
|
1599
1598
|
max_steps: int = 20,
|
|
1600
1599
|
planning_interval: int = 3,
|
|
1601
|
-
executor_type: str = "local",
|
|
1602
1600
|
additional_authorized_imports: list[str] | None = None,
|
|
1603
1601
|
step_callbacks: list[Callable[[object, object], None]] | None = None,
|
|
1604
1602
|
final_answer_checks: list[Callable[..., bool]] | None = None,
|
|
@@ -1613,10 +1611,7 @@ def build_agent(
|
|
|
1613
1611
|
yields an empty tool list: the supervisor builds the fullmap-bound tools (US-009) and passes
|
|
1614
1612
|
them in, since they need a fullmap this factory does not have.
|
|
1615
1613
|
|
|
1616
|
-
|
|
1617
|
-
boundary; ``executor_type="docker"`` is the HARDENED option (sandboxed executor). Pass
|
|
1618
|
-
``executor_type`` straight through (``local``/``docker``/``e2b``). ``verbosity_level`` (a
|
|
1619
|
-
smolagents ``LogLevel``) is forwarded only when not None.
|
|
1614
|
+
``verbosity_level`` (a smolagents ``LogLevel``) is forwarded only when not None.
|
|
1620
1615
|
"""
|
|
1621
1616
|
_require("smolagents")
|
|
1622
1617
|
from smolagents import CodeAgent # local import keeps module import lazy # pyright: ignore[reportMissingImports]
|
|
@@ -1634,7 +1629,7 @@ def build_agent(
|
|
|
1634
1629
|
"additional_authorized_imports": imports,
|
|
1635
1630
|
"step_callbacks": callbacks,
|
|
1636
1631
|
"final_answer_checks": checks,
|
|
1637
|
-
"executor_type":
|
|
1632
|
+
"executor_type": "local",
|
|
1638
1633
|
}
|
|
1639
1634
|
if verbosity_level is not None:
|
|
1640
1635
|
agent_kwargs["verbosity_level"] = verbosity_level
|
|
@@ -1975,14 +1970,11 @@ def run_supervisor(
|
|
|
1975
1970
|
*,
|
|
1976
1971
|
fullmap: Path,
|
|
1977
1972
|
build_model_factory: Callable[[], object],
|
|
1978
|
-
map_threshold: float = 0.
|
|
1979
|
-
qc_threshold: float = 0.9,
|
|
1973
|
+
map_threshold: float = 0.25,
|
|
1980
1974
|
max_improve_iters: int = 3,
|
|
1981
1975
|
max_steps: int = 20,
|
|
1982
|
-
state_dir: Path = Path(".tablassert
|
|
1983
|
-
executor: str = "local",
|
|
1976
|
+
state_dir: Path = Path(".tablassert") / "agent",
|
|
1984
1977
|
workdir: Path | None = None,
|
|
1985
|
-
fetch: bool = True,
|
|
1986
1978
|
name: str = "agent",
|
|
1987
1979
|
version: str = "0.0.1",
|
|
1988
1980
|
) -> dict[str, object]:
|
|
@@ -1990,8 +1982,7 @@ def run_supervisor(
|
|
|
1990
1982
|
|
|
1991
1983
|
For each pmc id (resume-aware: terminal DONE/MAPPED/SKIPPED records are skipped):
|
|
1992
1984
|
1. mark RUNNING + checkpoint; fetch the latest-version article payload (``fetch_pmc_article``, the
|
|
1993
|
-
single seam tests monkeypatch
|
|
1994
|
-
candidate tables + the main-text path to the agent;
|
|
1985
|
+
single seam tests monkeypatch) and present ALL candidate tables + the main-text path to the agent;
|
|
1995
1986
|
2. run the INNER agent (``build_agent`` + ``build_model_factory()``) whose schema-gated
|
|
1996
1987
|
final answer is the initial Section config;
|
|
1997
1988
|
3. ``build_and_audit`` it for coverage, then run the deterministic IMPROVE loop
|
|
@@ -2036,14 +2027,7 @@ def run_supervisor(
|
|
|
2036
2027
|
rec.attempts += 1
|
|
2037
2028
|
save_state(state_dir, state)
|
|
2038
2029
|
|
|
2039
|
-
files: list[Path]
|
|
2040
|
-
if fetch:
|
|
2041
|
-
files = fetch_pmc_article(pmc_id, pmc_download_dir(art_root, pmc_id))
|
|
2042
|
-
else: # --no-fetch: resolve an already-fetched snapshot (the `fetch` param was previously dead)
|
|
2043
|
-
snapshot: Path = pmc_download_dir(art_root, pmc_id)
|
|
2044
|
-
files = sorted(path for path in snapshot.rglob("*") if path.is_file())
|
|
2045
|
-
if not files:
|
|
2046
|
-
raise FileNotFoundError(f"--no-fetch but no snapshot files under {snapshot}.")
|
|
2030
|
+
files: list[Path] = fetch_pmc_article(pmc_id, pmc_download_dir(art_root, pmc_id))
|
|
2047
2031
|
tables: list[Path] = candidate_tables(files)
|
|
2048
2032
|
table_list: str = "\n".join(f" - {path}" for path in tables)
|
|
2049
2033
|
article_xml: Path | None = next((path for path in files if path.suffix.lower() in {".xml", ".nxml"}), None)
|
|
@@ -2053,7 +2037,6 @@ def run_supervisor(
|
|
|
2053
2037
|
model=build_model_factory(),
|
|
2054
2038
|
tools=make_tools(fullmap=fullmap, table_path=tables[0], name=name, version=version),
|
|
2055
2039
|
max_steps=max_steps,
|
|
2056
|
-
executor_type=executor,
|
|
2057
2040
|
step_callbacks=[make_step_callback(metrics)],
|
|
2058
2041
|
verbosity_level=verbosity,
|
|
2059
2042
|
)
|
|
@@ -2159,7 +2142,6 @@ def run_supervisor(
|
|
|
2159
2142
|
|
|
2160
2143
|
state.metrics = {
|
|
2161
2144
|
"map_threshold": map_threshold,
|
|
2162
|
-
"qc_threshold": qc_threshold,
|
|
2163
2145
|
"mapped": mapped,
|
|
2164
2146
|
"skipped": skipped,
|
|
2165
2147
|
"mean_best_coverage": mean_best,
|
|
@@ -492,13 +492,13 @@ def _download_detail(downloaded: int, total: int) -> str:
|
|
|
492
492
|
|
|
493
493
|
@APP.command(name="build-kg")
|
|
494
494
|
def build_kg(
|
|
495
|
-
configuration_file: Path,
|
|
495
|
+
configuration_file: Annotated[Path, cyclopts.Parameter(name=["--configuration-file", "-f"])],
|
|
496
496
|
release: Annotated[bool, cyclopts.Parameter(name=["--release", "-r"], negative="")] = False,
|
|
497
497
|
qc: Annotated[bool, cyclopts.Parameter(name=["--qc", "-q"], negative="")] = False,
|
|
498
498
|
log: Annotated[bool, cyclopts.Parameter(name=["--log", "-l"], negative="")] = False,
|
|
499
499
|
head: Annotated[bool, cyclopts.Parameter(name=["--head", "-hd"], negative="")] = False,
|
|
500
500
|
table_config: Annotated[bool, cyclopts.Parameter(name=["--table-config", "-tc"], negative="")] = False,
|
|
501
|
-
fullmap: Annotated[Path, cyclopts.Parameter(name=["--fullmap", "-
|
|
501
|
+
fullmap: Annotated[Path, cyclopts.Parameter(name=["--fullmap", "-fm"])] = Path("./fullmap"),
|
|
502
502
|
) -> None:
|
|
503
503
|
"""Build a knowledge graph from a YAML configuration file.
|
|
504
504
|
|
|
@@ -536,13 +536,10 @@ def agent(
|
|
|
536
536
|
api_base: Annotated[str | None, cyclopts.Parameter(name=["--api-base", "-ab"])] = None,
|
|
537
537
|
api_key: Annotated[str | None, cyclopts.Parameter(name=["--api-key", "-ak"])] = None,
|
|
538
538
|
max_steps: Annotated[int, cyclopts.Parameter(name=["--max-steps", "-ms"])] = 20,
|
|
539
|
-
map_threshold: Annotated[float, cyclopts.Parameter(name=["--map-threshold", "-mt"])] = 0.
|
|
540
|
-
qc_threshold: Annotated[float, cyclopts.Parameter(name=["--qc-threshold", "-qt"])] = 0.9,
|
|
539
|
+
map_threshold: Annotated[float, cyclopts.Parameter(name=["--map-threshold", "-mt"])] = 0.25,
|
|
541
540
|
max_improve_iters: Annotated[int, cyclopts.Parameter(name=["--max-improve-iters", "-mi"])] = 3,
|
|
542
|
-
state_dir: Annotated[Path, cyclopts.Parameter(name=["--state-dir", "-sd"])] = Path(".tablassert
|
|
543
|
-
executor: Annotated[Literal["local", "docker"], cyclopts.Parameter(name=["--executor", "-e"])] = "local",
|
|
541
|
+
state_dir: Annotated[Path, cyclopts.Parameter(name=["--state-dir", "-sd"])] = Path(".tablassert") / "agent",
|
|
544
542
|
backend: Annotated[Literal["openai", "litellm"], cyclopts.Parameter(name=["--backend", "-b"])] = "openai",
|
|
545
|
-
no_fetch: Annotated[bool, cyclopts.Parameter(name=["--no-fetch", "-nf"], negative="")] = False,
|
|
546
543
|
) -> None:
|
|
547
544
|
"""Autonomously derive, build, audit, and improve KG configs from PMC articles.
|
|
548
545
|
|
|
@@ -556,8 +553,7 @@ def agent(
|
|
|
556
553
|
Model config comes from ``--model-id``/``--api-base``/``--api-key`` OR the ``TABLASSERT_AGENT_MODEL_ID``
|
|
557
554
|
/ ``TABLASSERT_AGENT_API_BASE`` / ``TABLASSERT_AGENT_API_KEY`` environment variables (explicit flags win).
|
|
558
555
|
Secrets are NEVER hardcoded or defaulted: a missing value fails loud (exit 2) BEFORE any model is built.
|
|
559
|
-
|
|
560
|
-
not a security boundary). Requires the ``[agent]`` extra (``pip install tablassert[agent]``).
|
|
556
|
+
Requires the ``[agent]`` extra (``pip install tablassert[agent]``).
|
|
561
557
|
|
|
562
558
|
Args:
|
|
563
559
|
pmc_ids: One or more PMC article ids (positional).
|
|
@@ -567,12 +563,9 @@ def agent(
|
|
|
567
563
|
api_key: API key (falls back to ``TABLASSERT_AGENT_API_KEY``).
|
|
568
564
|
max_steps: Max inner-agent steps per article.
|
|
569
565
|
map_threshold: Coverage an article must reach to be MAPPED.
|
|
570
|
-
qc_threshold: Target QC pass rate.
|
|
571
566
|
max_improve_iters: Max deterministic improve iterations per article.
|
|
572
567
|
state_dir: Checkpoint/resume directory.
|
|
573
|
-
executor: Code-execution backend; ``docker`` is the hardened sandbox.
|
|
574
568
|
backend: Model backend (``openai`` or ``litellm``).
|
|
575
|
-
no_fetch: Skip PMC download (use already-fetched snapshot tables).
|
|
576
569
|
"""
|
|
577
570
|
from tablassert import agent as agent_mod
|
|
578
571
|
|
|
@@ -596,12 +589,9 @@ def agent(
|
|
|
596
589
|
fullmap=fullmap,
|
|
597
590
|
build_model_factory=build_model_factory,
|
|
598
591
|
map_threshold=map_threshold,
|
|
599
|
-
qc_threshold=qc_threshold,
|
|
600
592
|
max_improve_iters=max_improve_iters,
|
|
601
593
|
max_steps=max_steps,
|
|
602
594
|
state_dir=state_dir,
|
|
603
|
-
executor=executor,
|
|
604
|
-
fetch=not no_fetch,
|
|
605
595
|
)
|
|
606
596
|
|
|
607
597
|
metrics_raw: object = result.get("metrics")
|
|
@@ -712,7 +712,7 @@ class Tcode(Section):
|
|
|
712
712
|
then the trim/format/write finalize ops.
|
|
713
713
|
"""
|
|
714
714
|
override = self.provenance.override
|
|
715
|
-
primary_knowledge_source
|
|
715
|
+
primary_knowledge_source: str | None = self.infores or (infores(self.name) if self.name else None)
|
|
716
716
|
upstream_ids = override.upstream_resource_ids if override else upstream_resource_ids(self.provenance.repo)
|
|
717
717
|
knowledge_level = override.knowledge_level if override else self.provenance.knowledge_level
|
|
718
718
|
agent_type = override.agent_type if override else self.provenance.agent_type
|
|
@@ -288,16 +288,14 @@ class ManualProvenance(TablaBase):
|
|
|
288
288
|
|
|
289
289
|
When present under :class:`Provenance`, these values replace the legacy
|
|
290
290
|
repo/publication-derived provenance while keeping the same KL/AT defaults.
|
|
291
|
+
The edge ``primary_knowledge_source`` always derives from the graph-level
|
|
292
|
+
``infores`` (or ``infores:<graph-name>``); manual infores CURIEs belong in
|
|
293
|
+
``upstream_resource_ids``.
|
|
291
294
|
"""
|
|
292
295
|
|
|
293
|
-
infores: str | None = Field(
|
|
294
|
-
None,
|
|
295
|
-
description="Optional per-section primary knowledge source infores CURIE; defaults to the graph infores when omitted.",
|
|
296
|
-
examples=["infores:my-kg"],
|
|
297
|
-
)
|
|
298
296
|
upstream_resource_ids: list[str] = Field(
|
|
299
297
|
default_factory=list,
|
|
300
|
-
description="
|
|
298
|
+
description="Manual upstream source infores CURIEs emitted instead of the repo-derived source map; the sanctioned place for manual infores.",
|
|
301
299
|
examples=[["infores:my-upstream"]],
|
|
302
300
|
)
|
|
303
301
|
publications: list[str] | None = Field(
|
|
@@ -310,13 +308,6 @@ class ManualProvenance(TablaBase):
|
|
|
310
308
|
)
|
|
311
309
|
agent_type: AgentTypes = Field(AgentTypes.DATA_ANALYSIS_PIPELINE, description="Biolink KL/AT agent type responsible for produced edges.")
|
|
312
310
|
|
|
313
|
-
@field_validator("infores", mode="after")
|
|
314
|
-
@classmethod
|
|
315
|
-
def infores_curie(cls, infores: str | None) -> str | None:
|
|
316
|
-
if infores is None:
|
|
317
|
-
return None
|
|
318
|
-
return validate_infores_curie(infores, "override-bad-infores")
|
|
319
|
-
|
|
320
311
|
@field_validator("upstream_resource_ids", mode="after")
|
|
321
312
|
@classmethod
|
|
322
313
|
def upstream_infores_curies(cls, values: list[str]) -> list[str]:
|
tablassert-8.0.0/README.md
DELETED
|
@@ -1,85 +0,0 @@
|
|
|
1
|
-
# Tablassert
|
|
2
|
-
|
|
3
|
-
[](https://pypi.org/project/tablassert/)
|
|
4
|
-
[](https://pypi.org/project/tablassert/)
|
|
5
|
-
[](https://github.com/SkyeAv/Tablassert/blob/main/LICENSE)
|
|
6
|
-
[](https://skyeav.github.io/Tablassert/)
|
|
7
|
-
|
|
8
|
-
Extract knowledge assertions from tabular data into NCATS Translator-compliant KGX NDJSON — declaratively, with entity resolution built in and optional quality control.
|
|
9
|
-
|
|
10
|
-
```bash
|
|
11
|
-
pip install tablassert
|
|
12
|
-
tablassert build-kg config.yaml
|
|
13
|
-
```
|
|
14
|
-
|
|
15
|
-
**[Full Documentation](https://skyeav.github.io/Tablassert/)** — installation guides, tutorials, configuration reference, and API docs.
|
|
16
|
-
|
|
17
|
-
## Installation
|
|
18
|
-
|
|
19
|
-
```bash
|
|
20
|
-
pip install tablassert
|
|
21
|
-
```
|
|
22
|
-
|
|
23
|
-
The base install includes everything needed to build knowledge graphs from CSV/TSV sources. Optional extras are available for CPU compatibility and quality control:
|
|
24
|
-
|
|
25
|
-
```bash
|
|
26
|
-
pip install "tablassert[rt]" # Polars build for CPUs without the required instructions
|
|
27
|
-
pip install "tablassert[qc]" # Enable QC (torch + sentence-transformers BioBERT, scikit-learn)
|
|
28
|
-
```
|
|
29
|
-
|
|
30
|
-
Excel (`.xlsx`) inputs are read through Polars' `calamine` engine and additionally require `python-calamine` (`pip install python-calamine`).
|
|
31
|
-
|
|
32
|
-
QC is opt-in: pass `--qc` to `build-kg` to run the three-stage audit (exact → fuzzy → BioBERT). See the [CLI Reference](https://skyeav.github.io/Tablassert/cli/) for the full flag reference.
|
|
33
|
-
|
|
34
|
-
## Quick Demo
|
|
35
|
-
|
|
36
|
-
```python
|
|
37
|
-
from pathlib import Path
|
|
38
|
-
from tablassert.lib import resolve_many
|
|
39
|
-
|
|
40
|
-
# Resolve gene names to CURIEs against a fullmap database
|
|
41
|
-
results = resolve_many(
|
|
42
|
-
col="gene",
|
|
43
|
-
entities=["TP53", "BRCA1", "EGFR"],
|
|
44
|
-
fullmap=Path("/path/to/fullmap"),
|
|
45
|
-
taxon="9606",
|
|
46
|
-
)
|
|
47
|
-
|
|
48
|
-
for row in results:
|
|
49
|
-
print(f"{row['original_gene']} → {row['gene']} ({row['gene_name']})")
|
|
50
|
-
# TP53 → HGNC:11998 (TP53)
|
|
51
|
-
# BRCA1 → HGNC:1100 (BRCA1)
|
|
52
|
-
# EGFR → HGNC:3236 (EGFR)
|
|
53
|
-
```
|
|
54
|
-
|
|
55
|
-
Point `resolve_many()` at a fullmap database and resolve any iterable of entity strings to CURIEs — no LazyFrame setup or NLP preprocessing required. For full pipeline builds with YAML configuration, use `tablassert build-kg config.yaml`.
|
|
56
|
-
|
|
57
|
-
## Key Features
|
|
58
|
-
|
|
59
|
-
- **Declarative Configuration** — YAML-based, no code required
|
|
60
|
-
- **Entity Resolution** — Maps text to biological entities (genes, diseases, chemicals)
|
|
61
|
-
- **Quality Control** — Optional three-stage validation (exact → fuzzy → BERT embeddings)
|
|
62
|
-
- **KGX Compliance** — NCATS Translator-compatible NDJSON output
|
|
63
|
-
- **Performance** — Lazy evaluation pipelines with Polars and an embedded redb-accelerated entity resolution database
|
|
64
|
-
|
|
65
|
-
## Developing
|
|
66
|
-
|
|
67
|
-
```bash
|
|
68
|
-
uv sync --group dev --extra qc
|
|
69
|
-
uv run maturin develop --manifest-path rust/Cargo.toml
|
|
70
|
-
make check
|
|
71
|
-
```
|
|
72
|
-
|
|
73
|
-
See **[CONTRIBUTING.md](CONTRIBUTING.md)** for the full development loop, quality gates, and pull request guidelines.
|
|
74
|
-
|
|
75
|
-
## License
|
|
76
|
-
|
|
77
|
-
[Apache License 2.0](LICENSE)
|
|
78
|
-
|
|
79
|
-
## Contributors
|
|
80
|
-
|
|
81
|
-
[Skye Lane Goetz](mailto:sgoetz@isbscience.org) — Institute for Systems Biology
|
|
82
|
-
|
|
83
|
-
[Gwênlyn Glusman](mailto:gglusman@isbscience.org) — Institute for Systems Biology
|
|
84
|
-
|
|
85
|
-
Jared C. Roach — Institute for Systems Biology
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|