file2records 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- file2records-0.2.0/.gitignore +29 -0
- file2records-0.2.0/CITATION.cff +26 -0
- file2records-0.2.0/LICENSE +21 -0
- file2records-0.2.0/PKG-INFO +75 -0
- file2records-0.2.0/README.md +38 -0
- file2records-0.2.0/pyproject.toml +70 -0
- file2records-0.2.0/src/file2records/__init__.py +15 -0
- file2records-0.2.0/src/file2records/__main__.py +5 -0
- file2records-0.2.0/src/file2records/bundle.py +155 -0
- file2records-0.2.0/src/file2records/cli.py +207 -0
- file2records-0.2.0/src/file2records/config.py +96 -0
- file2records-0.2.0/src/file2records/demo/NOTICE.md +34 -0
- file2records-0.2.0/src/file2records/demo/config/extract_prompt.txt +51 -0
- file2records-0.2.0/src/file2records/demo/config/few_shot.json +1 -0
- file2records-0.2.0/src/file2records/demo/config/judge_prompt.txt +103 -0
- file2records-0.2.0/src/file2records/demo/config/schema.json +54 -0
- file2records-0.2.0/src/file2records/demo/config/settings.json +3 -0
- file2records-0.2.0/src/file2records/demo/extracted/lewis-acidic-ionic-liquids-for-pet-glycolysis-polymers-2013-bc017719.json +268 -0
- file2records-0.2.0/src/file2records/demo/judged/lewis-acidic-ionic-liquids-for-pet-glycolysis-polymers-2013-bc017719.json +517 -0
- file2records-0.2.0/src/file2records/demo/parsed/lewis-acidic-ionic-liquids-for-pet-glycolysis-polymers-2013-bc017719.json +408 -0
- file2records-0.2.0/src/file2records/demo/pdfs/lewis-acidic-ionic-liquids-for-pet-glycolysis-polymers-2013-bc017719.pdf +0 -0
- file2records-0.2.0/src/file2records/demo.py +155 -0
- file2records-0.2.0/src/file2records/dynschema.py +37 -0
- file2records-0.2.0/src/file2records/exemplar.py +171 -0
- file2records-0.2.0/src/file2records/extraction.py +40 -0
- file2records-0.2.0/src/file2records/filters.py +81 -0
- file2records-0.2.0/src/file2records/judge.py +94 -0
- file2records-0.2.0/src/file2records/llm.py +79 -0
- file2records-0.2.0/src/file2records/main.py +768 -0
- file2records-0.2.0/src/file2records/models.py +105 -0
- file2records-0.2.0/src/file2records/parsing.py +129 -0
- file2records-0.2.0/src/file2records/pipeline.py +206 -0
- file2records-0.2.0/src/file2records/project.py +203 -0
- file2records-0.2.0/src/file2records/readers.py +385 -0
- file2records-0.2.0/src/file2records/report.py +230 -0
- file2records-0.2.0/src/file2records/storage.py +132 -0
- file2records-0.2.0/src/file2records/timings.py +50 -0
- file2records-0.2.0/src/file2records/web/app.js +2340 -0
- file2records-0.2.0/src/file2records/web/index.html +28 -0
- file2records-0.2.0/src/file2records/web/style.css +483 -0
- file2records-0.2.0/tests/test_docs.py +31 -0
- file2records-0.2.0/tests/test_library.py +330 -0
- file2records-0.2.0/tests/test_smoke.py +665 -0
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
.venv/
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.pyc
|
|
4
|
+
.env
|
|
5
|
+
workspace/pdfs/*
|
|
6
|
+
workspace/parsed/*
|
|
7
|
+
workspace/extracted/*
|
|
8
|
+
workspace/judged/*
|
|
9
|
+
!workspace/pdfs/.gitkeep
|
|
10
|
+
!workspace/parsed/.gitkeep
|
|
11
|
+
!workspace/extracted/.gitkeep
|
|
12
|
+
!workspace/judged/.gitkeep
|
|
13
|
+
|
|
14
|
+
# the Docker workspace volume
|
|
15
|
+
data/*
|
|
16
|
+
!data/.gitkeep
|
|
17
|
+
|
|
18
|
+
# user config is local: prompts, schema and model choice are per-project, and timings are
|
|
19
|
+
# measurements of one machine
|
|
20
|
+
workspace/config/*
|
|
21
|
+
!workspace/config/.gitkeep
|
|
22
|
+
|
|
23
|
+
# the verification run's console log; its report (artifacts/overnight-report.md) is worth keeping
|
|
24
|
+
artifacts/overnight.log
|
|
25
|
+
|
|
26
|
+
# built docs site (zensical build)
|
|
27
|
+
site/
|
|
28
|
+
# downloaded by `vale sync`
|
|
29
|
+
.vale/styles/Google/
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
cff-version: 1.2.0
|
|
2
|
+
title: file2records
|
|
3
|
+
message: >-
|
|
4
|
+
If you use this software in work that you publish, please cite it.
|
|
5
|
+
type: software
|
|
6
|
+
authors:
|
|
7
|
+
- family-names: Battulga
|
|
8
|
+
given-names: Enkhnyam
|
|
9
|
+
affiliation: RWTH Aachen University
|
|
10
|
+
repository-code: https://github.com/Enkhnyam/chemistry-data-extractor-toolkit
|
|
11
|
+
abstract: >-
|
|
12
|
+
A local tool for building auditable experimental datasets from scientific papers (PDF, JATS and Elsevier XML, HTML, Word). One language
|
|
13
|
+
model extracts records against a schema you define; a second re-reads the paper and audits
|
|
14
|
+
every record against it; the reviewer checks what remains with the source text alongside and
|
|
15
|
+
the passage behind each value highlighted. Schema, prompts and worked examples are
|
|
16
|
+
configuration rather than code, so the same pipeline serves any chemistry. Generalised from a
|
|
17
|
+
PET-depolymerisation database built the same way.
|
|
18
|
+
keywords:
|
|
19
|
+
- information extraction
|
|
20
|
+
- large language models
|
|
21
|
+
- chemistry
|
|
22
|
+
- literature mining
|
|
23
|
+
- LLM-as-a-judge
|
|
24
|
+
- research data
|
|
25
|
+
license: MIT
|
|
26
|
+
version: 0.2.0
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Enkhnyam Battulga, RWTH Aachen University
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: file2records
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Turn scientific papers (PDF, JATS and Elsevier XML, HTML, Word) into a structured dataset: one model extracts, a second audits, you review against the source text.
|
|
5
|
+
Project-URL: Homepage, https://github.com/Enkhnyam/chemistry-data-extractor-toolkit
|
|
6
|
+
Project-URL: Repository, https://github.com/Enkhnyam/chemistry-data-extractor-toolkit
|
|
7
|
+
Project-URL: Issues, https://github.com/Enkhnyam/chemistry-data-extractor-toolkit/issues
|
|
8
|
+
Author: Enkhnyam Battulga
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: chemistry,data extraction,elsevier,jats,llm,pdf,scientific papers
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Framework :: FastAPI
|
|
14
|
+
Classifier: Intended Audience :: Science/Research
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Topic :: Scientific/Engineering :: Chemistry
|
|
22
|
+
Requires-Python: >=3.10
|
|
23
|
+
Requires-Dist: fastapi>=0.115.0
|
|
24
|
+
Requires-Dist: httpx>=0.27
|
|
25
|
+
Requires-Dist: litellm>=1.91.0
|
|
26
|
+
Requires-Dist: lxml>=5.0
|
|
27
|
+
Requires-Dist: markdown-it-py>=3.0.0
|
|
28
|
+
Requires-Dist: pydantic>=2.0
|
|
29
|
+
Requires-Dist: python-docx>=1.1
|
|
30
|
+
Requires-Dist: python-dotenv>=1.0.1
|
|
31
|
+
Requires-Dist: python-multipart>=0.0.12
|
|
32
|
+
Requires-Dist: tenacity>=8.0
|
|
33
|
+
Requires-Dist: uvicorn[standard]>=0.32.0
|
|
34
|
+
Provides-Extra: pdf
|
|
35
|
+
Requires-Dist: docling>=2.110.0; extra == 'pdf'
|
|
36
|
+
Description-Content-Type: text/markdown
|
|
37
|
+
|
|
38
|
+
# file2records
|
|
39
|
+
|
|
40
|
+
file2records builds a dataset from a folder of scientific papers. You say which fields a
|
|
41
|
+
record has. A language model reads each paper and fills them in, a second model checks every
|
|
42
|
+
record against the paper, and you review the result with the source passage next to each
|
|
43
|
+
value. It reads PDF, JATS XML, Elsevier XML, HTML, and Word files, runs on your own machine,
|
|
44
|
+
and comes set up for RWTH's free KI:connect models.
|
|
45
|
+
|
|
46
|
+
Documentation: <https://enkhnyam.github.io/chemistry-data-extractor-toolkit/>
|
|
47
|
+
|
|
48
|
+
```bash
|
|
49
|
+
pip install "file2records[pdf]" # or: pip install file2records (no PDFs, ~250 MB)
|
|
50
|
+
file2records serve my-first-project # opens a finished example in your browser
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
```python
|
|
54
|
+
import file2records as fr
|
|
55
|
+
|
|
56
|
+
project = fr.Project("my-review")
|
|
57
|
+
project.add("papers/") # PDF, XML, HTML, Word, Markdown
|
|
58
|
+
project.extract(model=fr.rwth(), only=r"glycoly[sz]is") # RWTH_API_KEY from the environment
|
|
59
|
+
project.judge(model=fr.rwth())
|
|
60
|
+
project.export("dataset.csv") # one row per record, with its DOI
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
Start with the [tutorial](https://enkhnyam.github.io/chemistry-data-extractor-toolkit/tutorial/):
|
|
64
|
+
a PET glycolysis dataset from three real papers in 15 minutes.
|
|
65
|
+
|
|
66
|
+
## Development
|
|
67
|
+
|
|
68
|
+
```bash
|
|
69
|
+
uv sync # includes docling for PDF tests
|
|
70
|
+
uv run python -m unittest discover -s tests # no network, no key
|
|
71
|
+
uvx zensical serve # the docs, at http://localhost:8000
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
Releasing: [RELEASING.md](RELEASING.md). Licence: the code is MIT; the demo paper is CC BY
|
|
75
|
+
([NOTICE](src/file2records/demo/NOTICE.md)). Please cite: [CITATION.cff](CITATION.cff).
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
# file2records
|
|
2
|
+
|
|
3
|
+
file2records builds a dataset from a folder of scientific papers. You say which fields a
|
|
4
|
+
record has. A language model reads each paper and fills them in, a second model checks every
|
|
5
|
+
record against the paper, and you review the result with the source passage next to each
|
|
6
|
+
value. It reads PDF, JATS XML, Elsevier XML, HTML, and Word files, runs on your own machine,
|
|
7
|
+
and comes set up for RWTH's free KI:connect models.
|
|
8
|
+
|
|
9
|
+
Documentation: <https://enkhnyam.github.io/chemistry-data-extractor-toolkit/>
|
|
10
|
+
|
|
11
|
+
```bash
|
|
12
|
+
pip install "file2records[pdf]" # or: pip install file2records (no PDFs, ~250 MB)
|
|
13
|
+
file2records serve my-first-project # opens a finished example in your browser
|
|
14
|
+
```
|
|
15
|
+
|
|
16
|
+
```python
|
|
17
|
+
import file2records as fr
|
|
18
|
+
|
|
19
|
+
project = fr.Project("my-review")
|
|
20
|
+
project.add("papers/") # PDF, XML, HTML, Word, Markdown
|
|
21
|
+
project.extract(model=fr.rwth(), only=r"glycoly[sz]is") # RWTH_API_KEY from the environment
|
|
22
|
+
project.judge(model=fr.rwth())
|
|
23
|
+
project.export("dataset.csv") # one row per record, with its DOI
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
Start with the [tutorial](https://enkhnyam.github.io/chemistry-data-extractor-toolkit/tutorial/):
|
|
27
|
+
a PET glycolysis dataset from three real papers in 15 minutes.
|
|
28
|
+
|
|
29
|
+
## Development
|
|
30
|
+
|
|
31
|
+
```bash
|
|
32
|
+
uv sync # includes docling for PDF tests
|
|
33
|
+
uv run python -m unittest discover -s tests # no network, no key
|
|
34
|
+
uvx zensical serve # the docs, at http://localhost:8000
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
Releasing: [RELEASING.md](RELEASING.md). Licence: the code is MIT; the demo paper is CC BY
|
|
38
|
+
([NOTICE](src/file2records/demo/NOTICE.md)). Please cite: [CITATION.cff](CITATION.cff).
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "file2records"
|
|
3
|
+
version = "0.2.0"
|
|
4
|
+
description = "Turn scientific papers (PDF, JATS and Elsevier XML, HTML, Word) into a structured dataset: one model extracts, a second audits, you review against the source text."
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.10"
|
|
7
|
+
license = "MIT"
|
|
8
|
+
license-files = ["LICENSE"]
|
|
9
|
+
authors = [{ name = "Enkhnyam Battulga" }]
|
|
10
|
+
keywords = ["chemistry", "data extraction", "llm", "scientific papers", "jats", "elsevier", "pdf"]
|
|
11
|
+
classifiers = [
|
|
12
|
+
"Development Status :: 4 - Beta",
|
|
13
|
+
"Intended Audience :: Science/Research",
|
|
14
|
+
"Topic :: Scientific/Engineering :: Chemistry",
|
|
15
|
+
"Programming Language :: Python :: 3",
|
|
16
|
+
"Programming Language :: Python :: 3.10",
|
|
17
|
+
"Programming Language :: Python :: 3.11",
|
|
18
|
+
"Programming Language :: Python :: 3.12",
|
|
19
|
+
"Programming Language :: Python :: 3.13",
|
|
20
|
+
"Operating System :: OS Independent",
|
|
21
|
+
"Framework :: FastAPI",
|
|
22
|
+
]
|
|
23
|
+
|
|
24
|
+
dependencies = [
|
|
25
|
+
"fastapi>=0.115.0",
|
|
26
|
+
"uvicorn[standard]>=0.32.0",
|
|
27
|
+
"python-multipart>=0.0.12",
|
|
28
|
+
"litellm>=1.91.0",
|
|
29
|
+
# litellm.completion(num_retries=...) imports tenacity lazily and litellm does not declare
|
|
30
|
+
# it, so every model call on a clean install died with "tenacity import failed". We pass
|
|
31
|
+
# num_retries on every call, which made that every call.
|
|
32
|
+
"tenacity>=8.0",
|
|
33
|
+
"pydantic>=2.0",
|
|
34
|
+
"python-dotenv>=1.0.1",
|
|
35
|
+
"markdown-it-py>=3.0.0",
|
|
36
|
+
# Only for asking an endpoint which models it serves. It arrives anyway under litellm, but
|
|
37
|
+
# a direct import belongs in a direct dependency -- otherwise it vanishes the day litellm
|
|
38
|
+
# changes its HTTP client.
|
|
39
|
+
"httpx>=0.27",
|
|
40
|
+
# The XML and HTML readers (JATS, Elsevier, publisher pages) and the Word reader. Small, and
|
|
41
|
+
# no machine-learning model is involved in reading a structured format.
|
|
42
|
+
"lxml>=5.0",
|
|
43
|
+
"python-docx>=1.1",
|
|
44
|
+
]
|
|
45
|
+
|
|
46
|
+
[project.optional-dependencies]
|
|
47
|
+
# PDF reading: docling and its layout/table models, via PyTorch -- a download of several GB.
|
|
48
|
+
# Kept out of the base install so the structured formats work everywhere in seconds.
|
|
49
|
+
pdf = ["docling>=2.110.0"]
|
|
50
|
+
|
|
51
|
+
[project.scripts]
|
|
52
|
+
file2records = "file2records.cli:main"
|
|
53
|
+
|
|
54
|
+
[project.urls]
|
|
55
|
+
Homepage = "https://github.com/Enkhnyam/chemistry-data-extractor-toolkit"
|
|
56
|
+
Repository = "https://github.com/Enkhnyam/chemistry-data-extractor-toolkit"
|
|
57
|
+
Issues = "https://github.com/Enkhnyam/chemistry-data-extractor-toolkit/issues"
|
|
58
|
+
|
|
59
|
+
[build-system]
|
|
60
|
+
requires = ["hatchling>=1.27"]
|
|
61
|
+
build-backend = "hatchling.build"
|
|
62
|
+
|
|
63
|
+
[tool.hatch.build.targets.wheel]
|
|
64
|
+
packages = ["src/file2records"]
|
|
65
|
+
|
|
66
|
+
[tool.hatch.build.targets.sdist]
|
|
67
|
+
include = ["src/file2records", "tests", "README.md", "LICENSE", "CITATION.cff"]
|
|
68
|
+
|
|
69
|
+
[dependency-groups]
|
|
70
|
+
dev = ["docling>=2.110.0"]
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""file2records: turn a folder of papers into a structured dataset you can check.
|
|
2
|
+
|
|
3
|
+
import file2records as fr
|
|
4
|
+
project = fr.Project("my-review")
|
|
5
|
+
project.add("papers/")
|
|
6
|
+
project.extract(model=fr.rwth())
|
|
7
|
+
project.export("dataset.csv")
|
|
8
|
+
|
|
9
|
+
See project.py for the API, cli.py for the command line, main.py for the web app.
|
|
10
|
+
"""
|
|
11
|
+
__version__ = "0.2.0"
|
|
12
|
+
|
|
13
|
+
from .project import Project, rwth # noqa: E402
|
|
14
|
+
|
|
15
|
+
__all__ = ["Project", "rwth", "__version__"]
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
"""The export bundle: a dataset together with what produced it.
|
|
2
|
+
|
|
3
|
+
A CSV of records on its own is not reproducible -- it cannot say which model wrote it, under
|
|
4
|
+
which prompt, against which schema, or which rows a human then corrected. The bundle is the data
|
|
5
|
+
beside the config and a manifest that pins both.
|
|
6
|
+
|
|
7
|
+
What it leaves out by default matters as much. Papers obtained through publishers' text-and-data
|
|
8
|
+
mining agreements may be read and mined, not redistributed -- and a bundle is made to be shared.
|
|
9
|
+
So the paper text and the source files go in only when asked for (`include_text`,
|
|
10
|
+
`include_files`), which is the right call for open-access papers and the caller's to make.
|
|
11
|
+
Without them every record still names its paper's DOI and the ids of the chunks it came from,
|
|
12
|
+
which is enough for anyone with access to the paper to check it.
|
|
13
|
+
|
|
14
|
+
No key ever enters the bundle; model settings are copied without their key variables.
|
|
15
|
+
"""
|
|
16
|
+
import csv
|
|
17
|
+
import io
|
|
18
|
+
import json
|
|
19
|
+
import subprocess
|
|
20
|
+
import time
|
|
21
|
+
import zipfile
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
|
|
24
|
+
from . import __version__, config, models, report, storage
|
|
25
|
+
from .storage import read_json
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def csv_bytes(columns, rows) -> bytes:
|
|
29
|
+
buffer = io.StringIO()
|
|
30
|
+
writer = csv.DictWriter(buffer, fieldnames=columns, extrasaction="ignore")
|
|
31
|
+
writer.writeheader()
|
|
32
|
+
writer.writerows(rows)
|
|
33
|
+
return buffer.getvalue().encode("utf-8")
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def git_commit() -> str | None:
|
|
37
|
+
"""Which checkout produced this, when run from one. None for an installed package, whose
|
|
38
|
+
version is recorded instead."""
|
|
39
|
+
try:
|
|
40
|
+
return subprocess.check_output(["git", "-C", str(Path(__file__).resolve().parent),
|
|
41
|
+
"rev-parse", "HEAD"],
|
|
42
|
+
text=True, stderr=subprocess.DEVNULL, timeout=5).strip()
|
|
43
|
+
except Exception:
|
|
44
|
+
return None
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def build(paper_ids: list[str] | None = None, *, include_text: bool = False,
|
|
48
|
+
include_files: bool = False, profiles: list[dict] | None = None) -> bytes:
|
|
49
|
+
"""The zip as bytes. `paper_ids` limits it to those papers (None: all of them)."""
|
|
50
|
+
record_columns, record_rows = report.flat_records(paper_ids)
|
|
51
|
+
paper_columns, paper_rows = report.papers_table(paper_ids)
|
|
52
|
+
summary = report.build()
|
|
53
|
+
settings = config.get_settings()
|
|
54
|
+
if profiles is None:
|
|
55
|
+
profiles = [{k: v for k, v in p.items()} for p in models.listing(lambda _: False)]
|
|
56
|
+
chosen = {r["paper_id"] for r in paper_rows}
|
|
57
|
+
|
|
58
|
+
manifest = {
|
|
59
|
+
"exported_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
|
|
60
|
+
"tool": "file2records",
|
|
61
|
+
"version": __version__,
|
|
62
|
+
"git_commit": git_commit(),
|
|
63
|
+
"papers": len(paper_rows),
|
|
64
|
+
"records": len(record_rows),
|
|
65
|
+
"includes_paper_text": include_text,
|
|
66
|
+
"models": {
|
|
67
|
+
"extract": (models.get(settings.get("extract_model", "")) or {}).get("model"),
|
|
68
|
+
"judge": (models.get(settings.get("judge_model", "")) or {}).get("model"),
|
|
69
|
+
# per paper too: a corpus is often built across more than one model
|
|
70
|
+
"per_paper": {r["paper_id"]: {"extract": r["extract_model"], "judge": r["judge_model"]}
|
|
71
|
+
for r in paper_rows},
|
|
72
|
+
},
|
|
73
|
+
"spend": summary.get("totals", {}).get("spend", {}),
|
|
74
|
+
"source_tracking_default": settings.get("source_tracking_default", True),
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
buffer = io.BytesIO()
|
|
78
|
+
with zipfile.ZipFile(buffer, "w", zipfile.ZIP_DEFLATED) as bundle:
|
|
79
|
+
bundle.writestr("manifest.json", json.dumps(manifest, indent=2, ensure_ascii=False))
|
|
80
|
+
bundle.writestr("README.md", _readme(manifest))
|
|
81
|
+
bundle.writestr("data/records.csv", csv_bytes(record_columns, record_rows))
|
|
82
|
+
bundle.writestr("data/records.json", json.dumps(record_rows, indent=1, ensure_ascii=False))
|
|
83
|
+
bundle.writestr("data/papers.csv", csv_bytes(paper_columns, paper_rows))
|
|
84
|
+
bundle.writestr("data/report.json", json.dumps(summary, indent=2, ensure_ascii=False))
|
|
85
|
+
|
|
86
|
+
# config: everything that decides what a run produces, and nothing that authenticates it
|
|
87
|
+
bundle.writestr("config/schema.json",
|
|
88
|
+
json.dumps({"fields": config.get_schema()}, indent=2, ensure_ascii=False))
|
|
89
|
+
bundle.writestr("config/extract_prompt.txt", config.get_extract_prompt())
|
|
90
|
+
bundle.writestr("config/judge_prompt.txt", config.get_judge_prompt())
|
|
91
|
+
bundle.writestr("config/few_shot.json",
|
|
92
|
+
json.dumps(config.get_few_shot(), indent=2, ensure_ascii=False))
|
|
93
|
+
bundle.writestr("config/models.json", json.dumps(
|
|
94
|
+
[{k: v for k, v in m.items()
|
|
95
|
+
if k in ("id", "name", "model", "api_base", "api_version")} for m in profiles],
|
|
96
|
+
indent=2, ensure_ascii=False))
|
|
97
|
+
|
|
98
|
+
# the per-paper working files, so a reviewer can trace any row back to its chunk
|
|
99
|
+
for stage, directory in (("extracted", storage.EXTRACTED), ("judged", storage.JUDGED)):
|
|
100
|
+
for path in sorted(directory.glob("*.json")):
|
|
101
|
+
if path.stem in chosen:
|
|
102
|
+
bundle.write(path, f"{stage}/{path.name}")
|
|
103
|
+
for path in sorted(storage.PARSED.glob("*.json")):
|
|
104
|
+
if path.stem not in chosen:
|
|
105
|
+
continue
|
|
106
|
+
paper = read_json(path, {})
|
|
107
|
+
if not include_text:
|
|
108
|
+
paper["chunks"] = [{"id": c["id"]} for c in paper.get("chunks", [])]
|
|
109
|
+
bundle.writestr(f"parsed/{path.name}", json.dumps(paper, indent=1, ensure_ascii=False))
|
|
110
|
+
if include_files:
|
|
111
|
+
for pid in sorted(chosen):
|
|
112
|
+
if (source := storage.source_file(pid)) is not None:
|
|
113
|
+
bundle.write(source, f"papers/{source.name}")
|
|
114
|
+
return buffer.getvalue()
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _readme(manifest: dict) -> str:
|
|
118
|
+
m = manifest["models"]
|
|
119
|
+
text_note = (
|
|
120
|
+
"`chunks[].text` holds the paper text, because this bundle was exported with it. Check "
|
|
121
|
+
"the papers' licences before sharing it."
|
|
122
|
+
if manifest["includes_paper_text"] else
|
|
123
|
+
"Paper text is not included -- papers obtained under text-and-data-mining terms may be "
|
|
124
|
+
"mined, not redistributed. Each chunk keeps its `id`, and each record its paper's DOI, "
|
|
125
|
+
"so anyone with access to the paper can check a value against its source.")
|
|
126
|
+
return f"""# file2records bundle
|
|
127
|
+
|
|
128
|
+
{manifest['records']} records from {manifest['papers']} paper(s), exported
|
|
129
|
+
{manifest['exported_at']} by file2records {manifest['version']}.
|
|
130
|
+
|
|
131
|
+
## What is here
|
|
132
|
+
|
|
133
|
+
- `data/records.csv`, `data/records.json` — one row per record, with the paper's DOI, the
|
|
134
|
+
judge's verdict and any reviewer flag or note. `extract_model` and `judge_model` say what
|
|
135
|
+
produced each row.
|
|
136
|
+
- `data/papers.csv` — one row per paper: DOI, chunks in, records out, model, tokens, cost.
|
|
137
|
+
- `data/report.json` — completeness by field, what the judge changed, totals.
|
|
138
|
+
- `config/` — the schema, both prompts, the worked examples and the model settings that
|
|
139
|
+
produced this. API keys are not included.
|
|
140
|
+
- `parsed/`, `extracted/`, `judged/` — the working files, so any row can be traced back to the
|
|
141
|
+
chunk it came from. `source_chunk_ids` on a record refers to `chunks[].id` in `parsed/`.
|
|
142
|
+
- `papers/` — the source files, if you exported with them.
|
|
143
|
+
|
|
144
|
+
{text_note}
|
|
145
|
+
|
|
146
|
+
## Reading it
|
|
147
|
+
|
|
148
|
+
Extraction model: `{m['extract'] or 'not set'}` · judge model: `{m['judge'] or 'not set'}`.
|
|
149
|
+
Papers may differ from these if the corpus was built across more than one model; see
|
|
150
|
+
`models.per_paper` in `manifest.json`.
|
|
151
|
+
|
|
152
|
+
`model_records` in `extracted/*.json` is what the model originally said, kept beside the
|
|
153
|
+
corrected records the moment anything was edited. A corrected dataset that cannot be diffed
|
|
154
|
+
against the model's own output is not evidence of anything.
|
|
155
|
+
"""
|
|
@@ -0,0 +1,207 @@
|
|
|
1
|
+
"""The `file2records` command.
|
|
2
|
+
|
|
3
|
+
file2records serve my-review the web app on that project folder
|
|
4
|
+
file2records add my-review papers/ read files and folders into it
|
|
5
|
+
file2records papers my-review what is in it
|
|
6
|
+
file2records search my-review "glycoly[sz]is"
|
|
7
|
+
file2records check my-review what is missing before a run
|
|
8
|
+
file2records extract my-review --only "glycoly[sz]is" --exclude "positron|tomograph"
|
|
9
|
+
file2records judge my-review
|
|
10
|
+
file2records export my-review dataset.csv (.csv, .json, or .zip for the full bundle)
|
|
11
|
+
|
|
12
|
+
Every command takes the project folder first. Model commands use the model chosen in the web
|
|
13
|
+
app's Settings unless --model is given: any litellm model string, or rwth/<name> for RWTH's
|
|
14
|
+
KI:connect (key from RWTH_API_KEY).
|
|
15
|
+
"""
|
|
16
|
+
import argparse
|
|
17
|
+
import os
|
|
18
|
+
import sys
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
from dotenv import load_dotenv
|
|
22
|
+
|
|
23
|
+
from . import __version__
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _project(folder):
|
|
27
|
+
from .project import Project
|
|
28
|
+
project = Project(folder)
|
|
29
|
+
# Keys saved through the web app live in the project's .env; a .env in the current folder
|
|
30
|
+
# is read too. Neither overrides a variable already set in the shell.
|
|
31
|
+
load_dotenv(project.path / ".env")
|
|
32
|
+
load_dotenv(Path.cwd() / ".env")
|
|
33
|
+
return project
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _filters(p):
|
|
37
|
+
p.add_argument("--only", metavar="REGEX",
|
|
38
|
+
help="only papers whose full text matches this regular expression")
|
|
39
|
+
p.add_argument("--exclude", metavar="REGEX",
|
|
40
|
+
help="skip papers whose full text matches this regular expression")
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _print_result(r):
|
|
44
|
+
name = r.get("filename") or r["id"]
|
|
45
|
+
if r.get("error"):
|
|
46
|
+
print(f" ✗ {name}: {r['error']}")
|
|
47
|
+
elif "n_chunks" in r:
|
|
48
|
+
doi = f" doi:{r['doi']}" if r.get("doi") else ""
|
|
49
|
+
print(f" ✓ {name} [{r['format']}] {r['n_chunks']} chunks{doi}")
|
|
50
|
+
elif "n_records" in r:
|
|
51
|
+
print(f" ✓ {name}: {r['n_records']} records in {r['seconds']}s")
|
|
52
|
+
else:
|
|
53
|
+
print(f" ✓ {name}: {r['n_verdicts']} verdicts in {r['seconds']}s")
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def cmd_serve(args):
|
|
57
|
+
os.environ["WORKSPACE_DIR"] = str(Path(args.folder).resolve())
|
|
58
|
+
_project(args.folder) # creates the folder, opens it
|
|
59
|
+
import uvicorn
|
|
60
|
+
from .main import app
|
|
61
|
+
url = f"http://{args.host}:{args.port}"
|
|
62
|
+
print(f"file2records {__version__} — project {Path(args.folder).resolve()}\nOpen {url}")
|
|
63
|
+
if not args.no_browser:
|
|
64
|
+
import threading
|
|
65
|
+
import webbrowser
|
|
66
|
+
threading.Timer(1.0, webbrowser.open, [url]).start()
|
|
67
|
+
uvicorn.run(app, host=args.host, port=args.port, log_level="warning")
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def cmd_add(args):
|
|
71
|
+
project = _project(args.folder)
|
|
72
|
+
print(f"Reading into {project.path}")
|
|
73
|
+
results = project.add(*args.paths, source_tracking=not args.no_source_tracking,
|
|
74
|
+
on_file=_print_result)
|
|
75
|
+
failed = sum(1 for r in results if r.get("error"))
|
|
76
|
+
print(f"{len(results) - failed} added, {failed} failed")
|
|
77
|
+
return 1 if failed and failed == len(results) else 0
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def cmd_papers(args):
|
|
81
|
+
papers = _project(args.folder).papers()
|
|
82
|
+
if not papers:
|
|
83
|
+
print("No papers yet. Add some: file2records add <folder> <files or folders>")
|
|
84
|
+
for p in papers:
|
|
85
|
+
state = "judged" if p["judged"] else "extracted" if p["extracted"] else "parsed"
|
|
86
|
+
records = "" if p["n_records"] is None else f"{p['n_records']} records"
|
|
87
|
+
print(f"{p['id']:50.50} {p['format']:8} {state:9} {records:11} {p['doi']}")
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def cmd_search(args):
|
|
91
|
+
hits = _project(args.folder).search(args.pattern, ignore_case=not args.case_sensitive)
|
|
92
|
+
for h in hits:
|
|
93
|
+
print(f"\n{h['filename']} ({h['matches']} matches)")
|
|
94
|
+
for s in h["snippets"]:
|
|
95
|
+
print(f" …{s['before'][-60:]}[{s['match']}]{s['after'][:60]}…".replace("\n", " "))
|
|
96
|
+
total = sum(h["matches"] for h in hits)
|
|
97
|
+
print(f"\n{len(hits)} papers, {total} matches")
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def cmd_check(args):
|
|
101
|
+
project = _project(args.folder)
|
|
102
|
+
ok = True
|
|
103
|
+
for stage in ("extract", "judge"):
|
|
104
|
+
missing = project.check(stage, args.model)
|
|
105
|
+
ok &= stage == "judge" or not missing
|
|
106
|
+
print(f"{stage}: {'ready' if not missing else 'not ready'}")
|
|
107
|
+
for m in missing:
|
|
108
|
+
hint = " Or pass --model, e.g. rwth/gpt-oss-120b." if m.startswith("Choose a model") else ""
|
|
109
|
+
print(f" - {m}{hint}")
|
|
110
|
+
return 0 if ok else 1
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _run_stage(args, stage):
|
|
114
|
+
project = _project(args.folder)
|
|
115
|
+
run = project.extract if stage == "extract" else project.judge
|
|
116
|
+
results = run(args.model, only=args.only, exclude=args.exclude, redo=args.redo,
|
|
117
|
+
on_paper=_print_result)
|
|
118
|
+
if not results:
|
|
119
|
+
print(f"Nothing to {stage}: every chosen paper is done already (--redo to run again).")
|
|
120
|
+
failed = sum(1 for r in results if r.get("error"))
|
|
121
|
+
print(f"{len(results) - failed} done, {failed} failed")
|
|
122
|
+
return 1 if failed else 0
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def cmd_export(args):
|
|
126
|
+
path = _project(args.folder).export(args.output, only=args.only, exclude=args.exclude,
|
|
127
|
+
include_text=args.include_text,
|
|
128
|
+
include_files=args.include_files)
|
|
129
|
+
print(f"Wrote {path}")
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
FOLDER = "the project folder (created if it does not exist)"
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
136
|
+
parser = argparse.ArgumentParser(
|
|
137
|
+
prog="file2records",
|
|
138
|
+
description="Turn papers (PDF, XML, HTML, Word, Markdown) into a structured dataset.",
|
|
139
|
+
epilog="examples:\n" + "\n".join(__doc__.splitlines()[2:10]) +
|
|
140
|
+
"\n\ndocs: https://enkhnyam.github.io/chemistry-data-extractor-toolkit/",
|
|
141
|
+
formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
142
|
+
parser.add_argument("--version", action="version", version=f"file2records {__version__}")
|
|
143
|
+
sub = parser.add_subparsers(dest="command", required=True, metavar="COMMAND")
|
|
144
|
+
|
|
145
|
+
p = sub.add_parser("serve", help="open the web app on a project folder")
|
|
146
|
+
p.add_argument("folder", nargs="?", default="workspace", help=FOLDER + " (default: %(default)s)")
|
|
147
|
+
p.add_argument("--port", type=int, default=8000)
|
|
148
|
+
p.add_argument("--host", default="127.0.0.1",
|
|
149
|
+
help="keep the default unless you know why: the app has no login")
|
|
150
|
+
p.add_argument("--no-browser", action="store_true")
|
|
151
|
+
p.set_defaults(func=cmd_serve)
|
|
152
|
+
|
|
153
|
+
p = sub.add_parser("add", help="read files or folders of papers into a project")
|
|
154
|
+
p.add_argument("folder", help=FOLDER)
|
|
155
|
+
p.add_argument("paths", nargs="+", help="paper files, or folders of them")
|
|
156
|
+
p.add_argument("--no-source-tracking", action="store_true",
|
|
157
|
+
help="don't tag chunks, so records won't cite the passage they came from")
|
|
158
|
+
p.set_defaults(func=cmd_add)
|
|
159
|
+
|
|
160
|
+
p = sub.add_parser("papers", help="list the papers in a project")
|
|
161
|
+
p.add_argument("folder", help=FOLDER)
|
|
162
|
+
p.set_defaults(func=cmd_papers)
|
|
163
|
+
|
|
164
|
+
p = sub.add_parser("search", help="search the full text of every paper with a regex")
|
|
165
|
+
p.add_argument("folder", help=FOLDER)
|
|
166
|
+
p.add_argument("pattern", help="a regular expression, e.g. \"glycoly[sz]is\"")
|
|
167
|
+
p.add_argument("--case-sensitive", action="store_true")
|
|
168
|
+
p.set_defaults(func=cmd_search)
|
|
169
|
+
|
|
170
|
+
p = sub.add_parser("check", help="say what is missing before a run")
|
|
171
|
+
p.add_argument("folder", help=FOLDER)
|
|
172
|
+
p.add_argument("--model")
|
|
173
|
+
p.set_defaults(func=cmd_check)
|
|
174
|
+
|
|
175
|
+
for stage, text in (("extract", "extract records from papers not extracted yet"),
|
|
176
|
+
("judge", "audit extracted records with a second model")):
|
|
177
|
+
p = sub.add_parser(stage, help=text)
|
|
178
|
+
p.add_argument("folder", help=FOLDER)
|
|
179
|
+
p.add_argument("--model", help='litellm model string, or rwth/<name> '
|
|
180
|
+
'(default: the model chosen in Settings)')
|
|
181
|
+
p.add_argument("--redo", action="store_true", help="also re-run papers already done")
|
|
182
|
+
_filters(p)
|
|
183
|
+
p.set_defaults(func=lambda a, s=stage: _run_stage(a, s))
|
|
184
|
+
|
|
185
|
+
p = sub.add_parser("export", help="write records to .csv / .json, or a .zip bundle")
|
|
186
|
+
p.add_argument("folder", help=FOLDER)
|
|
187
|
+
p.add_argument("output", help="a .csv or .json file of records, or a .zip bundle")
|
|
188
|
+
p.add_argument("--include-text", action="store_true",
|
|
189
|
+
help="put the paper text in a .zip bundle (check the papers' licences)")
|
|
190
|
+
p.add_argument("--include-files", action="store_true",
|
|
191
|
+
help="put the source files in a .zip bundle (check the papers' licences)")
|
|
192
|
+
_filters(p)
|
|
193
|
+
p.set_defaults(func=cmd_export)
|
|
194
|
+
return parser
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def main(argv=None) -> int:
|
|
198
|
+
args = build_parser().parse_args(argv)
|
|
199
|
+
try:
|
|
200
|
+
return args.func(args) or 0
|
|
201
|
+
except (ValueError, RuntimeError, FileNotFoundError) as e:
|
|
202
|
+
print(f"error: {e}", file=sys.stderr)
|
|
203
|
+
return 2
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
if __name__ == "__main__":
|
|
207
|
+
sys.exit(main())
|