file2records 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. file2records-0.2.0/.gitignore +29 -0
  2. file2records-0.2.0/CITATION.cff +26 -0
  3. file2records-0.2.0/LICENSE +21 -0
  4. file2records-0.2.0/PKG-INFO +75 -0
  5. file2records-0.2.0/README.md +38 -0
  6. file2records-0.2.0/pyproject.toml +70 -0
  7. file2records-0.2.0/src/file2records/__init__.py +15 -0
  8. file2records-0.2.0/src/file2records/__main__.py +5 -0
  9. file2records-0.2.0/src/file2records/bundle.py +155 -0
  10. file2records-0.2.0/src/file2records/cli.py +207 -0
  11. file2records-0.2.0/src/file2records/config.py +96 -0
  12. file2records-0.2.0/src/file2records/demo/NOTICE.md +34 -0
  13. file2records-0.2.0/src/file2records/demo/config/extract_prompt.txt +51 -0
  14. file2records-0.2.0/src/file2records/demo/config/few_shot.json +1 -0
  15. file2records-0.2.0/src/file2records/demo/config/judge_prompt.txt +103 -0
  16. file2records-0.2.0/src/file2records/demo/config/schema.json +54 -0
  17. file2records-0.2.0/src/file2records/demo/config/settings.json +3 -0
  18. file2records-0.2.0/src/file2records/demo/extracted/lewis-acidic-ionic-liquids-for-pet-glycolysis-polymers-2013-bc017719.json +268 -0
  19. file2records-0.2.0/src/file2records/demo/judged/lewis-acidic-ionic-liquids-for-pet-glycolysis-polymers-2013-bc017719.json +517 -0
  20. file2records-0.2.0/src/file2records/demo/parsed/lewis-acidic-ionic-liquids-for-pet-glycolysis-polymers-2013-bc017719.json +408 -0
  21. file2records-0.2.0/src/file2records/demo/pdfs/lewis-acidic-ionic-liquids-for-pet-glycolysis-polymers-2013-bc017719.pdf +0 -0
  22. file2records-0.2.0/src/file2records/demo.py +155 -0
  23. file2records-0.2.0/src/file2records/dynschema.py +37 -0
  24. file2records-0.2.0/src/file2records/exemplar.py +171 -0
  25. file2records-0.2.0/src/file2records/extraction.py +40 -0
  26. file2records-0.2.0/src/file2records/filters.py +81 -0
  27. file2records-0.2.0/src/file2records/judge.py +94 -0
  28. file2records-0.2.0/src/file2records/llm.py +79 -0
  29. file2records-0.2.0/src/file2records/main.py +768 -0
  30. file2records-0.2.0/src/file2records/models.py +105 -0
  31. file2records-0.2.0/src/file2records/parsing.py +129 -0
  32. file2records-0.2.0/src/file2records/pipeline.py +206 -0
  33. file2records-0.2.0/src/file2records/project.py +203 -0
  34. file2records-0.2.0/src/file2records/readers.py +385 -0
  35. file2records-0.2.0/src/file2records/report.py +230 -0
  36. file2records-0.2.0/src/file2records/storage.py +132 -0
  37. file2records-0.2.0/src/file2records/timings.py +50 -0
  38. file2records-0.2.0/src/file2records/web/app.js +2340 -0
  39. file2records-0.2.0/src/file2records/web/index.html +28 -0
  40. file2records-0.2.0/src/file2records/web/style.css +483 -0
  41. file2records-0.2.0/tests/test_docs.py +31 -0
  42. file2records-0.2.0/tests/test_library.py +330 -0
  43. file2records-0.2.0/tests/test_smoke.py +665 -0
@@ -0,0 +1,29 @@
1
+ .venv/
2
+ __pycache__/
3
+ *.pyc
4
+ .env
5
+ workspace/pdfs/*
6
+ workspace/parsed/*
7
+ workspace/extracted/*
8
+ workspace/judged/*
9
+ !workspace/pdfs/.gitkeep
10
+ !workspace/parsed/.gitkeep
11
+ !workspace/extracted/.gitkeep
12
+ !workspace/judged/.gitkeep
13
+
14
+ # the Docker workspace volume
15
+ data/*
16
+ !data/.gitkeep
17
+
18
+ # user config is local: prompts, schema and model choice are per-project, and timings are
19
+ # measurements of one machine
20
+ workspace/config/*
21
+ !workspace/config/.gitkeep
22
+
23
+ # the verification run's console log; its report (artifacts/overnight-report.md) is worth keeping
24
+ artifacts/overnight.log
25
+
26
+ # built docs site (zensical build)
27
+ site/
28
+ # downloaded by `vale sync`
29
+ .vale/styles/Google/
@@ -0,0 +1,26 @@
1
+ cff-version: 1.2.0
2
+ title: file2records
3
+ message: >-
4
+ If you use this software in work that you publish, please cite it.
5
+ type: software
6
+ authors:
7
+ - family-names: Battulga
8
+ given-names: Enkhnyam
9
+ affiliation: RWTH Aachen University
10
+ repository-code: https://github.com/Enkhnyam/chemistry-data-extractor-toolkit
11
+ abstract: >-
12
+ A local tool for building auditable experimental datasets from scientific papers (PDF, JATS and Elsevier XML, HTML, Word). One language
13
+ model extracts records against a schema you define; a second re-reads the paper and audits
14
+ every record against it; the reviewer checks what remains with the source text alongside and
15
+ the passage behind each value highlighted. Schema, prompts and worked examples are
16
+ configuration rather than code, so the same pipeline serves any chemistry. Generalised from a
17
+ PET-depolymerisation database built the same way.
18
+ keywords:
19
+ - information extraction
20
+ - large language models
21
+ - chemistry
22
+ - literature mining
23
+ - LLM-as-a-judge
24
+ - research data
25
+ license: MIT
26
+ version: 0.2.0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Enkhnyam Battulga, RWTH Aachen University
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,75 @@
1
+ Metadata-Version: 2.5
2
+ Name: file2records
3
+ Version: 0.2.0
4
+ Summary: Turn scientific papers (PDF, JATS and Elsevier XML, HTML, Word) into a structured dataset: one model extracts, a second audits, you review against the source text.
5
+ Project-URL: Homepage, https://github.com/Enkhnyam/chemistry-data-extractor-toolkit
6
+ Project-URL: Repository, https://github.com/Enkhnyam/chemistry-data-extractor-toolkit
7
+ Project-URL: Issues, https://github.com/Enkhnyam/chemistry-data-extractor-toolkit/issues
8
+ Author: Enkhnyam Battulga
9
+ License-Expression: MIT
10
+ License-File: LICENSE
11
+ Keywords: chemistry,data extraction,elsevier,jats,llm,pdf,scientific papers
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Framework :: FastAPI
14
+ Classifier: Intended Audience :: Science/Research
15
+ Classifier: Operating System :: OS Independent
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Topic :: Scientific/Engineering :: Chemistry
22
+ Requires-Python: >=3.10
23
+ Requires-Dist: fastapi>=0.115.0
24
+ Requires-Dist: httpx>=0.27
25
+ Requires-Dist: litellm>=1.91.0
26
+ Requires-Dist: lxml>=5.0
27
+ Requires-Dist: markdown-it-py>=3.0.0
28
+ Requires-Dist: pydantic>=2.0
29
+ Requires-Dist: python-docx>=1.1
30
+ Requires-Dist: python-dotenv>=1.0.1
31
+ Requires-Dist: python-multipart>=0.0.12
32
+ Requires-Dist: tenacity>=8.0
33
+ Requires-Dist: uvicorn[standard]>=0.32.0
34
+ Provides-Extra: pdf
35
+ Requires-Dist: docling>=2.110.0; extra == 'pdf'
36
+ Description-Content-Type: text/markdown
37
+
38
+ # file2records
39
+
40
+ file2records builds a dataset from a folder of scientific papers. You say which fields a
41
+ record has. A language model reads each paper and fills them in, a second model checks every
42
+ record against the paper, and you review the result with the source passage next to each
43
+ value. It reads PDF, JATS XML, Elsevier XML, HTML, and Word files, runs on your own machine,
44
+ and comes set up for RWTH's free KI:connect models.
45
+
46
+ Documentation: <https://enkhnyam.github.io/chemistry-data-extractor-toolkit/>
47
+
48
+ ```bash
49
+ pip install "file2records[pdf]" # or: pip install file2records (no PDFs, ~250 MB)
50
+ file2records serve my-first-project # opens a finished example in your browser
51
+ ```
52
+
53
+ ```python
54
+ import file2records as fr
55
+
56
+ project = fr.Project("my-review")
57
+ project.add("papers/") # PDF, XML, HTML, Word, Markdown
58
+ project.extract(model=fr.rwth(), only=r"glycoly[sz]is") # RWTH_API_KEY from the environment
59
+ project.judge(model=fr.rwth())
60
+ project.export("dataset.csv") # one row per record, with its DOI
61
+ ```
62
+
63
+ Start with the [tutorial](https://enkhnyam.github.io/chemistry-data-extractor-toolkit/tutorial/):
64
+ a PET glycolysis dataset from three real papers in 15 minutes.
65
+
66
+ ## Development
67
+
68
+ ```bash
69
+ uv sync # includes docling for PDF tests
70
+ uv run python -m unittest discover -s tests # no network, no key
71
+ uvx zensical serve # the docs, at http://localhost:8000
72
+ ```
73
+
74
+ Releasing: [RELEASING.md](RELEASING.md). Licence: the code is MIT; the demo paper is CC BY
75
+ ([NOTICE](src/file2records/demo/NOTICE.md)). Please cite: [CITATION.cff](CITATION.cff).
@@ -0,0 +1,38 @@
1
+ # file2records
2
+
3
+ file2records builds a dataset from a folder of scientific papers. You say which fields a
4
+ record has. A language model reads each paper and fills them in, a second model checks every
5
+ record against the paper, and you review the result with the source passage next to each
6
+ value. It reads PDF, JATS XML, Elsevier XML, HTML, and Word files, runs on your own machine,
7
+ and comes set up for RWTH's free KI:connect models.
8
+
9
+ Documentation: <https://enkhnyam.github.io/chemistry-data-extractor-toolkit/>
10
+
11
+ ```bash
12
+ pip install "file2records[pdf]" # or: pip install file2records (no PDFs, ~250 MB)
13
+ file2records serve my-first-project # opens a finished example in your browser
14
+ ```
15
+
16
+ ```python
17
+ import file2records as fr
18
+
19
+ project = fr.Project("my-review")
20
+ project.add("papers/") # PDF, XML, HTML, Word, Markdown
21
+ project.extract(model=fr.rwth(), only=r"glycoly[sz]is") # RWTH_API_KEY from the environment
22
+ project.judge(model=fr.rwth())
23
+ project.export("dataset.csv") # one row per record, with its DOI
24
+ ```
25
+
26
+ Start with the [tutorial](https://enkhnyam.github.io/chemistry-data-extractor-toolkit/tutorial/):
27
+ a PET glycolysis dataset from three real papers in 15 minutes.
28
+
29
+ ## Development
30
+
31
+ ```bash
32
+ uv sync # includes docling for PDF tests
33
+ uv run python -m unittest discover -s tests # no network, no key
34
+ uvx zensical serve # the docs, at http://localhost:8000
35
+ ```
36
+
37
+ Releasing: [RELEASING.md](RELEASING.md). Licence: the code is MIT; the demo paper is CC BY
38
+ ([NOTICE](src/file2records/demo/NOTICE.md)). Please cite: [CITATION.cff](CITATION.cff).
@@ -0,0 +1,70 @@
1
+ [project]
2
+ name = "file2records"
3
+ version = "0.2.0"
4
+ description = "Turn scientific papers (PDF, JATS and Elsevier XML, HTML, Word) into a structured dataset: one model extracts, a second audits, you review against the source text."
5
+ readme = "README.md"
6
+ requires-python = ">=3.10"
7
+ license = "MIT"
8
+ license-files = ["LICENSE"]
9
+ authors = [{ name = "Enkhnyam Battulga" }]
10
+ keywords = ["chemistry", "data extraction", "llm", "scientific papers", "jats", "elsevier", "pdf"]
11
+ classifiers = [
12
+ "Development Status :: 4 - Beta",
13
+ "Intended Audience :: Science/Research",
14
+ "Topic :: Scientific/Engineering :: Chemistry",
15
+ "Programming Language :: Python :: 3",
16
+ "Programming Language :: Python :: 3.10",
17
+ "Programming Language :: Python :: 3.11",
18
+ "Programming Language :: Python :: 3.12",
19
+ "Programming Language :: Python :: 3.13",
20
+ "Operating System :: OS Independent",
21
+ "Framework :: FastAPI",
22
+ ]
23
+
24
+ dependencies = [
25
+ "fastapi>=0.115.0",
26
+ "uvicorn[standard]>=0.32.0",
27
+ "python-multipart>=0.0.12",
28
+ "litellm>=1.91.0",
29
+ # litellm.completion(num_retries=...) imports tenacity lazily and litellm does not declare
30
+ # it, so every model call on a clean install died with "tenacity import failed". We pass
31
+ # num_retries on every call, which made that every call.
32
+ "tenacity>=8.0",
33
+ "pydantic>=2.0",
34
+ "python-dotenv>=1.0.1",
35
+ "markdown-it-py>=3.0.0",
36
+ # Only for asking an endpoint which models it serves. It arrives anyway under litellm, but
37
+ # a direct import belongs in a direct dependency -- otherwise it vanishes the day litellm
38
+ # changes its HTTP client.
39
+ "httpx>=0.27",
40
+ # The XML and HTML readers (JATS, Elsevier, publisher pages) and the Word reader. Small, and
41
+ # no machine-learning model is involved in reading a structured format.
42
+ "lxml>=5.0",
43
+ "python-docx>=1.1",
44
+ ]
45
+
46
+ [project.optional-dependencies]
47
+ # PDF reading: docling and its layout/table models, via PyTorch -- a download of several GB.
48
+ # Kept out of the base install so the structured formats work everywhere in seconds.
49
+ pdf = ["docling>=2.110.0"]
50
+
51
+ [project.scripts]
52
+ file2records = "file2records.cli:main"
53
+
54
+ [project.urls]
55
+ Homepage = "https://github.com/Enkhnyam/chemistry-data-extractor-toolkit"
56
+ Repository = "https://github.com/Enkhnyam/chemistry-data-extractor-toolkit"
57
+ Issues = "https://github.com/Enkhnyam/chemistry-data-extractor-toolkit/issues"
58
+
59
+ [build-system]
60
+ requires = ["hatchling>=1.27"]
61
+ build-backend = "hatchling.build"
62
+
63
+ [tool.hatch.build.targets.wheel]
64
+ packages = ["src/file2records"]
65
+
66
+ [tool.hatch.build.targets.sdist]
67
+ include = ["src/file2records", "tests", "README.md", "LICENSE", "CITATION.cff"]
68
+
69
+ [dependency-groups]
70
+ dev = ["docling>=2.110.0"]
@@ -0,0 +1,15 @@
1
+ """file2records: turn a folder of papers into a structured dataset you can check.
2
+
3
+ import file2records as fr
4
+ project = fr.Project("my-review")
5
+ project.add("papers/")
6
+ project.extract(model=fr.rwth())
7
+ project.export("dataset.csv")
8
+
9
+ See project.py for the API, cli.py for the command line, main.py for the web app.
10
+ """
11
+ __version__ = "0.2.0"
12
+
13
+ from .project import Project, rwth # noqa: E402
14
+
15
+ __all__ = ["Project", "rwth", "__version__"]
@@ -0,0 +1,5 @@
1
+ import sys
2
+
3
+ from .cli import main
4
+
5
+ sys.exit(main())
@@ -0,0 +1,155 @@
1
+ """The export bundle: a dataset together with what produced it.
2
+
3
+ A CSV of records on its own is not reproducible -- it cannot say which model wrote it, under
4
+ which prompt, against which schema, or which rows a human then corrected. The bundle is the data
5
+ beside the config and a manifest that pins both.
6
+
7
+ What it leaves out by default matters as much. Papers obtained through publishers' text-and-data
8
+ mining agreements may be read and mined, not redistributed -- and a bundle is made to be shared.
9
+ So the paper text and the source files go in only when asked for (`include_text`,
10
+ `include_files`), which is the right call for open-access papers and the caller's to make.
11
+ Without them every record still names its paper's DOI and the ids of the chunks it came from,
12
+ which is enough for anyone with access to the paper to check it.
13
+
14
+ No key ever enters the bundle; model settings are copied without their key variables.
15
+ """
16
+ import csv
17
+ import io
18
+ import json
19
+ import subprocess
20
+ import time
21
+ import zipfile
22
+ from pathlib import Path
23
+
24
+ from . import __version__, config, models, report, storage
25
+ from .storage import read_json
26
+
27
+
28
+ def csv_bytes(columns, rows) -> bytes:
29
+ buffer = io.StringIO()
30
+ writer = csv.DictWriter(buffer, fieldnames=columns, extrasaction="ignore")
31
+ writer.writeheader()
32
+ writer.writerows(rows)
33
+ return buffer.getvalue().encode("utf-8")
34
+
35
+
36
+ def git_commit() -> str | None:
37
+ """Which checkout produced this, when run from one. None for an installed package, whose
38
+ version is recorded instead."""
39
+ try:
40
+ return subprocess.check_output(["git", "-C", str(Path(__file__).resolve().parent),
41
+ "rev-parse", "HEAD"],
42
+ text=True, stderr=subprocess.DEVNULL, timeout=5).strip()
43
+ except Exception:
44
+ return None
45
+
46
+
47
+ def build(paper_ids: list[str] | None = None, *, include_text: bool = False,
48
+ include_files: bool = False, profiles: list[dict] | None = None) -> bytes:
49
+ """The zip as bytes. `paper_ids` limits it to those papers (None: all of them)."""
50
+ record_columns, record_rows = report.flat_records(paper_ids)
51
+ paper_columns, paper_rows = report.papers_table(paper_ids)
52
+ summary = report.build()
53
+ settings = config.get_settings()
54
+ if profiles is None:
55
+ profiles = [{k: v for k, v in p.items()} for p in models.listing(lambda _: False)]
56
+ chosen = {r["paper_id"] for r in paper_rows}
57
+
58
+ manifest = {
59
+ "exported_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
60
+ "tool": "file2records",
61
+ "version": __version__,
62
+ "git_commit": git_commit(),
63
+ "papers": len(paper_rows),
64
+ "records": len(record_rows),
65
+ "includes_paper_text": include_text,
66
+ "models": {
67
+ "extract": (models.get(settings.get("extract_model", "")) or {}).get("model"),
68
+ "judge": (models.get(settings.get("judge_model", "")) or {}).get("model"),
69
+ # per paper too: a corpus is often built across more than one model
70
+ "per_paper": {r["paper_id"]: {"extract": r["extract_model"], "judge": r["judge_model"]}
71
+ for r in paper_rows},
72
+ },
73
+ "spend": summary.get("totals", {}).get("spend", {}),
74
+ "source_tracking_default": settings.get("source_tracking_default", True),
75
+ }
76
+
77
+ buffer = io.BytesIO()
78
+ with zipfile.ZipFile(buffer, "w", zipfile.ZIP_DEFLATED) as bundle:
79
+ bundle.writestr("manifest.json", json.dumps(manifest, indent=2, ensure_ascii=False))
80
+ bundle.writestr("README.md", _readme(manifest))
81
+ bundle.writestr("data/records.csv", csv_bytes(record_columns, record_rows))
82
+ bundle.writestr("data/records.json", json.dumps(record_rows, indent=1, ensure_ascii=False))
83
+ bundle.writestr("data/papers.csv", csv_bytes(paper_columns, paper_rows))
84
+ bundle.writestr("data/report.json", json.dumps(summary, indent=2, ensure_ascii=False))
85
+
86
+ # config: everything that decides what a run produces, and nothing that authenticates it
87
+ bundle.writestr("config/schema.json",
88
+ json.dumps({"fields": config.get_schema()}, indent=2, ensure_ascii=False))
89
+ bundle.writestr("config/extract_prompt.txt", config.get_extract_prompt())
90
+ bundle.writestr("config/judge_prompt.txt", config.get_judge_prompt())
91
+ bundle.writestr("config/few_shot.json",
92
+ json.dumps(config.get_few_shot(), indent=2, ensure_ascii=False))
93
+ bundle.writestr("config/models.json", json.dumps(
94
+ [{k: v for k, v in m.items()
95
+ if k in ("id", "name", "model", "api_base", "api_version")} for m in profiles],
96
+ indent=2, ensure_ascii=False))
97
+
98
+ # the per-paper working files, so a reviewer can trace any row back to its chunk
99
+ for stage, directory in (("extracted", storage.EXTRACTED), ("judged", storage.JUDGED)):
100
+ for path in sorted(directory.glob("*.json")):
101
+ if path.stem in chosen:
102
+ bundle.write(path, f"{stage}/{path.name}")
103
+ for path in sorted(storage.PARSED.glob("*.json")):
104
+ if path.stem not in chosen:
105
+ continue
106
+ paper = read_json(path, {})
107
+ if not include_text:
108
+ paper["chunks"] = [{"id": c["id"]} for c in paper.get("chunks", [])]
109
+ bundle.writestr(f"parsed/{path.name}", json.dumps(paper, indent=1, ensure_ascii=False))
110
+ if include_files:
111
+ for pid in sorted(chosen):
112
+ if (source := storage.source_file(pid)) is not None:
113
+ bundle.write(source, f"papers/{source.name}")
114
+ return buffer.getvalue()
115
+
116
+
117
+ def _readme(manifest: dict) -> str:
118
+ m = manifest["models"]
119
+ text_note = (
120
+ "`chunks[].text` holds the paper text, because this bundle was exported with it. Check "
121
+ "the papers' licences before sharing it."
122
+ if manifest["includes_paper_text"] else
123
+ "Paper text is not included -- papers obtained under text-and-data-mining terms may be "
124
+ "mined, not redistributed. Each chunk keeps its `id`, and each record its paper's DOI, "
125
+ "so anyone with access to the paper can check a value against its source.")
126
+ return f"""# file2records bundle
127
+
128
+ {manifest['records']} records from {manifest['papers']} paper(s), exported
129
+ {manifest['exported_at']} by file2records {manifest['version']}.
130
+
131
+ ## What is here
132
+
133
+ - `data/records.csv`, `data/records.json` — one row per record, with the paper's DOI, the
134
+ judge's verdict and any reviewer flag or note. `extract_model` and `judge_model` say what
135
+ produced each row.
136
+ - `data/papers.csv` — one row per paper: DOI, chunks in, records out, model, tokens, cost.
137
+ - `data/report.json` — completeness by field, what the judge changed, totals.
138
+ - `config/` — the schema, both prompts, the worked examples and the model settings that
139
+ produced this. API keys are not included.
140
+ - `parsed/`, `extracted/`, `judged/` — the working files, so any row can be traced back to the
141
+ chunk it came from. `source_chunk_ids` on a record refers to `chunks[].id` in `parsed/`.
142
+ - `papers/` — the source files, if you exported with them.
143
+
144
+ {text_note}
145
+
146
+ ## Reading it
147
+
148
+ Extraction model: `{m['extract'] or 'not set'}` · judge model: `{m['judge'] or 'not set'}`.
149
+ Papers may differ from these if the corpus was built across more than one model; see
150
+ `models.per_paper` in `manifest.json`.
151
+
152
+ `model_records` in `extracted/*.json` is what the model originally said, kept beside the
153
+ corrected records the moment anything was edited. A corrected dataset that cannot be diffed
154
+ against the model's own output is not evidence of anything.
155
+ """
@@ -0,0 +1,207 @@
1
+ """The `file2records` command.
2
+
3
+ file2records serve my-review the web app on that project folder
4
+ file2records add my-review papers/ read files and folders into it
5
+ file2records papers my-review what is in it
6
+ file2records search my-review "glycoly[sz]is"
7
+ file2records check my-review what is missing before a run
8
+ file2records extract my-review --only "glycoly[sz]is" --exclude "positron|tomograph"
9
+ file2records judge my-review
10
+ file2records export my-review dataset.csv (.csv, .json, or .zip for the full bundle)
11
+
12
+ Every command takes the project folder first. Model commands use the model chosen in the web
13
+ app's Settings unless --model is given: any litellm model string, or rwth/<name> for RWTH's
14
+ KI:connect (key from RWTH_API_KEY).
15
+ """
16
+ import argparse
17
+ import os
18
+ import sys
19
+ from pathlib import Path
20
+
21
+ from dotenv import load_dotenv
22
+
23
+ from . import __version__
24
+
25
+
26
+ def _project(folder):
27
+ from .project import Project
28
+ project = Project(folder)
29
+ # Keys saved through the web app live in the project's .env; a .env in the current folder
30
+ # is read too. Neither overrides a variable already set in the shell.
31
+ load_dotenv(project.path / ".env")
32
+ load_dotenv(Path.cwd() / ".env")
33
+ return project
34
+
35
+
36
+ def _filters(p):
37
+ p.add_argument("--only", metavar="REGEX",
38
+ help="only papers whose full text matches this regular expression")
39
+ p.add_argument("--exclude", metavar="REGEX",
40
+ help="skip papers whose full text matches this regular expression")
41
+
42
+
43
+ def _print_result(r):
44
+ name = r.get("filename") or r["id"]
45
+ if r.get("error"):
46
+ print(f" ✗ {name}: {r['error']}")
47
+ elif "n_chunks" in r:
48
+ doi = f" doi:{r['doi']}" if r.get("doi") else ""
49
+ print(f" ✓ {name} [{r['format']}] {r['n_chunks']} chunks{doi}")
50
+ elif "n_records" in r:
51
+ print(f" ✓ {name}: {r['n_records']} records in {r['seconds']}s")
52
+ else:
53
+ print(f" ✓ {name}: {r['n_verdicts']} verdicts in {r['seconds']}s")
54
+
55
+
56
+ def cmd_serve(args):
57
+ os.environ["WORKSPACE_DIR"] = str(Path(args.folder).resolve())
58
+ _project(args.folder) # creates the folder, opens it
59
+ import uvicorn
60
+ from .main import app
61
+ url = f"http://{args.host}:{args.port}"
62
+ print(f"file2records {__version__} — project {Path(args.folder).resolve()}\nOpen {url}")
63
+ if not args.no_browser:
64
+ import threading
65
+ import webbrowser
66
+ threading.Timer(1.0, webbrowser.open, [url]).start()
67
+ uvicorn.run(app, host=args.host, port=args.port, log_level="warning")
68
+
69
+
70
+ def cmd_add(args):
71
+ project = _project(args.folder)
72
+ print(f"Reading into {project.path}")
73
+ results = project.add(*args.paths, source_tracking=not args.no_source_tracking,
74
+ on_file=_print_result)
75
+ failed = sum(1 for r in results if r.get("error"))
76
+ print(f"{len(results) - failed} added, {failed} failed")
77
+ return 1 if failed and failed == len(results) else 0
78
+
79
+
80
+ def cmd_papers(args):
81
+ papers = _project(args.folder).papers()
82
+ if not papers:
83
+ print("No papers yet. Add some: file2records add <folder> <files or folders>")
84
+ for p in papers:
85
+ state = "judged" if p["judged"] else "extracted" if p["extracted"] else "parsed"
86
+ records = "" if p["n_records"] is None else f"{p['n_records']} records"
87
+ print(f"{p['id']:50.50} {p['format']:8} {state:9} {records:11} {p['doi']}")
88
+
89
+
90
+ def cmd_search(args):
91
+ hits = _project(args.folder).search(args.pattern, ignore_case=not args.case_sensitive)
92
+ for h in hits:
93
+ print(f"\n{h['filename']} ({h['matches']} matches)")
94
+ for s in h["snippets"]:
95
+ print(f" …{s['before'][-60:]}[{s['match']}]{s['after'][:60]}…".replace("\n", " "))
96
+ total = sum(h["matches"] for h in hits)
97
+ print(f"\n{len(hits)} papers, {total} matches")
98
+
99
+
100
+ def cmd_check(args):
101
+ project = _project(args.folder)
102
+ ok = True
103
+ for stage in ("extract", "judge"):
104
+ missing = project.check(stage, args.model)
105
+ ok &= stage == "judge" or not missing
106
+ print(f"{stage}: {'ready' if not missing else 'not ready'}")
107
+ for m in missing:
108
+ hint = " Or pass --model, e.g. rwth/gpt-oss-120b." if m.startswith("Choose a model") else ""
109
+ print(f" - {m}{hint}")
110
+ return 0 if ok else 1
111
+
112
+
113
+ def _run_stage(args, stage):
114
+ project = _project(args.folder)
115
+ run = project.extract if stage == "extract" else project.judge
116
+ results = run(args.model, only=args.only, exclude=args.exclude, redo=args.redo,
117
+ on_paper=_print_result)
118
+ if not results:
119
+ print(f"Nothing to {stage}: every chosen paper is done already (--redo to run again).")
120
+ failed = sum(1 for r in results if r.get("error"))
121
+ print(f"{len(results) - failed} done, {failed} failed")
122
+ return 1 if failed else 0
123
+
124
+
125
+ def cmd_export(args):
126
+ path = _project(args.folder).export(args.output, only=args.only, exclude=args.exclude,
127
+ include_text=args.include_text,
128
+ include_files=args.include_files)
129
+ print(f"Wrote {path}")
130
+
131
+
132
+ FOLDER = "the project folder (created if it does not exist)"
133
+
134
+
135
+ def build_parser() -> argparse.ArgumentParser:
136
+ parser = argparse.ArgumentParser(
137
+ prog="file2records",
138
+ description="Turn papers (PDF, XML, HTML, Word, Markdown) into a structured dataset.",
139
+ epilog="examples:\n" + "\n".join(__doc__.splitlines()[2:10]) +
140
+ "\n\ndocs: https://enkhnyam.github.io/chemistry-data-extractor-toolkit/",
141
+ formatter_class=argparse.RawDescriptionHelpFormatter)
142
+ parser.add_argument("--version", action="version", version=f"file2records {__version__}")
143
+ sub = parser.add_subparsers(dest="command", required=True, metavar="COMMAND")
144
+
145
+ p = sub.add_parser("serve", help="open the web app on a project folder")
146
+ p.add_argument("folder", nargs="?", default="workspace", help=FOLDER + " (default: %(default)s)")
147
+ p.add_argument("--port", type=int, default=8000)
148
+ p.add_argument("--host", default="127.0.0.1",
149
+ help="keep the default unless you know why: the app has no login")
150
+ p.add_argument("--no-browser", action="store_true")
151
+ p.set_defaults(func=cmd_serve)
152
+
153
+ p = sub.add_parser("add", help="read files or folders of papers into a project")
154
+ p.add_argument("folder", help=FOLDER)
155
+ p.add_argument("paths", nargs="+", help="paper files, or folders of them")
156
+ p.add_argument("--no-source-tracking", action="store_true",
157
+ help="don't tag chunks, so records won't cite the passage they came from")
158
+ p.set_defaults(func=cmd_add)
159
+
160
+ p = sub.add_parser("papers", help="list the papers in a project")
161
+ p.add_argument("folder", help=FOLDER)
162
+ p.set_defaults(func=cmd_papers)
163
+
164
+ p = sub.add_parser("search", help="search the full text of every paper with a regex")
165
+ p.add_argument("folder", help=FOLDER)
166
+ p.add_argument("pattern", help="a regular expression, e.g. \"glycoly[sz]is\"")
167
+ p.add_argument("--case-sensitive", action="store_true")
168
+ p.set_defaults(func=cmd_search)
169
+
170
+ p = sub.add_parser("check", help="say what is missing before a run")
171
+ p.add_argument("folder", help=FOLDER)
172
+ p.add_argument("--model")
173
+ p.set_defaults(func=cmd_check)
174
+
175
+ for stage, text in (("extract", "extract records from papers not extracted yet"),
176
+ ("judge", "audit extracted records with a second model")):
177
+ p = sub.add_parser(stage, help=text)
178
+ p.add_argument("folder", help=FOLDER)
179
+ p.add_argument("--model", help='litellm model string, or rwth/<name> '
180
+ '(default: the model chosen in Settings)')
181
+ p.add_argument("--redo", action="store_true", help="also re-run papers already done")
182
+ _filters(p)
183
+ p.set_defaults(func=lambda a, s=stage: _run_stage(a, s))
184
+
185
+ p = sub.add_parser("export", help="write records to .csv / .json, or a .zip bundle")
186
+ p.add_argument("folder", help=FOLDER)
187
+ p.add_argument("output", help="a .csv or .json file of records, or a .zip bundle")
188
+ p.add_argument("--include-text", action="store_true",
189
+ help="put the paper text in a .zip bundle (check the papers' licences)")
190
+ p.add_argument("--include-files", action="store_true",
191
+ help="put the source files in a .zip bundle (check the papers' licences)")
192
+ _filters(p)
193
+ p.set_defaults(func=cmd_export)
194
+ return parser
195
+
196
+
197
+ def main(argv=None) -> int:
198
+ args = build_parser().parse_args(argv)
199
+ try:
200
+ return args.func(args) or 0
201
+ except (ValueError, RuntimeError, FileNotFoundError) as e:
202
+ print(f"error: {e}", file=sys.stderr)
203
+ return 2
204
+
205
+
206
+ if __name__ == "__main__":
207
+ sys.exit(main())