file2records 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- file2records/__init__.py +15 -0
- file2records/__main__.py +5 -0
- file2records/bundle.py +155 -0
- file2records/cli.py +207 -0
- file2records/config.py +96 -0
- file2records/demo/NOTICE.md +34 -0
- file2records/demo/config/extract_prompt.txt +51 -0
- file2records/demo/config/few_shot.json +1 -0
- file2records/demo/config/judge_prompt.txt +103 -0
- file2records/demo/config/schema.json +54 -0
- file2records/demo/config/settings.json +3 -0
- file2records/demo/extracted/lewis-acidic-ionic-liquids-for-pet-glycolysis-polymers-2013-bc017719.json +268 -0
- file2records/demo/judged/lewis-acidic-ionic-liquids-for-pet-glycolysis-polymers-2013-bc017719.json +517 -0
- file2records/demo/parsed/lewis-acidic-ionic-liquids-for-pet-glycolysis-polymers-2013-bc017719.json +408 -0
- file2records/demo/pdfs/lewis-acidic-ionic-liquids-for-pet-glycolysis-polymers-2013-bc017719.pdf +0 -0
- file2records/demo.py +155 -0
- file2records/dynschema.py +37 -0
- file2records/exemplar.py +171 -0
- file2records/extraction.py +40 -0
- file2records/filters.py +81 -0
- file2records/judge.py +94 -0
- file2records/llm.py +79 -0
- file2records/main.py +768 -0
- file2records/models.py +105 -0
- file2records/parsing.py +129 -0
- file2records/pipeline.py +206 -0
- file2records/project.py +203 -0
- file2records/readers.py +385 -0
- file2records/report.py +230 -0
- file2records/storage.py +132 -0
- file2records/timings.py +50 -0
- file2records/web/app.js +2340 -0
- file2records/web/index.html +28 -0
- file2records/web/style.css +483 -0
- file2records-0.2.0.dist-info/METADATA +75 -0
- file2records-0.2.0.dist-info/RECORD +39 -0
- file2records-0.2.0.dist-info/WHEEL +4 -0
- file2records-0.2.0.dist-info/entry_points.txt +2 -0
- file2records-0.2.0.dist-info/licenses/LICENSE +21 -0
file2records/__init__.py
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""file2records: turn a folder of papers into a structured dataset you can check.
|
|
2
|
+
|
|
3
|
+
import file2records as fr
|
|
4
|
+
project = fr.Project("my-review")
|
|
5
|
+
project.add("papers/")
|
|
6
|
+
project.extract(model=fr.rwth())
|
|
7
|
+
project.export("dataset.csv")
|
|
8
|
+
|
|
9
|
+
See project.py for the API, cli.py for the command line, main.py for the web app.
|
|
10
|
+
"""
|
|
11
|
+
__version__ = "0.2.0"
|
|
12
|
+
|
|
13
|
+
from .project import Project, rwth # noqa: E402
|
|
14
|
+
|
|
15
|
+
__all__ = ["Project", "rwth", "__version__"]
|
file2records/__main__.py
ADDED
file2records/bundle.py
ADDED
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
"""The export bundle: a dataset together with what produced it.
|
|
2
|
+
|
|
3
|
+
A CSV of records on its own is not reproducible -- it cannot say which model wrote it, under
|
|
4
|
+
which prompt, against which schema, or which rows a human then corrected. The bundle is the data
|
|
5
|
+
beside the config and a manifest that pins both.
|
|
6
|
+
|
|
7
|
+
What it leaves out by default matters as much. Papers obtained through publishers' text-and-data
|
|
8
|
+
mining agreements may be read and mined, not redistributed -- and a bundle is made to be shared.
|
|
9
|
+
So the paper text and the source files go in only when asked for (`include_text`,
|
|
10
|
+
`include_files`), which is the right call for open-access papers and the caller's to make.
|
|
11
|
+
Without them every record still names its paper's DOI and the ids of the chunks it came from,
|
|
12
|
+
which is enough for anyone with access to the paper to check it.
|
|
13
|
+
|
|
14
|
+
No key ever enters the bundle; model settings are copied without their key variables.
|
|
15
|
+
"""
|
|
16
|
+
import csv
|
|
17
|
+
import io
|
|
18
|
+
import json
|
|
19
|
+
import subprocess
|
|
20
|
+
import time
|
|
21
|
+
import zipfile
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
|
|
24
|
+
from . import __version__, config, models, report, storage
|
|
25
|
+
from .storage import read_json
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def csv_bytes(columns, rows) -> bytes:
|
|
29
|
+
buffer = io.StringIO()
|
|
30
|
+
writer = csv.DictWriter(buffer, fieldnames=columns, extrasaction="ignore")
|
|
31
|
+
writer.writeheader()
|
|
32
|
+
writer.writerows(rows)
|
|
33
|
+
return buffer.getvalue().encode("utf-8")
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def git_commit() -> str | None:
|
|
37
|
+
"""Which checkout produced this, when run from one. None for an installed package, whose
|
|
38
|
+
version is recorded instead."""
|
|
39
|
+
try:
|
|
40
|
+
return subprocess.check_output(["git", "-C", str(Path(__file__).resolve().parent),
|
|
41
|
+
"rev-parse", "HEAD"],
|
|
42
|
+
text=True, stderr=subprocess.DEVNULL, timeout=5).strip()
|
|
43
|
+
except Exception:
|
|
44
|
+
return None
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def build(paper_ids: list[str] | None = None, *, include_text: bool = False,
|
|
48
|
+
include_files: bool = False, profiles: list[dict] | None = None) -> bytes:
|
|
49
|
+
"""The zip as bytes. `paper_ids` limits it to those papers (None: all of them)."""
|
|
50
|
+
record_columns, record_rows = report.flat_records(paper_ids)
|
|
51
|
+
paper_columns, paper_rows = report.papers_table(paper_ids)
|
|
52
|
+
summary = report.build()
|
|
53
|
+
settings = config.get_settings()
|
|
54
|
+
if profiles is None:
|
|
55
|
+
profiles = [{k: v for k, v in p.items()} for p in models.listing(lambda _: False)]
|
|
56
|
+
chosen = {r["paper_id"] for r in paper_rows}
|
|
57
|
+
|
|
58
|
+
manifest = {
|
|
59
|
+
"exported_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
|
|
60
|
+
"tool": "file2records",
|
|
61
|
+
"version": __version__,
|
|
62
|
+
"git_commit": git_commit(),
|
|
63
|
+
"papers": len(paper_rows),
|
|
64
|
+
"records": len(record_rows),
|
|
65
|
+
"includes_paper_text": include_text,
|
|
66
|
+
"models": {
|
|
67
|
+
"extract": (models.get(settings.get("extract_model", "")) or {}).get("model"),
|
|
68
|
+
"judge": (models.get(settings.get("judge_model", "")) or {}).get("model"),
|
|
69
|
+
# per paper too: a corpus is often built across more than one model
|
|
70
|
+
"per_paper": {r["paper_id"]: {"extract": r["extract_model"], "judge": r["judge_model"]}
|
|
71
|
+
for r in paper_rows},
|
|
72
|
+
},
|
|
73
|
+
"spend": summary.get("totals", {}).get("spend", {}),
|
|
74
|
+
"source_tracking_default": settings.get("source_tracking_default", True),
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
buffer = io.BytesIO()
|
|
78
|
+
with zipfile.ZipFile(buffer, "w", zipfile.ZIP_DEFLATED) as bundle:
|
|
79
|
+
bundle.writestr("manifest.json", json.dumps(manifest, indent=2, ensure_ascii=False))
|
|
80
|
+
bundle.writestr("README.md", _readme(manifest))
|
|
81
|
+
bundle.writestr("data/records.csv", csv_bytes(record_columns, record_rows))
|
|
82
|
+
bundle.writestr("data/records.json", json.dumps(record_rows, indent=1, ensure_ascii=False))
|
|
83
|
+
bundle.writestr("data/papers.csv", csv_bytes(paper_columns, paper_rows))
|
|
84
|
+
bundle.writestr("data/report.json", json.dumps(summary, indent=2, ensure_ascii=False))
|
|
85
|
+
|
|
86
|
+
# config: everything that decides what a run produces, and nothing that authenticates it
|
|
87
|
+
bundle.writestr("config/schema.json",
|
|
88
|
+
json.dumps({"fields": config.get_schema()}, indent=2, ensure_ascii=False))
|
|
89
|
+
bundle.writestr("config/extract_prompt.txt", config.get_extract_prompt())
|
|
90
|
+
bundle.writestr("config/judge_prompt.txt", config.get_judge_prompt())
|
|
91
|
+
bundle.writestr("config/few_shot.json",
|
|
92
|
+
json.dumps(config.get_few_shot(), indent=2, ensure_ascii=False))
|
|
93
|
+
bundle.writestr("config/models.json", json.dumps(
|
|
94
|
+
[{k: v for k, v in m.items()
|
|
95
|
+
if k in ("id", "name", "model", "api_base", "api_version")} for m in profiles],
|
|
96
|
+
indent=2, ensure_ascii=False))
|
|
97
|
+
|
|
98
|
+
# the per-paper working files, so a reviewer can trace any row back to its chunk
|
|
99
|
+
for stage, directory in (("extracted", storage.EXTRACTED), ("judged", storage.JUDGED)):
|
|
100
|
+
for path in sorted(directory.glob("*.json")):
|
|
101
|
+
if path.stem in chosen:
|
|
102
|
+
bundle.write(path, f"{stage}/{path.name}")
|
|
103
|
+
for path in sorted(storage.PARSED.glob("*.json")):
|
|
104
|
+
if path.stem not in chosen:
|
|
105
|
+
continue
|
|
106
|
+
paper = read_json(path, {})
|
|
107
|
+
if not include_text:
|
|
108
|
+
paper["chunks"] = [{"id": c["id"]} for c in paper.get("chunks", [])]
|
|
109
|
+
bundle.writestr(f"parsed/{path.name}", json.dumps(paper, indent=1, ensure_ascii=False))
|
|
110
|
+
if include_files:
|
|
111
|
+
for pid in sorted(chosen):
|
|
112
|
+
if (source := storage.source_file(pid)) is not None:
|
|
113
|
+
bundle.write(source, f"papers/{source.name}")
|
|
114
|
+
return buffer.getvalue()
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _readme(manifest: dict) -> str:
|
|
118
|
+
m = manifest["models"]
|
|
119
|
+
text_note = (
|
|
120
|
+
"`chunks[].text` holds the paper text, because this bundle was exported with it. Check "
|
|
121
|
+
"the papers' licences before sharing it."
|
|
122
|
+
if manifest["includes_paper_text"] else
|
|
123
|
+
"Paper text is not included -- papers obtained under text-and-data-mining terms may be "
|
|
124
|
+
"mined, not redistributed. Each chunk keeps its `id`, and each record its paper's DOI, "
|
|
125
|
+
"so anyone with access to the paper can check a value against its source.")
|
|
126
|
+
return f"""# file2records bundle
|
|
127
|
+
|
|
128
|
+
{manifest['records']} records from {manifest['papers']} paper(s), exported
|
|
129
|
+
{manifest['exported_at']} by file2records {manifest['version']}.
|
|
130
|
+
|
|
131
|
+
## What is here
|
|
132
|
+
|
|
133
|
+
- `data/records.csv`, `data/records.json` — one row per record, with the paper's DOI, the
|
|
134
|
+
judge's verdict and any reviewer flag or note. `extract_model` and `judge_model` say what
|
|
135
|
+
produced each row.
|
|
136
|
+
- `data/papers.csv` — one row per paper: DOI, chunks in, records out, model, tokens, cost.
|
|
137
|
+
- `data/report.json` — completeness by field, what the judge changed, totals.
|
|
138
|
+
- `config/` — the schema, both prompts, the worked examples and the model settings that
|
|
139
|
+
produced this. API keys are not included.
|
|
140
|
+
- `parsed/`, `extracted/`, `judged/` — the working files, so any row can be traced back to the
|
|
141
|
+
chunk it came from. `source_chunk_ids` on a record refers to `chunks[].id` in `parsed/`.
|
|
142
|
+
- `papers/` — the source files, if you exported with them.
|
|
143
|
+
|
|
144
|
+
{text_note}
|
|
145
|
+
|
|
146
|
+
## Reading it
|
|
147
|
+
|
|
148
|
+
Extraction model: `{m['extract'] or 'not set'}` · judge model: `{m['judge'] or 'not set'}`.
|
|
149
|
+
Papers may differ from these if the corpus was built across more than one model; see
|
|
150
|
+
`models.per_paper` in `manifest.json`.
|
|
151
|
+
|
|
152
|
+
`model_records` in `extracted/*.json` is what the model originally said, kept beside the
|
|
153
|
+
corrected records the moment anything was edited. A corrected dataset that cannot be diffed
|
|
154
|
+
against the model's own output is not evidence of anything.
|
|
155
|
+
"""
|
file2records/cli.py
ADDED
|
@@ -0,0 +1,207 @@
|
|
|
1
|
+
"""The `file2records` command.
|
|
2
|
+
|
|
3
|
+
file2records serve my-review the web app on that project folder
|
|
4
|
+
file2records add my-review papers/ read files and folders into it
|
|
5
|
+
file2records papers my-review what is in it
|
|
6
|
+
file2records search my-review "glycoly[sz]is"
|
|
7
|
+
file2records check my-review what is missing before a run
|
|
8
|
+
file2records extract my-review --only "glycoly[sz]is" --exclude "positron|tomograph"
|
|
9
|
+
file2records judge my-review
|
|
10
|
+
file2records export my-review dataset.csv (.csv, .json, or .zip for the full bundle)
|
|
11
|
+
|
|
12
|
+
Every command takes the project folder first. Model commands use the model chosen in the web
|
|
13
|
+
app's Settings unless --model is given: any litellm model string, or rwth/<name> for RWTH's
|
|
14
|
+
KI:connect (key from RWTH_API_KEY).
|
|
15
|
+
"""
|
|
16
|
+
import argparse
|
|
17
|
+
import os
|
|
18
|
+
import sys
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
from dotenv import load_dotenv
|
|
22
|
+
|
|
23
|
+
from . import __version__
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _project(folder):
|
|
27
|
+
from .project import Project
|
|
28
|
+
project = Project(folder)
|
|
29
|
+
# Keys saved through the web app live in the project's .env; a .env in the current folder
|
|
30
|
+
# is read too. Neither overrides a variable already set in the shell.
|
|
31
|
+
load_dotenv(project.path / ".env")
|
|
32
|
+
load_dotenv(Path.cwd() / ".env")
|
|
33
|
+
return project
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _filters(p):
|
|
37
|
+
p.add_argument("--only", metavar="REGEX",
|
|
38
|
+
help="only papers whose full text matches this regular expression")
|
|
39
|
+
p.add_argument("--exclude", metavar="REGEX",
|
|
40
|
+
help="skip papers whose full text matches this regular expression")
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _print_result(r):
|
|
44
|
+
name = r.get("filename") or r["id"]
|
|
45
|
+
if r.get("error"):
|
|
46
|
+
print(f" ✗ {name}: {r['error']}")
|
|
47
|
+
elif "n_chunks" in r:
|
|
48
|
+
doi = f" doi:{r['doi']}" if r.get("doi") else ""
|
|
49
|
+
print(f" ✓ {name} [{r['format']}] {r['n_chunks']} chunks{doi}")
|
|
50
|
+
elif "n_records" in r:
|
|
51
|
+
print(f" ✓ {name}: {r['n_records']} records in {r['seconds']}s")
|
|
52
|
+
else:
|
|
53
|
+
print(f" ✓ {name}: {r['n_verdicts']} verdicts in {r['seconds']}s")
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def cmd_serve(args):
|
|
57
|
+
os.environ["WORKSPACE_DIR"] = str(Path(args.folder).resolve())
|
|
58
|
+
_project(args.folder) # creates the folder, opens it
|
|
59
|
+
import uvicorn
|
|
60
|
+
from .main import app
|
|
61
|
+
url = f"http://{args.host}:{args.port}"
|
|
62
|
+
print(f"file2records {__version__} — project {Path(args.folder).resolve()}\nOpen {url}")
|
|
63
|
+
if not args.no_browser:
|
|
64
|
+
import threading
|
|
65
|
+
import webbrowser
|
|
66
|
+
threading.Timer(1.0, webbrowser.open, [url]).start()
|
|
67
|
+
uvicorn.run(app, host=args.host, port=args.port, log_level="warning")
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def cmd_add(args):
|
|
71
|
+
project = _project(args.folder)
|
|
72
|
+
print(f"Reading into {project.path}")
|
|
73
|
+
results = project.add(*args.paths, source_tracking=not args.no_source_tracking,
|
|
74
|
+
on_file=_print_result)
|
|
75
|
+
failed = sum(1 for r in results if r.get("error"))
|
|
76
|
+
print(f"{len(results) - failed} added, {failed} failed")
|
|
77
|
+
return 1 if failed and failed == len(results) else 0
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def cmd_papers(args):
|
|
81
|
+
papers = _project(args.folder).papers()
|
|
82
|
+
if not papers:
|
|
83
|
+
print("No papers yet. Add some: file2records add <folder> <files or folders>")
|
|
84
|
+
for p in papers:
|
|
85
|
+
state = "judged" if p["judged"] else "extracted" if p["extracted"] else "parsed"
|
|
86
|
+
records = "" if p["n_records"] is None else f"{p['n_records']} records"
|
|
87
|
+
print(f"{p['id']:50.50} {p['format']:8} {state:9} {records:11} {p['doi']}")
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def cmd_search(args):
|
|
91
|
+
hits = _project(args.folder).search(args.pattern, ignore_case=not args.case_sensitive)
|
|
92
|
+
for h in hits:
|
|
93
|
+
print(f"\n{h['filename']} ({h['matches']} matches)")
|
|
94
|
+
for s in h["snippets"]:
|
|
95
|
+
print(f" …{s['before'][-60:]}[{s['match']}]{s['after'][:60]}…".replace("\n", " "))
|
|
96
|
+
total = sum(h["matches"] for h in hits)
|
|
97
|
+
print(f"\n{len(hits)} papers, {total} matches")
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def cmd_check(args):
|
|
101
|
+
project = _project(args.folder)
|
|
102
|
+
ok = True
|
|
103
|
+
for stage in ("extract", "judge"):
|
|
104
|
+
missing = project.check(stage, args.model)
|
|
105
|
+
ok &= stage == "judge" or not missing
|
|
106
|
+
print(f"{stage}: {'ready' if not missing else 'not ready'}")
|
|
107
|
+
for m in missing:
|
|
108
|
+
hint = " Or pass --model, e.g. rwth/gpt-oss-120b." if m.startswith("Choose a model") else ""
|
|
109
|
+
print(f" - {m}{hint}")
|
|
110
|
+
return 0 if ok else 1
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _run_stage(args, stage):
|
|
114
|
+
project = _project(args.folder)
|
|
115
|
+
run = project.extract if stage == "extract" else project.judge
|
|
116
|
+
results = run(args.model, only=args.only, exclude=args.exclude, redo=args.redo,
|
|
117
|
+
on_paper=_print_result)
|
|
118
|
+
if not results:
|
|
119
|
+
print(f"Nothing to {stage}: every chosen paper is done already (--redo to run again).")
|
|
120
|
+
failed = sum(1 for r in results if r.get("error"))
|
|
121
|
+
print(f"{len(results) - failed} done, {failed} failed")
|
|
122
|
+
return 1 if failed else 0
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def cmd_export(args):
|
|
126
|
+
path = _project(args.folder).export(args.output, only=args.only, exclude=args.exclude,
|
|
127
|
+
include_text=args.include_text,
|
|
128
|
+
include_files=args.include_files)
|
|
129
|
+
print(f"Wrote {path}")
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
FOLDER = "the project folder (created if it does not exist)"
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
136
|
+
parser = argparse.ArgumentParser(
|
|
137
|
+
prog="file2records",
|
|
138
|
+
description="Turn papers (PDF, XML, HTML, Word, Markdown) into a structured dataset.",
|
|
139
|
+
epilog="examples:\n" + "\n".join(__doc__.splitlines()[2:10]) +
|
|
140
|
+
"\n\ndocs: https://enkhnyam.github.io/chemistry-data-extractor-toolkit/",
|
|
141
|
+
formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
142
|
+
parser.add_argument("--version", action="version", version=f"file2records {__version__}")
|
|
143
|
+
sub = parser.add_subparsers(dest="command", required=True, metavar="COMMAND")
|
|
144
|
+
|
|
145
|
+
p = sub.add_parser("serve", help="open the web app on a project folder")
|
|
146
|
+
p.add_argument("folder", nargs="?", default="workspace", help=FOLDER + " (default: %(default)s)")
|
|
147
|
+
p.add_argument("--port", type=int, default=8000)
|
|
148
|
+
p.add_argument("--host", default="127.0.0.1",
|
|
149
|
+
help="keep the default unless you know why: the app has no login")
|
|
150
|
+
p.add_argument("--no-browser", action="store_true")
|
|
151
|
+
p.set_defaults(func=cmd_serve)
|
|
152
|
+
|
|
153
|
+
p = sub.add_parser("add", help="read files or folders of papers into a project")
|
|
154
|
+
p.add_argument("folder", help=FOLDER)
|
|
155
|
+
p.add_argument("paths", nargs="+", help="paper files, or folders of them")
|
|
156
|
+
p.add_argument("--no-source-tracking", action="store_true",
|
|
157
|
+
help="don't tag chunks, so records won't cite the passage they came from")
|
|
158
|
+
p.set_defaults(func=cmd_add)
|
|
159
|
+
|
|
160
|
+
p = sub.add_parser("papers", help="list the papers in a project")
|
|
161
|
+
p.add_argument("folder", help=FOLDER)
|
|
162
|
+
p.set_defaults(func=cmd_papers)
|
|
163
|
+
|
|
164
|
+
p = sub.add_parser("search", help="search the full text of every paper with a regex")
|
|
165
|
+
p.add_argument("folder", help=FOLDER)
|
|
166
|
+
p.add_argument("pattern", help="a regular expression, e.g. \"glycoly[sz]is\"")
|
|
167
|
+
p.add_argument("--case-sensitive", action="store_true")
|
|
168
|
+
p.set_defaults(func=cmd_search)
|
|
169
|
+
|
|
170
|
+
p = sub.add_parser("check", help="say what is missing before a run")
|
|
171
|
+
p.add_argument("folder", help=FOLDER)
|
|
172
|
+
p.add_argument("--model")
|
|
173
|
+
p.set_defaults(func=cmd_check)
|
|
174
|
+
|
|
175
|
+
for stage, text in (("extract", "extract records from papers not extracted yet"),
|
|
176
|
+
("judge", "audit extracted records with a second model")):
|
|
177
|
+
p = sub.add_parser(stage, help=text)
|
|
178
|
+
p.add_argument("folder", help=FOLDER)
|
|
179
|
+
p.add_argument("--model", help='litellm model string, or rwth/<name> '
|
|
180
|
+
'(default: the model chosen in Settings)')
|
|
181
|
+
p.add_argument("--redo", action="store_true", help="also re-run papers already done")
|
|
182
|
+
_filters(p)
|
|
183
|
+
p.set_defaults(func=lambda a, s=stage: _run_stage(a, s))
|
|
184
|
+
|
|
185
|
+
p = sub.add_parser("export", help="write records to .csv / .json, or a .zip bundle")
|
|
186
|
+
p.add_argument("folder", help=FOLDER)
|
|
187
|
+
p.add_argument("output", help="a .csv or .json file of records, or a .zip bundle")
|
|
188
|
+
p.add_argument("--include-text", action="store_true",
|
|
189
|
+
help="put the paper text in a .zip bundle (check the papers' licences)")
|
|
190
|
+
p.add_argument("--include-files", action="store_true",
|
|
191
|
+
help="put the source files in a .zip bundle (check the papers' licences)")
|
|
192
|
+
_filters(p)
|
|
193
|
+
p.set_defaults(func=cmd_export)
|
|
194
|
+
return parser
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def main(argv=None) -> int:
|
|
198
|
+
args = build_parser().parse_args(argv)
|
|
199
|
+
try:
|
|
200
|
+
return args.func(args) or 0
|
|
201
|
+
except (ValueError, RuntimeError, FileNotFoundError) as e:
|
|
202
|
+
print(f"error: {e}", file=sys.stderr)
|
|
203
|
+
return 2
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
if __name__ == "__main__":
|
|
207
|
+
sys.exit(main())
|
file2records/config.py
ADDED
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
"""Reads/writes the four things that make this tool generic: the schema, the extraction
|
|
2
|
+
prompt, the judge rubric, and few-shot examples. All under workspace/config/, all editable
|
|
3
|
+
from the Settings page -- change these four and the same pipeline runs on a different
|
|
4
|
+
domain, no code changes."""
|
|
5
|
+
from . import exemplar
|
|
6
|
+
from . import storage
|
|
7
|
+
from .storage import read_json, write_json
|
|
8
|
+
from .dynschema import PLACEHOLDER_SCHEMA
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def settings_file():
|
|
13
|
+
return storage.CONFIG / "settings.json"
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def schema_file():
|
|
17
|
+
return storage.CONFIG / "schema.json"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def extract_prompt_file():
|
|
21
|
+
return storage.CONFIG / "extract_prompt.txt"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def judge_prompt_file():
|
|
25
|
+
return storage.CONFIG / "judge_prompt.txt"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def few_shot_file():
|
|
29
|
+
return storage.CONFIG / "few_shot.json"
|
|
30
|
+
|
|
31
|
+
DEFAULT_SETTINGS = {
|
|
32
|
+
# Which named model profile each stage calls; see server/models.py. Empty until chosen.
|
|
33
|
+
"extract_model": "",
|
|
34
|
+
"judge_model": "",
|
|
35
|
+
"source_tracking_default": True, # a real default: provenance on unless turned off
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
def get_settings() -> dict:
|
|
39
|
+
return {**DEFAULT_SETTINGS, **read_json(settings_file(), {})}
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def save_settings(patch: dict) -> dict:
|
|
43
|
+
cfg = get_settings()
|
|
44
|
+
cfg.update(patch)
|
|
45
|
+
write_json(settings_file(), cfg)
|
|
46
|
+
return cfg
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def get_schema() -> list[dict]:
|
|
50
|
+
"""Empty until someone defines one. A shipped default meant every new install began with
|
|
51
|
+
ten PET fields nobody chose, and an extraction that looked like it had worked."""
|
|
52
|
+
return read_json(schema_file(), {"fields": []})["fields"]
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def save_schema(fields: list[dict]) -> None:
|
|
56
|
+
write_json(schema_file(), {"fields": fields})
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _get_text(path) -> str:
|
|
60
|
+
"""A prompt nobody has written yet is empty, not somebody else's.
|
|
61
|
+
|
|
62
|
+
This used to fall back to a built-in PET prompt, which meant a first run quietly extracted
|
|
63
|
+
ionic-liquid chemistry from whatever you uploaded and looked like it had worked. The
|
|
64
|
+
example text is still shown -- as placeholder text in the box -- but it is never the value.
|
|
65
|
+
"""
|
|
66
|
+
return path.read_text(encoding="utf-8") if path.exists() else ""
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def get_extract_prompt() -> str:
|
|
70
|
+
return _get_text(extract_prompt_file())
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def save_extract_prompt(text: str) -> None:
|
|
74
|
+
extract_prompt_file().write_text(text, encoding="utf-8")
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def get_judge_prompt() -> str:
|
|
78
|
+
return _get_text(judge_prompt_file())
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def placeholders() -> dict:
|
|
82
|
+
"""Everything shown as grey example text in an empty field. None of it is ever a value."""
|
|
83
|
+
return {"extract": exemplar.EXTRACT_PROMPT, "judge": exemplar.JUDGE_PROMPT,
|
|
84
|
+
"model": "gpt-4o-mini", "schema": PLACEHOLDER_SCHEMA}
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def save_judge_prompt(text: str) -> None:
|
|
88
|
+
judge_prompt_file().write_text(text, encoding="utf-8")
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def get_few_shot() -> list[dict]:
|
|
92
|
+
return read_json(few_shot_file(), [])
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def save_few_shot(examples: list[dict]) -> None:
|
|
96
|
+
write_json(few_shot_file(), examples)
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
# The demo paper, and why it is here
|
|
2
|
+
|
|
3
|
+
This folder ships one real paper so that a fresh clone has something to run on. It is open
|
|
4
|
+
access under **CC BY 3.0**, on the *publisher's own published version* — which is what makes
|
|
5
|
+
redistributing the PDF itself lawful, rather than only the right to read it.
|
|
6
|
+
|
|
7
|
+
| File | Citation | Licence |
|
|
8
|
+
|---|---|---|
|
|
9
|
+
| `pdfs/…lewis-acidic…pdf` | Qun Feng Yue, Lin Fei Xiao, Mi Lin Zhang and Xue Feng Bai, *The Glycolysis of Poly(ethylene terephthalate) Waste: Lewis Acidic Ionic Liquids as High Efficient Catalysts*, **Polymers** 2013, **5**, 1258–1271. [10.3390/polym5041258](https://doi.org/10.3390/polym5041258) | [CC BY 3.0](https://creativecommons.org/licenses/by/3.0/) |
|
|
10
|
+
|
|
11
|
+
The PDF and the text derived from it under `parsed/` are redistributed unmodified in substance.
|
|
12
|
+
The row above is the attribution the licence requires: creator, title, source and licence, each
|
|
13
|
+
with a link. CC BY 3.0 requires attribution and nothing else: you may redistribute and adapt,
|
|
14
|
+
including commercially, provided the creators are credited as above.
|
|
15
|
+
|
|
16
|
+
The rest of this repository is MIT (see `LICENSE`). This PDF is not MIT; it stays under CC BY,
|
|
17
|
+
and the MIT licence does not extend to it.
|
|
18
|
+
|
|
19
|
+
## Why only one
|
|
20
|
+
|
|
21
|
+
The PET database this toolkit was generalised out of uses seven papers as redistributable
|
|
22
|
+
worked examples. Five of those seven are open access only as a **submitted-version deposit** in
|
|
23
|
+
an institutional repository (mostly `ir.ipe.ac.cn`), under CC BY-NC-SA. That licence covers the
|
|
24
|
+
deposited manuscript, not the publisher's typeset PDF, and it is the publisher's PDF that exists
|
|
25
|
+
on disk. Redistributing the text of those five is fine and is what the database does;
|
|
26
|
+
redistributing the publisher's PDF is not, so they are not shipped here.
|
|
27
|
+
|
|
28
|
+
An earlier version of this demo also shipped *Amino Acid-Based Cholinium Ionic Liquids as
|
|
29
|
+
Sustainable Catalysts for PET Depolymerization* (ACS Sustainable Chem. Eng. 2021,
|
|
30
|
+
[10.1021/acssuschemeng.1c04060](https://doi.org/10.1021/acssuschemeng.1c04060)). That article is
|
|
31
|
+
CC BY 4.0 and redistributing it was lawful, but ACS stamps every PDF download with the IP address
|
|
32
|
+
it was fetched from, and publishing that line alongside the file served nobody. Re-downloading
|
|
33
|
+
does not help — the stamp is applied to every copy — so the paper was removed rather than
|
|
34
|
+
replaced. One paper is enough to show what the tool does.
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
You extract experimental data for PET depolymerisation from a paper's text and tables.
|
|
2
|
+
The reaction is glycolysis, methanolysis or hydrolysis, and the catalyst may be an ionic liquid,
|
|
3
|
+
a metal salt, an acid or base, or none. Treat all three routes the same way.
|
|
4
|
+
SKIP papers and rows for any other route, including aminolysis and enzymatic degradation.
|
|
5
|
+
Extract only what is stated; leave anything unreported as null. The output structure is
|
|
6
|
+
enforced by the response schema, so focus on getting the values and their sources right.
|
|
7
|
+
|
|
8
|
+
## What to extract
|
|
9
|
+
- INCLUDE the current study's ("this work") depolymerisation experiments reported in text and
|
|
10
|
+
tables, whichever route they use.
|
|
11
|
+
- SKIP these rows entirely:
|
|
12
|
+
- literature / cited results — a Source or Reference column, "Ref. N", "[N]", "et al.";
|
|
13
|
+
- RSM / Box-Behnken / optimization tables (coded factors and a single modeled response);
|
|
14
|
+
- data that appears only in a figure/image;
|
|
15
|
+
- characterization or kinetics tables (NMR shifts, bond lengths, Arrhenius) — not depolymerisation runs.
|
|
16
|
+
- Extract EVERY qualifying row: N current-study rows → N records. Don't merge or stop early.
|
|
17
|
+
|
|
18
|
+
## Catalyst naming
|
|
19
|
+
- Use the paper's shorthand in bracket notation, with digits inline rather than subscripts (a
|
|
20
|
+
formula printed with subscript counts is recorded with normal digits).
|
|
21
|
+
- Join physical mixtures as written. No catalyst / uncatalyzed → "none".
|
|
22
|
+
- A catalyst named only by a code from a synthesis recipe (e.g. "IL-1", "Cat-A"): keep the code
|
|
23
|
+
verbatim unless the paper equates it with a clean chemical name. (Reading the code literally is
|
|
24
|
+
what lets the extraction match the ground truth.)
|
|
25
|
+
|
|
26
|
+
## Global reaction conditions
|
|
27
|
+
Conditions stated once — in the methods or a table footnote — apply to EVERY row of that table.
|
|
28
|
+
Propagate them to all records, and convert amounts to grams:
|
|
29
|
+
- catalyst wt% → g: catalyst_g = wt%/100 × PET_g
|
|
30
|
+
- solvent:PET mass ratio → g: solvent_g = ratio × PET_g (the solvent is whichever reagent the
|
|
31
|
+
reaction uses — ethylene glycol, methanol, water, an amine)
|
|
32
|
+
- molar ratio → g via molecular weight (PET repeat unit ≈ 192 g/mol)
|
|
33
|
+
|
|
34
|
+
## Values
|
|
35
|
+
- Ignore placeholder wording ("a certain amount", "specific temperature", a bare symbol) — take
|
|
36
|
+
the concrete value from the table; if there is none, use null.
|
|
37
|
+
- temperature_c: copy exactly, never convert. reaction_time_min: hours × 60.
|
|
38
|
+
- yield_percent, selectivity_percent, conversion_percent are DIFFERENT metrics — put each value
|
|
39
|
+
only in the field the paper labels it as (a conversion column → conversion_percent, a
|
|
40
|
+
selectivity column → selectivity_percent, a yield column → yield_percent).
|
|
41
|
+
- null (not 0) when a value is unreported. pressure_atm: null unless a number is stated.
|
|
42
|
+
|
|
43
|
+
## Skip-these examples (schematic — illustrate the row shape, not real data)
|
|
44
|
+
Literature (cited Source): RSM / optimization (coded factors):
|
|
45
|
+
| Source | Catalyst | Temp | | Run | A:factor | B:factor | Response |
|
|
46
|
+
| Ref. N | <cited> | ... | | 1 | lo | hi | ... |
|
|
47
|
+
|
|
48
|
+
## Source chunks
|
|
49
|
+
The text is split into chunks tagged "ID: <uuid>". For each record, list in source_chunk_ids
|
|
50
|
+
every chunk that supplied a value — usually the table chunk, the footnote/conditions chunk, and
|
|
51
|
+
any chunk that defined a catalyst code.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
[]
|