file2records 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. file2records/__init__.py +15 -0
  2. file2records/__main__.py +5 -0
  3. file2records/bundle.py +155 -0
  4. file2records/cli.py +207 -0
  5. file2records/config.py +96 -0
  6. file2records/demo/NOTICE.md +34 -0
  7. file2records/demo/config/extract_prompt.txt +51 -0
  8. file2records/demo/config/few_shot.json +1 -0
  9. file2records/demo/config/judge_prompt.txt +103 -0
  10. file2records/demo/config/schema.json +54 -0
  11. file2records/demo/config/settings.json +3 -0
  12. file2records/demo/extracted/lewis-acidic-ionic-liquids-for-pet-glycolysis-polymers-2013-bc017719.json +268 -0
  13. file2records/demo/judged/lewis-acidic-ionic-liquids-for-pet-glycolysis-polymers-2013-bc017719.json +517 -0
  14. file2records/demo/parsed/lewis-acidic-ionic-liquids-for-pet-glycolysis-polymers-2013-bc017719.json +408 -0
  15. file2records/demo/pdfs/lewis-acidic-ionic-liquids-for-pet-glycolysis-polymers-2013-bc017719.pdf +0 -0
  16. file2records/demo.py +155 -0
  17. file2records/dynschema.py +37 -0
  18. file2records/exemplar.py +171 -0
  19. file2records/extraction.py +40 -0
  20. file2records/filters.py +81 -0
  21. file2records/judge.py +94 -0
  22. file2records/llm.py +79 -0
  23. file2records/main.py +768 -0
  24. file2records/models.py +105 -0
  25. file2records/parsing.py +129 -0
  26. file2records/pipeline.py +206 -0
  27. file2records/project.py +203 -0
  28. file2records/readers.py +385 -0
  29. file2records/report.py +230 -0
  30. file2records/storage.py +132 -0
  31. file2records/timings.py +50 -0
  32. file2records/web/app.js +2340 -0
  33. file2records/web/index.html +28 -0
  34. file2records/web/style.css +483 -0
  35. file2records-0.2.0.dist-info/METADATA +75 -0
  36. file2records-0.2.0.dist-info/RECORD +39 -0
  37. file2records-0.2.0.dist-info/WHEEL +4 -0
  38. file2records-0.2.0.dist-info/entry_points.txt +2 -0
  39. file2records-0.2.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,15 @@
1
+ """file2records: turn a folder of papers into a structured dataset you can check.
2
+
3
+ import file2records as fr
4
+ project = fr.Project("my-review")
5
+ project.add("papers/")
6
+ project.extract(model=fr.rwth())
7
+ project.export("dataset.csv")
8
+
9
+ See project.py for the API, cli.py for the command line, main.py for the web app.
10
+ """
11
+ __version__ = "0.2.0"
12
+
13
+ from .project import Project, rwth # noqa: E402
14
+
15
+ __all__ = ["Project", "rwth", "__version__"]
@@ -0,0 +1,5 @@
1
+ import sys
2
+
3
+ from .cli import main
4
+
5
+ sys.exit(main())
file2records/bundle.py ADDED
@@ -0,0 +1,155 @@
1
+ """The export bundle: a dataset together with what produced it.
2
+
3
+ A CSV of records on its own is not reproducible -- it cannot say which model wrote it, under
4
+ which prompt, against which schema, or which rows a human then corrected. The bundle is the data
5
+ beside the config and a manifest that pins both.
6
+
7
+ What it leaves out by default matters as much. Papers obtained through publishers' text-and-data
8
+ mining agreements may be read and mined, not redistributed -- and a bundle is made to be shared.
9
+ So the paper text and the source files go in only when asked for (`include_text`,
10
+ `include_files`), which is the right call for open-access papers and the caller's to make.
11
+ Without them every record still names its paper's DOI and the ids of the chunks it came from,
12
+ which is enough for anyone with access to the paper to check it.
13
+
14
+ No key ever enters the bundle; model settings are copied without their key variables.
15
+ """
16
+ import csv
17
+ import io
18
+ import json
19
+ import subprocess
20
+ import time
21
+ import zipfile
22
+ from pathlib import Path
23
+
24
+ from . import __version__, config, models, report, storage
25
+ from .storage import read_json
26
+
27
+
28
+ def csv_bytes(columns, rows) -> bytes:
29
+ buffer = io.StringIO()
30
+ writer = csv.DictWriter(buffer, fieldnames=columns, extrasaction="ignore")
31
+ writer.writeheader()
32
+ writer.writerows(rows)
33
+ return buffer.getvalue().encode("utf-8")
34
+
35
+
36
+ def git_commit() -> str | None:
37
+ """Which checkout produced this, when run from one. None for an installed package, whose
38
+ version is recorded instead."""
39
+ try:
40
+ return subprocess.check_output(["git", "-C", str(Path(__file__).resolve().parent),
41
+ "rev-parse", "HEAD"],
42
+ text=True, stderr=subprocess.DEVNULL, timeout=5).strip()
43
+ except Exception:
44
+ return None
45
+
46
+
47
+ def build(paper_ids: list[str] | None = None, *, include_text: bool = False,
48
+ include_files: bool = False, profiles: list[dict] | None = None) -> bytes:
49
+ """The zip as bytes. `paper_ids` limits it to those papers (None: all of them)."""
50
+ record_columns, record_rows = report.flat_records(paper_ids)
51
+ paper_columns, paper_rows = report.papers_table(paper_ids)
52
+ summary = report.build()
53
+ settings = config.get_settings()
54
+ if profiles is None:
55
+ profiles = [{k: v for k, v in p.items()} for p in models.listing(lambda _: False)]
56
+ chosen = {r["paper_id"] for r in paper_rows}
57
+
58
+ manifest = {
59
+ "exported_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
60
+ "tool": "file2records",
61
+ "version": __version__,
62
+ "git_commit": git_commit(),
63
+ "papers": len(paper_rows),
64
+ "records": len(record_rows),
65
+ "includes_paper_text": include_text,
66
+ "models": {
67
+ "extract": (models.get(settings.get("extract_model", "")) or {}).get("model"),
68
+ "judge": (models.get(settings.get("judge_model", "")) or {}).get("model"),
69
+ # per paper too: a corpus is often built across more than one model
70
+ "per_paper": {r["paper_id"]: {"extract": r["extract_model"], "judge": r["judge_model"]}
71
+ for r in paper_rows},
72
+ },
73
+ "spend": summary.get("totals", {}).get("spend", {}),
74
+ "source_tracking_default": settings.get("source_tracking_default", True),
75
+ }
76
+
77
+ buffer = io.BytesIO()
78
+ with zipfile.ZipFile(buffer, "w", zipfile.ZIP_DEFLATED) as bundle:
79
+ bundle.writestr("manifest.json", json.dumps(manifest, indent=2, ensure_ascii=False))
80
+ bundle.writestr("README.md", _readme(manifest))
81
+ bundle.writestr("data/records.csv", csv_bytes(record_columns, record_rows))
82
+ bundle.writestr("data/records.json", json.dumps(record_rows, indent=1, ensure_ascii=False))
83
+ bundle.writestr("data/papers.csv", csv_bytes(paper_columns, paper_rows))
84
+ bundle.writestr("data/report.json", json.dumps(summary, indent=2, ensure_ascii=False))
85
+
86
+ # config: everything that decides what a run produces, and nothing that authenticates it
87
+ bundle.writestr("config/schema.json",
88
+ json.dumps({"fields": config.get_schema()}, indent=2, ensure_ascii=False))
89
+ bundle.writestr("config/extract_prompt.txt", config.get_extract_prompt())
90
+ bundle.writestr("config/judge_prompt.txt", config.get_judge_prompt())
91
+ bundle.writestr("config/few_shot.json",
92
+ json.dumps(config.get_few_shot(), indent=2, ensure_ascii=False))
93
+ bundle.writestr("config/models.json", json.dumps(
94
+ [{k: v for k, v in m.items()
95
+ if k in ("id", "name", "model", "api_base", "api_version")} for m in profiles],
96
+ indent=2, ensure_ascii=False))
97
+
98
+ # the per-paper working files, so a reviewer can trace any row back to its chunk
99
+ for stage, directory in (("extracted", storage.EXTRACTED), ("judged", storage.JUDGED)):
100
+ for path in sorted(directory.glob("*.json")):
101
+ if path.stem in chosen:
102
+ bundle.write(path, f"{stage}/{path.name}")
103
+ for path in sorted(storage.PARSED.glob("*.json")):
104
+ if path.stem not in chosen:
105
+ continue
106
+ paper = read_json(path, {})
107
+ if not include_text:
108
+ paper["chunks"] = [{"id": c["id"]} for c in paper.get("chunks", [])]
109
+ bundle.writestr(f"parsed/{path.name}", json.dumps(paper, indent=1, ensure_ascii=False))
110
+ if include_files:
111
+ for pid in sorted(chosen):
112
+ if (source := storage.source_file(pid)) is not None:
113
+ bundle.write(source, f"papers/{source.name}")
114
+ return buffer.getvalue()
115
+
116
+
117
+ def _readme(manifest: dict) -> str:
118
+ m = manifest["models"]
119
+ text_note = (
120
+ "`chunks[].text` holds the paper text, because this bundle was exported with it. Check "
121
+ "the papers' licences before sharing it."
122
+ if manifest["includes_paper_text"] else
123
+ "Paper text is not included -- papers obtained under text-and-data-mining terms may be "
124
+ "mined, not redistributed. Each chunk keeps its `id`, and each record its paper's DOI, "
125
+ "so anyone with access to the paper can check a value against its source.")
126
+ return f"""# file2records bundle
127
+
128
+ {manifest['records']} records from {manifest['papers']} paper(s), exported
129
+ {manifest['exported_at']} by file2records {manifest['version']}.
130
+
131
+ ## What is here
132
+
133
+ - `data/records.csv`, `data/records.json` — one row per record, with the paper's DOI, the
134
+ judge's verdict and any reviewer flag or note. `extract_model` and `judge_model` say what
135
+ produced each row.
136
+ - `data/papers.csv` — one row per paper: DOI, chunks in, records out, model, tokens, cost.
137
+ - `data/report.json` — completeness by field, what the judge changed, totals.
138
+ - `config/` — the schema, both prompts, the worked examples and the model settings that
139
+ produced this. API keys are not included.
140
+ - `parsed/`, `extracted/`, `judged/` — the working files, so any row can be traced back to the
141
+ chunk it came from. `source_chunk_ids` on a record refers to `chunks[].id` in `parsed/`.
142
+ - `papers/` — the source files, if you exported with them.
143
+
144
+ {text_note}
145
+
146
+ ## Reading it
147
+
148
+ Extraction model: `{m['extract'] or 'not set'}` · judge model: `{m['judge'] or 'not set'}`.
149
+ Papers may differ from these if the corpus was built across more than one model; see
150
+ `models.per_paper` in `manifest.json`.
151
+
152
+ `model_records` in `extracted/*.json` is what the model originally said, kept beside the
153
+ corrected records the moment anything was edited. A corrected dataset that cannot be diffed
154
+ against the model's own output is not evidence of anything.
155
+ """
file2records/cli.py ADDED
@@ -0,0 +1,207 @@
1
+ """The `file2records` command.
2
+
3
+ file2records serve my-review the web app on that project folder
4
+ file2records add my-review papers/ read files and folders into it
5
+ file2records papers my-review what is in it
6
+ file2records search my-review "glycoly[sz]is"
7
+ file2records check my-review what is missing before a run
8
+ file2records extract my-review --only "glycoly[sz]is" --exclude "positron|tomograph"
9
+ file2records judge my-review
10
+ file2records export my-review dataset.csv (.csv, .json, or .zip for the full bundle)
11
+
12
+ Every command takes the project folder first. Model commands use the model chosen in the web
13
+ app's Settings unless --model is given: any litellm model string, or rwth/<name> for RWTH's
14
+ KI:connect (key from RWTH_API_KEY).
15
+ """
16
+ import argparse
17
+ import os
18
+ import sys
19
+ from pathlib import Path
20
+
21
+ from dotenv import load_dotenv
22
+
23
+ from . import __version__
24
+
25
+
26
+ def _project(folder):
27
+ from .project import Project
28
+ project = Project(folder)
29
+ # Keys saved through the web app live in the project's .env; a .env in the current folder
30
+ # is read too. Neither overrides a variable already set in the shell.
31
+ load_dotenv(project.path / ".env")
32
+ load_dotenv(Path.cwd() / ".env")
33
+ return project
34
+
35
+
36
+ def _filters(p):
37
+ p.add_argument("--only", metavar="REGEX",
38
+ help="only papers whose full text matches this regular expression")
39
+ p.add_argument("--exclude", metavar="REGEX",
40
+ help="skip papers whose full text matches this regular expression")
41
+
42
+
43
+ def _print_result(r):
44
+ name = r.get("filename") or r["id"]
45
+ if r.get("error"):
46
+ print(f" ✗ {name}: {r['error']}")
47
+ elif "n_chunks" in r:
48
+ doi = f" doi:{r['doi']}" if r.get("doi") else ""
49
+ print(f" ✓ {name} [{r['format']}] {r['n_chunks']} chunks{doi}")
50
+ elif "n_records" in r:
51
+ print(f" ✓ {name}: {r['n_records']} records in {r['seconds']}s")
52
+ else:
53
+ print(f" ✓ {name}: {r['n_verdicts']} verdicts in {r['seconds']}s")
54
+
55
+
56
+ def cmd_serve(args):
57
+ os.environ["WORKSPACE_DIR"] = str(Path(args.folder).resolve())
58
+ _project(args.folder) # creates the folder, opens it
59
+ import uvicorn
60
+ from .main import app
61
+ url = f"http://{args.host}:{args.port}"
62
+ print(f"file2records {__version__} — project {Path(args.folder).resolve()}\nOpen {url}")
63
+ if not args.no_browser:
64
+ import threading
65
+ import webbrowser
66
+ threading.Timer(1.0, webbrowser.open, [url]).start()
67
+ uvicorn.run(app, host=args.host, port=args.port, log_level="warning")
68
+
69
+
70
+ def cmd_add(args):
71
+ project = _project(args.folder)
72
+ print(f"Reading into {project.path}")
73
+ results = project.add(*args.paths, source_tracking=not args.no_source_tracking,
74
+ on_file=_print_result)
75
+ failed = sum(1 for r in results if r.get("error"))
76
+ print(f"{len(results) - failed} added, {failed} failed")
77
+ return 1 if failed and failed == len(results) else 0
78
+
79
+
80
+ def cmd_papers(args):
81
+ papers = _project(args.folder).papers()
82
+ if not papers:
83
+ print("No papers yet. Add some: file2records add <folder> <files or folders>")
84
+ for p in papers:
85
+ state = "judged" if p["judged"] else "extracted" if p["extracted"] else "parsed"
86
+ records = "" if p["n_records"] is None else f"{p['n_records']} records"
87
+ print(f"{p['id']:50.50} {p['format']:8} {state:9} {records:11} {p['doi']}")
88
+
89
+
90
+ def cmd_search(args):
91
+ hits = _project(args.folder).search(args.pattern, ignore_case=not args.case_sensitive)
92
+ for h in hits:
93
+ print(f"\n{h['filename']} ({h['matches']} matches)")
94
+ for s in h["snippets"]:
95
+ print(f" …{s['before'][-60:]}[{s['match']}]{s['after'][:60]}…".replace("\n", " "))
96
+ total = sum(h["matches"] for h in hits)
97
+ print(f"\n{len(hits)} papers, {total} matches")
98
+
99
+
100
+ def cmd_check(args):
101
+ project = _project(args.folder)
102
+ ok = True
103
+ for stage in ("extract", "judge"):
104
+ missing = project.check(stage, args.model)
105
+ ok &= stage == "judge" or not missing
106
+ print(f"{stage}: {'ready' if not missing else 'not ready'}")
107
+ for m in missing:
108
+ hint = " Or pass --model, e.g. rwth/gpt-oss-120b." if m.startswith("Choose a model") else ""
109
+ print(f" - {m}{hint}")
110
+ return 0 if ok else 1
111
+
112
+
113
+ def _run_stage(args, stage):
114
+ project = _project(args.folder)
115
+ run = project.extract if stage == "extract" else project.judge
116
+ results = run(args.model, only=args.only, exclude=args.exclude, redo=args.redo,
117
+ on_paper=_print_result)
118
+ if not results:
119
+ print(f"Nothing to {stage}: every chosen paper is done already (--redo to run again).")
120
+ failed = sum(1 for r in results if r.get("error"))
121
+ print(f"{len(results) - failed} done, {failed} failed")
122
+ return 1 if failed else 0
123
+
124
+
125
+ def cmd_export(args):
126
+ path = _project(args.folder).export(args.output, only=args.only, exclude=args.exclude,
127
+ include_text=args.include_text,
128
+ include_files=args.include_files)
129
+ print(f"Wrote {path}")
130
+
131
+
132
+ FOLDER = "the project folder (created if it does not exist)"
133
+
134
+
135
+ def build_parser() -> argparse.ArgumentParser:
136
+ parser = argparse.ArgumentParser(
137
+ prog="file2records",
138
+ description="Turn papers (PDF, XML, HTML, Word, Markdown) into a structured dataset.",
139
+ epilog="examples:\n" + "\n".join(__doc__.splitlines()[2:10]) +
140
+ "\n\ndocs: https://enkhnyam.github.io/chemistry-data-extractor-toolkit/",
141
+ formatter_class=argparse.RawDescriptionHelpFormatter)
142
+ parser.add_argument("--version", action="version", version=f"file2records {__version__}")
143
+ sub = parser.add_subparsers(dest="command", required=True, metavar="COMMAND")
144
+
145
+ p = sub.add_parser("serve", help="open the web app on a project folder")
146
+ p.add_argument("folder", nargs="?", default="workspace", help=FOLDER + " (default: %(default)s)")
147
+ p.add_argument("--port", type=int, default=8000)
148
+ p.add_argument("--host", default="127.0.0.1",
149
+ help="keep the default unless you know why: the app has no login")
150
+ p.add_argument("--no-browser", action="store_true")
151
+ p.set_defaults(func=cmd_serve)
152
+
153
+ p = sub.add_parser("add", help="read files or folders of papers into a project")
154
+ p.add_argument("folder", help=FOLDER)
155
+ p.add_argument("paths", nargs="+", help="paper files, or folders of them")
156
+ p.add_argument("--no-source-tracking", action="store_true",
157
+ help="don't tag chunks, so records won't cite the passage they came from")
158
+ p.set_defaults(func=cmd_add)
159
+
160
+ p = sub.add_parser("papers", help="list the papers in a project")
161
+ p.add_argument("folder", help=FOLDER)
162
+ p.set_defaults(func=cmd_papers)
163
+
164
+ p = sub.add_parser("search", help="search the full text of every paper with a regex")
165
+ p.add_argument("folder", help=FOLDER)
166
+ p.add_argument("pattern", help="a regular expression, e.g. \"glycoly[sz]is\"")
167
+ p.add_argument("--case-sensitive", action="store_true")
168
+ p.set_defaults(func=cmd_search)
169
+
170
+ p = sub.add_parser("check", help="say what is missing before a run")
171
+ p.add_argument("folder", help=FOLDER)
172
+ p.add_argument("--model")
173
+ p.set_defaults(func=cmd_check)
174
+
175
+ for stage, text in (("extract", "extract records from papers not extracted yet"),
176
+ ("judge", "audit extracted records with a second model")):
177
+ p = sub.add_parser(stage, help=text)
178
+ p.add_argument("folder", help=FOLDER)
179
+ p.add_argument("--model", help='litellm model string, or rwth/<name> '
180
+ '(default: the model chosen in Settings)')
181
+ p.add_argument("--redo", action="store_true", help="also re-run papers already done")
182
+ _filters(p)
183
+ p.set_defaults(func=lambda a, s=stage: _run_stage(a, s))
184
+
185
+ p = sub.add_parser("export", help="write records to .csv / .json, or a .zip bundle")
186
+ p.add_argument("folder", help=FOLDER)
187
+ p.add_argument("output", help="a .csv or .json file of records, or a .zip bundle")
188
+ p.add_argument("--include-text", action="store_true",
189
+ help="put the paper text in a .zip bundle (check the papers' licences)")
190
+ p.add_argument("--include-files", action="store_true",
191
+ help="put the source files in a .zip bundle (check the papers' licences)")
192
+ _filters(p)
193
+ p.set_defaults(func=cmd_export)
194
+ return parser
195
+
196
+
197
+ def main(argv=None) -> int:
198
+ args = build_parser().parse_args(argv)
199
+ try:
200
+ return args.func(args) or 0
201
+ except (ValueError, RuntimeError, FileNotFoundError) as e:
202
+ print(f"error: {e}", file=sys.stderr)
203
+ return 2
204
+
205
+
206
+ if __name__ == "__main__":
207
+ sys.exit(main())
file2records/config.py ADDED
@@ -0,0 +1,96 @@
1
+ """Reads/writes the four things that make this tool generic: the schema, the extraction
2
+ prompt, the judge rubric, and few-shot examples. All under workspace/config/, all editable
3
+ from the Settings page -- change these four and the same pipeline runs on a different
4
+ domain, no code changes."""
5
+ from . import exemplar
6
+ from . import storage
7
+ from .storage import read_json, write_json
8
+ from .dynschema import PLACEHOLDER_SCHEMA
9
+
10
+
11
+
12
+ def settings_file():
13
+ return storage.CONFIG / "settings.json"
14
+
15
+
16
+ def schema_file():
17
+ return storage.CONFIG / "schema.json"
18
+
19
+
20
+ def extract_prompt_file():
21
+ return storage.CONFIG / "extract_prompt.txt"
22
+
23
+
24
+ def judge_prompt_file():
25
+ return storage.CONFIG / "judge_prompt.txt"
26
+
27
+
28
+ def few_shot_file():
29
+ return storage.CONFIG / "few_shot.json"
30
+
31
+ DEFAULT_SETTINGS = {
32
+ # Which named model profile each stage calls; see server/models.py. Empty until chosen.
33
+ "extract_model": "",
34
+ "judge_model": "",
35
+ "source_tracking_default": True, # a real default: provenance on unless turned off
36
+ }
37
+
38
+ def get_settings() -> dict:
39
+ return {**DEFAULT_SETTINGS, **read_json(settings_file(), {})}
40
+
41
+
42
+ def save_settings(patch: dict) -> dict:
43
+ cfg = get_settings()
44
+ cfg.update(patch)
45
+ write_json(settings_file(), cfg)
46
+ return cfg
47
+
48
+
49
+ def get_schema() -> list[dict]:
50
+ """Empty until someone defines one. A shipped default meant every new install began with
51
+ ten PET fields nobody chose, and an extraction that looked like it had worked."""
52
+ return read_json(schema_file(), {"fields": []})["fields"]
53
+
54
+
55
+ def save_schema(fields: list[dict]) -> None:
56
+ write_json(schema_file(), {"fields": fields})
57
+
58
+
59
+ def _get_text(path) -> str:
60
+ """A prompt nobody has written yet is empty, not somebody else's.
61
+
62
+ This used to fall back to a built-in PET prompt, which meant a first run quietly extracted
63
+ ionic-liquid chemistry from whatever you uploaded and looked like it had worked. The
64
+ example text is still shown -- as placeholder text in the box -- but it is never the value.
65
+ """
66
+ return path.read_text(encoding="utf-8") if path.exists() else ""
67
+
68
+
69
+ def get_extract_prompt() -> str:
70
+ return _get_text(extract_prompt_file())
71
+
72
+
73
+ def save_extract_prompt(text: str) -> None:
74
+ extract_prompt_file().write_text(text, encoding="utf-8")
75
+
76
+
77
+ def get_judge_prompt() -> str:
78
+ return _get_text(judge_prompt_file())
79
+
80
+
81
+ def placeholders() -> dict:
82
+ """Everything shown as grey example text in an empty field. None of it is ever a value."""
83
+ return {"extract": exemplar.EXTRACT_PROMPT, "judge": exemplar.JUDGE_PROMPT,
84
+ "model": "gpt-4o-mini", "schema": PLACEHOLDER_SCHEMA}
85
+
86
+
87
+ def save_judge_prompt(text: str) -> None:
88
+ judge_prompt_file().write_text(text, encoding="utf-8")
89
+
90
+
91
+ def get_few_shot() -> list[dict]:
92
+ return read_json(few_shot_file(), [])
93
+
94
+
95
+ def save_few_shot(examples: list[dict]) -> None:
96
+ write_json(few_shot_file(), examples)
@@ -0,0 +1,34 @@
1
+ # The demo paper, and why it is here
2
+
3
+ This folder ships one real paper so that a fresh clone has something to run on. It is open
4
+ access under **CC BY 3.0**, on the *publisher's own published version* — which is what makes
5
+ redistributing the PDF itself lawful, rather than only the right to read it.
6
+
7
+ | File | Citation | Licence |
8
+ |---|---|---|
9
+ | `pdfs/…lewis-acidic…pdf` | Qun Feng Yue, Lin Fei Xiao, Mi Lin Zhang and Xue Feng Bai, *The Glycolysis of Poly(ethylene terephthalate) Waste: Lewis Acidic Ionic Liquids as High Efficient Catalysts*, **Polymers** 2013, **5**, 1258–1271. [10.3390/polym5041258](https://doi.org/10.3390/polym5041258) | [CC BY 3.0](https://creativecommons.org/licenses/by/3.0/) |
10
+
11
+ The PDF and the text derived from it under `parsed/` are redistributed unmodified in substance.
12
+ The row above is the attribution the licence requires: creator, title, source and licence, each
13
+ with a link. CC BY 3.0 requires attribution and nothing else: you may redistribute and adapt,
14
+ including commercially, provided the creators are credited as above.
15
+
16
+ The rest of this repository is MIT (see `LICENSE`). This PDF is not MIT; it stays under CC BY,
17
+ and the MIT licence does not extend to it.
18
+
19
+ ## Why only one
20
+
21
+ The PET database this toolkit was generalised out of uses seven papers as redistributable
22
+ worked examples. Five of those seven are open access only as a **submitted-version deposit** in
23
+ an institutional repository (mostly `ir.ipe.ac.cn`), under CC BY-NC-SA. That licence covers the
24
+ deposited manuscript, not the publisher's typeset PDF, and it is the publisher's PDF that exists
25
+ on disk. Redistributing the text of those five is fine and is what the database does;
26
+ redistributing the publisher's PDF is not, so they are not shipped here.
27
+
28
+ An earlier version of this demo also shipped *Amino Acid-Based Cholinium Ionic Liquids as
29
+ Sustainable Catalysts for PET Depolymerization* (ACS Sustainable Chem. Eng. 2021,
30
+ [10.1021/acssuschemeng.1c04060](https://doi.org/10.1021/acssuschemeng.1c04060)). That article is
31
+ CC BY 4.0 and redistributing it was lawful, but ACS stamps every PDF download with the IP address
32
+ it was fetched from, and publishing that line alongside the file served nobody. Re-downloading
33
+ does not help — the stamp is applied to every copy — so the paper was removed rather than
34
+ replaced. One paper is enough to show what the tool does.
@@ -0,0 +1,51 @@
1
+ You extract experimental data for PET depolymerisation from a paper's text and tables.
2
+ The reaction is glycolysis, methanolysis or hydrolysis, and the catalyst may be an ionic liquid,
3
+ a metal salt, an acid or base, or none. Treat all three routes the same way.
4
+ SKIP papers and rows for any other route, including aminolysis and enzymatic degradation.
5
+ Extract only what is stated; leave anything unreported as null. The output structure is
6
+ enforced by the response schema, so focus on getting the values and their sources right.
7
+
8
+ ## What to extract
9
+ - INCLUDE the current study's ("this work") depolymerisation experiments reported in text and
10
+ tables, whichever route they use.
11
+ - SKIP these rows entirely:
12
+ - literature / cited results — a Source or Reference column, "Ref. N", "[N]", "et al.";
13
+ - RSM / Box-Behnken / optimization tables (coded factors and a single modeled response);
14
+ - data that appears only in a figure/image;
15
+ - characterization or kinetics tables (NMR shifts, bond lengths, Arrhenius) — not depolymerisation runs.
16
+ - Extract EVERY qualifying row: N current-study rows → N records. Don't merge or stop early.
17
+
18
+ ## Catalyst naming
19
+ - Use the paper's shorthand in bracket notation, with digits inline rather than subscripts (a
20
+ formula printed with subscript counts is recorded with normal digits).
21
+ - Join physical mixtures as written. No catalyst / uncatalyzed → "none".
22
+ - A catalyst named only by a code from a synthesis recipe (e.g. "IL-1", "Cat-A"): keep the code
23
+ verbatim unless the paper equates it with a clean chemical name. (Reading the code literally is
24
+ what lets the extraction match the ground truth.)
25
+
26
+ ## Global reaction conditions
27
+ Conditions stated once — in the methods or a table footnote — apply to EVERY row of that table.
28
+ Propagate them to all records, and convert amounts to grams:
29
+ - catalyst wt% → g: catalyst_g = wt%/100 × PET_g
30
+ - solvent:PET mass ratio → g: solvent_g = ratio × PET_g (the solvent is whichever reagent the
31
+ reaction uses — ethylene glycol, methanol, water, an amine)
32
+ - molar ratio → g via molecular weight (PET repeat unit ≈ 192 g/mol)
33
+
34
+ ## Values
35
+ - Ignore placeholder wording ("a certain amount", "specific temperature", a bare symbol) — take
36
+ the concrete value from the table; if there is none, use null.
37
+ - temperature_c: copy exactly, never convert. reaction_time_min: hours × 60.
38
+ - yield_percent, selectivity_percent, conversion_percent are DIFFERENT metrics — put each value
39
+ only in the field the paper labels it as (a conversion column → conversion_percent, a
40
+ selectivity column → selectivity_percent, a yield column → yield_percent).
41
+ - null (not 0) when a value is unreported. pressure_atm: null unless a number is stated.
42
+
43
+ ## Skip-these examples (schematic — illustrate the row shape, not real data)
44
+ Literature (cited Source): RSM / optimization (coded factors):
45
+ | Source | Catalyst | Temp | | Run | A:factor | B:factor | Response |
46
+ | Ref. N | <cited> | ... | | 1 | lo | hi | ... |
47
+
48
+ ## Source chunks
49
+ The text is split into chunks tagged "ID: <uuid>". For each record, list in source_chunk_ids
50
+ every chunk that supplied a value — usually the table chunk, the footnote/conditions chunk, and
51
+ any chunk that defined a catalyst code.
@@ -0,0 +1 @@
1
+ []