risforge 0.3.2__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- risforge-0.4.0/PKG-INFO +143 -0
- risforge-0.4.0/README.md +102 -0
- {risforge-0.3.2 → risforge-0.4.0}/pyproject.toml +30 -27
- risforge-0.4.0/src/risforge/__init__.py +22 -0
- risforge-0.4.0/src/risforge/cleaning.py +310 -0
- {risforge-0.3.2 → risforge-0.4.0}/src/risforge/cli.py +50 -128
- {risforge-0.3.2 → risforge-0.4.0}/src/risforge/enrichment.py +56 -134
- risforge-0.4.0/src/risforge/exceptions.py +15 -0
- risforge-0.4.0/src/risforge/merging.py +74 -0
- risforge-0.4.0/src/risforge/pipeline.py +143 -0
- risforge-0.4.0/src/risforge.egg-info/PKG-INFO +143 -0
- {risforge-0.3.2 → risforge-0.4.0}/src/risforge.egg-info/requires.txt +7 -8
- risforge-0.4.0/src/risforge_gui/__init__.py +3 -0
- risforge-0.4.0/src/risforge_gui/app.py +41 -0
- {risforge-0.3.2 → risforge-0.4.0}/src/risforge_gui/dialogs.py +6 -9
- {risforge-0.3.2 → risforge-0.4.0}/src/risforge_gui/file_counter.py +6 -14
- risforge-0.4.0/src/risforge_gui/main_window.py +324 -0
- {risforge-0.3.2 → risforge-0.4.0}/src/risforge_gui/models.py +39 -32
- risforge-0.4.0/src/risforge_gui/theme.py +197 -0
- risforge-0.4.0/src/risforge_gui/widgets/__init__.py +17 -0
- {risforge-0.3.2 → risforge-0.4.0}/src/risforge_gui/widgets/config_panel.py +23 -14
- {risforge-0.3.2 → risforge-0.4.0}/src/risforge_gui/widgets/input_panel.py +63 -46
- {risforge-0.3.2 → risforge-0.4.0}/src/risforge_gui/widgets/log_panel.py +7 -3
- {risforge-0.3.2 → risforge-0.4.0}/src/risforge_gui/widgets/output_panel.py +19 -24
- {risforge-0.3.2 → risforge-0.4.0}/src/risforge_gui/widgets/progress_panel.py +29 -13
- {risforge-0.3.2 → risforge-0.4.0}/src/risforge_gui/widgets/results_panel.py +13 -11
- {risforge-0.3.2 → risforge-0.4.0}/src/risforge_gui/worker.py +84 -75
- {risforge-0.3.2 → risforge-0.4.0}/tests/test_cli.py +6 -21
- {risforge-0.3.2 → risforge-0.4.0}/tests/test_enrichment.py +1 -4
- {risforge-0.3.2 → risforge-0.4.0}/tests/test_merging.py +20 -19
- {risforge-0.3.2 → risforge-0.4.0}/tests/test_pipeline.py +2 -6
- risforge-0.3.2/PKG-INFO +0 -296
- risforge-0.3.2/README.md +0 -255
- risforge-0.3.2/src/risforge/__init__.py +0 -67
- risforge-0.3.2/src/risforge/cleaning.py +0 -370
- risforge-0.3.2/src/risforge/exceptions.py +0 -38
- risforge-0.3.2/src/risforge/merging.py +0 -130
- risforge-0.3.2/src/risforge/pipeline.py +0 -233
- risforge-0.3.2/src/risforge.egg-info/PKG-INFO +0 -296
- risforge-0.3.2/src/risforge_gui/__init__.py +0 -21
- risforge-0.3.2/src/risforge_gui/app.py +0 -34
- risforge-0.3.2/src/risforge_gui/main_window.py +0 -375
- risforge-0.3.2/src/risforge_gui/theme.py +0 -236
- risforge-0.3.2/src/risforge_gui/widgets/__init__.py +0 -1
- {risforge-0.3.2 → risforge-0.4.0}/LICENSE +0 -0
- {risforge-0.3.2 → risforge-0.4.0}/setup.cfg +0 -0
- {risforge-0.3.2 → risforge-0.4.0}/src/risforge.egg-info/SOURCES.txt +0 -0
- {risforge-0.3.2 → risforge-0.4.0}/src/risforge.egg-info/dependency_links.txt +0 -0
- {risforge-0.3.2 → risforge-0.4.0}/src/risforge.egg-info/entry_points.txt +0 -0
- {risforge-0.3.2 → risforge-0.4.0}/src/risforge.egg-info/top_level.txt +0 -0
- {risforge-0.3.2 → risforge-0.4.0}/tests/test_cleaning.py +0 -0
risforge-0.4.0/PKG-INFO
ADDED
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: risforge
|
|
3
|
+
Version: 0.4.0
|
|
4
|
+
Summary: Merge, clean, deduplicate, and enrich RIS bibliographic files.
|
|
5
|
+
Author-email: Amyr <amyrhexa@gmail.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/amyr/risforge
|
|
8
|
+
Project-URL: Repository, https://github.com/amyr/risforge
|
|
9
|
+
Project-URL: Issues, https://github.com/amyr/risforge/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/amyr/risforge/blob/main/CHANGELOG.md
|
|
11
|
+
Keywords: ris,bibliography,systematic-review,deduplication,enrichment,pyside6
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering
|
|
15
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Operating System :: OS Independent
|
|
22
|
+
Classifier: Environment :: X11 Applications :: Qt
|
|
23
|
+
Requires-Python: >=3.10
|
|
24
|
+
Description-Content-Type: text/markdown
|
|
25
|
+
License-File: LICENSE
|
|
26
|
+
Requires-Dist: rispy>=0.10.0
|
|
27
|
+
Requires-Dist: requests>=2.31.0
|
|
28
|
+
Requires-Dist: requests-cache>=1.2.0
|
|
29
|
+
Provides-Extra: gui
|
|
30
|
+
Requires-Dist: PySide6>=6.5; extra == "gui"
|
|
31
|
+
Provides-Extra: dev
|
|
32
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
33
|
+
Requires-Dist: pytest-cov>=5.0; extra == "dev"
|
|
34
|
+
Requires-Dist: pytest-qt>=4.4; extra == "dev"
|
|
35
|
+
Requires-Dist: build>=1.2; extra == "dev"
|
|
36
|
+
Requires-Dist: twine>=5.0; extra == "dev"
|
|
37
|
+
Requires-Dist: ruff>=0.4.0; extra == "dev"
|
|
38
|
+
Requires-Dist: black>=24.0; extra == "dev"
|
|
39
|
+
Requires-Dist: mypy>=1.10; extra == "dev"
|
|
40
|
+
Dynamic: license-file
|
|
41
|
+
|
|
42
|
+
<h1 align="center">risforge</h1>
|
|
43
|
+
|
|
44
|
+
<p align="center">
|
|
45
|
+
<strong>Clean, deduplicate, and enrich RIS bibliographic files for systematic reviews.</strong>
|
|
46
|
+
</p>
|
|
47
|
+
|
|
48
|
+
`risforge` transforms messy, overlapping `.ris` exports from Scopus, PubMed, Web of Science, or EndNote into a single, clean, and metadata-rich file ready for screening.
|
|
49
|
+
<p align="center">
|
|
50
|
+
<img src="docs/Screenshot.png" alt="RisForge GUI Screenshot">
|
|
51
|
+
</p>
|
|
52
|
+
|
|
53
|
+
## How it works
|
|
54
|
+
|
|
55
|
+
| Stage | What it does | What it doesn't do |
|
|
56
|
+
| :--- | :--- | :--- |
|
|
57
|
+
| **Merge** | Combines multiple `.ris` files into one. | Decide if records are duplicates. |
|
|
58
|
+
| **Clean** | Deduplicates via DOI and fuzzy title/author matching. Zero data loss. | Fetch new data from the internet. |
|
|
59
|
+
| **Enrich** | Fills missing metadata (abstracts, PDFs, keywords) via Crossref, OpenAlex, Semantic Scholar, and Unpaywall. | Overwrite your existing data. |
|
|
60
|
+
|
|
61
|
+
## Features
|
|
62
|
+
|
|
63
|
+
- **Smart Deduplication**: DOI-first clustering with a normalized title + first-author fallback.
|
|
64
|
+
- **Zero Data Loss**: Merges duplicate clusters by keeping the most complete record and appending missing fields from others.
|
|
65
|
+
- **Non-Destructive Enrichment**: Only fills empty fields; existing metadata is never overwritten.
|
|
66
|
+
- **Resilient HTTP**: Automatic retries, exponential backoff, and a 7-day on-disk cache.
|
|
67
|
+
- **Flexible Interfaces**: Use it as a Python library, a CLI tool, or a cross-platform desktop GUI.
|
|
68
|
+
|
|
69
|
+
## Installation
|
|
70
|
+
|
|
71
|
+
Requires Python 3.10+.
|
|
72
|
+
|
|
73
|
+
```bash
|
|
74
|
+
# Core library and CLI
|
|
75
|
+
pip install risforge
|
|
76
|
+
|
|
77
|
+
# With the optional desktop GUI
|
|
78
|
+
pip install "risforge[gui]"
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
## Quick Start
|
|
82
|
+
|
|
83
|
+
### Desktop GUI
|
|
84
|
+
|
|
85
|
+
Launch the graphical interface for a drag-and-drop workflow:
|
|
86
|
+
|
|
87
|
+
```bash
|
|
88
|
+
risforge-gui
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
### Command Line
|
|
92
|
+
|
|
93
|
+
Run the full pipeline (merge, clean, and enrich) in a single command:
|
|
94
|
+
|
|
95
|
+
```bash
|
|
96
|
+
risforge pipeline scopus.ris pubmed.ris wos.ris \
|
|
97
|
+
--email you@example.com \
|
|
98
|
+
--output final.ris
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
*Note: An email is required for API "polite pool" access. It is only sent in request headers and never stored.*
|
|
102
|
+
|
|
103
|
+
Run individual steps if you need to inspect intermediate files:
|
|
104
|
+
|
|
105
|
+
```bash
|
|
106
|
+
risforge merge scopus.ris pubmed.ris wos.ris merged.ris
|
|
107
|
+
risforge clean merged.ris clean.ris
|
|
108
|
+
risforge enrich clean.ris enriched.ris --email you@example.com
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
### Python API
|
|
112
|
+
|
|
113
|
+
```python
|
|
114
|
+
from risforge import risforge
|
|
115
|
+
|
|
116
|
+
# Automatically merges multiple files, cleans, and enriches
|
|
117
|
+
result = risforge(
|
|
118
|
+
input_paths=["scopus.ris", "pubmed.ris", "wos.ris"],
|
|
119
|
+
dedup_path="clean.ris",
|
|
120
|
+
enriched_path="enriched.ris",
|
|
121
|
+
email="you@example.com",
|
|
122
|
+
)
|
|
123
|
+
|
|
124
|
+
print(f"Merged: {result.merged_record_count}")
|
|
125
|
+
print(f"Unique: {result.cleaned_record_count}")
|
|
126
|
+
print(f"Enriched: {result.enrichment_stats['enriched']}")
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
## Good to Know
|
|
130
|
+
|
|
131
|
+
- **Unresolved DOIs**: Records without a DOI that fail title-matching are left unmodified and logged to `failed_records.json`.
|
|
132
|
+
- **Merge vs. Clean**: `merge` simply concatenates files. Run `clean` afterward to actually collapse duplicates.
|
|
133
|
+
- **Caching**: API responses are cached for 7 days. Re-running the pipeline on the same data is nearly instant.
|
|
134
|
+
|
|
135
|
+
## Contributing
|
|
136
|
+
|
|
137
|
+
1. Clone and install dev dependencies: `pip install -e ".[dev,gui]"`
|
|
138
|
+
2. Run tests: `pytest`
|
|
139
|
+
3. Open a PR. Please include tests for behavioral changes and update `CHANGELOG.md`. Predictability is a core feature; silent behavior changes are treated as bugs.
|
|
140
|
+
|
|
141
|
+
## License
|
|
142
|
+
|
|
143
|
+
MIT — see [LICENSE](LICENSE).
|
risforge-0.4.0/README.md
ADDED
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
<h1 align="center">risforge</h1>
|
|
2
|
+
|
|
3
|
+
<p align="center">
|
|
4
|
+
<strong>Clean, deduplicate, and enrich RIS bibliographic files for systematic reviews.</strong>
|
|
5
|
+
</p>
|
|
6
|
+
|
|
7
|
+
`risforge` transforms messy, overlapping `.ris` exports from Scopus, PubMed, Web of Science, or EndNote into a single, clean, and metadata-rich file ready for screening.
|
|
8
|
+
<p align="center">
|
|
9
|
+
<img src="docs/Screenshot.png" alt="RisForge GUI Screenshot">
|
|
10
|
+
</p>
|
|
11
|
+
|
|
12
|
+
## How it works
|
|
13
|
+
|
|
14
|
+
| Stage | What it does | What it doesn't do |
|
|
15
|
+
| :--- | :--- | :--- |
|
|
16
|
+
| **Merge** | Combines multiple `.ris` files into one. | Decide if records are duplicates. |
|
|
17
|
+
| **Clean** | Deduplicates via DOI and fuzzy title/author matching. Zero data loss. | Fetch new data from the internet. |
|
|
18
|
+
| **Enrich** | Fills missing metadata (abstracts, PDFs, keywords) via Crossref, OpenAlex, Semantic Scholar, and Unpaywall. | Overwrite your existing data. |
|
|
19
|
+
|
|
20
|
+
## Features
|
|
21
|
+
|
|
22
|
+
- **Smart Deduplication**: DOI-first clustering with a normalized title + first-author fallback.
|
|
23
|
+
- **Zero Data Loss**: Merges duplicate clusters by keeping the most complete record and appending missing fields from others.
|
|
24
|
+
- **Non-Destructive Enrichment**: Only fills empty fields; existing metadata is never overwritten.
|
|
25
|
+
- **Resilient HTTP**: Automatic retries, exponential backoff, and a 7-day on-disk cache.
|
|
26
|
+
- **Flexible Interfaces**: Use it as a Python library, a CLI tool, or a cross-platform desktop GUI.
|
|
27
|
+
|
|
28
|
+
## Installation
|
|
29
|
+
|
|
30
|
+
Requires Python 3.10+.
|
|
31
|
+
|
|
32
|
+
```bash
|
|
33
|
+
# Core library and CLI
|
|
34
|
+
pip install risforge
|
|
35
|
+
|
|
36
|
+
# With the optional desktop GUI
|
|
37
|
+
pip install "risforge[gui]"
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
## Quick Start
|
|
41
|
+
|
|
42
|
+
### Desktop GUI
|
|
43
|
+
|
|
44
|
+
Launch the graphical interface for a drag-and-drop workflow:
|
|
45
|
+
|
|
46
|
+
```bash
|
|
47
|
+
risforge-gui
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
### Command Line
|
|
51
|
+
|
|
52
|
+
Run the full pipeline (merge, clean, and enrich) in a single command:
|
|
53
|
+
|
|
54
|
+
```bash
|
|
55
|
+
risforge pipeline scopus.ris pubmed.ris wos.ris \
|
|
56
|
+
--email you@example.com \
|
|
57
|
+
--output final.ris
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
*Note: An email is required for API "polite pool" access. It is only sent in request headers and never stored.*
|
|
61
|
+
|
|
62
|
+
Run individual steps if you need to inspect intermediate files:
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
risforge merge scopus.ris pubmed.ris wos.ris merged.ris
|
|
66
|
+
risforge clean merged.ris clean.ris
|
|
67
|
+
risforge enrich clean.ris enriched.ris --email you@example.com
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
### Python API
|
|
71
|
+
|
|
72
|
+
```python
|
|
73
|
+
from risforge import risforge
|
|
74
|
+
|
|
75
|
+
# Automatically merges multiple files, cleans, and enriches
|
|
76
|
+
result = risforge(
|
|
77
|
+
input_paths=["scopus.ris", "pubmed.ris", "wos.ris"],
|
|
78
|
+
dedup_path="clean.ris",
|
|
79
|
+
enriched_path="enriched.ris",
|
|
80
|
+
email="you@example.com",
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
print(f"Merged: {result.merged_record_count}")
|
|
84
|
+
print(f"Unique: {result.cleaned_record_count}")
|
|
85
|
+
print(f"Enriched: {result.enrichment_stats['enriched']}")
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
## Good to Know
|
|
89
|
+
|
|
90
|
+
- **Unresolved DOIs**: Records without a DOI that fail title-matching are left unmodified and logged to `failed_records.json`.
|
|
91
|
+
- **Merge vs. Clean**: `merge` simply concatenates files. Run `clean` afterward to actually collapse duplicates.
|
|
92
|
+
- **Caching**: API responses are cached for 7 days. Re-running the pipeline on the same data is nearly instant.
|
|
93
|
+
|
|
94
|
+
## Contributing
|
|
95
|
+
|
|
96
|
+
1. Clone and install dev dependencies: `pip install -e ".[dev,gui]"`
|
|
97
|
+
2. Run tests: `pytest`
|
|
98
|
+
3. Open a PR. Please include tests for behavioral changes and update `CHANGELOG.md`. Predictability is a core feature; silent behavior changes are treated as bugs.
|
|
99
|
+
|
|
100
|
+
## License
|
|
101
|
+
|
|
102
|
+
MIT — see [LICENSE](LICENSE).
|
|
@@ -4,29 +4,24 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "risforge"
|
|
7
|
-
version = "0.
|
|
8
|
-
description = "
|
|
7
|
+
version = "0.4.0"
|
|
8
|
+
description = "Merge, clean, deduplicate, and enrich RIS bibliographic files."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
11
11
|
license = { text = "MIT" }
|
|
12
|
-
authors = [
|
|
13
|
-
{ name = "Amyr", email = "amyrhexa@gmail.com" }
|
|
14
|
-
]
|
|
12
|
+
authors = [{ name = "Amyr", email = "amyrhexa@gmail.com" }]
|
|
15
13
|
keywords = [
|
|
16
14
|
"ris",
|
|
17
15
|
"bibliography",
|
|
18
16
|
"systematic-review",
|
|
19
17
|
"deduplication",
|
|
20
|
-
"
|
|
21
|
-
"
|
|
22
|
-
"citation-management",
|
|
23
|
-
"prisma",
|
|
18
|
+
"enrichment",
|
|
19
|
+
"pyside6",
|
|
24
20
|
]
|
|
25
21
|
classifiers = [
|
|
26
22
|
"Development Status :: 4 - Beta",
|
|
27
23
|
"Intended Audience :: Science/Research",
|
|
28
24
|
"Topic :: Scientific/Engineering",
|
|
29
|
-
"Topic :: Text Processing :: Filters",
|
|
30
25
|
"License :: OSI Approved :: MIT License",
|
|
31
26
|
"Programming Language :: Python :: 3",
|
|
32
27
|
"Programming Language :: Python :: 3.10",
|
|
@@ -34,35 +29,29 @@ classifiers = [
|
|
|
34
29
|
"Programming Language :: Python :: 3.12",
|
|
35
30
|
"Programming Language :: Python :: 3.13",
|
|
36
31
|
"Operating System :: OS Independent",
|
|
32
|
+
"Environment :: X11 Applications :: Qt",
|
|
37
33
|
]
|
|
38
34
|
|
|
39
|
-
dependencies = [
|
|
40
|
-
"rispy>=0.9.0",
|
|
41
|
-
"requests>=2.31",
|
|
42
|
-
"requests-cache>=1.1",
|
|
43
|
-
"urllib3>=2.0",
|
|
44
|
-
]
|
|
35
|
+
dependencies = ["rispy>=0.10.0", "requests>=2.31.0", "requests-cache>=1.2.0"]
|
|
45
36
|
|
|
46
37
|
[project.optional-dependencies]
|
|
38
|
+
gui = ["PySide6>=6.5"]
|
|
47
39
|
dev = [
|
|
48
40
|
"pytest>=8.0",
|
|
49
41
|
"pytest-cov>=5.0",
|
|
42
|
+
"pytest-qt>=4.4",
|
|
50
43
|
"build>=1.2",
|
|
51
44
|
"twine>=5.0",
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
"
|
|
55
|
-
]
|
|
56
|
-
gui-dev = [
|
|
57
|
-
"risforge[gui]",
|
|
58
|
-
"pytest-qt>=4.4",
|
|
45
|
+
"ruff>=0.4.0",
|
|
46
|
+
"black>=24.0",
|
|
47
|
+
"mypy>=1.10",
|
|
59
48
|
]
|
|
60
49
|
|
|
61
50
|
[project.urls]
|
|
62
|
-
Homepage = "https://github.com/
|
|
63
|
-
Repository = "https://github.com/
|
|
64
|
-
Issues = "https://github.com/
|
|
65
|
-
Changelog = "https://github.com/
|
|
51
|
+
Homepage = "https://github.com/amyr/risforge"
|
|
52
|
+
Repository = "https://github.com/amyr/risforge"
|
|
53
|
+
Issues = "https://github.com/amyr/risforge/issues"
|
|
54
|
+
Changelog = "https://github.com/amyr/risforge/blob/main/CHANGELOG.md"
|
|
66
55
|
|
|
67
56
|
[project.scripts]
|
|
68
57
|
risforge = "risforge.cli:main"
|
|
@@ -72,12 +61,26 @@ risforge-gui = "risforge_gui.app:main"
|
|
|
72
61
|
|
|
73
62
|
[tool.setuptools.packages.find]
|
|
74
63
|
where = ["src"]
|
|
64
|
+
include = ["risforge*", "risforge_gui*"]
|
|
75
65
|
|
|
76
66
|
[tool.setuptools.package-dir]
|
|
77
67
|
"" = "src"
|
|
78
68
|
|
|
79
69
|
[tool.pytest.ini_options]
|
|
80
70
|
testpaths = ["tests"]
|
|
71
|
+
qt_api = "pyside6"
|
|
81
72
|
|
|
82
73
|
[tool.coverage.run]
|
|
83
74
|
concurrency = ["thread"]
|
|
75
|
+
source = ["risforge", "risforge_gui"]
|
|
76
|
+
|
|
77
|
+
[tool.black]
|
|
78
|
+
line-length = 100
|
|
79
|
+
target-version = ["py310", "py311", "py312"]
|
|
80
|
+
|
|
81
|
+
[tool.ruff]
|
|
82
|
+
line-length = 100
|
|
83
|
+
target-version = "py310"
|
|
84
|
+
|
|
85
|
+
[tool.ruff.lint]
|
|
86
|
+
select = ["E", "F", "I", "UP", "B", "SIM"]
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
"""risforge: merge, clean, deduplicate, and enrich RIS bibliographic files."""
|
|
2
|
+
|
|
3
|
+
from risforge.cleaning import clean_ris_file, process_ris_file
|
|
4
|
+
from risforge.enrichment import RisEnricher
|
|
5
|
+
from risforge.exceptions import RisForgeError, RisParsingError
|
|
6
|
+
from risforge.merging import MergeResult, merge_ris_files
|
|
7
|
+
from risforge.pipeline import PipelineResult, risforge, run_pipeline
|
|
8
|
+
|
|
9
|
+
__version__ = "0.4.0"
|
|
10
|
+
|
|
11
|
+
__all__ = [
|
|
12
|
+
"MergeResult",
|
|
13
|
+
"PipelineResult",
|
|
14
|
+
"RisEnricher",
|
|
15
|
+
"RisForgeError",
|
|
16
|
+
"RisParsingError",
|
|
17
|
+
"clean_ris_file",
|
|
18
|
+
"merge_ris_files",
|
|
19
|
+
"process_ris_file",
|
|
20
|
+
"risforge",
|
|
21
|
+
"run_pipeline",
|
|
22
|
+
]
|
|
@@ -0,0 +1,310 @@
|
|
|
1
|
+
"""Cleaning, normalization, and deduplication for RIS records."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
import re
|
|
7
|
+
import unicodedata
|
|
8
|
+
from collections import defaultdict
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
import rispy
|
|
13
|
+
import rispy.writer
|
|
14
|
+
|
|
15
|
+
from risforge.exceptions import RisParsingError
|
|
16
|
+
|
|
17
|
+
logger = logging.getLogger(__name__)
|
|
18
|
+
|
|
19
|
+
RisRecord = dict[str, Any]
|
|
20
|
+
|
|
21
|
+
_RECORD_START = re.compile(r"(?m)^TY\s+-")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class CleanRisWriter(rispy.writer.RisWriter):
|
|
25
|
+
"""RIS writer without record numbering and with CRLF line endings."""
|
|
26
|
+
|
|
27
|
+
NEWLINE = "\r\n"
|
|
28
|
+
|
|
29
|
+
def set_header(self, count: int) -> str:
|
|
30
|
+
return ""
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
_CleanRisWriter = CleanRisWriter
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def normalize_title(title: str | None) -> str:
|
|
37
|
+
"""Normalize a title for deduplication comparison."""
|
|
38
|
+
if not isinstance(title, str) or not title:
|
|
39
|
+
return ""
|
|
40
|
+
|
|
41
|
+
normalized = unicodedata.normalize("NFKD", title).lower()
|
|
42
|
+
normalized = re.sub(r"[^a-z0-9\s]", "", normalized)
|
|
43
|
+
return re.sub(r"\s+", " ", normalized).strip()
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def extract_first_author(author_list: list[str] | None) -> str:
|
|
47
|
+
"""Extract a normalized first-author key: surname plus initials."""
|
|
48
|
+
if not author_list or not isinstance(author_list[0], str):
|
|
49
|
+
return ""
|
|
50
|
+
|
|
51
|
+
parts = author_list[0].split(",")
|
|
52
|
+
last_name = parts[0].strip()
|
|
53
|
+
|
|
54
|
+
initials = ""
|
|
55
|
+
if len(parts) > 1:
|
|
56
|
+
tokens = re.split(r"[\s.\-]+", parts[1].strip())
|
|
57
|
+
initials = "".join(token[0] for token in tokens if token)
|
|
58
|
+
|
|
59
|
+
name = f"{last_name} {initials}"
|
|
60
|
+
name = unicodedata.normalize("NFKD", name).lower()
|
|
61
|
+
name = re.sub(r"[^a-z0-9\s]", "", name)
|
|
62
|
+
return re.sub(r"\s+", " ", name).strip()
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def normalize_doi(doi: str | None) -> str:
|
|
66
|
+
"""Normalize a DOI by removing URL prefixes and trailing punctuation."""
|
|
67
|
+
if not isinstance(doi, str) or not doi:
|
|
68
|
+
return ""
|
|
69
|
+
|
|
70
|
+
cleaned = doi.strip().lower()
|
|
71
|
+
cleaned = re.sub(r"^(https?://)?(dx\.)?doi\.org/|^doi:", "", cleaned)
|
|
72
|
+
return re.sub(r"[\s.,;:]+$", "", cleaned)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def count_fields(record: RisRecord) -> int:
|
|
76
|
+
"""Count populated fields in a record."""
|
|
77
|
+
total = 0
|
|
78
|
+
|
|
79
|
+
for key, value in record.items():
|
|
80
|
+
if key == "unknown_tag":
|
|
81
|
+
for values in value.values():
|
|
82
|
+
total += sum(bool(item) for item in values)
|
|
83
|
+
elif isinstance(value, list):
|
|
84
|
+
total += sum(bool(item) for item in value)
|
|
85
|
+
elif value:
|
|
86
|
+
total += 1
|
|
87
|
+
|
|
88
|
+
return total
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _copy_record(record: RisRecord) -> RisRecord:
|
|
92
|
+
"""Copy a record deeply enough for safe cluster merging."""
|
|
93
|
+
copied: RisRecord = {}
|
|
94
|
+
|
|
95
|
+
for key, value in record.items():
|
|
96
|
+
if key == "unknown_tag":
|
|
97
|
+
copied[key] = defaultdict(
|
|
98
|
+
list,
|
|
99
|
+
{tag: list(items) for tag, items in value.items()},
|
|
100
|
+
)
|
|
101
|
+
elif isinstance(value, list):
|
|
102
|
+
copied[key] = list(value)
|
|
103
|
+
else:
|
|
104
|
+
copied[key] = value
|
|
105
|
+
|
|
106
|
+
return copied
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def merge_cluster(cluster: list[RisRecord]) -> RisRecord:
|
|
110
|
+
"""Merge duplicate records into one record without losing fields."""
|
|
111
|
+
if len(cluster) == 1:
|
|
112
|
+
return cluster[0]
|
|
113
|
+
|
|
114
|
+
best = max(cluster, key=count_fields)
|
|
115
|
+
merged = _copy_record(best)
|
|
116
|
+
|
|
117
|
+
unknown = merged.get("unknown_tag")
|
|
118
|
+
if unknown is None:
|
|
119
|
+
unknown = defaultdict(list)
|
|
120
|
+
merged["unknown_tag"] = unknown
|
|
121
|
+
|
|
122
|
+
for record in cluster:
|
|
123
|
+
if record is best:
|
|
124
|
+
continue
|
|
125
|
+
|
|
126
|
+
for key, value in record.items():
|
|
127
|
+
if key == "unknown_tag":
|
|
128
|
+
for tag, items in value.items():
|
|
129
|
+
target = unknown.setdefault(tag, [])
|
|
130
|
+
for item in items:
|
|
131
|
+
if item not in target:
|
|
132
|
+
target.append(item)
|
|
133
|
+
elif key not in merged or not merged[key]:
|
|
134
|
+
merged[key] = list(value) if isinstance(value, list) else value
|
|
135
|
+
elif isinstance(merged[key], list) and isinstance(value, list):
|
|
136
|
+
for item in value:
|
|
137
|
+
if item not in merged[key]:
|
|
138
|
+
merged[key].append(item)
|
|
139
|
+
|
|
140
|
+
return merged
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
class RecordUnionFind:
|
|
144
|
+
"""Disjoint-set structure for clustering duplicate records."""
|
|
145
|
+
|
|
146
|
+
def __init__(self, records: list[RisRecord]) -> None:
|
|
147
|
+
self._parents = list(range(len(records)))
|
|
148
|
+
self._root_dois = [normalize_doi(record.get("doi", "")) for record in records]
|
|
149
|
+
|
|
150
|
+
def find(self, index: int) -> int:
|
|
151
|
+
root = index
|
|
152
|
+
while self._parents[root] != root:
|
|
153
|
+
root = self._parents[root]
|
|
154
|
+
|
|
155
|
+
current = index
|
|
156
|
+
while current != root:
|
|
157
|
+
next_index = self._parents[current]
|
|
158
|
+
self._parents[current] = root
|
|
159
|
+
current = next_index
|
|
160
|
+
|
|
161
|
+
return root
|
|
162
|
+
|
|
163
|
+
def union(self, first: int, second: int) -> bool:
|
|
164
|
+
root_first = self.find(first)
|
|
165
|
+
root_second = self.find(second)
|
|
166
|
+
|
|
167
|
+
if root_first == root_second:
|
|
168
|
+
return False
|
|
169
|
+
|
|
170
|
+
doi_first = self._root_dois[root_first]
|
|
171
|
+
doi_second = self._root_dois[root_second]
|
|
172
|
+
|
|
173
|
+
if doi_first and doi_second and doi_first != doi_second:
|
|
174
|
+
return False
|
|
175
|
+
|
|
176
|
+
self._parents[root_first] = root_second
|
|
177
|
+
self._root_dois[root_second] = doi_first or doi_second
|
|
178
|
+
return True
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
_RecordUnionFind = RecordUnionFind
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def _read_ris_text(input_path: Path) -> str:
|
|
185
|
+
"""Read a RIS file as UTF-8 text, stripping a BOM if present."""
|
|
186
|
+
if not input_path.exists():
|
|
187
|
+
raise FileNotFoundError(f"Input file '{input_path}' not found.")
|
|
188
|
+
|
|
189
|
+
try:
|
|
190
|
+
return input_path.read_text(encoding="utf-8-sig")
|
|
191
|
+
except UnicodeDecodeError as error:
|
|
192
|
+
raise RisParsingError(
|
|
193
|
+
f"Could not read '{input_path}' as UTF-8 text. "
|
|
194
|
+
"The file may be saved in a different encoding; re-save it as UTF-8."
|
|
195
|
+
) from error
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def parse_ris_records(
|
|
199
|
+
input_path: str | Path,
|
|
200
|
+
) -> tuple[list[RisRecord], list[tuple[int, str]]]:
|
|
201
|
+
"""Parse RIS records block by block, tolerating malformed blocks."""
|
|
202
|
+
input_path = Path(input_path)
|
|
203
|
+
text = _read_ris_text(input_path)
|
|
204
|
+
|
|
205
|
+
records: list[RisRecord] = []
|
|
206
|
+
errors: list[tuple[int, str]] = []
|
|
207
|
+
|
|
208
|
+
for block_number, block in enumerate(_RECORD_START.split(text), start=1):
|
|
209
|
+
if not block.strip():
|
|
210
|
+
continue
|
|
211
|
+
|
|
212
|
+
block_text = f"TY -{block}"
|
|
213
|
+
|
|
214
|
+
try:
|
|
215
|
+
parsed = rispy.loads(block_text)
|
|
216
|
+
except (ValueError, TypeError, KeyError, AttributeError, IndexError) as error:
|
|
217
|
+
errors.append((block_number, str(error)))
|
|
218
|
+
continue
|
|
219
|
+
|
|
220
|
+
if parsed:
|
|
221
|
+
records.extend(parsed)
|
|
222
|
+
else:
|
|
223
|
+
errors.append((block_number, "Empty parse result"))
|
|
224
|
+
|
|
225
|
+
return records, errors
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def write_ris_records(records: list[RisRecord], output_path: str | Path) -> None:
|
|
229
|
+
"""Write records to disk using the project's canonical RIS writer."""
|
|
230
|
+
output_path = Path(output_path)
|
|
231
|
+
text = rispy.dumps(records, implementation=CleanRisWriter)
|
|
232
|
+
output_path.write_text(text, encoding="utf-8")
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def clean_ris_file(
|
|
236
|
+
input_path: str | Path,
|
|
237
|
+
output_path: str | Path,
|
|
238
|
+
) -> tuple[list[RisRecord], list[tuple[int, str]]]:
|
|
239
|
+
"""Clean, deduplicate, and write a RIS file."""
|
|
240
|
+
input_path = Path(input_path)
|
|
241
|
+
output_path = Path(output_path)
|
|
242
|
+
|
|
243
|
+
records, errors = parse_ris_records(input_path)
|
|
244
|
+
|
|
245
|
+
if errors:
|
|
246
|
+
logger.warning("Skipped %d malformed record block(s) in %s.", len(errors), input_path)
|
|
247
|
+
|
|
248
|
+
if not records:
|
|
249
|
+
logger.warning("No valid records found in %s. Writing empty output.", input_path)
|
|
250
|
+
write_ris_records([], output_path)
|
|
251
|
+
return [], errors
|
|
252
|
+
|
|
253
|
+
union_find = RecordUnionFind(records)
|
|
254
|
+
|
|
255
|
+
doi_map: dict[str, int] = {}
|
|
256
|
+
doi_duplicates_removed = 0
|
|
257
|
+
|
|
258
|
+
for index, record in enumerate(records):
|
|
259
|
+
doi = normalize_doi(record.get("doi", ""))
|
|
260
|
+
if not doi:
|
|
261
|
+
continue
|
|
262
|
+
|
|
263
|
+
if doi in doi_map:
|
|
264
|
+
if union_find.union(index, doi_map[doi]):
|
|
265
|
+
doi_duplicates_removed += 1
|
|
266
|
+
else:
|
|
267
|
+
doi_map[doi] = index
|
|
268
|
+
|
|
269
|
+
composite_key_map: dict[str, int] = {}
|
|
270
|
+
title_author_duplicates_removed = 0
|
|
271
|
+
|
|
272
|
+
for index, record in enumerate(records):
|
|
273
|
+
title = normalize_title(record.get("title", ""))
|
|
274
|
+
author = extract_first_author(record.get("authors", []))
|
|
275
|
+
|
|
276
|
+
if not title or not author:
|
|
277
|
+
continue
|
|
278
|
+
|
|
279
|
+
composite_key = f"{title}|{author}"
|
|
280
|
+
|
|
281
|
+
if composite_key in composite_key_map:
|
|
282
|
+
if union_find.union(index, composite_key_map[composite_key]):
|
|
283
|
+
title_author_duplicates_removed += 1
|
|
284
|
+
composite_key_map[composite_key] = union_find.find(index)
|
|
285
|
+
else:
|
|
286
|
+
composite_key_map[composite_key] = union_find.find(index)
|
|
287
|
+
|
|
288
|
+
clusters: dict[int, list[int]] = defaultdict(list)
|
|
289
|
+
for index in range(len(records)):
|
|
290
|
+
clusters[union_find.find(index)].append(index)
|
|
291
|
+
|
|
292
|
+
final_records = [merge_cluster([records[i] for i in indices]) for indices in clusters.values()]
|
|
293
|
+
|
|
294
|
+
write_ris_records(final_records, output_path)
|
|
295
|
+
|
|
296
|
+
total_duplicates_removed = doi_duplicates_removed + title_author_duplicates_removed
|
|
297
|
+
|
|
298
|
+
logger.info(
|
|
299
|
+
"Deduplication complete: input=%d malformed=%d duplicates_removed=%d output=%s",
|
|
300
|
+
len(records),
|
|
301
|
+
len(errors),
|
|
302
|
+
total_duplicates_removed,
|
|
303
|
+
output_path,
|
|
304
|
+
)
|
|
305
|
+
|
|
306
|
+
return final_records, errors
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
# Backward-compatible alias.
|
|
310
|
+
process_ris_file = clean_ris_file
|