risforge 0.3.2__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. risforge-0.4.0/PKG-INFO +143 -0
  2. risforge-0.4.0/README.md +102 -0
  3. {risforge-0.3.2 → risforge-0.4.0}/pyproject.toml +30 -27
  4. risforge-0.4.0/src/risforge/__init__.py +22 -0
  5. risforge-0.4.0/src/risforge/cleaning.py +310 -0
  6. {risforge-0.3.2 → risforge-0.4.0}/src/risforge/cli.py +50 -128
  7. {risforge-0.3.2 → risforge-0.4.0}/src/risforge/enrichment.py +56 -134
  8. risforge-0.4.0/src/risforge/exceptions.py +15 -0
  9. risforge-0.4.0/src/risforge/merging.py +74 -0
  10. risforge-0.4.0/src/risforge/pipeline.py +143 -0
  11. risforge-0.4.0/src/risforge.egg-info/PKG-INFO +143 -0
  12. {risforge-0.3.2 → risforge-0.4.0}/src/risforge.egg-info/requires.txt +7 -8
  13. risforge-0.4.0/src/risforge_gui/__init__.py +3 -0
  14. risforge-0.4.0/src/risforge_gui/app.py +41 -0
  15. {risforge-0.3.2 → risforge-0.4.0}/src/risforge_gui/dialogs.py +6 -9
  16. {risforge-0.3.2 → risforge-0.4.0}/src/risforge_gui/file_counter.py +6 -14
  17. risforge-0.4.0/src/risforge_gui/main_window.py +324 -0
  18. {risforge-0.3.2 → risforge-0.4.0}/src/risforge_gui/models.py +39 -32
  19. risforge-0.4.0/src/risforge_gui/theme.py +197 -0
  20. risforge-0.4.0/src/risforge_gui/widgets/__init__.py +17 -0
  21. {risforge-0.3.2 → risforge-0.4.0}/src/risforge_gui/widgets/config_panel.py +23 -14
  22. {risforge-0.3.2 → risforge-0.4.0}/src/risforge_gui/widgets/input_panel.py +63 -46
  23. {risforge-0.3.2 → risforge-0.4.0}/src/risforge_gui/widgets/log_panel.py +7 -3
  24. {risforge-0.3.2 → risforge-0.4.0}/src/risforge_gui/widgets/output_panel.py +19 -24
  25. {risforge-0.3.2 → risforge-0.4.0}/src/risforge_gui/widgets/progress_panel.py +29 -13
  26. {risforge-0.3.2 → risforge-0.4.0}/src/risforge_gui/widgets/results_panel.py +13 -11
  27. {risforge-0.3.2 → risforge-0.4.0}/src/risforge_gui/worker.py +84 -75
  28. {risforge-0.3.2 → risforge-0.4.0}/tests/test_cli.py +6 -21
  29. {risforge-0.3.2 → risforge-0.4.0}/tests/test_enrichment.py +1 -4
  30. {risforge-0.3.2 → risforge-0.4.0}/tests/test_merging.py +20 -19
  31. {risforge-0.3.2 → risforge-0.4.0}/tests/test_pipeline.py +2 -6
  32. risforge-0.3.2/PKG-INFO +0 -296
  33. risforge-0.3.2/README.md +0 -255
  34. risforge-0.3.2/src/risforge/__init__.py +0 -67
  35. risforge-0.3.2/src/risforge/cleaning.py +0 -370
  36. risforge-0.3.2/src/risforge/exceptions.py +0 -38
  37. risforge-0.3.2/src/risforge/merging.py +0 -130
  38. risforge-0.3.2/src/risforge/pipeline.py +0 -233
  39. risforge-0.3.2/src/risforge.egg-info/PKG-INFO +0 -296
  40. risforge-0.3.2/src/risforge_gui/__init__.py +0 -21
  41. risforge-0.3.2/src/risforge_gui/app.py +0 -34
  42. risforge-0.3.2/src/risforge_gui/main_window.py +0 -375
  43. risforge-0.3.2/src/risforge_gui/theme.py +0 -236
  44. risforge-0.3.2/src/risforge_gui/widgets/__init__.py +0 -1
  45. {risforge-0.3.2 → risforge-0.4.0}/LICENSE +0 -0
  46. {risforge-0.3.2 → risforge-0.4.0}/setup.cfg +0 -0
  47. {risforge-0.3.2 → risforge-0.4.0}/src/risforge.egg-info/SOURCES.txt +0 -0
  48. {risforge-0.3.2 → risforge-0.4.0}/src/risforge.egg-info/dependency_links.txt +0 -0
  49. {risforge-0.3.2 → risforge-0.4.0}/src/risforge.egg-info/entry_points.txt +0 -0
  50. {risforge-0.3.2 → risforge-0.4.0}/src/risforge.egg-info/top_level.txt +0 -0
  51. {risforge-0.3.2 → risforge-0.4.0}/tests/test_cleaning.py +0 -0
@@ -0,0 +1,143 @@
1
+ Metadata-Version: 2.4
2
+ Name: risforge
3
+ Version: 0.4.0
4
+ Summary: Merge, clean, deduplicate, and enrich RIS bibliographic files.
5
+ Author-email: Amyr <amyrhexa@gmail.com>
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/amyr/risforge
8
+ Project-URL: Repository, https://github.com/amyr/risforge
9
+ Project-URL: Issues, https://github.com/amyr/risforge/issues
10
+ Project-URL: Changelog, https://github.com/amyr/risforge/blob/main/CHANGELOG.md
11
+ Keywords: ris,bibliography,systematic-review,deduplication,enrichment,pyside6
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: Topic :: Scientific/Engineering
15
+ Classifier: License :: OSI Approved :: MIT License
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Operating System :: OS Independent
22
+ Classifier: Environment :: X11 Applications :: Qt
23
+ Requires-Python: >=3.10
24
+ Description-Content-Type: text/markdown
25
+ License-File: LICENSE
26
+ Requires-Dist: rispy>=0.10.0
27
+ Requires-Dist: requests>=2.31.0
28
+ Requires-Dist: requests-cache>=1.2.0
29
+ Provides-Extra: gui
30
+ Requires-Dist: PySide6>=6.5; extra == "gui"
31
+ Provides-Extra: dev
32
+ Requires-Dist: pytest>=8.0; extra == "dev"
33
+ Requires-Dist: pytest-cov>=5.0; extra == "dev"
34
+ Requires-Dist: pytest-qt>=4.4; extra == "dev"
35
+ Requires-Dist: build>=1.2; extra == "dev"
36
+ Requires-Dist: twine>=5.0; extra == "dev"
37
+ Requires-Dist: ruff>=0.4.0; extra == "dev"
38
+ Requires-Dist: black>=24.0; extra == "dev"
39
+ Requires-Dist: mypy>=1.10; extra == "dev"
40
+ Dynamic: license-file
41
+
42
+ <h1 align="center">risforge</h1>
43
+
44
+ <p align="center">
45
+ <strong>Clean, deduplicate, and enrich RIS bibliographic files for systematic reviews.</strong>
46
+ </p>
47
+
48
+ `risforge` transforms messy, overlapping `.ris` exports from Scopus, PubMed, Web of Science, or EndNote into a single, clean, and metadata-rich file ready for screening.
49
+ <p align="center">
50
+ <img src="docs/Screenshot.png" alt="RisForge GUI Screenshot">
51
+ </p>
52
+
53
+ ## How it works
54
+
55
+ | Stage | What it does | What it doesn't do |
56
+ | :--- | :--- | :--- |
57
+ | **Merge** | Combines multiple `.ris` files into one. | Decide if records are duplicates. |
58
+ | **Clean** | Deduplicates via DOI and fuzzy title/author matching. Zero data loss. | Fetch new data from the internet. |
59
+ | **Enrich** | Fills missing metadata (abstracts, PDFs, keywords) via Crossref, OpenAlex, Semantic Scholar, and Unpaywall. | Overwrite your existing data. |
60
+
61
+ ## Features
62
+
63
+ - **Smart Deduplication**: DOI-first clustering with a normalized title + first-author fallback.
64
+ - **Zero Data Loss**: Merges duplicate clusters by keeping the most complete record and appending missing fields from others.
65
+ - **Non-Destructive Enrichment**: Only fills empty fields; existing metadata is never overwritten.
66
+ - **Resilient HTTP**: Automatic retries, exponential backoff, and a 7-day on-disk cache.
67
+ - **Flexible Interfaces**: Use it as a Python library, a CLI tool, or a cross-platform desktop GUI.
68
+
69
+ ## Installation
70
+
71
+ Requires Python 3.10+.
72
+
73
+ ```bash
74
+ # Core library and CLI
75
+ pip install risforge
76
+
77
+ # With the optional desktop GUI
78
+ pip install "risforge[gui]"
79
+ ```
80
+
81
+ ## Quick Start
82
+
83
+ ### Desktop GUI
84
+
85
+ Launch the graphical interface for a drag-and-drop workflow:
86
+
87
+ ```bash
88
+ risforge-gui
89
+ ```
90
+
91
+ ### Command Line
92
+
93
+ Run the full pipeline (merge, clean, and enrich) in a single command:
94
+
95
+ ```bash
96
+ risforge pipeline scopus.ris pubmed.ris wos.ris \
97
+ --email you@example.com \
98
+ --output final.ris
99
+ ```
100
+
101
+ *Note: An email is required for API "polite pool" access. It is only sent in request headers and never stored.*
102
+
103
+ Run individual steps if you need to inspect intermediate files:
104
+
105
+ ```bash
106
+ risforge merge scopus.ris pubmed.ris wos.ris merged.ris
107
+ risforge clean merged.ris clean.ris
108
+ risforge enrich clean.ris enriched.ris --email you@example.com
109
+ ```
110
+
111
+ ### Python API
112
+
113
+ ```python
114
+ from risforge import risforge
115
+
116
+ # Automatically merges multiple files, cleans, and enriches
117
+ result = risforge(
118
+ input_paths=["scopus.ris", "pubmed.ris", "wos.ris"],
119
+ dedup_path="clean.ris",
120
+ enriched_path="enriched.ris",
121
+ email="you@example.com",
122
+ )
123
+
124
+ print(f"Merged: {result.merged_record_count}")
125
+ print(f"Unique: {result.cleaned_record_count}")
126
+ print(f"Enriched: {result.enrichment_stats['enriched']}")
127
+ ```
128
+
129
+ ## Good to Know
130
+
131
+ - **Unresolved DOIs**: Records without a DOI that fail title-matching are left unmodified and logged to `failed_records.json`.
132
+ - **Merge vs. Clean**: `merge` simply concatenates files. Run `clean` afterward to actually collapse duplicates.
133
+ - **Caching**: API responses are cached for 7 days. Re-running the pipeline on the same data is nearly instant.
134
+
135
+ ## Contributing
136
+
137
+ 1. Clone and install dev dependencies: `pip install -e ".[dev,gui]"`
138
+ 2. Run tests: `pytest`
139
+ 3. Open a PR. Please include tests for behavioral changes and update `CHANGELOG.md`. Predictability is a core feature; silent behavior changes are treated as bugs.
140
+
141
+ ## License
142
+
143
+ MIT — see [LICENSE](LICENSE).
@@ -0,0 +1,102 @@
1
+ <h1 align="center">risforge</h1>
2
+
3
+ <p align="center">
4
+ <strong>Clean, deduplicate, and enrich RIS bibliographic files for systematic reviews.</strong>
5
+ </p>
6
+
7
+ `risforge` transforms messy, overlapping `.ris` exports from Scopus, PubMed, Web of Science, or EndNote into a single, clean, and metadata-rich file ready for screening.
8
+ <p align="center">
9
+ <img src="docs/Screenshot.png" alt="RisForge GUI Screenshot">
10
+ </p>
11
+
12
+ ## How it works
13
+
14
+ | Stage | What it does | What it doesn't do |
15
+ | :--- | :--- | :--- |
16
+ | **Merge** | Combines multiple `.ris` files into one. | Decide if records are duplicates. |
17
+ | **Clean** | Deduplicates via DOI and fuzzy title/author matching. Zero data loss. | Fetch new data from the internet. |
18
+ | **Enrich** | Fills missing metadata (abstracts, PDFs, keywords) via Crossref, OpenAlex, Semantic Scholar, and Unpaywall. | Overwrite your existing data. |
19
+
20
+ ## Features
21
+
22
+ - **Smart Deduplication**: DOI-first clustering with a normalized title + first-author fallback.
23
+ - **Zero Data Loss**: Merges duplicate clusters by keeping the most complete record and appending missing fields from others.
24
+ - **Non-Destructive Enrichment**: Only fills empty fields; existing metadata is never overwritten.
25
+ - **Resilient HTTP**: Automatic retries, exponential backoff, and a 7-day on-disk cache.
26
+ - **Flexible Interfaces**: Use it as a Python library, a CLI tool, or a cross-platform desktop GUI.
27
+
28
+ ## Installation
29
+
30
+ Requires Python 3.10+.
31
+
32
+ ```bash
33
+ # Core library and CLI
34
+ pip install risforge
35
+
36
+ # With the optional desktop GUI
37
+ pip install "risforge[gui]"
38
+ ```
39
+
40
+ ## Quick Start
41
+
42
+ ### Desktop GUI
43
+
44
+ Launch the graphical interface for a drag-and-drop workflow:
45
+
46
+ ```bash
47
+ risforge-gui
48
+ ```
49
+
50
+ ### Command Line
51
+
52
+ Run the full pipeline (merge, clean, and enrich) in a single command:
53
+
54
+ ```bash
55
+ risforge pipeline scopus.ris pubmed.ris wos.ris \
56
+ --email you@example.com \
57
+ --output final.ris
58
+ ```
59
+
60
+ *Note: An email is required for API "polite pool" access. It is only sent in request headers and never stored.*
61
+
62
+ Run individual steps if you need to inspect intermediate files:
63
+
64
+ ```bash
65
+ risforge merge scopus.ris pubmed.ris wos.ris merged.ris
66
+ risforge clean merged.ris clean.ris
67
+ risforge enrich clean.ris enriched.ris --email you@example.com
68
+ ```
69
+
70
+ ### Python API
71
+
72
+ ```python
73
+ from risforge import risforge
74
+
75
+ # Automatically merges multiple files, cleans, and enriches
76
+ result = risforge(
77
+ input_paths=["scopus.ris", "pubmed.ris", "wos.ris"],
78
+ dedup_path="clean.ris",
79
+ enriched_path="enriched.ris",
80
+ email="you@example.com",
81
+ )
82
+
83
+ print(f"Merged: {result.merged_record_count}")
84
+ print(f"Unique: {result.cleaned_record_count}")
85
+ print(f"Enriched: {result.enrichment_stats['enriched']}")
86
+ ```
87
+
88
+ ## Good to Know
89
+
90
+ - **Unresolved DOIs**: Records without a DOI that fail title-matching are left unmodified and logged to `failed_records.json`.
91
+ - **Merge vs. Clean**: `merge` simply concatenates files. Run `clean` afterward to actually collapse duplicates.
92
+ - **Caching**: API responses are cached for 7 days. Re-running the pipeline on the same data is nearly instant.
93
+
94
+ ## Contributing
95
+
96
+ 1. Clone and install dev dependencies: `pip install -e ".[dev,gui]"`
97
+ 2. Run tests: `pytest`
98
+ 3. Open a PR. Please include tests for behavioral changes and update `CHANGELOG.md`. Predictability is a core feature; silent behavior changes are treated as bugs.
99
+
100
+ ## License
101
+
102
+ MIT — see [LICENSE](LICENSE).
@@ -4,29 +4,24 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "risforge"
7
- version = "0.3.2"
8
- description = "Clean, deduplicate, and enrich RIS bibliographic files for systematic reviews."
7
+ version = "0.4.0"
8
+ description = "Merge, clean, deduplicate, and enrich RIS bibliographic files."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
11
11
  license = { text = "MIT" }
12
- authors = [
13
- { name = "Amyr", email = "amyrhexa@gmail.com" }
14
- ]
12
+ authors = [{ name = "Amyr", email = "amyrhexa@gmail.com" }]
15
13
  keywords = [
16
14
  "ris",
17
15
  "bibliography",
18
16
  "systematic-review",
19
17
  "deduplication",
20
- "crossref",
21
- "openalex",
22
- "citation-management",
23
- "prisma",
18
+ "enrichment",
19
+ "pyside6",
24
20
  ]
25
21
  classifiers = [
26
22
  "Development Status :: 4 - Beta",
27
23
  "Intended Audience :: Science/Research",
28
24
  "Topic :: Scientific/Engineering",
29
- "Topic :: Text Processing :: Filters",
30
25
  "License :: OSI Approved :: MIT License",
31
26
  "Programming Language :: Python :: 3",
32
27
  "Programming Language :: Python :: 3.10",
@@ -34,35 +29,29 @@ classifiers = [
34
29
  "Programming Language :: Python :: 3.12",
35
30
  "Programming Language :: Python :: 3.13",
36
31
  "Operating System :: OS Independent",
32
+ "Environment :: X11 Applications :: Qt",
37
33
  ]
38
34
 
39
- dependencies = [
40
- "rispy>=0.9.0",
41
- "requests>=2.31",
42
- "requests-cache>=1.1",
43
- "urllib3>=2.0",
44
- ]
35
+ dependencies = ["rispy>=0.10.0", "requests>=2.31.0", "requests-cache>=1.2.0"]
45
36
 
46
37
  [project.optional-dependencies]
38
+ gui = ["PySide6>=6.5"]
47
39
  dev = [
48
40
  "pytest>=8.0",
49
41
  "pytest-cov>=5.0",
42
+ "pytest-qt>=4.4",
50
43
  "build>=1.2",
51
44
  "twine>=5.0",
52
- ]
53
- gui = [
54
- "PySide6>=6.5",
55
- ]
56
- gui-dev = [
57
- "risforge[gui]",
58
- "pytest-qt>=4.4",
45
+ "ruff>=0.4.0",
46
+ "black>=24.0",
47
+ "mypy>=1.10",
59
48
  ]
60
49
 
61
50
  [project.urls]
62
- Homepage = "https://github.com/pythyn/risforge"
63
- Repository = "https://github.com/pythyn/risforge"
64
- Issues = "https://github.com/pythyn/risforge/issues"
65
- Changelog = "https://github.com/pythyn/risforge/blob/main/CHANGELOG.md"
51
+ Homepage = "https://github.com/amyr/risforge"
52
+ Repository = "https://github.com/amyr/risforge"
53
+ Issues = "https://github.com/amyr/risforge/issues"
54
+ Changelog = "https://github.com/amyr/risforge/blob/main/CHANGELOG.md"
66
55
 
67
56
  [project.scripts]
68
57
  risforge = "risforge.cli:main"
@@ -72,12 +61,26 @@ risforge-gui = "risforge_gui.app:main"
72
61
 
73
62
  [tool.setuptools.packages.find]
74
63
  where = ["src"]
64
+ include = ["risforge*", "risforge_gui*"]
75
65
 
76
66
  [tool.setuptools.package-dir]
77
67
  "" = "src"
78
68
 
79
69
  [tool.pytest.ini_options]
80
70
  testpaths = ["tests"]
71
+ qt_api = "pyside6"
81
72
 
82
73
  [tool.coverage.run]
83
74
  concurrency = ["thread"]
75
+ source = ["risforge", "risforge_gui"]
76
+
77
+ [tool.black]
78
+ line-length = 100
79
+ target-version = ["py310", "py311", "py312"]
80
+
81
+ [tool.ruff]
82
+ line-length = 100
83
+ target-version = "py310"
84
+
85
+ [tool.ruff.lint]
86
+ select = ["E", "F", "I", "UP", "B", "SIM"]
@@ -0,0 +1,22 @@
1
+ """risforge: merge, clean, deduplicate, and enrich RIS bibliographic files."""
2
+
3
+ from risforge.cleaning import clean_ris_file, process_ris_file
4
+ from risforge.enrichment import RisEnricher
5
+ from risforge.exceptions import RisForgeError, RisParsingError
6
+ from risforge.merging import MergeResult, merge_ris_files
7
+ from risforge.pipeline import PipelineResult, risforge, run_pipeline
8
+
9
+ __version__ = "0.4.0"
10
+
11
+ __all__ = [
12
+ "MergeResult",
13
+ "PipelineResult",
14
+ "RisEnricher",
15
+ "RisForgeError",
16
+ "RisParsingError",
17
+ "clean_ris_file",
18
+ "merge_ris_files",
19
+ "process_ris_file",
20
+ "risforge",
21
+ "run_pipeline",
22
+ ]
@@ -0,0 +1,310 @@
1
+ """Cleaning, normalization, and deduplication for RIS records."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ import re
7
+ import unicodedata
8
+ from collections import defaultdict
9
+ from pathlib import Path
10
+ from typing import Any
11
+
12
+ import rispy
13
+ import rispy.writer
14
+
15
+ from risforge.exceptions import RisParsingError
16
+
17
+ logger = logging.getLogger(__name__)
18
+
19
+ RisRecord = dict[str, Any]
20
+
21
+ _RECORD_START = re.compile(r"(?m)^TY\s+-")
22
+
23
+
24
+ class CleanRisWriter(rispy.writer.RisWriter):
25
+ """RIS writer without record numbering and with CRLF line endings."""
26
+
27
+ NEWLINE = "\r\n"
28
+
29
+ def set_header(self, count: int) -> str:
30
+ return ""
31
+
32
+
33
+ _CleanRisWriter = CleanRisWriter
34
+
35
+
36
+ def normalize_title(title: str | None) -> str:
37
+ """Normalize a title for deduplication comparison."""
38
+ if not isinstance(title, str) or not title:
39
+ return ""
40
+
41
+ normalized = unicodedata.normalize("NFKD", title).lower()
42
+ normalized = re.sub(r"[^a-z0-9\s]", "", normalized)
43
+ return re.sub(r"\s+", " ", normalized).strip()
44
+
45
+
46
+ def extract_first_author(author_list: list[str] | None) -> str:
47
+ """Extract a normalized first-author key: surname plus initials."""
48
+ if not author_list or not isinstance(author_list[0], str):
49
+ return ""
50
+
51
+ parts = author_list[0].split(",")
52
+ last_name = parts[0].strip()
53
+
54
+ initials = ""
55
+ if len(parts) > 1:
56
+ tokens = re.split(r"[\s.\-]+", parts[1].strip())
57
+ initials = "".join(token[0] for token in tokens if token)
58
+
59
+ name = f"{last_name} {initials}"
60
+ name = unicodedata.normalize("NFKD", name).lower()
61
+ name = re.sub(r"[^a-z0-9\s]", "", name)
62
+ return re.sub(r"\s+", " ", name).strip()
63
+
64
+
65
+ def normalize_doi(doi: str | None) -> str:
66
+ """Normalize a DOI by removing URL prefixes and trailing punctuation."""
67
+ if not isinstance(doi, str) or not doi:
68
+ return ""
69
+
70
+ cleaned = doi.strip().lower()
71
+ cleaned = re.sub(r"^(https?://)?(dx\.)?doi\.org/|^doi:", "", cleaned)
72
+ return re.sub(r"[\s.,;:]+$", "", cleaned)
73
+
74
+
75
+ def count_fields(record: RisRecord) -> int:
76
+ """Count populated fields in a record."""
77
+ total = 0
78
+
79
+ for key, value in record.items():
80
+ if key == "unknown_tag":
81
+ for values in value.values():
82
+ total += sum(bool(item) for item in values)
83
+ elif isinstance(value, list):
84
+ total += sum(bool(item) for item in value)
85
+ elif value:
86
+ total += 1
87
+
88
+ return total
89
+
90
+
91
+ def _copy_record(record: RisRecord) -> RisRecord:
92
+ """Copy a record deeply enough for safe cluster merging."""
93
+ copied: RisRecord = {}
94
+
95
+ for key, value in record.items():
96
+ if key == "unknown_tag":
97
+ copied[key] = defaultdict(
98
+ list,
99
+ {tag: list(items) for tag, items in value.items()},
100
+ )
101
+ elif isinstance(value, list):
102
+ copied[key] = list(value)
103
+ else:
104
+ copied[key] = value
105
+
106
+ return copied
107
+
108
+
109
+ def merge_cluster(cluster: list[RisRecord]) -> RisRecord:
110
+ """Merge duplicate records into one record without losing fields."""
111
+ if len(cluster) == 1:
112
+ return cluster[0]
113
+
114
+ best = max(cluster, key=count_fields)
115
+ merged = _copy_record(best)
116
+
117
+ unknown = merged.get("unknown_tag")
118
+ if unknown is None:
119
+ unknown = defaultdict(list)
120
+ merged["unknown_tag"] = unknown
121
+
122
+ for record in cluster:
123
+ if record is best:
124
+ continue
125
+
126
+ for key, value in record.items():
127
+ if key == "unknown_tag":
128
+ for tag, items in value.items():
129
+ target = unknown.setdefault(tag, [])
130
+ for item in items:
131
+ if item not in target:
132
+ target.append(item)
133
+ elif key not in merged or not merged[key]:
134
+ merged[key] = list(value) if isinstance(value, list) else value
135
+ elif isinstance(merged[key], list) and isinstance(value, list):
136
+ for item in value:
137
+ if item not in merged[key]:
138
+ merged[key].append(item)
139
+
140
+ return merged
141
+
142
+
143
+ class RecordUnionFind:
144
+ """Disjoint-set structure for clustering duplicate records."""
145
+
146
+ def __init__(self, records: list[RisRecord]) -> None:
147
+ self._parents = list(range(len(records)))
148
+ self._root_dois = [normalize_doi(record.get("doi", "")) for record in records]
149
+
150
+ def find(self, index: int) -> int:
151
+ root = index
152
+ while self._parents[root] != root:
153
+ root = self._parents[root]
154
+
155
+ current = index
156
+ while current != root:
157
+ next_index = self._parents[current]
158
+ self._parents[current] = root
159
+ current = next_index
160
+
161
+ return root
162
+
163
+ def union(self, first: int, second: int) -> bool:
164
+ root_first = self.find(first)
165
+ root_second = self.find(second)
166
+
167
+ if root_first == root_second:
168
+ return False
169
+
170
+ doi_first = self._root_dois[root_first]
171
+ doi_second = self._root_dois[root_second]
172
+
173
+ if doi_first and doi_second and doi_first != doi_second:
174
+ return False
175
+
176
+ self._parents[root_first] = root_second
177
+ self._root_dois[root_second] = doi_first or doi_second
178
+ return True
179
+
180
+
181
+ _RecordUnionFind = RecordUnionFind
182
+
183
+
184
+ def _read_ris_text(input_path: Path) -> str:
185
+ """Read a RIS file as UTF-8 text, stripping a BOM if present."""
186
+ if not input_path.exists():
187
+ raise FileNotFoundError(f"Input file '{input_path}' not found.")
188
+
189
+ try:
190
+ return input_path.read_text(encoding="utf-8-sig")
191
+ except UnicodeDecodeError as error:
192
+ raise RisParsingError(
193
+ f"Could not read '{input_path}' as UTF-8 text. "
194
+ "The file may be saved in a different encoding; re-save it as UTF-8."
195
+ ) from error
196
+
197
+
198
+ def parse_ris_records(
199
+ input_path: str | Path,
200
+ ) -> tuple[list[RisRecord], list[tuple[int, str]]]:
201
+ """Parse RIS records block by block, tolerating malformed blocks."""
202
+ input_path = Path(input_path)
203
+ text = _read_ris_text(input_path)
204
+
205
+ records: list[RisRecord] = []
206
+ errors: list[tuple[int, str]] = []
207
+
208
+ for block_number, block in enumerate(_RECORD_START.split(text), start=1):
209
+ if not block.strip():
210
+ continue
211
+
212
+ block_text = f"TY -{block}"
213
+
214
+ try:
215
+ parsed = rispy.loads(block_text)
216
+ except (ValueError, TypeError, KeyError, AttributeError, IndexError) as error:
217
+ errors.append((block_number, str(error)))
218
+ continue
219
+
220
+ if parsed:
221
+ records.extend(parsed)
222
+ else:
223
+ errors.append((block_number, "Empty parse result"))
224
+
225
+ return records, errors
226
+
227
+
228
+ def write_ris_records(records: list[RisRecord], output_path: str | Path) -> None:
229
+ """Write records to disk using the project's canonical RIS writer."""
230
+ output_path = Path(output_path)
231
+ text = rispy.dumps(records, implementation=CleanRisWriter)
232
+ output_path.write_text(text, encoding="utf-8")
233
+
234
+
235
+ def clean_ris_file(
236
+ input_path: str | Path,
237
+ output_path: str | Path,
238
+ ) -> tuple[list[RisRecord], list[tuple[int, str]]]:
239
+ """Clean, deduplicate, and write a RIS file."""
240
+ input_path = Path(input_path)
241
+ output_path = Path(output_path)
242
+
243
+ records, errors = parse_ris_records(input_path)
244
+
245
+ if errors:
246
+ logger.warning("Skipped %d malformed record block(s) in %s.", len(errors), input_path)
247
+
248
+ if not records:
249
+ logger.warning("No valid records found in %s. Writing empty output.", input_path)
250
+ write_ris_records([], output_path)
251
+ return [], errors
252
+
253
+ union_find = RecordUnionFind(records)
254
+
255
+ doi_map: dict[str, int] = {}
256
+ doi_duplicates_removed = 0
257
+
258
+ for index, record in enumerate(records):
259
+ doi = normalize_doi(record.get("doi", ""))
260
+ if not doi:
261
+ continue
262
+
263
+ if doi in doi_map:
264
+ if union_find.union(index, doi_map[doi]):
265
+ doi_duplicates_removed += 1
266
+ else:
267
+ doi_map[doi] = index
268
+
269
+ composite_key_map: dict[str, int] = {}
270
+ title_author_duplicates_removed = 0
271
+
272
+ for index, record in enumerate(records):
273
+ title = normalize_title(record.get("title", ""))
274
+ author = extract_first_author(record.get("authors", []))
275
+
276
+ if not title or not author:
277
+ continue
278
+
279
+ composite_key = f"{title}|{author}"
280
+
281
+ if composite_key in composite_key_map:
282
+ if union_find.union(index, composite_key_map[composite_key]):
283
+ title_author_duplicates_removed += 1
284
+ composite_key_map[composite_key] = union_find.find(index)
285
+ else:
286
+ composite_key_map[composite_key] = union_find.find(index)
287
+
288
+ clusters: dict[int, list[int]] = defaultdict(list)
289
+ for index in range(len(records)):
290
+ clusters[union_find.find(index)].append(index)
291
+
292
+ final_records = [merge_cluster([records[i] for i in indices]) for indices in clusters.values()]
293
+
294
+ write_ris_records(final_records, output_path)
295
+
296
+ total_duplicates_removed = doi_duplicates_removed + title_author_duplicates_removed
297
+
298
+ logger.info(
299
+ "Deduplication complete: input=%d malformed=%d duplicates_removed=%d output=%s",
300
+ len(records),
301
+ len(errors),
302
+ total_duplicates_removed,
303
+ output_path,
304
+ )
305
+
306
+ return final_records, errors
307
+
308
+
309
+ # Backward-compatible alias.
310
+ process_ris_file = clean_ris_file