risforge 0.3.0__tar.gz → 0.3.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {risforge-0.3.0/src/risforge.egg-info → risforge-0.3.2}/PKG-INFO +5 -5
- {risforge-0.3.0 → risforge-0.3.2}/pyproject.toml +5 -5
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge/__init__.py +1 -1
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge/cleaning.py +22 -1
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge/enrichment.py +31 -2
- risforge-0.3.2/src/risforge/exceptions.py +38 -0
- {risforge-0.3.0 → risforge-0.3.2/src/risforge.egg-info}/PKG-INFO +5 -5
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/main_window.py +2 -0
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/widgets/results_panel.py +12 -1
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/worker.py +6 -1
- {risforge-0.3.0 → risforge-0.3.2}/tests/test_cleaning.py +76 -0
- {risforge-0.3.0 → risforge-0.3.2}/tests/test_enrichment.py +90 -0
- risforge-0.3.0/src/risforge/exceptions.py +0 -23
- {risforge-0.3.0 → risforge-0.3.2}/LICENSE +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/README.md +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/setup.cfg +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge/cli.py +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge/merging.py +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge/pipeline.py +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge.egg-info/SOURCES.txt +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge.egg-info/dependency_links.txt +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge.egg-info/entry_points.txt +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge.egg-info/requires.txt +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge.egg-info/top_level.txt +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/__init__.py +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/app.py +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/dialogs.py +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/file_counter.py +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/models.py +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/theme.py +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/widgets/__init__.py +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/widgets/config_panel.py +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/widgets/input_panel.py +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/widgets/log_panel.py +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/widgets/output_panel.py +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/widgets/progress_panel.py +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/tests/test_cli.py +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/tests/test_merging.py +0 -0
- {risforge-0.3.0 → risforge-0.3.2}/tests/test_pipeline.py +0 -0
|
@@ -1,13 +1,13 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: risforge
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.2
|
|
4
4
|
Summary: Clean, deduplicate, and enrich RIS bibliographic files for systematic reviews.
|
|
5
5
|
Author-email: Amyr <amyrhexa@gmail.com>
|
|
6
6
|
License: MIT
|
|
7
|
-
Project-URL: Homepage, https://github.com/
|
|
8
|
-
Project-URL: Repository, https://github.com/
|
|
9
|
-
Project-URL: Issues, https://github.com/
|
|
10
|
-
Project-URL: Changelog, https://github.com/
|
|
7
|
+
Project-URL: Homepage, https://github.com/pythyn/risforge
|
|
8
|
+
Project-URL: Repository, https://github.com/pythyn/risforge
|
|
9
|
+
Project-URL: Issues, https://github.com/pythyn/risforge/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/pythyn/risforge/blob/main/CHANGELOG.md
|
|
11
11
|
Keywords: ris,bibliography,systematic-review,deduplication,crossref,openalex,citation-management,prisma
|
|
12
12
|
Classifier: Development Status :: 4 - Beta
|
|
13
13
|
Classifier: Intended Audience :: Science/Research
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "risforge"
|
|
7
|
-
version = "0.3.
|
|
7
|
+
version = "0.3.2"
|
|
8
8
|
description = "Clean, deduplicate, and enrich RIS bibliographic files for systematic reviews."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -59,10 +59,10 @@ gui-dev = [
|
|
|
59
59
|
]
|
|
60
60
|
|
|
61
61
|
[project.urls]
|
|
62
|
-
Homepage = "https://github.com/
|
|
63
|
-
Repository = "https://github.com/
|
|
64
|
-
Issues = "https://github.com/
|
|
65
|
-
Changelog = "https://github.com/
|
|
62
|
+
Homepage = "https://github.com/pythyn/risforge"
|
|
63
|
+
Repository = "https://github.com/pythyn/risforge"
|
|
64
|
+
Issues = "https://github.com/pythyn/risforge/issues"
|
|
65
|
+
Changelog = "https://github.com/pythyn/risforge/blob/main/CHANGELOG.md"
|
|
66
66
|
|
|
67
67
|
[project.scripts]
|
|
68
68
|
risforge = "risforge.cli:main"
|
|
@@ -51,7 +51,7 @@ from risforge.exceptions import RisForgeError, RisParsingError
|
|
|
51
51
|
from risforge.merging import MergeResult, merge_ris_files
|
|
52
52
|
from risforge.pipeline import PipelineResult, risforge, run_pipeline
|
|
53
53
|
|
|
54
|
-
__version__ = "0.3.
|
|
54
|
+
__version__ = "0.3.2"
|
|
55
55
|
|
|
56
56
|
__all__ = [
|
|
57
57
|
"clean_ris_file",
|
|
@@ -24,6 +24,8 @@ from typing import Any
|
|
|
24
24
|
import rispy
|
|
25
25
|
import rispy.writer
|
|
26
26
|
|
|
27
|
+
from risforge.exceptions import RisParsingError
|
|
28
|
+
|
|
27
29
|
logger = logging.getLogger(__name__)
|
|
28
30
|
|
|
29
31
|
RisRecord = dict[str, Any]
|
|
@@ -223,13 +225,32 @@ def parse_ris_records(
|
|
|
223
225
|
|
|
224
226
|
Raises:
|
|
225
227
|
FileNotFoundError: If ``input_path`` does not exist.
|
|
228
|
+
RisParsingError: If the file's bytes can't be decoded as text
|
|
229
|
+
(for example, a non-UTF-8 file saved by an older Windows
|
|
230
|
+
reference manager). This does not cover malformed
|
|
231
|
+
*individual records* -- those are tolerated and reported
|
|
232
|
+
in ``errors`` instead.
|
|
226
233
|
"""
|
|
227
234
|
input_path = Path(input_path)
|
|
228
235
|
|
|
229
236
|
if not input_path.exists():
|
|
230
237
|
raise FileNotFoundError(f"Input file '{input_path}' not found.")
|
|
231
238
|
|
|
232
|
-
|
|
239
|
+
try:
|
|
240
|
+
# utf-8-sig: identical to plain utf-8 for files with no BOM,
|
|
241
|
+
# but also transparently strips a leading UTF-8 byte-order
|
|
242
|
+
# mark if one is present. Windows text editors and some
|
|
243
|
+
# reference managers commonly save UTF-8 files with a BOM;
|
|
244
|
+
# left in place, it silently prevents the very first "TY" tag
|
|
245
|
+
# in the file from being recognized at all (no crash, just an
|
|
246
|
+
# empty result), which is worse than being explicit about it.
|
|
247
|
+
text = input_path.read_text(encoding="utf-8-sig")
|
|
248
|
+
except UnicodeDecodeError as error:
|
|
249
|
+
raise RisParsingError(
|
|
250
|
+
f"Could not read '{input_path}' as UTF-8 text. The file may be "
|
|
251
|
+
"saved in a different encoding (common with exports from older "
|
|
252
|
+
"reference managers) -- try re-saving it as UTF-8."
|
|
253
|
+
) from error
|
|
233
254
|
|
|
234
255
|
blocks = re.split(r"(?m)^TY\s+-", text)
|
|
235
256
|
records: list[RisRecord] = []
|
|
@@ -32,6 +32,24 @@ file, and ``extract_doi()`` could never find a DOI that was already
|
|
|
32
32
|
present on the record either, since it looked for ``"DO"`` instead of
|
|
33
33
|
``"doi"``. :data:`RISPY_FIELD_MAP` below uses rispy's real field
|
|
34
34
|
names, and :meth:`RisEnricher.extract_doi` reads ``"doi"``/``"urls"``.
|
|
35
|
+
|
|
36
|
+
A second bug was found and fixed the same way: :meth:`RisEnricher.enrich_file`
|
|
37
|
+
used to call ``rispy.load()`` directly on the whole input file in one
|
|
38
|
+
pass. ``rispy`` (as of 0.10.0) has its own bug where it tracks the
|
|
39
|
+
"last tag seen" as parser-wide state that is never reset between
|
|
40
|
+
records -- so a single stray blank or otherwise non-tag-pattern line
|
|
41
|
+
positioned early in one record, right after a record boundary, can
|
|
42
|
+
make it try to extend a field from the *previous* record onto the new
|
|
43
|
+
record's (fresh, and therefore missing that key) dict, raising a
|
|
44
|
+
``KeyError`` for whatever field that happened to be and aborting the
|
|
45
|
+
entire file's enrichment. ``risforge.cleaning.parse_ris_records()``
|
|
46
|
+
already sidesteps this by parsing each record block independently (a
|
|
47
|
+
fresh parser instance per block, so there's no cross-record state to
|
|
48
|
+
leak) -- ``enrich_file()`` now reuses that same function instead of
|
|
49
|
+
calling ``rispy.load()`` itself, which both fixes the crash and means
|
|
50
|
+
a malformed record is tolerated and reported exactly the way
|
|
51
|
+
:func:`risforge.cleaning.clean_ris_file` already tolerates and reports
|
|
52
|
+
one, rather than each module handling malformed input differently.
|
|
35
53
|
"""
|
|
36
54
|
|
|
37
55
|
from __future__ import annotations
|
|
@@ -52,6 +70,8 @@ import rispy
|
|
|
52
70
|
from requests.adapters import HTTPAdapter
|
|
53
71
|
from urllib3.util.retry import Retry
|
|
54
72
|
|
|
73
|
+
from risforge.cleaning import parse_ris_records
|
|
74
|
+
|
|
55
75
|
TITLE_MATCH_THRESHOLD = 0.90
|
|
56
76
|
CACHE_EXPIRE_DAYS = 7
|
|
57
77
|
|
|
@@ -125,6 +145,7 @@ class RisEnricher:
|
|
|
125
145
|
"processed": 0,
|
|
126
146
|
"enriched": 0,
|
|
127
147
|
"failed": 0,
|
|
148
|
+
"skipped_malformed": 0,
|
|
128
149
|
"api_calls": {
|
|
129
150
|
"crossref": 0,
|
|
130
151
|
"openalex": 0,
|
|
@@ -407,12 +428,20 @@ class RisEnricher:
|
|
|
407
428
|
|
|
408
429
|
logger.info("Loading %s...", input_path)
|
|
409
430
|
try:
|
|
410
|
-
|
|
411
|
-
records = list(rispy.load(file))
|
|
431
|
+
records, parse_errors = parse_ris_records(input_path)
|
|
412
432
|
except (OSError, TypeError, ValueError) as error:
|
|
413
433
|
logger.error("Failed to parse RIS file: %s", error)
|
|
414
434
|
return self.stats
|
|
415
435
|
|
|
436
|
+
if parse_errors:
|
|
437
|
+
logger.warning(
|
|
438
|
+
"Encountered %d malformed record block(s) in %s, skipped: %s",
|
|
439
|
+
len(parse_errors),
|
|
440
|
+
input_path,
|
|
441
|
+
"; ".join(f"block {n}: {msg}" for n, msg in parse_errors),
|
|
442
|
+
)
|
|
443
|
+
self.stats["skipped_malformed"] = len(parse_errors)
|
|
444
|
+
|
|
416
445
|
total = len(records)
|
|
417
446
|
enriched_records = []
|
|
418
447
|
for index, record in enumerate(records):
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""Exception types for risforge.
|
|
2
|
+
|
|
3
|
+
We deliberately keep this hierarchy small. Most failure modes in this
|
|
4
|
+
package are already well described by the standard library's own
|
|
5
|
+
exceptions (``FileNotFoundError``, ``ValueError``, etc.), and the
|
|
6
|
+
original scripts caught those precisely rather than reaching for
|
|
7
|
+
generic ``Exception``. We keep that pattern. ``RisForgeError`` exists
|
|
8
|
+
only as a common base for the handful of errors that are specific to
|
|
9
|
+
this package's domain logic, so library users can catch
|
|
10
|
+
``RisForgeError`` if they want a single net for "something in risforge
|
|
11
|
+
itself went wrong" without having to also catch unrelated stdlib
|
|
12
|
+
errors they may want to handle differently.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class RisForgeError(Exception):
|
|
19
|
+
"""Base class for errors raised directly by risforge's own logic."""
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class RisParsingError(RisForgeError, ValueError):
|
|
23
|
+
"""Raised when a RIS file's content can't be decoded/read at all.
|
|
24
|
+
|
|
25
|
+
Deliberately inherits from ``ValueError`` too (like
|
|
26
|
+
``json.JSONDecodeError`` does in the standard library) so every
|
|
27
|
+
existing ``except (OSError, ValueError, RuntimeError)`` clause
|
|
28
|
+
throughout this codebase (CLI, pipeline, GUI worker) already
|
|
29
|
+
catches it correctly, with no call site changes required. Catch
|
|
30
|
+
``RisParsingError`` specifically, or ``RisForgeError`` generally,
|
|
31
|
+
for finer-grained handling.
|
|
32
|
+
|
|
33
|
+
This is for whole-file failures only (e.g. the file's bytes can't
|
|
34
|
+
be decoded as text) -- a malformed *individual record* within an
|
|
35
|
+
otherwise-readable file is not an error at this level; see
|
|
36
|
+
:func:`risforge.cleaning.parse_ris_records`, which tolerates and
|
|
37
|
+
reports those per-block instead of raising.
|
|
38
|
+
"""
|
|
@@ -1,13 +1,13 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: risforge
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.2
|
|
4
4
|
Summary: Clean, deduplicate, and enrich RIS bibliographic files for systematic reviews.
|
|
5
5
|
Author-email: Amyr <amyrhexa@gmail.com>
|
|
6
6
|
License: MIT
|
|
7
|
-
Project-URL: Homepage, https://github.com/
|
|
8
|
-
Project-URL: Repository, https://github.com/
|
|
9
|
-
Project-URL: Issues, https://github.com/
|
|
10
|
-
Project-URL: Changelog, https://github.com/
|
|
7
|
+
Project-URL: Homepage, https://github.com/pythyn/risforge
|
|
8
|
+
Project-URL: Repository, https://github.com/pythyn/risforge
|
|
9
|
+
Project-URL: Issues, https://github.com/pythyn/risforge/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/pythyn/risforge/blob/main/CHANGELOG.md
|
|
11
11
|
Keywords: ris,bibliography,systematic-review,deduplication,crossref,openalex,citation-management,prisma
|
|
12
12
|
Classifier: Development Status :: 4 - Beta
|
|
13
13
|
Classifier: Intended Audience :: Science/Research
|
|
@@ -353,6 +353,7 @@ class MainWindow(QMainWindow):
|
|
|
353
353
|
"input_records": stats.get("processed", 0), # type: ignore[union-attr]
|
|
354
354
|
"enriched_records": stats.get("enriched", 0), # type: ignore[union-attr]
|
|
355
355
|
"failed_enrichment": stats.get("failed", 0), # type: ignore[union-attr]
|
|
356
|
+
"skipped_malformed": stats.get("skipped_malformed", 0), # type: ignore[union-attr]
|
|
356
357
|
}
|
|
357
358
|
return summary, config.enriched_output_path, config.fail_report_path
|
|
358
359
|
|
|
@@ -363,6 +364,7 @@ class MainWindow(QMainWindow):
|
|
|
363
364
|
"unique_records": result.cleaned_record_count, # type: ignore[attr-defined]
|
|
364
365
|
"enriched_records": result.enrichment_stats.get("enriched", 0), # type: ignore[attr-defined]
|
|
365
366
|
"failed_enrichment": result.enrichment_stats.get("failed", 0), # type: ignore[attr-defined]
|
|
367
|
+
"skipped_malformed": result.enrichment_stats.get("skipped_malformed", 0), # type: ignore[attr-defined]
|
|
366
368
|
}
|
|
367
369
|
return summary, config.enriched_output_path, config.fail_report_path
|
|
368
370
|
|
|
@@ -23,6 +23,7 @@ _SUMMARY_LABELS = {
|
|
|
23
23
|
"unique_records": "Unique records",
|
|
24
24
|
"enriched_records": "Enriched records",
|
|
25
25
|
"failed_enrichment": "Unresolved records",
|
|
26
|
+
"skipped_malformed": "Skipped (unreadable records)",
|
|
26
27
|
}
|
|
27
28
|
|
|
28
29
|
|
|
@@ -47,9 +48,11 @@ class ResultsPanel(QWidget):
|
|
|
47
48
|
summary_box = QGroupBox("Summary")
|
|
48
49
|
self._summary_form = QFormLayout(summary_box)
|
|
49
50
|
self._value_labels: dict[str, QLabel] = {}
|
|
50
|
-
|
|
51
|
+
self._summary_rows: dict[str, int] = {}
|
|
52
|
+
for row, (key, label_text) in enumerate(_SUMMARY_LABELS.items()):
|
|
51
53
|
value_label = QLabel("\u2014")
|
|
52
54
|
self._value_labels[key] = value_label
|
|
55
|
+
self._summary_rows[key] = row
|
|
53
56
|
self._summary_form.addRow(QLabel(label_text), value_label)
|
|
54
57
|
layout.addWidget(summary_box)
|
|
55
58
|
|
|
@@ -95,6 +98,14 @@ class ResultsPanel(QWidget):
|
|
|
95
98
|
value = summary.get(key)
|
|
96
99
|
label.setText(f"{value:,}" if isinstance(value, int) else "\u2014")
|
|
97
100
|
|
|
101
|
+
# Keep the summary uncluttered for the common case: only show
|
|
102
|
+
# "Skipped (unreadable records)" when there's actually
|
|
103
|
+
# something to report, rather than a permanent "0" row.
|
|
104
|
+
skipped = summary.get("skipped_malformed")
|
|
105
|
+
row = self._summary_rows.get("skipped_malformed")
|
|
106
|
+
if row is not None:
|
|
107
|
+
self._summary_form.setRowVisible(row, bool(skipped))
|
|
108
|
+
|
|
98
109
|
self._output_dir = output_dir
|
|
99
110
|
self._final_ris_path = final_ris_path if final_ris_path and final_ris_path.exists() else None
|
|
100
111
|
self._fail_report_path = (
|
|
@@ -178,7 +178,11 @@ class PipelineWorker(QThread):
|
|
|
178
178
|
)
|
|
179
179
|
self.stage_changed.emit("enrich", "completed")
|
|
180
180
|
self.stats_changed.emit(
|
|
181
|
-
{
|
|
181
|
+
{
|
|
182
|
+
"enriched_records": stats.get("enriched", 0),
|
|
183
|
+
"failed_enrichment": stats.get("failed", 0),
|
|
184
|
+
"skipped_malformed": stats.get("skipped_malformed", 0),
|
|
185
|
+
}
|
|
182
186
|
)
|
|
183
187
|
self.finished_ok.emit(stats)
|
|
184
188
|
|
|
@@ -202,6 +206,7 @@ class PipelineWorker(QThread):
|
|
|
202
206
|
"unique_records": result.cleaned_record_count,
|
|
203
207
|
"enriched_records": result.enrichment_stats.get("enriched", 0),
|
|
204
208
|
"failed_enrichment": result.enrichment_stats.get("failed", 0),
|
|
209
|
+
"skipped_malformed": result.enrichment_stats.get("skipped_malformed", 0),
|
|
205
210
|
}
|
|
206
211
|
)
|
|
207
212
|
self.finished_ok.emit(result)
|
|
@@ -9,7 +9,9 @@ from risforge.cleaning import (
|
|
|
9
9
|
merge_cluster,
|
|
10
10
|
normalize_doi,
|
|
11
11
|
normalize_title,
|
|
12
|
+
parse_ris_records,
|
|
12
13
|
)
|
|
14
|
+
from risforge.exceptions import RisParsingError
|
|
13
15
|
|
|
14
16
|
|
|
15
17
|
class TestNormalizeTitle:
|
|
@@ -104,3 +106,77 @@ class TestCleanRisFile:
|
|
|
104
106
|
def test_missing_input_raises(self, tmp_path) -> None:
|
|
105
107
|
with pytest.raises(FileNotFoundError):
|
|
106
108
|
clean_ris_file(tmp_path / "does_not_exist.ris", tmp_path / "out.ris")
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
class TestParseRisRecordsEncoding:
|
|
112
|
+
"""Regression tests for encoding-related parsing robustness.
|
|
113
|
+
|
|
114
|
+
These are not Windows-specific fixes (see the equivalent tests in
|
|
115
|
+
test_enrichment.py for the actual crash this was found alongside),
|
|
116
|
+
but a UTF-8 BOM in particular is disproportionately common in
|
|
117
|
+
files saved by Windows text editors and some reference managers,
|
|
118
|
+
so it's worth covering explicitly here too.
|
|
119
|
+
"""
|
|
120
|
+
|
|
121
|
+
def test_utf8_bom_is_stripped_transparently(self, tmp_path) -> None:
|
|
122
|
+
ris_text = "TY - JOUR\nAU - Smith, John\nTI - BOM paper\nER - \n"
|
|
123
|
+
path = tmp_path / "bom.ris"
|
|
124
|
+
path.write_bytes(b"\xef\xbb\xbf" + ris_text.encode("utf-8"))
|
|
125
|
+
|
|
126
|
+
records, errors = parse_ris_records(path)
|
|
127
|
+
|
|
128
|
+
assert len(records) == 1
|
|
129
|
+
assert records[0]["title"] == "BOM paper"
|
|
130
|
+
assert errors == []
|
|
131
|
+
|
|
132
|
+
def test_non_utf8_bytes_raise_a_clear_parsing_error(self, tmp_path) -> None:
|
|
133
|
+
path = tmp_path / "bad_encoding.ris"
|
|
134
|
+
path.write_bytes(b"TY - JOUR\nTI - Bad \xff\xfe byte sequence\nER - \n")
|
|
135
|
+
|
|
136
|
+
with pytest.raises(RisParsingError, match="bad_encoding.ris"):
|
|
137
|
+
parse_ris_records(path)
|
|
138
|
+
|
|
139
|
+
def test_parsing_error_is_also_a_value_error(self, tmp_path) -> None:
|
|
140
|
+
# RisParsingError intentionally also subclasses ValueError so
|
|
141
|
+
# every existing `except (..., ValueError, ...)` call site
|
|
142
|
+
# (CLI, pipeline, GUI worker) already catches it with no
|
|
143
|
+
# changes required at those call sites.
|
|
144
|
+
path = tmp_path / "bad_encoding.ris"
|
|
145
|
+
path.write_bytes(b"\xff\xfe\x00\x01")
|
|
146
|
+
|
|
147
|
+
with pytest.raises(ValueError):
|
|
148
|
+
parse_ris_records(path)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
class TestParseRisRecordsCrossRecordStateBug:
|
|
152
|
+
"""Regression test for the rispy 0.10.0 cross-record state bug.
|
|
153
|
+
|
|
154
|
+
See risforge.enrichment's module docstring and
|
|
155
|
+
tests/test_enrichment.py::TestEnrichFileMalformedInput for the
|
|
156
|
+
full explanation. clean_ris_file() was never actually vulnerable
|
|
157
|
+
to this (each record block is parsed independently), but this
|
|
158
|
+
test pins that guarantee down explicitly so a future refactor
|
|
159
|
+
can't accidentally reintroduce the whole-file single-parse
|
|
160
|
+
pattern that enrich_file() used to use.
|
|
161
|
+
"""
|
|
162
|
+
|
|
163
|
+
def test_stray_blank_line_after_record_boundary_does_not_crash(self, tmp_path) -> None:
|
|
164
|
+
ris_text = (
|
|
165
|
+
"TY - JOUR\n"
|
|
166
|
+
"AU - Smith, John\n"
|
|
167
|
+
"TI - First paper\n"
|
|
168
|
+
"LA - English\n"
|
|
169
|
+
"ER - \n"
|
|
170
|
+
"\n"
|
|
171
|
+
"TY - JOUR\n"
|
|
172
|
+
"\n"
|
|
173
|
+
"AU - Doe, Jane\n"
|
|
174
|
+
"TI - Second paper\n"
|
|
175
|
+
"ER - \n"
|
|
176
|
+
)
|
|
177
|
+
path = tmp_path / "malformed.ris"
|
|
178
|
+
path.write_text(ris_text, encoding="utf-8")
|
|
179
|
+
|
|
180
|
+
records, errors = parse_ris_records(path) # must not raise KeyError
|
|
181
|
+
|
|
182
|
+
assert len(records) == 2
|
|
@@ -152,3 +152,93 @@ class TestEnrichFile:
|
|
|
152
152
|
assert output_path.exists()
|
|
153
153
|
assert stats["processed"] == 1
|
|
154
154
|
assert stats["enriched"] == 1
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
class TestEnrichFileMalformedInput:
|
|
158
|
+
"""Regression tests for a rispy 0.10.0 bug: RisParser tracks the
|
|
159
|
+
|
|
160
|
+
"last tag seen" as state that persists across record boundaries
|
|
161
|
+
within a single ``rispy.load()``/``rispy.loads()`` call. A stray
|
|
162
|
+
blank (or otherwise non-tag-pattern) line positioned early in one
|
|
163
|
+
record -- right after a record boundary -- could make it try to
|
|
164
|
+
extend a field from the *previous* record onto the new record's
|
|
165
|
+
dict, which doesn't have that key yet, raising KeyError and
|
|
166
|
+
aborting the whole file. Reproduces identically on every platform
|
|
167
|
+
(verified on Linux); it is not a Windows-specific issue, just a
|
|
168
|
+
RIS-content issue that happened to be triggered first by files
|
|
169
|
+
tested on Windows. enrich_file() now parses via
|
|
170
|
+
risforge.cleaning.parse_ris_records() (block-isolated, immune to
|
|
171
|
+
this) instead of calling rispy.load() directly on the whole file.
|
|
172
|
+
"""
|
|
173
|
+
|
|
174
|
+
def test_stray_blank_line_after_record_boundary_does_not_crash(self, tmp_path) -> None:
|
|
175
|
+
# Record 1 ends with an "LA" (language) tag as its last real
|
|
176
|
+
# field before ER. Record 2 has a stray blank line immediately
|
|
177
|
+
# after "TY", before its own first tag -- this is exactly the
|
|
178
|
+
# combination that crashed rispy.load() with KeyError('language').
|
|
179
|
+
ris_text = (
|
|
180
|
+
"TY - JOUR\n"
|
|
181
|
+
"AU - Smith, John\n"
|
|
182
|
+
"TI - First paper\n"
|
|
183
|
+
"LA - English\n"
|
|
184
|
+
"DO - 10.1000/first\n"
|
|
185
|
+
"ER - \n"
|
|
186
|
+
"\n"
|
|
187
|
+
"TY - JOUR\n"
|
|
188
|
+
"\n"
|
|
189
|
+
"AU - Doe, Jane\n"
|
|
190
|
+
"TI - Second paper\n"
|
|
191
|
+
"DO - 10.1000/second\n"
|
|
192
|
+
"ER - \n"
|
|
193
|
+
)
|
|
194
|
+
input_path = tmp_path / "malformed.ris"
|
|
195
|
+
input_path.write_text(ris_text, encoding="utf-8")
|
|
196
|
+
|
|
197
|
+
enricher, _session = _make_enricher()
|
|
198
|
+
# Must not raise KeyError.
|
|
199
|
+
stats = enricher.enrich_file(str(input_path), str(tmp_path / "out.ris"))
|
|
200
|
+
|
|
201
|
+
assert stats["processed"] == 2
|
|
202
|
+
|
|
203
|
+
def test_genuinely_unparseable_block_is_skipped_and_counted(self, tmp_path) -> None:
|
|
204
|
+
ris_text = (
|
|
205
|
+
"TY - JOUR\n"
|
|
206
|
+
"AU - Smith, John\n"
|
|
207
|
+
"TI - Valid paper\n"
|
|
208
|
+
"DO - 10.1000/valid\n"
|
|
209
|
+
"ER - \n"
|
|
210
|
+
"\n"
|
|
211
|
+
"TY - JOUR\n"
|
|
212
|
+
"AU - Broken, Record\n"
|
|
213
|
+
"TI - Missing ER terminator entirely\n"
|
|
214
|
+
)
|
|
215
|
+
input_path = tmp_path / "genuinely_malformed.ris"
|
|
216
|
+
input_path.write_text(ris_text, encoding="utf-8")
|
|
217
|
+
|
|
218
|
+
enricher, _session = _make_enricher()
|
|
219
|
+
stats = enricher.enrich_file(str(input_path), str(tmp_path / "out.ris"))
|
|
220
|
+
|
|
221
|
+
assert stats["processed"] == 1
|
|
222
|
+
assert stats["skipped_malformed"] == 1
|
|
223
|
+
|
|
224
|
+
def test_utf8_bom_is_handled_transparently(self, tmp_path) -> None:
|
|
225
|
+
ris_text = "TY - JOUR\nAU - Smith, John\nTI - BOM paper\nDO - 10.1000/bom\nER - \n"
|
|
226
|
+
input_path = tmp_path / "bom.ris"
|
|
227
|
+
input_path.write_bytes(b"\xef\xbb\xbf" + ris_text.encode("utf-8"))
|
|
228
|
+
|
|
229
|
+
enricher, _session = _make_enricher()
|
|
230
|
+
stats = enricher.enrich_file(str(input_path), str(tmp_path / "out.ris"))
|
|
231
|
+
|
|
232
|
+
assert stats["processed"] == 1
|
|
233
|
+
|
|
234
|
+
def test_non_utf8_file_gives_a_clear_error_without_crashing(self, tmp_path) -> None:
|
|
235
|
+
input_path = tmp_path / "bad_encoding.ris"
|
|
236
|
+
input_path.write_bytes(b"TY - JOUR\nTI - Bad \xff\xfe byte sequence\nER - \n")
|
|
237
|
+
|
|
238
|
+
enricher, _session = _make_enricher()
|
|
239
|
+
# enrich_file()'s existing contract: parsing failures are
|
|
240
|
+
# logged and it returns the (unmodified) stats rather than
|
|
241
|
+
# raising, exactly as it already did for a missing input file.
|
|
242
|
+
stats = enricher.enrich_file(str(input_path), str(tmp_path / "out.ris"))
|
|
243
|
+
|
|
244
|
+
assert stats["processed"] == 0
|
|
@@ -1,23 +0,0 @@
|
|
|
1
|
-
"""Exception types for risforge.
|
|
2
|
-
|
|
3
|
-
We deliberately keep this hierarchy small. Most failure modes in this
|
|
4
|
-
package are already well described by the standard library's own
|
|
5
|
-
exceptions (``FileNotFoundError``, ``ValueError``, etc.), and the
|
|
6
|
-
original scripts caught those precisely rather than reaching for
|
|
7
|
-
generic ``Exception``. We keep that pattern. ``RisForgeError`` exists
|
|
8
|
-
only as a common base for the handful of errors that are specific to
|
|
9
|
-
this package's domain logic, so library users can catch
|
|
10
|
-
``RisForgeError`` if they want a single net for "something in risforge
|
|
11
|
-
itself went wrong" without having to also catch unrelated stdlib
|
|
12
|
-
errors they may want to handle differently.
|
|
13
|
-
"""
|
|
14
|
-
|
|
15
|
-
from __future__ import annotations
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
class RisForgeError(Exception):
|
|
19
|
-
"""Base class for errors raised directly by risforge's own logic."""
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
class RisParsingError(RisForgeError):
|
|
23
|
-
"""Raised when a RIS file cannot be parsed into any usable records."""
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|