risforge 0.3.0__tar.gz → 0.3.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. {risforge-0.3.0/src/risforge.egg-info → risforge-0.3.2}/PKG-INFO +5 -5
  2. {risforge-0.3.0 → risforge-0.3.2}/pyproject.toml +5 -5
  3. {risforge-0.3.0 → risforge-0.3.2}/src/risforge/__init__.py +1 -1
  4. {risforge-0.3.0 → risforge-0.3.2}/src/risforge/cleaning.py +22 -1
  5. {risforge-0.3.0 → risforge-0.3.2}/src/risforge/enrichment.py +31 -2
  6. risforge-0.3.2/src/risforge/exceptions.py +38 -0
  7. {risforge-0.3.0 → risforge-0.3.2/src/risforge.egg-info}/PKG-INFO +5 -5
  8. {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/main_window.py +2 -0
  9. {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/widgets/results_panel.py +12 -1
  10. {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/worker.py +6 -1
  11. {risforge-0.3.0 → risforge-0.3.2}/tests/test_cleaning.py +76 -0
  12. {risforge-0.3.0 → risforge-0.3.2}/tests/test_enrichment.py +90 -0
  13. risforge-0.3.0/src/risforge/exceptions.py +0 -23
  14. {risforge-0.3.0 → risforge-0.3.2}/LICENSE +0 -0
  15. {risforge-0.3.0 → risforge-0.3.2}/README.md +0 -0
  16. {risforge-0.3.0 → risforge-0.3.2}/setup.cfg +0 -0
  17. {risforge-0.3.0 → risforge-0.3.2}/src/risforge/cli.py +0 -0
  18. {risforge-0.3.0 → risforge-0.3.2}/src/risforge/merging.py +0 -0
  19. {risforge-0.3.0 → risforge-0.3.2}/src/risforge/pipeline.py +0 -0
  20. {risforge-0.3.0 → risforge-0.3.2}/src/risforge.egg-info/SOURCES.txt +0 -0
  21. {risforge-0.3.0 → risforge-0.3.2}/src/risforge.egg-info/dependency_links.txt +0 -0
  22. {risforge-0.3.0 → risforge-0.3.2}/src/risforge.egg-info/entry_points.txt +0 -0
  23. {risforge-0.3.0 → risforge-0.3.2}/src/risforge.egg-info/requires.txt +0 -0
  24. {risforge-0.3.0 → risforge-0.3.2}/src/risforge.egg-info/top_level.txt +0 -0
  25. {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/__init__.py +0 -0
  26. {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/app.py +0 -0
  27. {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/dialogs.py +0 -0
  28. {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/file_counter.py +0 -0
  29. {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/models.py +0 -0
  30. {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/theme.py +0 -0
  31. {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/widgets/__init__.py +0 -0
  32. {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/widgets/config_panel.py +0 -0
  33. {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/widgets/input_panel.py +0 -0
  34. {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/widgets/log_panel.py +0 -0
  35. {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/widgets/output_panel.py +0 -0
  36. {risforge-0.3.0 → risforge-0.3.2}/src/risforge_gui/widgets/progress_panel.py +0 -0
  37. {risforge-0.3.0 → risforge-0.3.2}/tests/test_cli.py +0 -0
  38. {risforge-0.3.0 → risforge-0.3.2}/tests/test_merging.py +0 -0
  39. {risforge-0.3.0 → risforge-0.3.2}/tests/test_pipeline.py +0 -0
@@ -1,13 +1,13 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: risforge
3
- Version: 0.3.0
3
+ Version: 0.3.2
4
4
  Summary: Clean, deduplicate, and enrich RIS bibliographic files for systematic reviews.
5
5
  Author-email: Amyr <amyrhexa@gmail.com>
6
6
  License: MIT
7
- Project-URL: Homepage, https://github.com/amyr/risforge
8
- Project-URL: Repository, https://github.com/amyr/risforge
9
- Project-URL: Issues, https://github.com/amyr/risforge/issues
10
- Project-URL: Changelog, https://github.com/amyr/risforge/blob/main/CHANGELOG.md
7
+ Project-URL: Homepage, https://github.com/pythyn/risforge
8
+ Project-URL: Repository, https://github.com/pythyn/risforge
9
+ Project-URL: Issues, https://github.com/pythyn/risforge/issues
10
+ Project-URL: Changelog, https://github.com/pythyn/risforge/blob/main/CHANGELOG.md
11
11
  Keywords: ris,bibliography,systematic-review,deduplication,crossref,openalex,citation-management,prisma
12
12
  Classifier: Development Status :: 4 - Beta
13
13
  Classifier: Intended Audience :: Science/Research
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "risforge"
7
- version = "0.3.0"
7
+ version = "0.3.2"
8
8
  description = "Clean, deduplicate, and enrich RIS bibliographic files for systematic reviews."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -59,10 +59,10 @@ gui-dev = [
59
59
  ]
60
60
 
61
61
  [project.urls]
62
- Homepage = "https://github.com/amyr/risforge"
63
- Repository = "https://github.com/amyr/risforge"
64
- Issues = "https://github.com/amyr/risforge/issues"
65
- Changelog = "https://github.com/amyr/risforge/blob/main/CHANGELOG.md"
62
+ Homepage = "https://github.com/pythyn/risforge"
63
+ Repository = "https://github.com/pythyn/risforge"
64
+ Issues = "https://github.com/pythyn/risforge/issues"
65
+ Changelog = "https://github.com/pythyn/risforge/blob/main/CHANGELOG.md"
66
66
 
67
67
  [project.scripts]
68
68
  risforge = "risforge.cli:main"
@@ -51,7 +51,7 @@ from risforge.exceptions import RisForgeError, RisParsingError
51
51
  from risforge.merging import MergeResult, merge_ris_files
52
52
  from risforge.pipeline import PipelineResult, risforge, run_pipeline
53
53
 
54
- __version__ = "0.3.0"
54
+ __version__ = "0.3.2"
55
55
 
56
56
  __all__ = [
57
57
  "clean_ris_file",
@@ -24,6 +24,8 @@ from typing import Any
24
24
  import rispy
25
25
  import rispy.writer
26
26
 
27
+ from risforge.exceptions import RisParsingError
28
+
27
29
  logger = logging.getLogger(__name__)
28
30
 
29
31
  RisRecord = dict[str, Any]
@@ -223,13 +225,32 @@ def parse_ris_records(
223
225
 
224
226
  Raises:
225
227
  FileNotFoundError: If ``input_path`` does not exist.
228
+ RisParsingError: If the file's bytes can't be decoded as text
229
+ (for example, a non-UTF-8 file saved by an older Windows
230
+ reference manager). This does not cover malformed
231
+ *individual records* -- those are tolerated and reported
232
+ in ``errors`` instead.
226
233
  """
227
234
  input_path = Path(input_path)
228
235
 
229
236
  if not input_path.exists():
230
237
  raise FileNotFoundError(f"Input file '{input_path}' not found.")
231
238
 
232
- text = input_path.read_text(encoding="utf-8")
239
+ try:
240
+ # utf-8-sig: identical to plain utf-8 for files with no BOM,
241
+ # but also transparently strips a leading UTF-8 byte-order
242
+ # mark if one is present. Windows text editors and some
243
+ # reference managers commonly save UTF-8 files with a BOM;
244
+ # left in place, it silently prevents the very first "TY" tag
245
+ # in the file from being recognized at all (no crash, just an
246
+ # empty result), which is worse than being explicit about it.
247
+ text = input_path.read_text(encoding="utf-8-sig")
248
+ except UnicodeDecodeError as error:
249
+ raise RisParsingError(
250
+ f"Could not read '{input_path}' as UTF-8 text. The file may be "
251
+ "saved in a different encoding (common with exports from older "
252
+ "reference managers) -- try re-saving it as UTF-8."
253
+ ) from error
233
254
 
234
255
  blocks = re.split(r"(?m)^TY\s+-", text)
235
256
  records: list[RisRecord] = []
@@ -32,6 +32,24 @@ file, and ``extract_doi()`` could never find a DOI that was already
32
32
  present on the record either, since it looked for ``"DO"`` instead of
33
33
  ``"doi"``. :data:`RISPY_FIELD_MAP` below uses rispy's real field
34
34
  names, and :meth:`RisEnricher.extract_doi` reads ``"doi"``/``"urls"``.
35
+
36
+ A second bug was found and fixed the same way: :meth:`RisEnricher.enrich_file`
37
+ used to call ``rispy.load()`` directly on the whole input file in one
38
+ pass. ``rispy`` (as of 0.10.0) has its own bug where it tracks the
39
+ "last tag seen" as parser-wide state that is never reset between
40
+ records -- so a single stray blank or otherwise non-tag-pattern line
41
+ positioned early in one record, right after a record boundary, can
42
+ make it try to extend a field from the *previous* record onto the new
43
+ record's (fresh, and therefore missing that key) dict, raising a
44
+ ``KeyError`` for whatever field that happened to be and aborting the
45
+ entire file's enrichment. ``risforge.cleaning.parse_ris_records()``
46
+ already sidesteps this by parsing each record block independently (a
47
+ fresh parser instance per block, so there's no cross-record state to
48
+ leak) -- ``enrich_file()`` now reuses that same function instead of
49
+ calling ``rispy.load()`` itself, which both fixes the crash and means
50
+ a malformed record is tolerated and reported exactly the way
51
+ :func:`risforge.cleaning.clean_ris_file` already tolerates and reports
52
+ one, rather than each module handling malformed input differently.
35
53
  """
36
54
 
37
55
  from __future__ import annotations
@@ -52,6 +70,8 @@ import rispy
52
70
  from requests.adapters import HTTPAdapter
53
71
  from urllib3.util.retry import Retry
54
72
 
73
+ from risforge.cleaning import parse_ris_records
74
+
55
75
  TITLE_MATCH_THRESHOLD = 0.90
56
76
  CACHE_EXPIRE_DAYS = 7
57
77
 
@@ -125,6 +145,7 @@ class RisEnricher:
125
145
  "processed": 0,
126
146
  "enriched": 0,
127
147
  "failed": 0,
148
+ "skipped_malformed": 0,
128
149
  "api_calls": {
129
150
  "crossref": 0,
130
151
  "openalex": 0,
@@ -407,12 +428,20 @@ class RisEnricher:
407
428
 
408
429
  logger.info("Loading %s...", input_path)
409
430
  try:
410
- with input_path.open("r", encoding="utf-8") as file:
411
- records = list(rispy.load(file))
431
+ records, parse_errors = parse_ris_records(input_path)
412
432
  except (OSError, TypeError, ValueError) as error:
413
433
  logger.error("Failed to parse RIS file: %s", error)
414
434
  return self.stats
415
435
 
436
+ if parse_errors:
437
+ logger.warning(
438
+ "Encountered %d malformed record block(s) in %s, skipped: %s",
439
+ len(parse_errors),
440
+ input_path,
441
+ "; ".join(f"block {n}: {msg}" for n, msg in parse_errors),
442
+ )
443
+ self.stats["skipped_malformed"] = len(parse_errors)
444
+
416
445
  total = len(records)
417
446
  enriched_records = []
418
447
  for index, record in enumerate(records):
@@ -0,0 +1,38 @@
1
+ """Exception types for risforge.
2
+
3
+ We deliberately keep this hierarchy small. Most failure modes in this
4
+ package are already well described by the standard library's own
5
+ exceptions (``FileNotFoundError``, ``ValueError``, etc.), and the
6
+ original scripts caught those precisely rather than reaching for
7
+ generic ``Exception``. We keep that pattern. ``RisForgeError`` exists
8
+ only as a common base for the handful of errors that are specific to
9
+ this package's domain logic, so library users can catch
10
+ ``RisForgeError`` if they want a single net for "something in risforge
11
+ itself went wrong" without having to also catch unrelated stdlib
12
+ errors they may want to handle differently.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+
18
+ class RisForgeError(Exception):
19
+ """Base class for errors raised directly by risforge's own logic."""
20
+
21
+
22
+ class RisParsingError(RisForgeError, ValueError):
23
+ """Raised when a RIS file's content can't be decoded/read at all.
24
+
25
+ Deliberately inherits from ``ValueError`` too (like
26
+ ``json.JSONDecodeError`` does in the standard library) so every
27
+ existing ``except (OSError, ValueError, RuntimeError)`` clause
28
+ throughout this codebase (CLI, pipeline, GUI worker) already
29
+ catches it correctly, with no call site changes required. Catch
30
+ ``RisParsingError`` specifically, or ``RisForgeError`` generally,
31
+ for finer-grained handling.
32
+
33
+ This is for whole-file failures only (e.g. the file's bytes can't
34
+ be decoded as text) -- a malformed *individual record* within an
35
+ otherwise-readable file is not an error at this level; see
36
+ :func:`risforge.cleaning.parse_ris_records`, which tolerates and
37
+ reports those per-block instead of raising.
38
+ """
@@ -1,13 +1,13 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: risforge
3
- Version: 0.3.0
3
+ Version: 0.3.2
4
4
  Summary: Clean, deduplicate, and enrich RIS bibliographic files for systematic reviews.
5
5
  Author-email: Amyr <amyrhexa@gmail.com>
6
6
  License: MIT
7
- Project-URL: Homepage, https://github.com/amyr/risforge
8
- Project-URL: Repository, https://github.com/amyr/risforge
9
- Project-URL: Issues, https://github.com/amyr/risforge/issues
10
- Project-URL: Changelog, https://github.com/amyr/risforge/blob/main/CHANGELOG.md
7
+ Project-URL: Homepage, https://github.com/pythyn/risforge
8
+ Project-URL: Repository, https://github.com/pythyn/risforge
9
+ Project-URL: Issues, https://github.com/pythyn/risforge/issues
10
+ Project-URL: Changelog, https://github.com/pythyn/risforge/blob/main/CHANGELOG.md
11
11
  Keywords: ris,bibliography,systematic-review,deduplication,crossref,openalex,citation-management,prisma
12
12
  Classifier: Development Status :: 4 - Beta
13
13
  Classifier: Intended Audience :: Science/Research
@@ -353,6 +353,7 @@ class MainWindow(QMainWindow):
353
353
  "input_records": stats.get("processed", 0), # type: ignore[union-attr]
354
354
  "enriched_records": stats.get("enriched", 0), # type: ignore[union-attr]
355
355
  "failed_enrichment": stats.get("failed", 0), # type: ignore[union-attr]
356
+ "skipped_malformed": stats.get("skipped_malformed", 0), # type: ignore[union-attr]
356
357
  }
357
358
  return summary, config.enriched_output_path, config.fail_report_path
358
359
 
@@ -363,6 +364,7 @@ class MainWindow(QMainWindow):
363
364
  "unique_records": result.cleaned_record_count, # type: ignore[attr-defined]
364
365
  "enriched_records": result.enrichment_stats.get("enriched", 0), # type: ignore[attr-defined]
365
366
  "failed_enrichment": result.enrichment_stats.get("failed", 0), # type: ignore[attr-defined]
367
+ "skipped_malformed": result.enrichment_stats.get("skipped_malformed", 0), # type: ignore[attr-defined]
366
368
  }
367
369
  return summary, config.enriched_output_path, config.fail_report_path
368
370
 
@@ -23,6 +23,7 @@ _SUMMARY_LABELS = {
23
23
  "unique_records": "Unique records",
24
24
  "enriched_records": "Enriched records",
25
25
  "failed_enrichment": "Unresolved records",
26
+ "skipped_malformed": "Skipped (unreadable records)",
26
27
  }
27
28
 
28
29
 
@@ -47,9 +48,11 @@ class ResultsPanel(QWidget):
47
48
  summary_box = QGroupBox("Summary")
48
49
  self._summary_form = QFormLayout(summary_box)
49
50
  self._value_labels: dict[str, QLabel] = {}
50
- for key, label_text in _SUMMARY_LABELS.items():
51
+ self._summary_rows: dict[str, int] = {}
52
+ for row, (key, label_text) in enumerate(_SUMMARY_LABELS.items()):
51
53
  value_label = QLabel("\u2014")
52
54
  self._value_labels[key] = value_label
55
+ self._summary_rows[key] = row
53
56
  self._summary_form.addRow(QLabel(label_text), value_label)
54
57
  layout.addWidget(summary_box)
55
58
 
@@ -95,6 +98,14 @@ class ResultsPanel(QWidget):
95
98
  value = summary.get(key)
96
99
  label.setText(f"{value:,}" if isinstance(value, int) else "\u2014")
97
100
 
101
+ # Keep the summary uncluttered for the common case: only show
102
+ # "Skipped (unreadable records)" when there's actually
103
+ # something to report, rather than a permanent "0" row.
104
+ skipped = summary.get("skipped_malformed")
105
+ row = self._summary_rows.get("skipped_malformed")
106
+ if row is not None:
107
+ self._summary_form.setRowVisible(row, bool(skipped))
108
+
98
109
  self._output_dir = output_dir
99
110
  self._final_ris_path = final_ris_path if final_ris_path and final_ris_path.exists() else None
100
111
  self._fail_report_path = (
@@ -178,7 +178,11 @@ class PipelineWorker(QThread):
178
178
  )
179
179
  self.stage_changed.emit("enrich", "completed")
180
180
  self.stats_changed.emit(
181
- {"enriched_records": stats.get("enriched", 0), "failed_enrichment": stats.get("failed", 0)}
181
+ {
182
+ "enriched_records": stats.get("enriched", 0),
183
+ "failed_enrichment": stats.get("failed", 0),
184
+ "skipped_malformed": stats.get("skipped_malformed", 0),
185
+ }
182
186
  )
183
187
  self.finished_ok.emit(stats)
184
188
 
@@ -202,6 +206,7 @@ class PipelineWorker(QThread):
202
206
  "unique_records": result.cleaned_record_count,
203
207
  "enriched_records": result.enrichment_stats.get("enriched", 0),
204
208
  "failed_enrichment": result.enrichment_stats.get("failed", 0),
209
+ "skipped_malformed": result.enrichment_stats.get("skipped_malformed", 0),
205
210
  }
206
211
  )
207
212
  self.finished_ok.emit(result)
@@ -9,7 +9,9 @@ from risforge.cleaning import (
9
9
  merge_cluster,
10
10
  normalize_doi,
11
11
  normalize_title,
12
+ parse_ris_records,
12
13
  )
14
+ from risforge.exceptions import RisParsingError
13
15
 
14
16
 
15
17
  class TestNormalizeTitle:
@@ -104,3 +106,77 @@ class TestCleanRisFile:
104
106
  def test_missing_input_raises(self, tmp_path) -> None:
105
107
  with pytest.raises(FileNotFoundError):
106
108
  clean_ris_file(tmp_path / "does_not_exist.ris", tmp_path / "out.ris")
109
+
110
+
111
+ class TestParseRisRecordsEncoding:
112
+ """Regression tests for encoding-related parsing robustness.
113
+
114
+ These are not Windows-specific fixes (see the equivalent tests in
115
+ test_enrichment.py for the actual crash this was found alongside),
116
+ but a UTF-8 BOM in particular is disproportionately common in
117
+ files saved by Windows text editors and some reference managers,
118
+ so it's worth covering explicitly here too.
119
+ """
120
+
121
+ def test_utf8_bom_is_stripped_transparently(self, tmp_path) -> None:
122
+ ris_text = "TY - JOUR\nAU - Smith, John\nTI - BOM paper\nER - \n"
123
+ path = tmp_path / "bom.ris"
124
+ path.write_bytes(b"\xef\xbb\xbf" + ris_text.encode("utf-8"))
125
+
126
+ records, errors = parse_ris_records(path)
127
+
128
+ assert len(records) == 1
129
+ assert records[0]["title"] == "BOM paper"
130
+ assert errors == []
131
+
132
+ def test_non_utf8_bytes_raise_a_clear_parsing_error(self, tmp_path) -> None:
133
+ path = tmp_path / "bad_encoding.ris"
134
+ path.write_bytes(b"TY - JOUR\nTI - Bad \xff\xfe byte sequence\nER - \n")
135
+
136
+ with pytest.raises(RisParsingError, match="bad_encoding.ris"):
137
+ parse_ris_records(path)
138
+
139
+ def test_parsing_error_is_also_a_value_error(self, tmp_path) -> None:
140
+ # RisParsingError intentionally also subclasses ValueError so
141
+ # every existing `except (..., ValueError, ...)` call site
142
+ # (CLI, pipeline, GUI worker) already catches it with no
143
+ # changes required at those call sites.
144
+ path = tmp_path / "bad_encoding.ris"
145
+ path.write_bytes(b"\xff\xfe\x00\x01")
146
+
147
+ with pytest.raises(ValueError):
148
+ parse_ris_records(path)
149
+
150
+
151
+ class TestParseRisRecordsCrossRecordStateBug:
152
+ """Regression test for the rispy 0.10.0 cross-record state bug.
153
+
154
+ See risforge.enrichment's module docstring and
155
+ tests/test_enrichment.py::TestEnrichFileMalformedInput for the
156
+ full explanation. clean_ris_file() was never actually vulnerable
157
+ to this (each record block is parsed independently), but this
158
+ test pins that guarantee down explicitly so a future refactor
159
+ can't accidentally reintroduce the whole-file single-parse
160
+ pattern that enrich_file() used to use.
161
+ """
162
+
163
+ def test_stray_blank_line_after_record_boundary_does_not_crash(self, tmp_path) -> None:
164
+ ris_text = (
165
+ "TY - JOUR\n"
166
+ "AU - Smith, John\n"
167
+ "TI - First paper\n"
168
+ "LA - English\n"
169
+ "ER - \n"
170
+ "\n"
171
+ "TY - JOUR\n"
172
+ "\n"
173
+ "AU - Doe, Jane\n"
174
+ "TI - Second paper\n"
175
+ "ER - \n"
176
+ )
177
+ path = tmp_path / "malformed.ris"
178
+ path.write_text(ris_text, encoding="utf-8")
179
+
180
+ records, errors = parse_ris_records(path) # must not raise KeyError
181
+
182
+ assert len(records) == 2
@@ -152,3 +152,93 @@ class TestEnrichFile:
152
152
  assert output_path.exists()
153
153
  assert stats["processed"] == 1
154
154
  assert stats["enriched"] == 1
155
+
156
+
157
+ class TestEnrichFileMalformedInput:
158
+ """Regression tests for a rispy 0.10.0 bug: RisParser tracks the
159
+
160
+ "last tag seen" as state that persists across record boundaries
161
+ within a single ``rispy.load()``/``rispy.loads()`` call. A stray
162
+ blank (or otherwise non-tag-pattern) line positioned early in one
163
+ record -- right after a record boundary -- could make it try to
164
+ extend a field from the *previous* record onto the new record's
165
+ dict, which doesn't have that key yet, raising KeyError and
166
+ aborting the whole file. Reproduces identically on every platform
167
+ (verified on Linux); it is not a Windows-specific issue, just a
168
+ RIS-content issue that happened to be triggered first by files
169
+ tested on Windows. enrich_file() now parses via
170
+ risforge.cleaning.parse_ris_records() (block-isolated, immune to
171
+ this) instead of calling rispy.load() directly on the whole file.
172
+ """
173
+
174
+ def test_stray_blank_line_after_record_boundary_does_not_crash(self, tmp_path) -> None:
175
+ # Record 1 ends with an "LA" (language) tag as its last real
176
+ # field before ER. Record 2 has a stray blank line immediately
177
+ # after "TY", before its own first tag -- this is exactly the
178
+ # combination that crashed rispy.load() with KeyError('language').
179
+ ris_text = (
180
+ "TY - JOUR\n"
181
+ "AU - Smith, John\n"
182
+ "TI - First paper\n"
183
+ "LA - English\n"
184
+ "DO - 10.1000/first\n"
185
+ "ER - \n"
186
+ "\n"
187
+ "TY - JOUR\n"
188
+ "\n"
189
+ "AU - Doe, Jane\n"
190
+ "TI - Second paper\n"
191
+ "DO - 10.1000/second\n"
192
+ "ER - \n"
193
+ )
194
+ input_path = tmp_path / "malformed.ris"
195
+ input_path.write_text(ris_text, encoding="utf-8")
196
+
197
+ enricher, _session = _make_enricher()
198
+ # Must not raise KeyError.
199
+ stats = enricher.enrich_file(str(input_path), str(tmp_path / "out.ris"))
200
+
201
+ assert stats["processed"] == 2
202
+
203
+ def test_genuinely_unparseable_block_is_skipped_and_counted(self, tmp_path) -> None:
204
+ ris_text = (
205
+ "TY - JOUR\n"
206
+ "AU - Smith, John\n"
207
+ "TI - Valid paper\n"
208
+ "DO - 10.1000/valid\n"
209
+ "ER - \n"
210
+ "\n"
211
+ "TY - JOUR\n"
212
+ "AU - Broken, Record\n"
213
+ "TI - Missing ER terminator entirely\n"
214
+ )
215
+ input_path = tmp_path / "genuinely_malformed.ris"
216
+ input_path.write_text(ris_text, encoding="utf-8")
217
+
218
+ enricher, _session = _make_enricher()
219
+ stats = enricher.enrich_file(str(input_path), str(tmp_path / "out.ris"))
220
+
221
+ assert stats["processed"] == 1
222
+ assert stats["skipped_malformed"] == 1
223
+
224
+ def test_utf8_bom_is_handled_transparently(self, tmp_path) -> None:
225
+ ris_text = "TY - JOUR\nAU - Smith, John\nTI - BOM paper\nDO - 10.1000/bom\nER - \n"
226
+ input_path = tmp_path / "bom.ris"
227
+ input_path.write_bytes(b"\xef\xbb\xbf" + ris_text.encode("utf-8"))
228
+
229
+ enricher, _session = _make_enricher()
230
+ stats = enricher.enrich_file(str(input_path), str(tmp_path / "out.ris"))
231
+
232
+ assert stats["processed"] == 1
233
+
234
+ def test_non_utf8_file_gives_a_clear_error_without_crashing(self, tmp_path) -> None:
235
+ input_path = tmp_path / "bad_encoding.ris"
236
+ input_path.write_bytes(b"TY - JOUR\nTI - Bad \xff\xfe byte sequence\nER - \n")
237
+
238
+ enricher, _session = _make_enricher()
239
+ # enrich_file()'s existing contract: parsing failures are
240
+ # logged and it returns the (unmodified) stats rather than
241
+ # raising, exactly as it already did for a missing input file.
242
+ stats = enricher.enrich_file(str(input_path), str(tmp_path / "out.ris"))
243
+
244
+ assert stats["processed"] == 0
@@ -1,23 +0,0 @@
1
- """Exception types for risforge.
2
-
3
- We deliberately keep this hierarchy small. Most failure modes in this
4
- package are already well described by the standard library's own
5
- exceptions (``FileNotFoundError``, ``ValueError``, etc.), and the
6
- original scripts caught those precisely rather than reaching for
7
- generic ``Exception``. We keep that pattern. ``RisForgeError`` exists
8
- only as a common base for the handful of errors that are specific to
9
- this package's domain logic, so library users can catch
10
- ``RisForgeError`` if they want a single net for "something in risforge
11
- itself went wrong" without having to also catch unrelated stdlib
12
- errors they may want to handle differently.
13
- """
14
-
15
- from __future__ import annotations
16
-
17
-
18
- class RisForgeError(Exception):
19
- """Base class for errors raised directly by risforge's own logic."""
20
-
21
-
22
- class RisParsingError(RisForgeError):
23
- """Raised when a RIS file cannot be parsed into any usable records."""
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes