pydfdoi 0.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pydfdoi/__init__.py ADDED
@@ -0,0 +1,57 @@
1
+ """Tools for DOI-aware PDF metadata and citation-based PDF renaming."""
2
+
3
+ __version__ = "0.1.1"
4
+
5
+ from .citation import build_citation_filename, clean_filename_piece
6
+ from .crossref import fetch_crossref_item
7
+ from .doi import normalize_doi
8
+ from .metadata import (
9
+ DoiMetadata,
10
+ capture_book_doi,
11
+ capture_doi_metadata,
12
+ capture_journal_doi,
13
+ classify_by_page_count,
14
+ classify_crossref_type,
15
+ )
16
+ from .page_labels import (
17
+ ArticlePageRange,
18
+ PageLabelPlan,
19
+ align_journal_article_page_labels,
20
+ build_page_label_plan,
21
+ parse_crossref_page_range,
22
+ )
23
+ from .pdf import capture_and_write_doi_metadata, count_pdf_pages, extract_doi, write_pdf_metadata
24
+ from .processing import PdfResult, process_journal_page_labels, process_pdf, process_pdf_metadata_only, rename_by_citation
25
+ from .publishers import PUBLISHERS, PublisherInfo, publisher_for_doi, publisher_for_name
26
+
27
+ __all__ = [
28
+ "PdfResult",
29
+ "DoiMetadata",
30
+ "PUBLISHERS",
31
+ "ArticlePageRange",
32
+ "PageLabelPlan",
33
+ "PublisherInfo",
34
+ "__version__",
35
+ "build_citation_filename",
36
+ "build_page_label_plan",
37
+ "capture_and_write_doi_metadata",
38
+ "capture_book_doi",
39
+ "capture_doi_metadata",
40
+ "capture_journal_doi",
41
+ "classify_by_page_count",
42
+ "classify_crossref_type",
43
+ "clean_filename_piece",
44
+ "count_pdf_pages",
45
+ "extract_doi",
46
+ "fetch_crossref_item",
47
+ "align_journal_article_page_labels",
48
+ "normalize_doi",
49
+ "parse_crossref_page_range",
50
+ "process_pdf",
51
+ "process_journal_page_labels",
52
+ "process_pdf_metadata_only",
53
+ "publisher_for_doi",
54
+ "publisher_for_name",
55
+ "rename_by_citation",
56
+ "write_pdf_metadata",
57
+ ]
pydfdoi/citation.py ADDED
@@ -0,0 +1,56 @@
1
+ """Citation metadata formatting for filenames."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ import unicodedata
7
+ from pathlib import Path
8
+
9
+ INVALID_FILENAME_CHARS = re.compile(r'[<>:"/\\|?*\x00-\x1f]')
10
+ WHITESPACE_RE = re.compile(r"\s+")
11
+
12
+
13
+ def clean_filename_piece(text: str) -> str:
14
+ text = unicodedata.normalize("NFKC", text)
15
+ text = INVALID_FILENAME_CHARS.sub(" ", text)
16
+ text = WHITESPACE_RE.sub(" ", text)
17
+ return text.strip(" .")
18
+
19
+
20
+ def first_author_label(item: dict) -> str:
21
+ authors = item.get("author") or []
22
+ if not authors:
23
+ return "Unknown"
24
+ first = authors[0]
25
+ family = clean_filename_piece(first.get("family") or first.get("name") or "Unknown")
26
+ return family if len(authors) == 1 else f"{family} et al."
27
+
28
+
29
+ def item_year(item: dict) -> str:
30
+ for key in ("published-print", "published-online", "published", "issued", "created"):
31
+ parts = ((item.get(key) or {}).get("date-parts") or [])
32
+ if parts and parts[0]:
33
+ return str(parts[0][0])
34
+ return "n.d."
35
+
36
+
37
+ def item_title(item: dict) -> str:
38
+ titles = item.get("title") or []
39
+ return clean_filename_piece(titles[0] if titles else "Untitled")
40
+
41
+
42
+ def build_citation_filename(item: dict, *, extension: str = ".pdf", max_name_chars: int = 180) -> str:
43
+ filename = clean_filename_piece(f"{first_author_label(item)} {item_year(item)} {item_title(item)}")
44
+ if len(filename) > max_name_chars:
45
+ filename = filename[:max_name_chars].rstrip(" .")
46
+ return f"{filename}{extension}"
47
+
48
+
49
+ def unique_path(path: Path) -> Path:
50
+ if not path.exists():
51
+ return path
52
+ for number in range(2, 1000):
53
+ candidate = path.with_name(f"{path.stem} ({number}){path.suffix}")
54
+ if not candidate.exists():
55
+ return candidate
56
+ raise FileExistsError(f"Could not find an unused filename for {path}")
pydfdoi/cli.py ADDED
@@ -0,0 +1,102 @@
1
+ """Command-line interface for pydfdoi."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import json
7
+ import sys
8
+ import time
9
+ from pathlib import Path
10
+ from typing import Iterable
11
+
12
+ from .processing import process_journal_page_labels, process_pdf, process_pdf_metadata_only
13
+
14
+
15
+ def choose_files_with_dialog() -> list[str]:
16
+ import tkinter as tk
17
+ from tkinter import filedialog
18
+
19
+ root = tk.Tk()
20
+ root.withdraw()
21
+ root.attributes("-topmost", True)
22
+ files = filedialog.askopenfilenames(
23
+ title="Select PDF files",
24
+ filetypes=[("PDF files", "*.pdf"), ("All files", "*.*")],
25
+ )
26
+ root.destroy()
27
+ return list(files)
28
+
29
+
30
+ def flatten_files(values: Iterable[str]) -> list[Path]:
31
+ paths: list[Path] = []
32
+ for value in values:
33
+ for piece in str(value).splitlines():
34
+ piece = piece.strip().strip('"')
35
+ if piece:
36
+ paths.append(Path(piece))
37
+ return paths
38
+
39
+
40
+ def parse_args(argv: list[str]) -> argparse.Namespace:
41
+ parser = argparse.ArgumentParser(description="PDF DOI metadata and citation rename tool")
42
+ parser.add_argument("files", nargs="*", help="PDF files to process")
43
+ parser.add_argument("--mode", choices=("rename", "metadata-only", "page-labels"), default="rename")
44
+ parser.add_argument("--max-pages", type=int, default=20, help="pages to scan for DOI")
45
+ parser.add_argument("--extra-placement", choices=("front", "back", "split"), default="front")
46
+ parser.add_argument("--skipped-label-prefix", default="skip-")
47
+ parser.add_argument("--dry-run", action="store_true", help="show actions without writing files")
48
+ parser.add_argument("--json", action="store_true", help="print machine-readable results")
49
+ return parser.parse_args(argv)
50
+
51
+
52
+ def main(argv: list[str] | None = None) -> int:
53
+ args = parse_args(list(sys.argv[1:] if argv is None else argv))
54
+ files = flatten_files(args.files) or flatten_files(choose_files_with_dialog())
55
+ if not files:
56
+ print("No files selected.")
57
+ return 1
58
+
59
+ if args.mode == "metadata-only":
60
+ results = [process_pdf_metadata_only(path, max_pages=args.max_pages, dry_run=args.dry_run) for path in files]
61
+ elif args.mode == "page-labels":
62
+ results = [
63
+ process_journal_page_labels(
64
+ path,
65
+ max_pages=args.max_pages,
66
+ extra_placement=args.extra_placement,
67
+ skipped_label_prefix=args.skipped_label_prefix,
68
+ dry_run=args.dry_run,
69
+ )
70
+ for path in files
71
+ ]
72
+ else:
73
+ results = [process_pdf(path, rename=True, max_pages=args.max_pages, dry_run=args.dry_run) for path in files]
74
+
75
+ if args.json:
76
+ print(
77
+ json.dumps(
78
+ [
79
+ {
80
+ "path": str(result.path),
81
+ "ok": result.ok,
82
+ "doi": result.doi,
83
+ "renamed_to": str(result.renamed_to) if result.renamed_to else None,
84
+ "error": result.error,
85
+ }
86
+ for result in results
87
+ ],
88
+ ensure_ascii=False,
89
+ indent=2,
90
+ )
91
+ )
92
+ else:
93
+ ok_count = sum(1 for result in results if result.ok)
94
+ fail_count = len(results) - ok_count
95
+ for result in results:
96
+ if not result.ok:
97
+ print(f"[Failed] {result.path.name}: {result.error}")
98
+ print(f"Done. Success: {ok_count}, failed: {fail_count}")
99
+
100
+ if len(results) == 1:
101
+ time.sleep(0.2)
102
+ return 0 if all(result.ok for result in results) else 2
pydfdoi/crossref.py ADDED
@@ -0,0 +1,22 @@
1
+ """Crossref client helpers."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from urllib.parse import quote
6
+
7
+ import requests
8
+
9
+ USER_AGENT = "pydfdoi/0.1.1"
10
+
11
+
12
+ def fetch_crossref_item(doi: str, timeout: float = 20.0) -> dict | None:
13
+ url = f"https://api.crossref.org/works/{quote(doi, safe='')}"
14
+ headers = {
15
+ "User-Agent": USER_AGENT,
16
+ "Accept": "application/json",
17
+ }
18
+ response = requests.get(url, headers=headers, timeout=timeout)
19
+ if response.status_code == 404:
20
+ return None
21
+ response.raise_for_status()
22
+ return response.json().get("message")
pydfdoi/doi.py ADDED
@@ -0,0 +1,36 @@
1
+ """DOI normalization and extraction helpers."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+
7
+ DOI_RE = re.compile(r"\b10\.\d{4,9}/[-._;()/:A-Z0-9]+", re.IGNORECASE)
8
+
9
+
10
+ def normalize_doi(raw: str | None) -> str | None:
11
+ """Return a bare DOI string from common DOI spellings and URLs."""
12
+ if not raw:
13
+ return None
14
+ text = raw.strip()
15
+ text = re.sub(r"^(?:doi\s*:?\s*|https?://(?:dx\.)?doi\.org/)", "", text, flags=re.I)
16
+ text = re.sub(r"\s*([./:_;()-])\s*", r"\1", text)
17
+ text = re.sub(r"\s+", "", text)
18
+ text = text.strip().strip(".,;:)])}>")
19
+ text = text.replace("\u200b", "")
20
+ return text or None
21
+
22
+
23
+ def find_doi(text: str | None) -> str | None:
24
+ """Find and normalize a DOI in text, tolerating simple PDF spacing noise."""
25
+ if not text:
26
+ return None
27
+ collapsed = re.sub(r"\s+", " ", text)
28
+ match = DOI_RE.search(collapsed)
29
+ if match:
30
+ return normalize_doi(match.group(0))
31
+
32
+ compact_punctuation = re.sub(r"\s*([./:_;()-])\s*", r"\1", collapsed)
33
+ match = DOI_RE.search(compact_punctuation)
34
+ if match:
35
+ return normalize_doi(match.group(0))
36
+ return None
pydfdoi/metadata.py ADDED
@@ -0,0 +1,182 @@
1
+ """DOI metadata capture for journal articles and books."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass, field
6
+ from pathlib import Path
7
+
8
+ from .crossref import fetch_crossref_item
9
+ from .pdf import count_pdf_pages, extract_doi
10
+ from .publishers import PublisherInfo, publisher_for_doi, publisher_for_name
11
+
12
+ JOURNAL_TYPES = frozenset({"journal-article", "journal-issue", "journal-volume"})
13
+ BOOK_TYPES = frozenset({"book", "book-chapter", "book-part", "book-section", "edited-book", "monograph", "reference-book"})
14
+ ARTICLE_PAGE_LIMIT = 80
15
+ BOOK_MAX_PAGES = 20
16
+
17
+
18
+ @dataclass
19
+ class DoiMetadata:
20
+ doi: str
21
+ source_path: Path
22
+ requested_kind: str
23
+ inferred_kind: str | None = None
24
+ classification_source: str | None = None
25
+ page_count: int | None = None
26
+ work_type: str | None = None
27
+ title: str | None = None
28
+ publisher: str | None = None
29
+ publisher_abbreviation: str | None = None
30
+ publisher_doi_prefixes: tuple[str, ...] = ()
31
+ crossref_item: dict | None = field(default=None, repr=False)
32
+
33
+
34
+ def _first_text(value: object) -> str | None:
35
+ if isinstance(value, list) and value:
36
+ return str(value[0])
37
+ if isinstance(value, str):
38
+ return value
39
+ return None
40
+
41
+
42
+ def _publisher_hint(doi: str, item: dict | None) -> PublisherInfo | None:
43
+ if item:
44
+ matched = publisher_for_name(item.get("publisher"))
45
+ if matched:
46
+ return matched
47
+ return publisher_for_doi(doi)
48
+
49
+
50
+ def classify_by_page_count(page_count: int | None, *, article_page_limit: int = ARTICLE_PAGE_LIMIT) -> str | None:
51
+ """Guess whether a PDF is article-like or book-like from its page count."""
52
+ if page_count is None:
53
+ return None
54
+ return "journal" if page_count <= article_page_limit else "book"
55
+
56
+
57
+ def classify_crossref_type(work_type: str | None) -> str | None:
58
+ """Map a Crossref work type to the package's journal/book categories."""
59
+ if work_type in JOURNAL_TYPES:
60
+ return "journal"
61
+ if work_type in BOOK_TYPES:
62
+ return "book"
63
+ return None
64
+
65
+
66
+ def _safe_page_count(path: Path) -> int | None:
67
+ try:
68
+ return count_pdf_pages(path)
69
+ except Exception:
70
+ return None
71
+
72
+
73
+ def _capture(
74
+ path: Path | str,
75
+ *,
76
+ requested_kind: str,
77
+ allowed_types: frozenset[str] | None,
78
+ max_pages: int,
79
+ query_crossref: bool,
80
+ use_page_guess: bool,
81
+ article_page_limit: int,
82
+ strict_type: bool,
83
+ ) -> DoiMetadata | None:
84
+ source_path = Path(path)
85
+ page_count = _safe_page_count(source_path) if use_page_guess else None
86
+ page_kind = classify_by_page_count(page_count, article_page_limit=article_page_limit)
87
+ doi = extract_doi(source_path, max_pages=max_pages)
88
+ if not doi:
89
+ return None
90
+
91
+ item = fetch_crossref_item(doi) if query_crossref else None
92
+ work_type = item.get("type") if item else None
93
+ crossref_kind = classify_crossref_type(work_type)
94
+ inferred_kind = crossref_kind or page_kind
95
+ classification_source = "crossref" if crossref_kind else ("page-count" if page_kind else None)
96
+ if allowed_types and work_type and work_type not in allowed_types:
97
+ if strict_type or crossref_kind is not None or page_kind != requested_kind:
98
+ return None
99
+
100
+ publisher = _publisher_hint(doi, item)
101
+ return DoiMetadata(
102
+ doi=doi,
103
+ source_path=source_path,
104
+ requested_kind=requested_kind,
105
+ inferred_kind=inferred_kind,
106
+ classification_source=classification_source,
107
+ page_count=page_count,
108
+ work_type=work_type,
109
+ title=_first_text(item.get("title")) if item else None,
110
+ publisher=item.get("publisher") if item else (publisher.name if publisher else None),
111
+ publisher_abbreviation=publisher.abbreviation if publisher else None,
112
+ publisher_doi_prefixes=publisher.doi_prefixes if publisher else (),
113
+ crossref_item=item,
114
+ )
115
+
116
+
117
+ def capture_journal_doi(path: Path | str, *, max_pages: int = 3, query_crossref: bool = True) -> DoiMetadata | None:
118
+ """Capture DOI metadata for a journal-like Crossref work."""
119
+ return _capture(
120
+ path,
121
+ requested_kind="journal",
122
+ allowed_types=JOURNAL_TYPES,
123
+ max_pages=max_pages,
124
+ query_crossref=query_crossref,
125
+ use_page_guess=False,
126
+ article_page_limit=ARTICLE_PAGE_LIMIT,
127
+ strict_type=True,
128
+ )
129
+
130
+
131
+ def capture_book_doi(
132
+ path: Path | str,
133
+ *,
134
+ max_pages: int = BOOK_MAX_PAGES,
135
+ query_crossref: bool = True,
136
+ article_page_limit: int = ARTICLE_PAGE_LIMIT,
137
+ strict_type: bool = False,
138
+ ) -> DoiMetadata | None:
139
+ """Capture DOI metadata for a book-like work.
140
+
141
+ Books often place DOI information after the cover and title pages, so this
142
+ function scans more pages than the journal helper by default. When
143
+ `strict_type` is false, an unknown Crossref type can still be accepted if
144
+ the page-count heuristic says the PDF is book-like.
145
+ """
146
+ return _capture(
147
+ path,
148
+ requested_kind="book",
149
+ allowed_types=BOOK_TYPES,
150
+ max_pages=max_pages,
151
+ query_crossref=query_crossref,
152
+ use_page_guess=True,
153
+ article_page_limit=article_page_limit,
154
+ strict_type=strict_type,
155
+ )
156
+
157
+
158
+ def capture_doi_metadata(
159
+ path: Path | str,
160
+ *,
161
+ max_pages: int = 3,
162
+ query_crossref: bool = True,
163
+ use_page_guess: bool = True,
164
+ article_page_limit: int = ARTICLE_PAGE_LIMIT,
165
+ ) -> DoiMetadata | None:
166
+ """Capture DOI metadata and infer whether the work is article-like or book-like.
167
+
168
+ The general classifier first makes a coarse page-count guess when enabled:
169
+ PDFs with `article_page_limit` pages or fewer are treated as journal-like,
170
+ and longer PDFs are treated as book-like. If Crossref is queried and returns
171
+ a known work type, the Crossref type overrides that page-count guess.
172
+ """
173
+ return _capture(
174
+ path,
175
+ requested_kind="any",
176
+ allowed_types=None,
177
+ max_pages=max_pages,
178
+ query_crossref=query_crossref,
179
+ use_page_guess=use_page_guess,
180
+ article_page_limit=article_page_limit,
181
+ strict_type=False,
182
+ )