pydfdoi 0.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pydfdoi/__init__.py +57 -0
- pydfdoi/citation.py +56 -0
- pydfdoi/cli.py +102 -0
- pydfdoi/crossref.py +22 -0
- pydfdoi/doi.py +36 -0
- pydfdoi/metadata.py +182 -0
- pydfdoi/page_labels.py +246 -0
- pydfdoi/pdf.py +124 -0
- pydfdoi/processing.py +122 -0
- pydfdoi/publishers.py +96 -0
- pydfdoi-0.1.1.dist-info/METADATA +392 -0
- pydfdoi-0.1.1.dist-info/RECORD +15 -0
- pydfdoi-0.1.1.dist-info/WHEEL +4 -0
- pydfdoi-0.1.1.dist-info/entry_points.txt +3 -0
- pydfdoi-0.1.1.dist-info/licenses/LICENSE +21 -0
pydfdoi/__init__.py
ADDED
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"""Tools for DOI-aware PDF metadata and citation-based PDF renaming."""
|
|
2
|
+
|
|
3
|
+
__version__ = "0.1.1"
|
|
4
|
+
|
|
5
|
+
from .citation import build_citation_filename, clean_filename_piece
|
|
6
|
+
from .crossref import fetch_crossref_item
|
|
7
|
+
from .doi import normalize_doi
|
|
8
|
+
from .metadata import (
|
|
9
|
+
DoiMetadata,
|
|
10
|
+
capture_book_doi,
|
|
11
|
+
capture_doi_metadata,
|
|
12
|
+
capture_journal_doi,
|
|
13
|
+
classify_by_page_count,
|
|
14
|
+
classify_crossref_type,
|
|
15
|
+
)
|
|
16
|
+
from .page_labels import (
|
|
17
|
+
ArticlePageRange,
|
|
18
|
+
PageLabelPlan,
|
|
19
|
+
align_journal_article_page_labels,
|
|
20
|
+
build_page_label_plan,
|
|
21
|
+
parse_crossref_page_range,
|
|
22
|
+
)
|
|
23
|
+
from .pdf import capture_and_write_doi_metadata, count_pdf_pages, extract_doi, write_pdf_metadata
|
|
24
|
+
from .processing import PdfResult, process_journal_page_labels, process_pdf, process_pdf_metadata_only, rename_by_citation
|
|
25
|
+
from .publishers import PUBLISHERS, PublisherInfo, publisher_for_doi, publisher_for_name
|
|
26
|
+
|
|
27
|
+
__all__ = [
|
|
28
|
+
"PdfResult",
|
|
29
|
+
"DoiMetadata",
|
|
30
|
+
"PUBLISHERS",
|
|
31
|
+
"ArticlePageRange",
|
|
32
|
+
"PageLabelPlan",
|
|
33
|
+
"PublisherInfo",
|
|
34
|
+
"__version__",
|
|
35
|
+
"build_citation_filename",
|
|
36
|
+
"build_page_label_plan",
|
|
37
|
+
"capture_and_write_doi_metadata",
|
|
38
|
+
"capture_book_doi",
|
|
39
|
+
"capture_doi_metadata",
|
|
40
|
+
"capture_journal_doi",
|
|
41
|
+
"classify_by_page_count",
|
|
42
|
+
"classify_crossref_type",
|
|
43
|
+
"clean_filename_piece",
|
|
44
|
+
"count_pdf_pages",
|
|
45
|
+
"extract_doi",
|
|
46
|
+
"fetch_crossref_item",
|
|
47
|
+
"align_journal_article_page_labels",
|
|
48
|
+
"normalize_doi",
|
|
49
|
+
"parse_crossref_page_range",
|
|
50
|
+
"process_pdf",
|
|
51
|
+
"process_journal_page_labels",
|
|
52
|
+
"process_pdf_metadata_only",
|
|
53
|
+
"publisher_for_doi",
|
|
54
|
+
"publisher_for_name",
|
|
55
|
+
"rename_by_citation",
|
|
56
|
+
"write_pdf_metadata",
|
|
57
|
+
]
|
pydfdoi/citation.py
ADDED
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
"""Citation metadata formatting for filenames."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
import unicodedata
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
INVALID_FILENAME_CHARS = re.compile(r'[<>:"/\\|?*\x00-\x1f]')
|
|
10
|
+
WHITESPACE_RE = re.compile(r"\s+")
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def clean_filename_piece(text: str) -> str:
|
|
14
|
+
text = unicodedata.normalize("NFKC", text)
|
|
15
|
+
text = INVALID_FILENAME_CHARS.sub(" ", text)
|
|
16
|
+
text = WHITESPACE_RE.sub(" ", text)
|
|
17
|
+
return text.strip(" .")
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def first_author_label(item: dict) -> str:
|
|
21
|
+
authors = item.get("author") or []
|
|
22
|
+
if not authors:
|
|
23
|
+
return "Unknown"
|
|
24
|
+
first = authors[0]
|
|
25
|
+
family = clean_filename_piece(first.get("family") or first.get("name") or "Unknown")
|
|
26
|
+
return family if len(authors) == 1 else f"{family} et al."
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def item_year(item: dict) -> str:
|
|
30
|
+
for key in ("published-print", "published-online", "published", "issued", "created"):
|
|
31
|
+
parts = ((item.get(key) or {}).get("date-parts") or [])
|
|
32
|
+
if parts and parts[0]:
|
|
33
|
+
return str(parts[0][0])
|
|
34
|
+
return "n.d."
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def item_title(item: dict) -> str:
|
|
38
|
+
titles = item.get("title") or []
|
|
39
|
+
return clean_filename_piece(titles[0] if titles else "Untitled")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def build_citation_filename(item: dict, *, extension: str = ".pdf", max_name_chars: int = 180) -> str:
|
|
43
|
+
filename = clean_filename_piece(f"{first_author_label(item)} {item_year(item)} {item_title(item)}")
|
|
44
|
+
if len(filename) > max_name_chars:
|
|
45
|
+
filename = filename[:max_name_chars].rstrip(" .")
|
|
46
|
+
return f"{filename}{extension}"
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def unique_path(path: Path) -> Path:
|
|
50
|
+
if not path.exists():
|
|
51
|
+
return path
|
|
52
|
+
for number in range(2, 1000):
|
|
53
|
+
candidate = path.with_name(f"{path.stem} ({number}){path.suffix}")
|
|
54
|
+
if not candidate.exists():
|
|
55
|
+
return candidate
|
|
56
|
+
raise FileExistsError(f"Could not find an unused filename for {path}")
|
pydfdoi/cli.py
ADDED
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
"""Command-line interface for pydfdoi."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import json
|
|
7
|
+
import sys
|
|
8
|
+
import time
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Iterable
|
|
11
|
+
|
|
12
|
+
from .processing import process_journal_page_labels, process_pdf, process_pdf_metadata_only
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def choose_files_with_dialog() -> list[str]:
|
|
16
|
+
import tkinter as tk
|
|
17
|
+
from tkinter import filedialog
|
|
18
|
+
|
|
19
|
+
root = tk.Tk()
|
|
20
|
+
root.withdraw()
|
|
21
|
+
root.attributes("-topmost", True)
|
|
22
|
+
files = filedialog.askopenfilenames(
|
|
23
|
+
title="Select PDF files",
|
|
24
|
+
filetypes=[("PDF files", "*.pdf"), ("All files", "*.*")],
|
|
25
|
+
)
|
|
26
|
+
root.destroy()
|
|
27
|
+
return list(files)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def flatten_files(values: Iterable[str]) -> list[Path]:
|
|
31
|
+
paths: list[Path] = []
|
|
32
|
+
for value in values:
|
|
33
|
+
for piece in str(value).splitlines():
|
|
34
|
+
piece = piece.strip().strip('"')
|
|
35
|
+
if piece:
|
|
36
|
+
paths.append(Path(piece))
|
|
37
|
+
return paths
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def parse_args(argv: list[str]) -> argparse.Namespace:
|
|
41
|
+
parser = argparse.ArgumentParser(description="PDF DOI metadata and citation rename tool")
|
|
42
|
+
parser.add_argument("files", nargs="*", help="PDF files to process")
|
|
43
|
+
parser.add_argument("--mode", choices=("rename", "metadata-only", "page-labels"), default="rename")
|
|
44
|
+
parser.add_argument("--max-pages", type=int, default=20, help="pages to scan for DOI")
|
|
45
|
+
parser.add_argument("--extra-placement", choices=("front", "back", "split"), default="front")
|
|
46
|
+
parser.add_argument("--skipped-label-prefix", default="skip-")
|
|
47
|
+
parser.add_argument("--dry-run", action="store_true", help="show actions without writing files")
|
|
48
|
+
parser.add_argument("--json", action="store_true", help="print machine-readable results")
|
|
49
|
+
return parser.parse_args(argv)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def main(argv: list[str] | None = None) -> int:
|
|
53
|
+
args = parse_args(list(sys.argv[1:] if argv is None else argv))
|
|
54
|
+
files = flatten_files(args.files) or flatten_files(choose_files_with_dialog())
|
|
55
|
+
if not files:
|
|
56
|
+
print("No files selected.")
|
|
57
|
+
return 1
|
|
58
|
+
|
|
59
|
+
if args.mode == "metadata-only":
|
|
60
|
+
results = [process_pdf_metadata_only(path, max_pages=args.max_pages, dry_run=args.dry_run) for path in files]
|
|
61
|
+
elif args.mode == "page-labels":
|
|
62
|
+
results = [
|
|
63
|
+
process_journal_page_labels(
|
|
64
|
+
path,
|
|
65
|
+
max_pages=args.max_pages,
|
|
66
|
+
extra_placement=args.extra_placement,
|
|
67
|
+
skipped_label_prefix=args.skipped_label_prefix,
|
|
68
|
+
dry_run=args.dry_run,
|
|
69
|
+
)
|
|
70
|
+
for path in files
|
|
71
|
+
]
|
|
72
|
+
else:
|
|
73
|
+
results = [process_pdf(path, rename=True, max_pages=args.max_pages, dry_run=args.dry_run) for path in files]
|
|
74
|
+
|
|
75
|
+
if args.json:
|
|
76
|
+
print(
|
|
77
|
+
json.dumps(
|
|
78
|
+
[
|
|
79
|
+
{
|
|
80
|
+
"path": str(result.path),
|
|
81
|
+
"ok": result.ok,
|
|
82
|
+
"doi": result.doi,
|
|
83
|
+
"renamed_to": str(result.renamed_to) if result.renamed_to else None,
|
|
84
|
+
"error": result.error,
|
|
85
|
+
}
|
|
86
|
+
for result in results
|
|
87
|
+
],
|
|
88
|
+
ensure_ascii=False,
|
|
89
|
+
indent=2,
|
|
90
|
+
)
|
|
91
|
+
)
|
|
92
|
+
else:
|
|
93
|
+
ok_count = sum(1 for result in results if result.ok)
|
|
94
|
+
fail_count = len(results) - ok_count
|
|
95
|
+
for result in results:
|
|
96
|
+
if not result.ok:
|
|
97
|
+
print(f"[Failed] {result.path.name}: {result.error}")
|
|
98
|
+
print(f"Done. Success: {ok_count}, failed: {fail_count}")
|
|
99
|
+
|
|
100
|
+
if len(results) == 1:
|
|
101
|
+
time.sleep(0.2)
|
|
102
|
+
return 0 if all(result.ok for result in results) else 2
|
pydfdoi/crossref.py
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
"""Crossref client helpers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from urllib.parse import quote
|
|
6
|
+
|
|
7
|
+
import requests
|
|
8
|
+
|
|
9
|
+
USER_AGENT = "pydfdoi/0.1.1"
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def fetch_crossref_item(doi: str, timeout: float = 20.0) -> dict | None:
|
|
13
|
+
url = f"https://api.crossref.org/works/{quote(doi, safe='')}"
|
|
14
|
+
headers = {
|
|
15
|
+
"User-Agent": USER_AGENT,
|
|
16
|
+
"Accept": "application/json",
|
|
17
|
+
}
|
|
18
|
+
response = requests.get(url, headers=headers, timeout=timeout)
|
|
19
|
+
if response.status_code == 404:
|
|
20
|
+
return None
|
|
21
|
+
response.raise_for_status()
|
|
22
|
+
return response.json().get("message")
|
pydfdoi/doi.py
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""DOI normalization and extraction helpers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
|
|
7
|
+
DOI_RE = re.compile(r"\b10\.\d{4,9}/[-._;()/:A-Z0-9]+", re.IGNORECASE)
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def normalize_doi(raw: str | None) -> str | None:
|
|
11
|
+
"""Return a bare DOI string from common DOI spellings and URLs."""
|
|
12
|
+
if not raw:
|
|
13
|
+
return None
|
|
14
|
+
text = raw.strip()
|
|
15
|
+
text = re.sub(r"^(?:doi\s*:?\s*|https?://(?:dx\.)?doi\.org/)", "", text, flags=re.I)
|
|
16
|
+
text = re.sub(r"\s*([./:_;()-])\s*", r"\1", text)
|
|
17
|
+
text = re.sub(r"\s+", "", text)
|
|
18
|
+
text = text.strip().strip(".,;:)])}>")
|
|
19
|
+
text = text.replace("\u200b", "")
|
|
20
|
+
return text or None
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def find_doi(text: str | None) -> str | None:
|
|
24
|
+
"""Find and normalize a DOI in text, tolerating simple PDF spacing noise."""
|
|
25
|
+
if not text:
|
|
26
|
+
return None
|
|
27
|
+
collapsed = re.sub(r"\s+", " ", text)
|
|
28
|
+
match = DOI_RE.search(collapsed)
|
|
29
|
+
if match:
|
|
30
|
+
return normalize_doi(match.group(0))
|
|
31
|
+
|
|
32
|
+
compact_punctuation = re.sub(r"\s*([./:_;()-])\s*", r"\1", collapsed)
|
|
33
|
+
match = DOI_RE.search(compact_punctuation)
|
|
34
|
+
if match:
|
|
35
|
+
return normalize_doi(match.group(0))
|
|
36
|
+
return None
|
pydfdoi/metadata.py
ADDED
|
@@ -0,0 +1,182 @@
|
|
|
1
|
+
"""DOI metadata capture for journal articles and books."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from .crossref import fetch_crossref_item
|
|
9
|
+
from .pdf import count_pdf_pages, extract_doi
|
|
10
|
+
from .publishers import PublisherInfo, publisher_for_doi, publisher_for_name
|
|
11
|
+
|
|
12
|
+
JOURNAL_TYPES = frozenset({"journal-article", "journal-issue", "journal-volume"})
|
|
13
|
+
BOOK_TYPES = frozenset({"book", "book-chapter", "book-part", "book-section", "edited-book", "monograph", "reference-book"})
|
|
14
|
+
ARTICLE_PAGE_LIMIT = 80
|
|
15
|
+
BOOK_MAX_PAGES = 20
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@dataclass
|
|
19
|
+
class DoiMetadata:
|
|
20
|
+
doi: str
|
|
21
|
+
source_path: Path
|
|
22
|
+
requested_kind: str
|
|
23
|
+
inferred_kind: str | None = None
|
|
24
|
+
classification_source: str | None = None
|
|
25
|
+
page_count: int | None = None
|
|
26
|
+
work_type: str | None = None
|
|
27
|
+
title: str | None = None
|
|
28
|
+
publisher: str | None = None
|
|
29
|
+
publisher_abbreviation: str | None = None
|
|
30
|
+
publisher_doi_prefixes: tuple[str, ...] = ()
|
|
31
|
+
crossref_item: dict | None = field(default=None, repr=False)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _first_text(value: object) -> str | None:
|
|
35
|
+
if isinstance(value, list) and value:
|
|
36
|
+
return str(value[0])
|
|
37
|
+
if isinstance(value, str):
|
|
38
|
+
return value
|
|
39
|
+
return None
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _publisher_hint(doi: str, item: dict | None) -> PublisherInfo | None:
|
|
43
|
+
if item:
|
|
44
|
+
matched = publisher_for_name(item.get("publisher"))
|
|
45
|
+
if matched:
|
|
46
|
+
return matched
|
|
47
|
+
return publisher_for_doi(doi)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def classify_by_page_count(page_count: int | None, *, article_page_limit: int = ARTICLE_PAGE_LIMIT) -> str | None:
|
|
51
|
+
"""Guess whether a PDF is article-like or book-like from its page count."""
|
|
52
|
+
if page_count is None:
|
|
53
|
+
return None
|
|
54
|
+
return "journal" if page_count <= article_page_limit else "book"
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def classify_crossref_type(work_type: str | None) -> str | None:
|
|
58
|
+
"""Map a Crossref work type to the package's journal/book categories."""
|
|
59
|
+
if work_type in JOURNAL_TYPES:
|
|
60
|
+
return "journal"
|
|
61
|
+
if work_type in BOOK_TYPES:
|
|
62
|
+
return "book"
|
|
63
|
+
return None
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _safe_page_count(path: Path) -> int | None:
|
|
67
|
+
try:
|
|
68
|
+
return count_pdf_pages(path)
|
|
69
|
+
except Exception:
|
|
70
|
+
return None
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _capture(
|
|
74
|
+
path: Path | str,
|
|
75
|
+
*,
|
|
76
|
+
requested_kind: str,
|
|
77
|
+
allowed_types: frozenset[str] | None,
|
|
78
|
+
max_pages: int,
|
|
79
|
+
query_crossref: bool,
|
|
80
|
+
use_page_guess: bool,
|
|
81
|
+
article_page_limit: int,
|
|
82
|
+
strict_type: bool,
|
|
83
|
+
) -> DoiMetadata | None:
|
|
84
|
+
source_path = Path(path)
|
|
85
|
+
page_count = _safe_page_count(source_path) if use_page_guess else None
|
|
86
|
+
page_kind = classify_by_page_count(page_count, article_page_limit=article_page_limit)
|
|
87
|
+
doi = extract_doi(source_path, max_pages=max_pages)
|
|
88
|
+
if not doi:
|
|
89
|
+
return None
|
|
90
|
+
|
|
91
|
+
item = fetch_crossref_item(doi) if query_crossref else None
|
|
92
|
+
work_type = item.get("type") if item else None
|
|
93
|
+
crossref_kind = classify_crossref_type(work_type)
|
|
94
|
+
inferred_kind = crossref_kind or page_kind
|
|
95
|
+
classification_source = "crossref" if crossref_kind else ("page-count" if page_kind else None)
|
|
96
|
+
if allowed_types and work_type and work_type not in allowed_types:
|
|
97
|
+
if strict_type or crossref_kind is not None or page_kind != requested_kind:
|
|
98
|
+
return None
|
|
99
|
+
|
|
100
|
+
publisher = _publisher_hint(doi, item)
|
|
101
|
+
return DoiMetadata(
|
|
102
|
+
doi=doi,
|
|
103
|
+
source_path=source_path,
|
|
104
|
+
requested_kind=requested_kind,
|
|
105
|
+
inferred_kind=inferred_kind,
|
|
106
|
+
classification_source=classification_source,
|
|
107
|
+
page_count=page_count,
|
|
108
|
+
work_type=work_type,
|
|
109
|
+
title=_first_text(item.get("title")) if item else None,
|
|
110
|
+
publisher=item.get("publisher") if item else (publisher.name if publisher else None),
|
|
111
|
+
publisher_abbreviation=publisher.abbreviation if publisher else None,
|
|
112
|
+
publisher_doi_prefixes=publisher.doi_prefixes if publisher else (),
|
|
113
|
+
crossref_item=item,
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def capture_journal_doi(path: Path | str, *, max_pages: int = 3, query_crossref: bool = True) -> DoiMetadata | None:
|
|
118
|
+
"""Capture DOI metadata for a journal-like Crossref work."""
|
|
119
|
+
return _capture(
|
|
120
|
+
path,
|
|
121
|
+
requested_kind="journal",
|
|
122
|
+
allowed_types=JOURNAL_TYPES,
|
|
123
|
+
max_pages=max_pages,
|
|
124
|
+
query_crossref=query_crossref,
|
|
125
|
+
use_page_guess=False,
|
|
126
|
+
article_page_limit=ARTICLE_PAGE_LIMIT,
|
|
127
|
+
strict_type=True,
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def capture_book_doi(
|
|
132
|
+
path: Path | str,
|
|
133
|
+
*,
|
|
134
|
+
max_pages: int = BOOK_MAX_PAGES,
|
|
135
|
+
query_crossref: bool = True,
|
|
136
|
+
article_page_limit: int = ARTICLE_PAGE_LIMIT,
|
|
137
|
+
strict_type: bool = False,
|
|
138
|
+
) -> DoiMetadata | None:
|
|
139
|
+
"""Capture DOI metadata for a book-like work.
|
|
140
|
+
|
|
141
|
+
Books often place DOI information after the cover and title pages, so this
|
|
142
|
+
function scans more pages than the journal helper by default. When
|
|
143
|
+
`strict_type` is false, an unknown Crossref type can still be accepted if
|
|
144
|
+
the page-count heuristic says the PDF is book-like.
|
|
145
|
+
"""
|
|
146
|
+
return _capture(
|
|
147
|
+
path,
|
|
148
|
+
requested_kind="book",
|
|
149
|
+
allowed_types=BOOK_TYPES,
|
|
150
|
+
max_pages=max_pages,
|
|
151
|
+
query_crossref=query_crossref,
|
|
152
|
+
use_page_guess=True,
|
|
153
|
+
article_page_limit=article_page_limit,
|
|
154
|
+
strict_type=strict_type,
|
|
155
|
+
)
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def capture_doi_metadata(
|
|
159
|
+
path: Path | str,
|
|
160
|
+
*,
|
|
161
|
+
max_pages: int = 3,
|
|
162
|
+
query_crossref: bool = True,
|
|
163
|
+
use_page_guess: bool = True,
|
|
164
|
+
article_page_limit: int = ARTICLE_PAGE_LIMIT,
|
|
165
|
+
) -> DoiMetadata | None:
|
|
166
|
+
"""Capture DOI metadata and infer whether the work is article-like or book-like.
|
|
167
|
+
|
|
168
|
+
The general classifier first makes a coarse page-count guess when enabled:
|
|
169
|
+
PDFs with `article_page_limit` pages or fewer are treated as journal-like,
|
|
170
|
+
and longer PDFs are treated as book-like. If Crossref is queried and returns
|
|
171
|
+
a known work type, the Crossref type overrides that page-count guess.
|
|
172
|
+
"""
|
|
173
|
+
return _capture(
|
|
174
|
+
path,
|
|
175
|
+
requested_kind="any",
|
|
176
|
+
allowed_types=None,
|
|
177
|
+
max_pages=max_pages,
|
|
178
|
+
query_crossref=query_crossref,
|
|
179
|
+
use_page_guess=use_page_guess,
|
|
180
|
+
article_page_limit=article_page_limit,
|
|
181
|
+
strict_type=False,
|
|
182
|
+
)
|