erga 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- erga/__init__.py +7 -0
- erga/cli.py +96 -0
- erga/config.py +139 -0
- erga/crossref.py +50 -0
- erga/curation.py +300 -0
- erga/dedup.py +120 -0
- erga/errors.py +13 -0
- erga/http.py +108 -0
- erga/model.py +100 -0
- erga/normalize.py +132 -0
- erga/openalex.py +167 -0
- erga/output.py +71 -0
- erga/pipeline.py +142 -0
- erga/verify.py +59 -0
- erga-0.1.0.dist-info/METADATA +91 -0
- erga-0.1.0.dist-info/RECORD +19 -0
- erga-0.1.0.dist-info/WHEEL +4 -0
- erga-0.1.0.dist-info/entry_points.txt +2 -0
- erga-0.1.0.dist-info/licenses/LICENSE +21 -0
erga/__init__.py
ADDED
erga/cli.py
ADDED
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
"""Console entry point: `erga build` and `erga verify`."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import os
|
|
7
|
+
import sys
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
from erga import __version__
|
|
11
|
+
from erga.config import Config, load_config
|
|
12
|
+
from erga.crossref import CrossrefClient
|
|
13
|
+
from erga.errors import ErgaError
|
|
14
|
+
from erga.http import UrlTransport
|
|
15
|
+
from erga.openalex import OpenAlexClient
|
|
16
|
+
from erga.pipeline import build
|
|
17
|
+
from erga.verify import verify_report
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _user_agent(mailto: str) -> str:
|
|
21
|
+
return f"erga/{__version__} (https://github.com/belalik/erga; mailto:{mailto})"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _clients(config: Config) -> tuple[OpenAlexClient, CrossrefClient]:
|
|
25
|
+
# One transport for both APIs. The key never touches disk: read from the
|
|
26
|
+
# configured env var, passed as a query parameter, nothing else. The
|
|
27
|
+
# transport redacts it from network-error text (requests embeds the full
|
|
28
|
+
# request URL in its exception messages).
|
|
29
|
+
api_key = os.environ.get(config.api_key_env) or None
|
|
30
|
+
if not api_key:
|
|
31
|
+
print(
|
|
32
|
+
f"erga: note: {config.api_key_env} not set; using OpenAlex's keyless per-IP quota",
|
|
33
|
+
file=sys.stderr,
|
|
34
|
+
)
|
|
35
|
+
transport = UrlTransport(_user_agent(config.mailto), secrets=[api_key] if api_key else [])
|
|
36
|
+
return (
|
|
37
|
+
OpenAlexClient(transport, mailto=config.mailto, api_key=api_key),
|
|
38
|
+
CrossrefClient(transport, mailto=config.mailto),
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _print_warnings(warnings: list[str]) -> None:
|
|
43
|
+
for warning in warnings:
|
|
44
|
+
print(f"erga: warning: {warning}", file=sys.stderr)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _run_build(config: Config, dry_run: bool) -> int:
|
|
48
|
+
openalex, crossref = _clients(config)
|
|
49
|
+
stats = build(config, openalex, crossref, dry_run=dry_run)
|
|
50
|
+
_print_warnings(stats.warnings)
|
|
51
|
+
if dry_run:
|
|
52
|
+
print(f"dry run: {stats.summary()}")
|
|
53
|
+
print(f"would write {stats.total} works to {config.output_path}")
|
|
54
|
+
else:
|
|
55
|
+
print(f"wrote {config.output_path} ({stats.total} works)")
|
|
56
|
+
return 0
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _run_verify(config: Config) -> int:
|
|
60
|
+
openalex, _ = _clients(config)
|
|
61
|
+
report, warnings = verify_report(config, openalex)
|
|
62
|
+
print(report, end="")
|
|
63
|
+
_print_warnings(warnings)
|
|
64
|
+
return 0
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def main(argv: list[str] | None = None) -> int:
|
|
68
|
+
parser = argparse.ArgumentParser(
|
|
69
|
+
prog="erga",
|
|
70
|
+
description="Keep a website's academic publications list current.",
|
|
71
|
+
)
|
|
72
|
+
parser.add_argument("--version", action="version", version=f"erga {__version__}")
|
|
73
|
+
subparsers = parser.add_subparsers(dest="command", required=True)
|
|
74
|
+
|
|
75
|
+
build_parser = subparsers.add_parser("build", help="run the pipeline and write the JSON")
|
|
76
|
+
build_parser.add_argument("--config", type=Path, default=Path("erga.yml"))
|
|
77
|
+
build_parser.add_argument(
|
|
78
|
+
"--dry-run", action="store_true", help="print a summary without writing"
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
verify_parser = subparsers.add_parser("verify", help="author-disambiguation report")
|
|
82
|
+
verify_parser.add_argument("--config", type=Path, default=Path("erga.yml"))
|
|
83
|
+
|
|
84
|
+
args = parser.parse_args(argv)
|
|
85
|
+
try:
|
|
86
|
+
config = load_config(args.config)
|
|
87
|
+
if args.command == "build":
|
|
88
|
+
return _run_build(config, args.dry_run)
|
|
89
|
+
return _run_verify(config)
|
|
90
|
+
except ErgaError as exc:
|
|
91
|
+
print(f"erga: {exc}", file=sys.stderr)
|
|
92
|
+
return 1
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
if __name__ == "__main__":
|
|
96
|
+
sys.exit(main())
|
erga/config.py
ADDED
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
"""Configuration loading and validation (requirements section 5)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from dataclasses import dataclass, field
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
import yaml
|
|
11
|
+
|
|
12
|
+
from erga.errors import ConfigError
|
|
13
|
+
from erga.model import normalize_orcid
|
|
14
|
+
|
|
15
|
+
_ORCID_RE = re.compile(r"^\d{4}-\d{4}-\d{4}-\d{3}[\dX]$")
|
|
16
|
+
_OPENALEX_AUTHOR_RE = re.compile(r"^A\d+$")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass
|
|
20
|
+
class AuthorConfig:
|
|
21
|
+
name: str
|
|
22
|
+
orcid: str | None = None
|
|
23
|
+
openalex_id: str | None = None
|
|
24
|
+
aliases: list[str] = field(default_factory=list)
|
|
25
|
+
|
|
26
|
+
def match_names(self) -> set[str]:
|
|
27
|
+
"""Casefolded name and aliases, for matching manual entries."""
|
|
28
|
+
return {n.casefold().strip() for n in [self.name, *self.aliases]}
|
|
29
|
+
|
|
30
|
+
@property
|
|
31
|
+
def tracking_only(self) -> bool:
|
|
32
|
+
"""No registrar ids: contributes names to tracking, fetches nothing."""
|
|
33
|
+
return self.orcid is None and self.openalex_id is None
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass
|
|
37
|
+
class Config:
|
|
38
|
+
mailto: str
|
|
39
|
+
authors: list[AuthorConfig]
|
|
40
|
+
api_key_env: str = "OPENALEX_API_KEY"
|
|
41
|
+
include_xpac: bool = False
|
|
42
|
+
output_path: Path = Path("publications.json")
|
|
43
|
+
manual_path: Path = Path("manual.yml")
|
|
44
|
+
overrides_path: Path = Path("overrides.yml")
|
|
45
|
+
tags_path: Path = Path("tags.yml")
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def load_yaml(path: Path, expect: type) -> Any:
|
|
49
|
+
"""Parse a YAML file and check its top-level type."""
|
|
50
|
+
try:
|
|
51
|
+
with open(path, encoding="utf-8") as fh:
|
|
52
|
+
data = yaml.safe_load(fh)
|
|
53
|
+
except OSError as exc:
|
|
54
|
+
raise ConfigError(f"{path}: {exc.strerror or exc}") from exc
|
|
55
|
+
except yaml.YAMLError as exc:
|
|
56
|
+
raise ConfigError(f"{path}: invalid YAML: {exc}") from exc
|
|
57
|
+
if data is None:
|
|
58
|
+
data = expect()
|
|
59
|
+
if not isinstance(data, expect):
|
|
60
|
+
raise ConfigError(f"{path}: expected a {expect.__name__} at top level")
|
|
61
|
+
return data
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def reject_unknown_keys(
|
|
65
|
+
mapping: dict[str, Any], allowed: set[str], where: str, noun: str = "keys"
|
|
66
|
+
) -> None:
|
|
67
|
+
unknown = set(mapping) - allowed
|
|
68
|
+
if unknown:
|
|
69
|
+
raise ConfigError(f"{where}: unknown {noun}: {', '.join(sorted(unknown))}")
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def expect_str_list(value: Any, where: str) -> list[str]:
|
|
73
|
+
if not isinstance(value, list) or not all(isinstance(item, str) for item in value):
|
|
74
|
+
raise ConfigError(f"{where}: must be a list of strings")
|
|
75
|
+
return list(value)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _parse_author(entry: Any, path: Path, index: int) -> AuthorConfig:
|
|
79
|
+
where = f"{path}: authors[{index}]"
|
|
80
|
+
if not isinstance(entry, dict):
|
|
81
|
+
raise ConfigError(f"{where}: expected a mapping")
|
|
82
|
+
reject_unknown_keys(entry, {"name", "orcid", "openalex_id", "aliases"}, where)
|
|
83
|
+
name = entry.get("name")
|
|
84
|
+
if not isinstance(name, str) or not name.strip():
|
|
85
|
+
raise ConfigError(f"{where}: 'name' is required")
|
|
86
|
+
orcid = entry.get("orcid")
|
|
87
|
+
if orcid is not None:
|
|
88
|
+
orcid = normalize_orcid(str(orcid))
|
|
89
|
+
if not _ORCID_RE.match(orcid):
|
|
90
|
+
raise ConfigError(f"{where}: invalid ORCID iD {orcid!r}")
|
|
91
|
+
openalex_id = entry.get("openalex_id")
|
|
92
|
+
if openalex_id is not None:
|
|
93
|
+
openalex_id = str(openalex_id).strip()
|
|
94
|
+
if not _OPENALEX_AUTHOR_RE.match(openalex_id):
|
|
95
|
+
raise ConfigError(f"{where}: invalid OpenAlex author id {openalex_id!r}")
|
|
96
|
+
# Neither id is fine: the entry contributes its names to the tracked
|
|
97
|
+
# flag but resolves and fetches nothing (authors without any registrar
|
|
98
|
+
# identity, or whose works OpenAlex misassigns to a conflated profile).
|
|
99
|
+
aliases = expect_str_list(entry.get("aliases", []), f"{where}: 'aliases'")
|
|
100
|
+
return AuthorConfig(name=name.strip(), orcid=orcid, openalex_id=openalex_id, aliases=aliases)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _section(data: dict[str, Any], key: str, path: Path, allowed: set[str]) -> dict[str, Any]:
|
|
104
|
+
section = data.get(key) or {}
|
|
105
|
+
if not isinstance(section, dict):
|
|
106
|
+
raise ConfigError(f"{path}: '{key}' must be a mapping")
|
|
107
|
+
reject_unknown_keys(section, allowed, f"{path}: {key}")
|
|
108
|
+
return section
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def load_config(path: Path) -> Config:
|
|
112
|
+
"""Load and validate erga.yml; relative paths resolve against its directory."""
|
|
113
|
+
data = load_yaml(path, dict)
|
|
114
|
+
base = path.resolve().parent
|
|
115
|
+
reject_unknown_keys(data, {"mailto", "authors", "openalex", "output", "curation"}, str(path))
|
|
116
|
+
|
|
117
|
+
mailto = data.get("mailto")
|
|
118
|
+
if not isinstance(mailto, str) or "@" not in mailto:
|
|
119
|
+
raise ConfigError(f"{path}: 'mailto' is required (identifies requests to the APIs)")
|
|
120
|
+
|
|
121
|
+
raw_authors = data.get("authors")
|
|
122
|
+
if not isinstance(raw_authors, list) or not raw_authors:
|
|
123
|
+
raise ConfigError(f"{path}: 'authors' must be a non-empty list")
|
|
124
|
+
authors = [_parse_author(entry, path, i) for i, entry in enumerate(raw_authors)]
|
|
125
|
+
|
|
126
|
+
openalex = _section(data, "openalex", path, {"api_key_env", "include_xpac"})
|
|
127
|
+
output = _section(data, "output", path, {"path"})
|
|
128
|
+
curation = _section(data, "curation", path, {"manual", "overrides", "tags"})
|
|
129
|
+
|
|
130
|
+
return Config(
|
|
131
|
+
mailto=mailto.strip(),
|
|
132
|
+
authors=authors,
|
|
133
|
+
api_key_env=str(openalex.get("api_key_env", "OPENALEX_API_KEY")),
|
|
134
|
+
include_xpac=bool(openalex.get("include_xpac", False)),
|
|
135
|
+
output_path=base / str(output.get("path", "publications.json")),
|
|
136
|
+
manual_path=base / str(curation.get("manual", "manual.yml")),
|
|
137
|
+
overrides_path=base / str(curation.get("overrides", "overrides.yml")),
|
|
138
|
+
tags_path=base / str(curation.get("tags", "tags.yml")),
|
|
139
|
+
)
|
erga/crossref.py
ADDED
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
"""Crossref venue lookup for the backfill stage.
|
|
2
|
+
|
|
3
|
+
Crossref still operates a mailto polite pool, so the mailto rides along both
|
|
4
|
+
as a query parameter and in the transport's User-Agent.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import time
|
|
10
|
+
import urllib.parse
|
|
11
|
+
from collections.abc import Callable
|
|
12
|
+
|
|
13
|
+
from erga.http import Pacer, Transport, request_with_retry
|
|
14
|
+
|
|
15
|
+
CROSSREF_BASE = "https://api.crossref.org/works/"
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class CrossrefClient:
|
|
19
|
+
def __init__(
|
|
20
|
+
self,
|
|
21
|
+
transport: Transport,
|
|
22
|
+
*,
|
|
23
|
+
mailto: str,
|
|
24
|
+
delay: float = 0.5,
|
|
25
|
+
sleep: Callable[[float], None] = time.sleep,
|
|
26
|
+
) -> None:
|
|
27
|
+
self._transport = transport
|
|
28
|
+
self._mailto = mailto
|
|
29
|
+
self._sleep = sleep
|
|
30
|
+
self._pacer = Pacer(delay, sleep)
|
|
31
|
+
|
|
32
|
+
def venue_for_doi(self, doi: str) -> str | None:
|
|
33
|
+
"""Container title for a bare DOI, or None.
|
|
34
|
+
|
|
35
|
+
DataCite DOIs 404 here and are skipped silently, as is any record
|
|
36
|
+
without a container title. Raises FetchError only when retryable
|
|
37
|
+
failures persist (the caller stops backfilling, keeping nulls).
|
|
38
|
+
"""
|
|
39
|
+
self._pacer.wait()
|
|
40
|
+
url = CROSSREF_BASE + urllib.parse.quote(doi, safe="")
|
|
41
|
+
response = request_with_retry(
|
|
42
|
+
self._transport, url, {"mailto": self._mailto}, sleep=self._sleep
|
|
43
|
+
)
|
|
44
|
+
if response.status_code != 200 or not isinstance(response.data, dict):
|
|
45
|
+
return None
|
|
46
|
+
message = response.data.get("message") or {}
|
|
47
|
+
titles = message.get("container-title") or []
|
|
48
|
+
if titles and isinstance(titles[0], str) and titles[0].strip():
|
|
49
|
+
return titles[0].strip()
|
|
50
|
+
return None
|
erga/curation.py
ADDED
|
@@ -0,0 +1,300 @@
|
|
|
1
|
+
"""Curation files: manual entries, overrides, tags.
|
|
2
|
+
|
|
3
|
+
All three survive every refresh; a missing file means "none". Typos fail
|
|
4
|
+
loudly: curation is the maintainer's reviewable artifact, and a silently
|
|
5
|
+
skipped patch is worse than an aborted run.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import copy
|
|
11
|
+
import datetime
|
|
12
|
+
from collections.abc import Iterator
|
|
13
|
+
from dataclasses import dataclass, field
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
from erga.config import AuthorConfig, expect_str_list, load_yaml, reject_unknown_keys
|
|
18
|
+
from erga.errors import ConfigError
|
|
19
|
+
from erga.model import WORK_TYPES, Work, WorkAuthor, doi_key, doi_url, slugify
|
|
20
|
+
|
|
21
|
+
MANUAL_KEYS = {"title", "authors", "venue", "year", "date", "doi", "type", "tags", "abstract"}
|
|
22
|
+
PATCH_KEYS = {
|
|
23
|
+
"title",
|
|
24
|
+
"authors",
|
|
25
|
+
"venue",
|
|
26
|
+
"year",
|
|
27
|
+
"date",
|
|
28
|
+
"doi",
|
|
29
|
+
"type",
|
|
30
|
+
"cited_by_count",
|
|
31
|
+
"abstract",
|
|
32
|
+
"open_access",
|
|
33
|
+
"tags",
|
|
34
|
+
"is_retracted",
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _parse_authors(value: Any, authors_cfg: list[AuthorConfig], where: str) -> list[WorkAuthor]:
|
|
39
|
+
"""Author strings matched to configured authors by name/alias."""
|
|
40
|
+
if isinstance(value, str):
|
|
41
|
+
value = [value]
|
|
42
|
+
if not isinstance(value, list):
|
|
43
|
+
raise ConfigError(f"{where}: 'authors' must be a string or a list of strings")
|
|
44
|
+
parsed = []
|
|
45
|
+
for name in value:
|
|
46
|
+
if not isinstance(name, str) or not name.strip():
|
|
47
|
+
raise ConfigError(f"{where}: 'authors' entries must be non-empty strings")
|
|
48
|
+
name = name.strip()
|
|
49
|
+
matched = next((a for a in authors_cfg if name.casefold() in a.match_names()), None)
|
|
50
|
+
parsed.append(
|
|
51
|
+
WorkAuthor(
|
|
52
|
+
name=name,
|
|
53
|
+
orcid=f"https://orcid.org/{matched.orcid}" if matched and matched.orcid else None,
|
|
54
|
+
tracked=matched is not None,
|
|
55
|
+
)
|
|
56
|
+
)
|
|
57
|
+
return parsed
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _parse_type(value: Any, where: str) -> str:
|
|
61
|
+
if not isinstance(value, str) or value not in WORK_TYPES:
|
|
62
|
+
allowed = ", ".join(sorted(WORK_TYPES))
|
|
63
|
+
raise ConfigError(f"{where}: type {value!r} is not one of: {allowed}")
|
|
64
|
+
return value
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _opt_str(entry: dict[str, Any], key: str) -> str | None:
|
|
68
|
+
value = entry.get(key)
|
|
69
|
+
return str(value) if value is not None else None
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def load_manual(path: Path, authors_cfg: list[AuthorConfig]) -> list[Work]:
|
|
73
|
+
"""Manual records the APIs miss; absent file means none."""
|
|
74
|
+
if not path.exists():
|
|
75
|
+
return []
|
|
76
|
+
entries = load_yaml(path, list)
|
|
77
|
+
works = []
|
|
78
|
+
used_ids: set[str] = set()
|
|
79
|
+
for index, entry in enumerate(entries):
|
|
80
|
+
where = f"{path}: entry {index + 1}"
|
|
81
|
+
if not isinstance(entry, dict):
|
|
82
|
+
raise ConfigError(f"{where}: expected a mapping")
|
|
83
|
+
reject_unknown_keys(entry, MANUAL_KEYS, where)
|
|
84
|
+
title = entry.get("title")
|
|
85
|
+
if not isinstance(title, str) or not title.strip():
|
|
86
|
+
raise ConfigError(f"{where}: 'title' is required")
|
|
87
|
+
title = title.strip()
|
|
88
|
+
|
|
89
|
+
base_id = "manual-" + slugify(title)
|
|
90
|
+
work_id, suffix = base_id, 2
|
|
91
|
+
while work_id in used_ids:
|
|
92
|
+
work_id, suffix = f"{base_id}-{suffix}", suffix + 1
|
|
93
|
+
used_ids.add(work_id)
|
|
94
|
+
|
|
95
|
+
year = entry.get("year")
|
|
96
|
+
if year is not None and not isinstance(year, int):
|
|
97
|
+
raise ConfigError(f"{where}: 'year' must be an integer")
|
|
98
|
+
|
|
99
|
+
works.append(
|
|
100
|
+
Work(
|
|
101
|
+
id=work_id,
|
|
102
|
+
title=title,
|
|
103
|
+
authors=_parse_authors(entry.get("authors", []), authors_cfg, where),
|
|
104
|
+
year=year,
|
|
105
|
+
date=_opt_str(entry, "date"),
|
|
106
|
+
venue=_opt_str(entry, "venue"),
|
|
107
|
+
type=_parse_type(entry.get("type", "other"), where),
|
|
108
|
+
doi=doi_url(str(entry["doi"])) if entry.get("doi") else None,
|
|
109
|
+
abstract=_opt_str(entry, "abstract"),
|
|
110
|
+
tags=expect_str_list(entry.get("tags", []), f"{where}: 'tags'"),
|
|
111
|
+
source="manual",
|
|
112
|
+
)
|
|
113
|
+
)
|
|
114
|
+
return works
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
@dataclass
|
|
118
|
+
class Override:
|
|
119
|
+
where: str
|
|
120
|
+
match_doi: str | None = None # doi_key form
|
|
121
|
+
match_id: str | None = None
|
|
122
|
+
exclude: bool = False
|
|
123
|
+
keep_distinct: bool = False
|
|
124
|
+
patch: dict[str, Any] = field(default_factory=dict)
|
|
125
|
+
matched: bool = False
|
|
126
|
+
changed: bool = False
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def load_overrides(path: Path) -> list[Override]:
|
|
130
|
+
if not path.exists():
|
|
131
|
+
return []
|
|
132
|
+
entries = load_yaml(path, list)
|
|
133
|
+
overrides = []
|
|
134
|
+
for index, entry in enumerate(entries):
|
|
135
|
+
where = f"{path}: entry {index + 1}"
|
|
136
|
+
if not isinstance(entry, dict):
|
|
137
|
+
raise ConfigError(f"{where}: expected a mapping")
|
|
138
|
+
if ("doi" in entry) == ("id" in entry):
|
|
139
|
+
raise ConfigError(f"{where}: needs exactly one of 'doi' or 'id' to match on")
|
|
140
|
+
patch = {
|
|
141
|
+
k: v for k, v in entry.items() if k not in {"doi", "id", "exclude", "keep_distinct"}
|
|
142
|
+
}
|
|
143
|
+
reject_unknown_keys(patch, PATCH_KEYS, where, noun="fields")
|
|
144
|
+
overrides.append(
|
|
145
|
+
Override(
|
|
146
|
+
where=where,
|
|
147
|
+
match_doi=doi_key(str(entry["doi"])) if "doi" in entry else None,
|
|
148
|
+
match_id=str(entry["id"]) if "id" in entry else None,
|
|
149
|
+
exclude=bool(entry.get("exclude", False)),
|
|
150
|
+
keep_distinct=bool(entry.get("keep_distinct", False)),
|
|
151
|
+
patch=patch,
|
|
152
|
+
)
|
|
153
|
+
)
|
|
154
|
+
return overrides
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def load_tags(path: Path) -> dict[str, list[str]]:
|
|
158
|
+
"""Mapping of tag name to list of DOI/id references."""
|
|
159
|
+
if not path.exists():
|
|
160
|
+
return {}
|
|
161
|
+
data = load_yaml(path, dict)
|
|
162
|
+
tags: dict[str, list[str]] = {}
|
|
163
|
+
for name, refs in data.items():
|
|
164
|
+
where = f"{path}: tag {name!r}"
|
|
165
|
+
if not isinstance(name, str):
|
|
166
|
+
raise ConfigError(f"{where}: tag names must be strings")
|
|
167
|
+
tags[name] = expect_str_list(refs, where)
|
|
168
|
+
return tags
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def _iter_matches(overrides: list[Override], works: list[Work]) -> Iterator[tuple[Override, Work]]:
|
|
172
|
+
"""Pair each override with the works it hits, in override file order.
|
|
173
|
+
|
|
174
|
+
Indexes the works once, and flips `matched` in this one place so the
|
|
175
|
+
stale-override detection cannot drift between callers. Pre-dedup, several
|
|
176
|
+
works can share a DOI, so the indexes map to lists.
|
|
177
|
+
"""
|
|
178
|
+
by_doi: dict[str, list[Work]] = {}
|
|
179
|
+
by_id: dict[str, list[Work]] = {}
|
|
180
|
+
for work in works:
|
|
181
|
+
if work.doi_key:
|
|
182
|
+
by_doi.setdefault(work.doi_key, []).append(work)
|
|
183
|
+
by_id.setdefault(work.id, []).append(work)
|
|
184
|
+
for override in overrides:
|
|
185
|
+
if override.match_doi is not None:
|
|
186
|
+
hits = by_doi.get(override.match_doi, [])
|
|
187
|
+
else:
|
|
188
|
+
hits = by_id.get(override.match_id or "", [])
|
|
189
|
+
for work in hits:
|
|
190
|
+
override.matched = True
|
|
191
|
+
yield override, work
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def mark_keep_distinct(works: list[Work], overrides: list[Override]) -> None:
|
|
195
|
+
"""Applied before title clustering, ahead of the override patch stage."""
|
|
196
|
+
keep = [o for o in overrides if o.keep_distinct]
|
|
197
|
+
for _, work in _iter_matches(keep, works):
|
|
198
|
+
work.keep_distinct = True
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
# Expected value shapes for the scalar patch fields; a mistyped value must
|
|
202
|
+
# fail as a ConfigError at apply time, not as a TypeError deep in the
|
|
203
|
+
# pipeline (sorting, clustering) where the file/entry context is lost.
|
|
204
|
+
_SCALAR_PATCH_TYPES: dict[str, tuple[str, tuple[type, ...]]] = {
|
|
205
|
+
"title": ("a string", (str,)),
|
|
206
|
+
"venue": ("a string or null", (str, type(None))),
|
|
207
|
+
"year": ("an integer or null", (int, type(None))),
|
|
208
|
+
"cited_by_count": ("an integer", (int,)),
|
|
209
|
+
"abstract": ("a string or null", (str, type(None))),
|
|
210
|
+
"is_retracted": ("a boolean", (bool,)),
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def _patch_work(
|
|
215
|
+
work: Work, patch: dict[str, Any], authors_cfg: list[AuthorConfig], where: str
|
|
216
|
+
) -> None:
|
|
217
|
+
for key, value in patch.items():
|
|
218
|
+
if key == "authors":
|
|
219
|
+
work.authors = _parse_authors(value, authors_cfg, where)
|
|
220
|
+
elif key == "open_access":
|
|
221
|
+
if isinstance(value, dict):
|
|
222
|
+
value = value.get("url")
|
|
223
|
+
work.open_access_url = str(value) if value else None
|
|
224
|
+
elif key == "doi":
|
|
225
|
+
work.doi = doi_url(str(value)) if value else None
|
|
226
|
+
elif key == "type":
|
|
227
|
+
work.type = _parse_type(value, where)
|
|
228
|
+
elif key == "tags":
|
|
229
|
+
work.tags = expect_str_list(value, f"{where}: 'tags'")
|
|
230
|
+
elif key == "date":
|
|
231
|
+
# YAML parses unquoted ISO dates as date objects; accept both.
|
|
232
|
+
if value is not None and not isinstance(value, (str, datetime.date)):
|
|
233
|
+
raise ConfigError(f"{where}: 'date' must be an ISO date string or null")
|
|
234
|
+
work.date = str(value) if value is not None else None
|
|
235
|
+
else:
|
|
236
|
+
description, types = _SCALAR_PATCH_TYPES[key]
|
|
237
|
+
if not isinstance(value, types) or (isinstance(value, bool) and bool not in types):
|
|
238
|
+
raise ConfigError(f"{where}: '{key}' must be {description}")
|
|
239
|
+
setattr(work, key, value)
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
def apply_overrides(
|
|
243
|
+
works: list[Work], overrides: list[Override], authors_cfg: list[AuthorConfig]
|
|
244
|
+
) -> tuple[list[Work], int]:
|
|
245
|
+
"""Patch or exclude merged records; returns (kept, excluded_count)."""
|
|
246
|
+
excluded: set[int] = set()
|
|
247
|
+
for override, work in _iter_matches(overrides, works):
|
|
248
|
+
if override.exclude:
|
|
249
|
+
excluded.add(id(work))
|
|
250
|
+
elif override.patch:
|
|
251
|
+
# Compare against the pre-patch record: comparing the override
|
|
252
|
+
# against the output would be circular, the output already has
|
|
253
|
+
# the override applied and every entry would look load-bearing.
|
|
254
|
+
# Deep copy so the check stays honest even if a patch branch
|
|
255
|
+
# ever mutates a list in place instead of reassigning it.
|
|
256
|
+
before = copy.deepcopy(work)
|
|
257
|
+
_patch_work(work, override.patch, authors_cfg, override.where)
|
|
258
|
+
if work != before:
|
|
259
|
+
override.changed = True
|
|
260
|
+
return [w for w in works if id(w) not in excluded], len(excluded)
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def unmatched_overrides(overrides: list[Override]) -> list[str]:
|
|
264
|
+
"""Locations of overrides that touched nothing (stale DOI or id)."""
|
|
265
|
+
return [o.where for o in overrides if not o.matched]
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
def redundant_overrides(overrides: list[Override]) -> list[str]:
|
|
269
|
+
"""Locations of field patches that no longer change anything.
|
|
270
|
+
|
|
271
|
+
Upstream caught up with the correction. Redundant is information, not
|
|
272
|
+
an instruction to delete: an override may stay as insurance against the
|
|
273
|
+
upstream regressing again. keep_distinct-only entries drop out via the
|
|
274
|
+
empty-patch check; `exclude` needs its explicit guard because an
|
|
275
|
+
exclude entry carrying patch fields never runs them.
|
|
276
|
+
"""
|
|
277
|
+
return [o.where for o in overrides if o.matched and o.patch and not o.exclude and not o.changed]
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
def apply_tags(works: list[Work], tags: dict[str, list[str]]) -> list[str]:
|
|
281
|
+
"""Attach curated tags; returns unmatched references for warnings.
|
|
282
|
+
|
|
283
|
+
References are matched as DOIs when they look like one (URL or 10.x
|
|
284
|
+
form), else as record ids. Each record's final tag list is sorted for
|
|
285
|
+
deterministic output.
|
|
286
|
+
"""
|
|
287
|
+
by_doi = {w.doi_key: w for w in works if w.doi_key}
|
|
288
|
+
by_id = {w.id: w for w in works}
|
|
289
|
+
unmatched = []
|
|
290
|
+
for name, refs in tags.items():
|
|
291
|
+
for ref in refs:
|
|
292
|
+
key = doi_key(ref)
|
|
293
|
+
work = by_doi.get(key) if key.startswith("10.") else by_id.get(ref)
|
|
294
|
+
if work is None:
|
|
295
|
+
unmatched.append(f"tag {name!r}: {ref}")
|
|
296
|
+
elif name not in work.tags:
|
|
297
|
+
work.tags.append(name)
|
|
298
|
+
for work in works:
|
|
299
|
+
work.tags.sort()
|
|
300
|
+
return unmatched
|