erga 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
erga/__init__.py ADDED
@@ -0,0 +1,7 @@
1
+ """erga: keep a website's academic publications list current.
2
+
3
+ Fetches works from OpenAlex, deduplicates them across registrars, applies
4
+ maintainer curation, and writes canonical JSON for any static site to render.
5
+ """
6
+
7
+ __version__ = "0.1.0"
erga/cli.py ADDED
@@ -0,0 +1,96 @@
1
+ """Console entry point: `erga build` and `erga verify`."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import os
7
+ import sys
8
+ from pathlib import Path
9
+
10
+ from erga import __version__
11
+ from erga.config import Config, load_config
12
+ from erga.crossref import CrossrefClient
13
+ from erga.errors import ErgaError
14
+ from erga.http import UrlTransport
15
+ from erga.openalex import OpenAlexClient
16
+ from erga.pipeline import build
17
+ from erga.verify import verify_report
18
+
19
+
20
+ def _user_agent(mailto: str) -> str:
21
+ return f"erga/{__version__} (https://github.com/belalik/erga; mailto:{mailto})"
22
+
23
+
24
+ def _clients(config: Config) -> tuple[OpenAlexClient, CrossrefClient]:
25
+ # One transport for both APIs. The key never touches disk: read from the
26
+ # configured env var, passed as a query parameter, nothing else. The
27
+ # transport redacts it from network-error text (requests embeds the full
28
+ # request URL in its exception messages).
29
+ api_key = os.environ.get(config.api_key_env) or None
30
+ if not api_key:
31
+ print(
32
+ f"erga: note: {config.api_key_env} not set; using OpenAlex's keyless per-IP quota",
33
+ file=sys.stderr,
34
+ )
35
+ transport = UrlTransport(_user_agent(config.mailto), secrets=[api_key] if api_key else [])
36
+ return (
37
+ OpenAlexClient(transport, mailto=config.mailto, api_key=api_key),
38
+ CrossrefClient(transport, mailto=config.mailto),
39
+ )
40
+
41
+
42
+ def _print_warnings(warnings: list[str]) -> None:
43
+ for warning in warnings:
44
+ print(f"erga: warning: {warning}", file=sys.stderr)
45
+
46
+
47
+ def _run_build(config: Config, dry_run: bool) -> int:
48
+ openalex, crossref = _clients(config)
49
+ stats = build(config, openalex, crossref, dry_run=dry_run)
50
+ _print_warnings(stats.warnings)
51
+ if dry_run:
52
+ print(f"dry run: {stats.summary()}")
53
+ print(f"would write {stats.total} works to {config.output_path}")
54
+ else:
55
+ print(f"wrote {config.output_path} ({stats.total} works)")
56
+ return 0
57
+
58
+
59
+ def _run_verify(config: Config) -> int:
60
+ openalex, _ = _clients(config)
61
+ report, warnings = verify_report(config, openalex)
62
+ print(report, end="")
63
+ _print_warnings(warnings)
64
+ return 0
65
+
66
+
67
+ def main(argv: list[str] | None = None) -> int:
68
+ parser = argparse.ArgumentParser(
69
+ prog="erga",
70
+ description="Keep a website's academic publications list current.",
71
+ )
72
+ parser.add_argument("--version", action="version", version=f"erga {__version__}")
73
+ subparsers = parser.add_subparsers(dest="command", required=True)
74
+
75
+ build_parser = subparsers.add_parser("build", help="run the pipeline and write the JSON")
76
+ build_parser.add_argument("--config", type=Path, default=Path("erga.yml"))
77
+ build_parser.add_argument(
78
+ "--dry-run", action="store_true", help="print a summary without writing"
79
+ )
80
+
81
+ verify_parser = subparsers.add_parser("verify", help="author-disambiguation report")
82
+ verify_parser.add_argument("--config", type=Path, default=Path("erga.yml"))
83
+
84
+ args = parser.parse_args(argv)
85
+ try:
86
+ config = load_config(args.config)
87
+ if args.command == "build":
88
+ return _run_build(config, args.dry_run)
89
+ return _run_verify(config)
90
+ except ErgaError as exc:
91
+ print(f"erga: {exc}", file=sys.stderr)
92
+ return 1
93
+
94
+
95
+ if __name__ == "__main__":
96
+ sys.exit(main())
erga/config.py ADDED
@@ -0,0 +1,139 @@
1
+ """Configuration loading and validation (requirements section 5)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from dataclasses import dataclass, field
7
+ from pathlib import Path
8
+ from typing import Any
9
+
10
+ import yaml
11
+
12
+ from erga.errors import ConfigError
13
+ from erga.model import normalize_orcid
14
+
15
+ _ORCID_RE = re.compile(r"^\d{4}-\d{4}-\d{4}-\d{3}[\dX]$")
16
+ _OPENALEX_AUTHOR_RE = re.compile(r"^A\d+$")
17
+
18
+
19
+ @dataclass
20
+ class AuthorConfig:
21
+ name: str
22
+ orcid: str | None = None
23
+ openalex_id: str | None = None
24
+ aliases: list[str] = field(default_factory=list)
25
+
26
+ def match_names(self) -> set[str]:
27
+ """Casefolded name and aliases, for matching manual entries."""
28
+ return {n.casefold().strip() for n in [self.name, *self.aliases]}
29
+
30
+ @property
31
+ def tracking_only(self) -> bool:
32
+ """No registrar ids: contributes names to tracking, fetches nothing."""
33
+ return self.orcid is None and self.openalex_id is None
34
+
35
+
36
+ @dataclass
37
+ class Config:
38
+ mailto: str
39
+ authors: list[AuthorConfig]
40
+ api_key_env: str = "OPENALEX_API_KEY"
41
+ include_xpac: bool = False
42
+ output_path: Path = Path("publications.json")
43
+ manual_path: Path = Path("manual.yml")
44
+ overrides_path: Path = Path("overrides.yml")
45
+ tags_path: Path = Path("tags.yml")
46
+
47
+
48
+ def load_yaml(path: Path, expect: type) -> Any:
49
+ """Parse a YAML file and check its top-level type."""
50
+ try:
51
+ with open(path, encoding="utf-8") as fh:
52
+ data = yaml.safe_load(fh)
53
+ except OSError as exc:
54
+ raise ConfigError(f"{path}: {exc.strerror or exc}") from exc
55
+ except yaml.YAMLError as exc:
56
+ raise ConfigError(f"{path}: invalid YAML: {exc}") from exc
57
+ if data is None:
58
+ data = expect()
59
+ if not isinstance(data, expect):
60
+ raise ConfigError(f"{path}: expected a {expect.__name__} at top level")
61
+ return data
62
+
63
+
64
+ def reject_unknown_keys(
65
+ mapping: dict[str, Any], allowed: set[str], where: str, noun: str = "keys"
66
+ ) -> None:
67
+ unknown = set(mapping) - allowed
68
+ if unknown:
69
+ raise ConfigError(f"{where}: unknown {noun}: {', '.join(sorted(unknown))}")
70
+
71
+
72
+ def expect_str_list(value: Any, where: str) -> list[str]:
73
+ if not isinstance(value, list) or not all(isinstance(item, str) for item in value):
74
+ raise ConfigError(f"{where}: must be a list of strings")
75
+ return list(value)
76
+
77
+
78
+ def _parse_author(entry: Any, path: Path, index: int) -> AuthorConfig:
79
+ where = f"{path}: authors[{index}]"
80
+ if not isinstance(entry, dict):
81
+ raise ConfigError(f"{where}: expected a mapping")
82
+ reject_unknown_keys(entry, {"name", "orcid", "openalex_id", "aliases"}, where)
83
+ name = entry.get("name")
84
+ if not isinstance(name, str) or not name.strip():
85
+ raise ConfigError(f"{where}: 'name' is required")
86
+ orcid = entry.get("orcid")
87
+ if orcid is not None:
88
+ orcid = normalize_orcid(str(orcid))
89
+ if not _ORCID_RE.match(orcid):
90
+ raise ConfigError(f"{where}: invalid ORCID iD {orcid!r}")
91
+ openalex_id = entry.get("openalex_id")
92
+ if openalex_id is not None:
93
+ openalex_id = str(openalex_id).strip()
94
+ if not _OPENALEX_AUTHOR_RE.match(openalex_id):
95
+ raise ConfigError(f"{where}: invalid OpenAlex author id {openalex_id!r}")
96
+ # Neither id is fine: the entry contributes its names to the tracked
97
+ # flag but resolves and fetches nothing (authors without any registrar
98
+ # identity, or whose works OpenAlex misassigns to a conflated profile).
99
+ aliases = expect_str_list(entry.get("aliases", []), f"{where}: 'aliases'")
100
+ return AuthorConfig(name=name.strip(), orcid=orcid, openalex_id=openalex_id, aliases=aliases)
101
+
102
+
103
+ def _section(data: dict[str, Any], key: str, path: Path, allowed: set[str]) -> dict[str, Any]:
104
+ section = data.get(key) or {}
105
+ if not isinstance(section, dict):
106
+ raise ConfigError(f"{path}: '{key}' must be a mapping")
107
+ reject_unknown_keys(section, allowed, f"{path}: {key}")
108
+ return section
109
+
110
+
111
+ def load_config(path: Path) -> Config:
112
+ """Load and validate erga.yml; relative paths resolve against its directory."""
113
+ data = load_yaml(path, dict)
114
+ base = path.resolve().parent
115
+ reject_unknown_keys(data, {"mailto", "authors", "openalex", "output", "curation"}, str(path))
116
+
117
+ mailto = data.get("mailto")
118
+ if not isinstance(mailto, str) or "@" not in mailto:
119
+ raise ConfigError(f"{path}: 'mailto' is required (identifies requests to the APIs)")
120
+
121
+ raw_authors = data.get("authors")
122
+ if not isinstance(raw_authors, list) or not raw_authors:
123
+ raise ConfigError(f"{path}: 'authors' must be a non-empty list")
124
+ authors = [_parse_author(entry, path, i) for i, entry in enumerate(raw_authors)]
125
+
126
+ openalex = _section(data, "openalex", path, {"api_key_env", "include_xpac"})
127
+ output = _section(data, "output", path, {"path"})
128
+ curation = _section(data, "curation", path, {"manual", "overrides", "tags"})
129
+
130
+ return Config(
131
+ mailto=mailto.strip(),
132
+ authors=authors,
133
+ api_key_env=str(openalex.get("api_key_env", "OPENALEX_API_KEY")),
134
+ include_xpac=bool(openalex.get("include_xpac", False)),
135
+ output_path=base / str(output.get("path", "publications.json")),
136
+ manual_path=base / str(curation.get("manual", "manual.yml")),
137
+ overrides_path=base / str(curation.get("overrides", "overrides.yml")),
138
+ tags_path=base / str(curation.get("tags", "tags.yml")),
139
+ )
erga/crossref.py ADDED
@@ -0,0 +1,50 @@
1
+ """Crossref venue lookup for the backfill stage.
2
+
3
+ Crossref still operates a mailto polite pool, so the mailto rides along both
4
+ as a query parameter and in the transport's User-Agent.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import time
10
+ import urllib.parse
11
+ from collections.abc import Callable
12
+
13
+ from erga.http import Pacer, Transport, request_with_retry
14
+
15
+ CROSSREF_BASE = "https://api.crossref.org/works/"
16
+
17
+
18
+ class CrossrefClient:
19
+ def __init__(
20
+ self,
21
+ transport: Transport,
22
+ *,
23
+ mailto: str,
24
+ delay: float = 0.5,
25
+ sleep: Callable[[float], None] = time.sleep,
26
+ ) -> None:
27
+ self._transport = transport
28
+ self._mailto = mailto
29
+ self._sleep = sleep
30
+ self._pacer = Pacer(delay, sleep)
31
+
32
+ def venue_for_doi(self, doi: str) -> str | None:
33
+ """Container title for a bare DOI, or None.
34
+
35
+ DataCite DOIs 404 here and are skipped silently, as is any record
36
+ without a container title. Raises FetchError only when retryable
37
+ failures persist (the caller stops backfilling, keeping nulls).
38
+ """
39
+ self._pacer.wait()
40
+ url = CROSSREF_BASE + urllib.parse.quote(doi, safe="")
41
+ response = request_with_retry(
42
+ self._transport, url, {"mailto": self._mailto}, sleep=self._sleep
43
+ )
44
+ if response.status_code != 200 or not isinstance(response.data, dict):
45
+ return None
46
+ message = response.data.get("message") or {}
47
+ titles = message.get("container-title") or []
48
+ if titles and isinstance(titles[0], str) and titles[0].strip():
49
+ return titles[0].strip()
50
+ return None
erga/curation.py ADDED
@@ -0,0 +1,300 @@
1
+ """Curation files: manual entries, overrides, tags.
2
+
3
+ All three survive every refresh; a missing file means "none". Typos fail
4
+ loudly: curation is the maintainer's reviewable artifact, and a silently
5
+ skipped patch is worse than an aborted run.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import copy
11
+ import datetime
12
+ from collections.abc import Iterator
13
+ from dataclasses import dataclass, field
14
+ from pathlib import Path
15
+ from typing import Any
16
+
17
+ from erga.config import AuthorConfig, expect_str_list, load_yaml, reject_unknown_keys
18
+ from erga.errors import ConfigError
19
+ from erga.model import WORK_TYPES, Work, WorkAuthor, doi_key, doi_url, slugify
20
+
21
+ MANUAL_KEYS = {"title", "authors", "venue", "year", "date", "doi", "type", "tags", "abstract"}
22
+ PATCH_KEYS = {
23
+ "title",
24
+ "authors",
25
+ "venue",
26
+ "year",
27
+ "date",
28
+ "doi",
29
+ "type",
30
+ "cited_by_count",
31
+ "abstract",
32
+ "open_access",
33
+ "tags",
34
+ "is_retracted",
35
+ }
36
+
37
+
38
+ def _parse_authors(value: Any, authors_cfg: list[AuthorConfig], where: str) -> list[WorkAuthor]:
39
+ """Author strings matched to configured authors by name/alias."""
40
+ if isinstance(value, str):
41
+ value = [value]
42
+ if not isinstance(value, list):
43
+ raise ConfigError(f"{where}: 'authors' must be a string or a list of strings")
44
+ parsed = []
45
+ for name in value:
46
+ if not isinstance(name, str) or not name.strip():
47
+ raise ConfigError(f"{where}: 'authors' entries must be non-empty strings")
48
+ name = name.strip()
49
+ matched = next((a for a in authors_cfg if name.casefold() in a.match_names()), None)
50
+ parsed.append(
51
+ WorkAuthor(
52
+ name=name,
53
+ orcid=f"https://orcid.org/{matched.orcid}" if matched and matched.orcid else None,
54
+ tracked=matched is not None,
55
+ )
56
+ )
57
+ return parsed
58
+
59
+
60
+ def _parse_type(value: Any, where: str) -> str:
61
+ if not isinstance(value, str) or value not in WORK_TYPES:
62
+ allowed = ", ".join(sorted(WORK_TYPES))
63
+ raise ConfigError(f"{where}: type {value!r} is not one of: {allowed}")
64
+ return value
65
+
66
+
67
+ def _opt_str(entry: dict[str, Any], key: str) -> str | None:
68
+ value = entry.get(key)
69
+ return str(value) if value is not None else None
70
+
71
+
72
+ def load_manual(path: Path, authors_cfg: list[AuthorConfig]) -> list[Work]:
73
+ """Manual records the APIs miss; absent file means none."""
74
+ if not path.exists():
75
+ return []
76
+ entries = load_yaml(path, list)
77
+ works = []
78
+ used_ids: set[str] = set()
79
+ for index, entry in enumerate(entries):
80
+ where = f"{path}: entry {index + 1}"
81
+ if not isinstance(entry, dict):
82
+ raise ConfigError(f"{where}: expected a mapping")
83
+ reject_unknown_keys(entry, MANUAL_KEYS, where)
84
+ title = entry.get("title")
85
+ if not isinstance(title, str) or not title.strip():
86
+ raise ConfigError(f"{where}: 'title' is required")
87
+ title = title.strip()
88
+
89
+ base_id = "manual-" + slugify(title)
90
+ work_id, suffix = base_id, 2
91
+ while work_id in used_ids:
92
+ work_id, suffix = f"{base_id}-{suffix}", suffix + 1
93
+ used_ids.add(work_id)
94
+
95
+ year = entry.get("year")
96
+ if year is not None and not isinstance(year, int):
97
+ raise ConfigError(f"{where}: 'year' must be an integer")
98
+
99
+ works.append(
100
+ Work(
101
+ id=work_id,
102
+ title=title,
103
+ authors=_parse_authors(entry.get("authors", []), authors_cfg, where),
104
+ year=year,
105
+ date=_opt_str(entry, "date"),
106
+ venue=_opt_str(entry, "venue"),
107
+ type=_parse_type(entry.get("type", "other"), where),
108
+ doi=doi_url(str(entry["doi"])) if entry.get("doi") else None,
109
+ abstract=_opt_str(entry, "abstract"),
110
+ tags=expect_str_list(entry.get("tags", []), f"{where}: 'tags'"),
111
+ source="manual",
112
+ )
113
+ )
114
+ return works
115
+
116
+
117
+ @dataclass
118
+ class Override:
119
+ where: str
120
+ match_doi: str | None = None # doi_key form
121
+ match_id: str | None = None
122
+ exclude: bool = False
123
+ keep_distinct: bool = False
124
+ patch: dict[str, Any] = field(default_factory=dict)
125
+ matched: bool = False
126
+ changed: bool = False
127
+
128
+
129
+ def load_overrides(path: Path) -> list[Override]:
130
+ if not path.exists():
131
+ return []
132
+ entries = load_yaml(path, list)
133
+ overrides = []
134
+ for index, entry in enumerate(entries):
135
+ where = f"{path}: entry {index + 1}"
136
+ if not isinstance(entry, dict):
137
+ raise ConfigError(f"{where}: expected a mapping")
138
+ if ("doi" in entry) == ("id" in entry):
139
+ raise ConfigError(f"{where}: needs exactly one of 'doi' or 'id' to match on")
140
+ patch = {
141
+ k: v for k, v in entry.items() if k not in {"doi", "id", "exclude", "keep_distinct"}
142
+ }
143
+ reject_unknown_keys(patch, PATCH_KEYS, where, noun="fields")
144
+ overrides.append(
145
+ Override(
146
+ where=where,
147
+ match_doi=doi_key(str(entry["doi"])) if "doi" in entry else None,
148
+ match_id=str(entry["id"]) if "id" in entry else None,
149
+ exclude=bool(entry.get("exclude", False)),
150
+ keep_distinct=bool(entry.get("keep_distinct", False)),
151
+ patch=patch,
152
+ )
153
+ )
154
+ return overrides
155
+
156
+
157
+ def load_tags(path: Path) -> dict[str, list[str]]:
158
+ """Mapping of tag name to list of DOI/id references."""
159
+ if not path.exists():
160
+ return {}
161
+ data = load_yaml(path, dict)
162
+ tags: dict[str, list[str]] = {}
163
+ for name, refs in data.items():
164
+ where = f"{path}: tag {name!r}"
165
+ if not isinstance(name, str):
166
+ raise ConfigError(f"{where}: tag names must be strings")
167
+ tags[name] = expect_str_list(refs, where)
168
+ return tags
169
+
170
+
171
+ def _iter_matches(overrides: list[Override], works: list[Work]) -> Iterator[tuple[Override, Work]]:
172
+ """Pair each override with the works it hits, in override file order.
173
+
174
+ Indexes the works once, and flips `matched` in this one place so the
175
+ stale-override detection cannot drift between callers. Pre-dedup, several
176
+ works can share a DOI, so the indexes map to lists.
177
+ """
178
+ by_doi: dict[str, list[Work]] = {}
179
+ by_id: dict[str, list[Work]] = {}
180
+ for work in works:
181
+ if work.doi_key:
182
+ by_doi.setdefault(work.doi_key, []).append(work)
183
+ by_id.setdefault(work.id, []).append(work)
184
+ for override in overrides:
185
+ if override.match_doi is not None:
186
+ hits = by_doi.get(override.match_doi, [])
187
+ else:
188
+ hits = by_id.get(override.match_id or "", [])
189
+ for work in hits:
190
+ override.matched = True
191
+ yield override, work
192
+
193
+
194
+ def mark_keep_distinct(works: list[Work], overrides: list[Override]) -> None:
195
+ """Applied before title clustering, ahead of the override patch stage."""
196
+ keep = [o for o in overrides if o.keep_distinct]
197
+ for _, work in _iter_matches(keep, works):
198
+ work.keep_distinct = True
199
+
200
+
201
+ # Expected value shapes for the scalar patch fields; a mistyped value must
202
+ # fail as a ConfigError at apply time, not as a TypeError deep in the
203
+ # pipeline (sorting, clustering) where the file/entry context is lost.
204
+ _SCALAR_PATCH_TYPES: dict[str, tuple[str, tuple[type, ...]]] = {
205
+ "title": ("a string", (str,)),
206
+ "venue": ("a string or null", (str, type(None))),
207
+ "year": ("an integer or null", (int, type(None))),
208
+ "cited_by_count": ("an integer", (int,)),
209
+ "abstract": ("a string or null", (str, type(None))),
210
+ "is_retracted": ("a boolean", (bool,)),
211
+ }
212
+
213
+
214
+ def _patch_work(
215
+ work: Work, patch: dict[str, Any], authors_cfg: list[AuthorConfig], where: str
216
+ ) -> None:
217
+ for key, value in patch.items():
218
+ if key == "authors":
219
+ work.authors = _parse_authors(value, authors_cfg, where)
220
+ elif key == "open_access":
221
+ if isinstance(value, dict):
222
+ value = value.get("url")
223
+ work.open_access_url = str(value) if value else None
224
+ elif key == "doi":
225
+ work.doi = doi_url(str(value)) if value else None
226
+ elif key == "type":
227
+ work.type = _parse_type(value, where)
228
+ elif key == "tags":
229
+ work.tags = expect_str_list(value, f"{where}: 'tags'")
230
+ elif key == "date":
231
+ # YAML parses unquoted ISO dates as date objects; accept both.
232
+ if value is not None and not isinstance(value, (str, datetime.date)):
233
+ raise ConfigError(f"{where}: 'date' must be an ISO date string or null")
234
+ work.date = str(value) if value is not None else None
235
+ else:
236
+ description, types = _SCALAR_PATCH_TYPES[key]
237
+ if not isinstance(value, types) or (isinstance(value, bool) and bool not in types):
238
+ raise ConfigError(f"{where}: '{key}' must be {description}")
239
+ setattr(work, key, value)
240
+
241
+
242
+ def apply_overrides(
243
+ works: list[Work], overrides: list[Override], authors_cfg: list[AuthorConfig]
244
+ ) -> tuple[list[Work], int]:
245
+ """Patch or exclude merged records; returns (kept, excluded_count)."""
246
+ excluded: set[int] = set()
247
+ for override, work in _iter_matches(overrides, works):
248
+ if override.exclude:
249
+ excluded.add(id(work))
250
+ elif override.patch:
251
+ # Compare against the pre-patch record: comparing the override
252
+ # against the output would be circular, the output already has
253
+ # the override applied and every entry would look load-bearing.
254
+ # Deep copy so the check stays honest even if a patch branch
255
+ # ever mutates a list in place instead of reassigning it.
256
+ before = copy.deepcopy(work)
257
+ _patch_work(work, override.patch, authors_cfg, override.where)
258
+ if work != before:
259
+ override.changed = True
260
+ return [w for w in works if id(w) not in excluded], len(excluded)
261
+
262
+
263
+ def unmatched_overrides(overrides: list[Override]) -> list[str]:
264
+ """Locations of overrides that touched nothing (stale DOI or id)."""
265
+ return [o.where for o in overrides if not o.matched]
266
+
267
+
268
+ def redundant_overrides(overrides: list[Override]) -> list[str]:
269
+ """Locations of field patches that no longer change anything.
270
+
271
+ Upstream caught up with the correction. Redundant is information, not
272
+ an instruction to delete: an override may stay as insurance against the
273
+ upstream regressing again. keep_distinct-only entries drop out via the
274
+ empty-patch check; `exclude` needs its explicit guard because an
275
+ exclude entry carrying patch fields never runs them.
276
+ """
277
+ return [o.where for o in overrides if o.matched and o.patch and not o.exclude and not o.changed]
278
+
279
+
280
+ def apply_tags(works: list[Work], tags: dict[str, list[str]]) -> list[str]:
281
+ """Attach curated tags; returns unmatched references for warnings.
282
+
283
+ References are matched as DOIs when they look like one (URL or 10.x
284
+ form), else as record ids. Each record's final tag list is sorted for
285
+ deterministic output.
286
+ """
287
+ by_doi = {w.doi_key: w for w in works if w.doi_key}
288
+ by_id = {w.id: w for w in works}
289
+ unmatched = []
290
+ for name, refs in tags.items():
291
+ for ref in refs:
292
+ key = doi_key(ref)
293
+ work = by_doi.get(key) if key.startswith("10.") else by_id.get(ref)
294
+ if work is None:
295
+ unmatched.append(f"tag {name!r}: {ref}")
296
+ elif name not in work.tags:
297
+ work.tags.append(name)
298
+ for work in works:
299
+ work.tags.sort()
300
+ return unmatched