zotkit 0.4.1__tar.gz → 0.4.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: zotkit
3
- Version: 0.4.1
3
+ Version: 0.4.2
4
4
  Summary: Headless Zotero library management: Web API CRUD plus direct WebDAV attachment upload/download — no desktop app required
5
5
  Author: Shawn
6
6
  License-Expression: MIT
@@ -125,9 +125,20 @@ zotkit lint field:physics topic:new-idea # offline tag check
125
125
  ```
126
126
 
127
127
  `--arxiv` takes ids or abs/pdf URLs (several, space- or comma-separated) and maps
128
- the full record (all authors, abstract, date, DOI, `preprint` item type); `--doi`
128
+ the full record (all authors, abstract, date, DOI); `--doi`
129
129
  maps CrossRef records (journal articles, conference papers, books, chapters, …)
130
- and refuses to guess on CrossRef types it doesn't know. Both accept `--collection`
130
+ and refuses to guess on CrossRef types it doesn't know.
131
+
132
+ **Version of record**: when arXiv reports a *journal* DOI (the paper was formally
133
+ published — as opposed to arXiv's own `10.48550/*` DataCite DOI), `--arxiv`
134
+ builds the journal record from CrossRef instead of a `preprint`: proper item
135
+ type, venue, volume/pages, formal date. The arXiv identity is kept — `arXiv: <id>`
136
+ goes in Extra, the `url` stays the open-access abs page (the journal link lives in
137
+ the DOI field), the arXiv abstract fills in when CrossRef has none, and the PDF
138
+ still comes from arXiv. If the CrossRef lookup fails, the item falls back to the
139
+ preprint record with a warning rather than failing. Items whose abstract zotkit
140
+ wrote also carry an `abstract-source: arxiv|crossref` line in Extra, naming where
141
+ it actually came from. Both accept `--collection`
131
142
  and `--tags`, and `--no-pdf` skips the arXiv PDFs. Rate limiting is built into the
132
143
  request layer — batches use one arXiv metadata request and space PDF downloads
133
144
  per arXiv's terms of use, so callers (humans or agents) never pace themselves. A
@@ -105,9 +105,20 @@ zotkit lint field:physics topic:new-idea # offline tag check
105
105
  ```
106
106
 
107
107
  `--arxiv` takes ids or abs/pdf URLs (several, space- or comma-separated) and maps
108
- the full record (all authors, abstract, date, DOI, `preprint` item type); `--doi`
108
+ the full record (all authors, abstract, date, DOI); `--doi`
109
109
  maps CrossRef records (journal articles, conference papers, books, chapters, …)
110
- and refuses to guess on CrossRef types it doesn't know. Both accept `--collection`
110
+ and refuses to guess on CrossRef types it doesn't know.
111
+
112
+ **Version of record**: when arXiv reports a *journal* DOI (the paper was formally
113
+ published — as opposed to arXiv's own `10.48550/*` DataCite DOI), `--arxiv`
114
+ builds the journal record from CrossRef instead of a `preprint`: proper item
115
+ type, venue, volume/pages, formal date. The arXiv identity is kept — `arXiv: <id>`
116
+ goes in Extra, the `url` stays the open-access abs page (the journal link lives in
117
+ the DOI field), the arXiv abstract fills in when CrossRef has none, and the PDF
118
+ still comes from arXiv. If the CrossRef lookup fails, the item falls back to the
119
+ preprint record with a warning rather than failing. Items whose abstract zotkit
120
+ wrote also carry an `abstract-source: arxiv|crossref` line in Extra, naming where
121
+ it actually came from. Both accept `--collection`
111
122
  and `--tags`, and `--no-pdf` skips the arXiv PDFs. Rate limiting is built into the
112
123
  request layer — batches use one arXiv metadata request and space PDF downloads
113
124
  per arXiv's terms of use, so callers (humans or agents) never pace themselves. A
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "zotkit"
7
- version = "0.4.1"
7
+ version = "0.4.2"
8
8
  description = "Headless Zotero library management: Web API CRUD plus direct WebDAV attachment upload/download — no desktop app required"
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -2,4 +2,4 @@
2
2
  from .core import Conventions, TagConventionError, Zot, lint_tags, load_conventions
3
3
 
4
4
  __all__ = ["Zot", "lint_tags", "load_conventions", "Conventions", "TagConventionError"]
5
- __version__ = "0.4.1"
5
+ __version__ = "0.4.2"
@@ -158,9 +158,13 @@ def main(argv=None):
158
158
  for r in fetch_arxiv_batch(wanted):
159
159
  if r.get("error"):
160
160
  failures.append((r["id"], r["error"]))
161
- else:
162
- items.append(r["item"])
163
- pdfs[r["item"]["title"]] = (r["id"], r["pdf_url"])
161
+ continue
162
+ if r.get("note"): # version-of-record upgrade happened
163
+ print(r["note"])
164
+ if r.get("warning"):
165
+ print(f"warning: {r['warning']}", file=sys.stderr)
166
+ items.append(r["item"])
167
+ pdfs[r["item"]["title"]] = (r["id"], r["pdf_url"])
164
168
  else:
165
169
  for d in wanted:
166
170
  try:
@@ -95,6 +95,11 @@ def _squash(s: str) -> str:
95
95
  return re.sub(r"\s+", " ", s or "").strip()
96
96
 
97
97
 
98
+ def _add_extra(item: dict, line: str) -> None:
99
+ """Append a line to the item's Extra field, never overwriting what's there."""
100
+ item["extra"] = f"{item['extra']}\n{line}" if item.get("extra") else line
101
+
102
+
98
103
  def _person(name: str) -> dict:
99
104
  """Split a display name into first/last on the final space; single-token
100
105
  names go in lastName alone (matches Zotero's single-field mode)."""
@@ -145,14 +150,39 @@ def _arxiv_item(entry, aid: str) -> tuple[dict, str]:
145
150
  "archiveID": f"arXiv:{aid}",
146
151
  "libraryCatalog": "arXiv.org",
147
152
  }
153
+ if item["abstractNote"]:
154
+ _add_extra(item, "abstract-source: arxiv")
148
155
  return item, f"https://arxiv.org/pdf/{aid}"
149
156
 
150
157
 
158
+ _DATACITE_PREFIX = "10.48550/" # arXiv's own DOIs; anything else is a journal DOI
159
+
160
+
161
+ def _journal_record(pre: dict, aid: str) -> dict:
162
+ """Version-of-record upgrade: the arXiv record carries a journal DOI, so
163
+ build the item from CrossRef instead (venue/volume/pages/formal date),
164
+ keeping the arXiv identity: abstract fallback when CrossRef has none,
165
+ `arXiv: <id>` in Extra, and the open-access abs page as url (the journal
166
+ link is carried by the DOI field). Raises MetadataError on lookup failure —
167
+ the caller falls back to the preprint record."""
168
+ j = fetch_doi(pre["DOI"])
169
+ if not j.get("abstractNote") and pre.get("abstractNote"):
170
+ j["abstractNote"] = pre["abstractNote"]
171
+ _add_extra(j, "abstract-source: arxiv")
172
+ j["url"] = pre["url"]
173
+ _add_extra(j, f"arXiv: {aid}")
174
+ return j
175
+
176
+
151
177
  def fetch_arxiv_batch(ids_or_urls: list[str]) -> list[dict]:
152
178
  """arXiv export API, one id_list request per ≤50 ids → results in input
153
179
  order, each {"id", "item", "pdf_url"} or {"id", "error"}. A bad id never
154
180
  fails the batch; only transport-level trouble raises MetadataError.
155
181
 
182
+ Version of record: when arXiv reports a journal DOI (not its own
183
+ 10.48550/* DataCite DOI) the item is rebuilt from CrossRef ("note" key
184
+ says so); if that lookup fails the preprint record stands ("warning" key).
185
+
156
186
  The response can't be trusted positionally: entries come back in arbitrary
157
187
  order (and with resolved version numbers), so they are mapped back to the
158
188
  requested ids by version-stripped id.
@@ -191,8 +221,17 @@ def fetch_arxiv_batch(ids_or_urls: list[str]) -> list[dict]:
191
221
  entry = emap.get(_bare(c["id"]))
192
222
  if entry is None:
193
223
  c["error"] = f"arXiv has no record for '{c['id']}' — check the id"
194
- else:
195
- c["item"], c["pdf_url"] = _arxiv_item(entry, c["id"])
224
+ continue
225
+ c["item"], c["pdf_url"] = _arxiv_item(entry, c["id"])
226
+ doi = c["item"]["DOI"]
227
+ if not doi.startswith(_DATACITE_PREFIX):
228
+ try:
229
+ c["item"] = _journal_record(c["item"], c["id"])
230
+ c["note"] = (f"{c['id']} → journal DOI {doi}, "
231
+ "building journal record")
232
+ except MetadataError as e:
233
+ c["warning"] = (f"{c['id']}: journal DOI {doi} lookup failed "
234
+ f"({e}) — keeping preprint record")
196
235
  return results
197
236
 
198
237
 
@@ -268,7 +307,7 @@ def fetch_doi(doi: str) -> dict:
268
307
  "url": msg.get("URL") or f"https://doi.org/{doi}",
269
308
  "volume": msg.get("volume", ""),
270
309
  "issue": msg.get("issue", ""),
271
- "pages": msg.get("page", ""),
310
+ "pages": msg.get("page") or msg.get("article-number", ""),
272
311
  "publisher": _squash(msg.get("publisher", "")),
273
312
  "language": msg.get("language", ""),
274
313
  "ISSN": (msg.get("ISSN") or [""])[0],
@@ -282,4 +321,7 @@ def fetch_doi(doi: str) -> dict:
282
321
  if not item["title"]:
283
322
  raise MetadataError(f"CrossRef record for '{doi}' has no title — refusing to "
284
323
  "create an empty item")
285
- return {k: v for k, v in item.items() if v not in ("", [])}
324
+ item = {k: v for k, v in item.items() if v not in ("", [])}
325
+ if item.get("abstractNote"):
326
+ _add_extra(item, "abstract-source: crossref")
327
+ return item
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: zotkit
3
- Version: 0.4.1
3
+ Version: 0.4.2
4
4
  Summary: Headless Zotero library management: Web API CRUD plus direct WebDAV attachment upload/download — no desktop app required
5
5
  Author: Shawn
6
6
  License-Expression: MIT
@@ -125,9 +125,20 @@ zotkit lint field:physics topic:new-idea # offline tag check
125
125
  ```
126
126
 
127
127
  `--arxiv` takes ids or abs/pdf URLs (several, space- or comma-separated) and maps
128
- the full record (all authors, abstract, date, DOI, `preprint` item type); `--doi`
128
+ the full record (all authors, abstract, date, DOI); `--doi`
129
129
  maps CrossRef records (journal articles, conference papers, books, chapters, …)
130
- and refuses to guess on CrossRef types it doesn't know. Both accept `--collection`
130
+ and refuses to guess on CrossRef types it doesn't know.
131
+
132
+ **Version of record**: when arXiv reports a *journal* DOI (the paper was formally
133
+ published — as opposed to arXiv's own `10.48550/*` DataCite DOI), `--arxiv`
134
+ builds the journal record from CrossRef instead of a `preprint`: proper item
135
+ type, venue, volume/pages, formal date. The arXiv identity is kept — `arXiv: <id>`
136
+ goes in Extra, the `url` stays the open-access abs page (the journal link lives in
137
+ the DOI field), the arXiv abstract fills in when CrossRef has none, and the PDF
138
+ still comes from arXiv. If the CrossRef lookup fails, the item falls back to the
139
+ preprint record with a warning rather than failing. Items whose abstract zotkit
140
+ wrote also carry an `abstract-source: arxiv|crossref` line in Extra, naming where
141
+ it actually came from. Both accept `--collection`
131
142
  and `--tags`, and `--no-pdf` skips the arXiv PDFs. Rate limiting is built into the
132
143
  request layer — batches use one arXiv metadata request and space PDF downloads
133
144
  per arXiv's terms of use, so callers (humans or agents) never pace themselves. A
File without changes
File without changes
File without changes