zotkit 0.4.1__tar.gz → 0.4.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {zotkit-0.4.1/zotkit.egg-info → zotkit-0.4.2}/PKG-INFO +14 -3
- {zotkit-0.4.1 → zotkit-0.4.2}/README.md +13 -2
- {zotkit-0.4.1 → zotkit-0.4.2}/pyproject.toml +1 -1
- {zotkit-0.4.1 → zotkit-0.4.2}/zotkit/__init__.py +1 -1
- {zotkit-0.4.1 → zotkit-0.4.2}/zotkit/cli.py +7 -3
- {zotkit-0.4.1 → zotkit-0.4.2}/zotkit/metadata.py +46 -4
- {zotkit-0.4.1 → zotkit-0.4.2/zotkit.egg-info}/PKG-INFO +14 -3
- {zotkit-0.4.1 → zotkit-0.4.2}/LICENSE +0 -0
- {zotkit-0.4.1 → zotkit-0.4.2}/setup.cfg +0 -0
- {zotkit-0.4.1 → zotkit-0.4.2}/zotkit/core.py +0 -0
- {zotkit-0.4.1 → zotkit-0.4.2}/zotkit.egg-info/SOURCES.txt +0 -0
- {zotkit-0.4.1 → zotkit-0.4.2}/zotkit.egg-info/dependency_links.txt +0 -0
- {zotkit-0.4.1 → zotkit-0.4.2}/zotkit.egg-info/entry_points.txt +0 -0
- {zotkit-0.4.1 → zotkit-0.4.2}/zotkit.egg-info/requires.txt +0 -0
- {zotkit-0.4.1 → zotkit-0.4.2}/zotkit.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: zotkit
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.2
|
|
4
4
|
Summary: Headless Zotero library management: Web API CRUD plus direct WebDAV attachment upload/download — no desktop app required
|
|
5
5
|
Author: Shawn
|
|
6
6
|
License-Expression: MIT
|
|
@@ -125,9 +125,20 @@ zotkit lint field:physics topic:new-idea # offline tag check
|
|
|
125
125
|
```
|
|
126
126
|
|
|
127
127
|
`--arxiv` takes ids or abs/pdf URLs (several, space- or comma-separated) and maps
|
|
128
|
-
the full record (all authors, abstract, date, DOI
|
|
128
|
+
the full record (all authors, abstract, date, DOI); `--doi`
|
|
129
129
|
maps CrossRef records (journal articles, conference papers, books, chapters, …)
|
|
130
|
-
and refuses to guess on CrossRef types it doesn't know.
|
|
130
|
+
and refuses to guess on CrossRef types it doesn't know.
|
|
131
|
+
|
|
132
|
+
**Version of record**: when arXiv reports a *journal* DOI (the paper was formally
|
|
133
|
+
published — as opposed to arXiv's own `10.48550/*` DataCite DOI), `--arxiv`
|
|
134
|
+
builds the journal record from CrossRef instead of a `preprint`: proper item
|
|
135
|
+
type, venue, volume/pages, formal date. The arXiv identity is kept — `arXiv: <id>`
|
|
136
|
+
goes in Extra, the `url` stays the open-access abs page (the journal link lives in
|
|
137
|
+
the DOI field), the arXiv abstract fills in when CrossRef has none, and the PDF
|
|
138
|
+
still comes from arXiv. If the CrossRef lookup fails, the item falls back to the
|
|
139
|
+
preprint record with a warning rather than failing. Items whose abstract zotkit
|
|
140
|
+
wrote also carry an `abstract-source: arxiv|crossref` line in Extra, naming where
|
|
141
|
+
it actually came from. Both accept `--collection`
|
|
131
142
|
and `--tags`, and `--no-pdf` skips the arXiv PDFs. Rate limiting is built into the
|
|
132
143
|
request layer — batches use one arXiv metadata request and space PDF downloads
|
|
133
144
|
per arXiv's terms of use, so callers (humans or agents) never pace themselves. A
|
|
@@ -105,9 +105,20 @@ zotkit lint field:physics topic:new-idea # offline tag check
|
|
|
105
105
|
```
|
|
106
106
|
|
|
107
107
|
`--arxiv` takes ids or abs/pdf URLs (several, space- or comma-separated) and maps
|
|
108
|
-
the full record (all authors, abstract, date, DOI
|
|
108
|
+
the full record (all authors, abstract, date, DOI); `--doi`
|
|
109
109
|
maps CrossRef records (journal articles, conference papers, books, chapters, …)
|
|
110
|
-
and refuses to guess on CrossRef types it doesn't know.
|
|
110
|
+
and refuses to guess on CrossRef types it doesn't know.
|
|
111
|
+
|
|
112
|
+
**Version of record**: when arXiv reports a *journal* DOI (the paper was formally
|
|
113
|
+
published — as opposed to arXiv's own `10.48550/*` DataCite DOI), `--arxiv`
|
|
114
|
+
builds the journal record from CrossRef instead of a `preprint`: proper item
|
|
115
|
+
type, venue, volume/pages, formal date. The arXiv identity is kept — `arXiv: <id>`
|
|
116
|
+
goes in Extra, the `url` stays the open-access abs page (the journal link lives in
|
|
117
|
+
the DOI field), the arXiv abstract fills in when CrossRef has none, and the PDF
|
|
118
|
+
still comes from arXiv. If the CrossRef lookup fails, the item falls back to the
|
|
119
|
+
preprint record with a warning rather than failing. Items whose abstract zotkit
|
|
120
|
+
wrote also carry an `abstract-source: arxiv|crossref` line in Extra, naming where
|
|
121
|
+
it actually came from. Both accept `--collection`
|
|
111
122
|
and `--tags`, and `--no-pdf` skips the arXiv PDFs. Rate limiting is built into the
|
|
112
123
|
request layer — batches use one arXiv metadata request and space PDF downloads
|
|
113
124
|
per arXiv's terms of use, so callers (humans or agents) never pace themselves. A
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "zotkit"
|
|
7
|
-
version = "0.4.
|
|
7
|
+
version = "0.4.2"
|
|
8
8
|
description = "Headless Zotero library management: Web API CRUD plus direct WebDAV attachment upload/download — no desktop app required"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "MIT"
|
|
@@ -158,9 +158,13 @@ def main(argv=None):
|
|
|
158
158
|
for r in fetch_arxiv_batch(wanted):
|
|
159
159
|
if r.get("error"):
|
|
160
160
|
failures.append((r["id"], r["error"]))
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
161
|
+
continue
|
|
162
|
+
if r.get("note"): # version-of-record upgrade happened
|
|
163
|
+
print(r["note"])
|
|
164
|
+
if r.get("warning"):
|
|
165
|
+
print(f"warning: {r['warning']}", file=sys.stderr)
|
|
166
|
+
items.append(r["item"])
|
|
167
|
+
pdfs[r["item"]["title"]] = (r["id"], r["pdf_url"])
|
|
164
168
|
else:
|
|
165
169
|
for d in wanted:
|
|
166
170
|
try:
|
|
@@ -95,6 +95,11 @@ def _squash(s: str) -> str:
|
|
|
95
95
|
return re.sub(r"\s+", " ", s or "").strip()
|
|
96
96
|
|
|
97
97
|
|
|
98
|
+
def _add_extra(item: dict, line: str) -> None:
|
|
99
|
+
"""Append a line to the item's Extra field, never overwriting what's there."""
|
|
100
|
+
item["extra"] = f"{item['extra']}\n{line}" if item.get("extra") else line
|
|
101
|
+
|
|
102
|
+
|
|
98
103
|
def _person(name: str) -> dict:
|
|
99
104
|
"""Split a display name into first/last on the final space; single-token
|
|
100
105
|
names go in lastName alone (matches Zotero's single-field mode)."""
|
|
@@ -145,14 +150,39 @@ def _arxiv_item(entry, aid: str) -> tuple[dict, str]:
|
|
|
145
150
|
"archiveID": f"arXiv:{aid}",
|
|
146
151
|
"libraryCatalog": "arXiv.org",
|
|
147
152
|
}
|
|
153
|
+
if item["abstractNote"]:
|
|
154
|
+
_add_extra(item, "abstract-source: arxiv")
|
|
148
155
|
return item, f"https://arxiv.org/pdf/{aid}"
|
|
149
156
|
|
|
150
157
|
|
|
158
|
+
_DATACITE_PREFIX = "10.48550/" # arXiv's own DOIs; anything else is a journal DOI
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _journal_record(pre: dict, aid: str) -> dict:
|
|
162
|
+
"""Version-of-record upgrade: the arXiv record carries a journal DOI, so
|
|
163
|
+
build the item from CrossRef instead (venue/volume/pages/formal date),
|
|
164
|
+
keeping the arXiv identity: abstract fallback when CrossRef has none,
|
|
165
|
+
`arXiv: <id>` in Extra, and the open-access abs page as url (the journal
|
|
166
|
+
link is carried by the DOI field). Raises MetadataError on lookup failure —
|
|
167
|
+
the caller falls back to the preprint record."""
|
|
168
|
+
j = fetch_doi(pre["DOI"])
|
|
169
|
+
if not j.get("abstractNote") and pre.get("abstractNote"):
|
|
170
|
+
j["abstractNote"] = pre["abstractNote"]
|
|
171
|
+
_add_extra(j, "abstract-source: arxiv")
|
|
172
|
+
j["url"] = pre["url"]
|
|
173
|
+
_add_extra(j, f"arXiv: {aid}")
|
|
174
|
+
return j
|
|
175
|
+
|
|
176
|
+
|
|
151
177
|
def fetch_arxiv_batch(ids_or_urls: list[str]) -> list[dict]:
|
|
152
178
|
"""arXiv export API, one id_list request per ≤50 ids → results in input
|
|
153
179
|
order, each {"id", "item", "pdf_url"} or {"id", "error"}. A bad id never
|
|
154
180
|
fails the batch; only transport-level trouble raises MetadataError.
|
|
155
181
|
|
|
182
|
+
Version of record: when arXiv reports a journal DOI (not its own
|
|
183
|
+
10.48550/* DataCite DOI) the item is rebuilt from CrossRef ("note" key
|
|
184
|
+
says so); if that lookup fails the preprint record stands ("warning" key).
|
|
185
|
+
|
|
156
186
|
The response can't be trusted positionally: entries come back in arbitrary
|
|
157
187
|
order (and with resolved version numbers), so they are mapped back to the
|
|
158
188
|
requested ids by version-stripped id.
|
|
@@ -191,8 +221,17 @@ def fetch_arxiv_batch(ids_or_urls: list[str]) -> list[dict]:
|
|
|
191
221
|
entry = emap.get(_bare(c["id"]))
|
|
192
222
|
if entry is None:
|
|
193
223
|
c["error"] = f"arXiv has no record for '{c['id']}' — check the id"
|
|
194
|
-
|
|
195
|
-
|
|
224
|
+
continue
|
|
225
|
+
c["item"], c["pdf_url"] = _arxiv_item(entry, c["id"])
|
|
226
|
+
doi = c["item"]["DOI"]
|
|
227
|
+
if not doi.startswith(_DATACITE_PREFIX):
|
|
228
|
+
try:
|
|
229
|
+
c["item"] = _journal_record(c["item"], c["id"])
|
|
230
|
+
c["note"] = (f"{c['id']} → journal DOI {doi}, "
|
|
231
|
+
"building journal record")
|
|
232
|
+
except MetadataError as e:
|
|
233
|
+
c["warning"] = (f"{c['id']}: journal DOI {doi} lookup failed "
|
|
234
|
+
f"({e}) — keeping preprint record")
|
|
196
235
|
return results
|
|
197
236
|
|
|
198
237
|
|
|
@@ -268,7 +307,7 @@ def fetch_doi(doi: str) -> dict:
|
|
|
268
307
|
"url": msg.get("URL") or f"https://doi.org/{doi}",
|
|
269
308
|
"volume": msg.get("volume", ""),
|
|
270
309
|
"issue": msg.get("issue", ""),
|
|
271
|
-
"pages": msg.get("page", ""),
|
|
310
|
+
"pages": msg.get("page") or msg.get("article-number", ""),
|
|
272
311
|
"publisher": _squash(msg.get("publisher", "")),
|
|
273
312
|
"language": msg.get("language", ""),
|
|
274
313
|
"ISSN": (msg.get("ISSN") or [""])[0],
|
|
@@ -282,4 +321,7 @@ def fetch_doi(doi: str) -> dict:
|
|
|
282
321
|
if not item["title"]:
|
|
283
322
|
raise MetadataError(f"CrossRef record for '{doi}' has no title — refusing to "
|
|
284
323
|
"create an empty item")
|
|
285
|
-
|
|
324
|
+
item = {k: v for k, v in item.items() if v not in ("", [])}
|
|
325
|
+
if item.get("abstractNote"):
|
|
326
|
+
_add_extra(item, "abstract-source: crossref")
|
|
327
|
+
return item
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: zotkit
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.2
|
|
4
4
|
Summary: Headless Zotero library management: Web API CRUD plus direct WebDAV attachment upload/download — no desktop app required
|
|
5
5
|
Author: Shawn
|
|
6
6
|
License-Expression: MIT
|
|
@@ -125,9 +125,20 @@ zotkit lint field:physics topic:new-idea # offline tag check
|
|
|
125
125
|
```
|
|
126
126
|
|
|
127
127
|
`--arxiv` takes ids or abs/pdf URLs (several, space- or comma-separated) and maps
|
|
128
|
-
the full record (all authors, abstract, date, DOI
|
|
128
|
+
the full record (all authors, abstract, date, DOI); `--doi`
|
|
129
129
|
maps CrossRef records (journal articles, conference papers, books, chapters, …)
|
|
130
|
-
and refuses to guess on CrossRef types it doesn't know.
|
|
130
|
+
and refuses to guess on CrossRef types it doesn't know.
|
|
131
|
+
|
|
132
|
+
**Version of record**: when arXiv reports a *journal* DOI (the paper was formally
|
|
133
|
+
published — as opposed to arXiv's own `10.48550/*` DataCite DOI), `--arxiv`
|
|
134
|
+
builds the journal record from CrossRef instead of a `preprint`: proper item
|
|
135
|
+
type, venue, volume/pages, formal date. The arXiv identity is kept — `arXiv: <id>`
|
|
136
|
+
goes in Extra, the `url` stays the open-access abs page (the journal link lives in
|
|
137
|
+
the DOI field), the arXiv abstract fills in when CrossRef has none, and the PDF
|
|
138
|
+
still comes from arXiv. If the CrossRef lookup fails, the item falls back to the
|
|
139
|
+
preprint record with a warning rather than failing. Items whose abstract zotkit
|
|
140
|
+
wrote also carry an `abstract-source: arxiv|crossref` line in Extra, naming where
|
|
141
|
+
it actually came from. Both accept `--collection`
|
|
131
142
|
and `--tags`, and `--no-pdf` skips the arXiv PDFs. Rate limiting is built into the
|
|
132
143
|
request layer — batches use one arXiv metadata request and space PDF downloads
|
|
133
144
|
per arXiv's terms of use, so callers (humans or agents) never pace themselves. A
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|