sourcelens 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sourcelens/__init__.py +7 -0
- sourcelens/__main__.py +7 -0
- sourcelens/buildcatalog/__init__.py +0 -0
- sourcelens/buildcatalog/build_progress.py +414 -0
- sourcelens/buildcatalog/summarise.py +162 -0
- sourcelens/buildcatalog/test_incremental.py +116 -0
- sourcelens/cli/__init__.py +0 -0
- sourcelens/cli/main.py +699 -0
- sourcelens/common/__init__.py +0 -0
- sourcelens/common/agelit.py +440 -0
- sourcelens/common/crossref.py +101 -0
- sourcelens/common/flags.py +47 -0
- sourcelens/common/hits.py +50 -0
- sourcelens/common/pubmed.py +135 -0
- sourcelens/common/settings.py +367 -0
- sourcelens/common/store.py +112 -0
- sourcelens/common/terms.py +217 -0
- sourcelens/defaults/default.yaml +733 -0
- sourcelens/localseeds/__init__.py +0 -0
- sourcelens/localseeds/extract_seeds.py +240 -0
- sourcelens/pullliturature/__init__.py +0 -0
- sourcelens/pullliturature/enrich_openalex.py +74 -0
- sourcelens/pullliturature/enrich_preprints.py +66 -0
- sourcelens/pullliturature/fetch_fulltext.py +662 -0
- sourcelens/pullliturature/jats.py +269 -0
- sourcelens/pullliturature/resolve_seeds.py +119 -0
- sourcelens/pullliturature/search_arxiv.py +141 -0
- sourcelens/pullliturature/search_europepmc.py +282 -0
- sourcelens/pullliturature/search_openalex.py +222 -0
- sourcelens/pullliturature/search_pubmed.py +111 -0
- sourcelens/pullrepos/__init__.py +0 -0
- sourcelens/pullrepos/build_repos.py +379 -0
- sourcelens/pullrepos/mine_links.py +146 -0
- sourcelens/pullrepos/search_github.py +108 -0
- sourcelens/query/__init__.py +0 -0
- sourcelens/query/catalog.py +191 -0
- sourcelens/query/export_refs.py +71 -0
- sourcelens/query/refs.py +566 -0
- sourcelens/query/search_catalog.py +53 -0
- sourcelens/websites/__init__.py +0 -0
- sourcelens/websites/build_websites.py +181 -0
- sourcelens-0.0.1.dist-info/METADATA +548 -0
- sourcelens-0.0.1.dist-info/RECORD +46 -0
- sourcelens-0.0.1.dist-info/WHEEL +4 -0
- sourcelens-0.0.1.dist-info/entry_points.txt +2 -0
- sourcelens-0.0.1.dist-info/licenses/LICENSE +674 -0
sourcelens/__init__.py
ADDED
sourcelens/__main__.py
ADDED
|
File without changes
|
|
@@ -0,0 +1,414 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Build <catalogue>/progress.csv: every resource, oldest first.
|
|
3
|
+
|
|
4
|
+
Merges the record store (articles and preprints from the literature searches
|
|
5
|
+
and the local seeds), repositories, packages and websites into one catalogue,
|
|
6
|
+
annotated with the classify section of the configuration.
|
|
7
|
+
|
|
8
|
+
INCREMENTAL BY DESIGN. progress.csv is never rebuilt from scratch:
|
|
9
|
+
* a row whose uid is already in the file keeps its `added_on` date and
|
|
10
|
+
its hand-edited columns (`notes`, `user_tags`) untouched;
|
|
11
|
+
* a resource not yet in the file is added with `added_on` = today;
|
|
12
|
+
* a row that no longer comes out of the pipeline (search terms changed,
|
|
13
|
+
source withdrew it) is kept and flagged in `status`, never deleted;
|
|
14
|
+
* the file is then re-sorted by date, so it stays chronological.
|
|
15
|
+
Every run writes <catalogue>/changelog/added_<date>.csv with only the rows it
|
|
16
|
+
added, and appends one line to <catalogue>/changelog/runs.csv.
|
|
17
|
+
|
|
18
|
+
Outputs
|
|
19
|
+
<catalogue>/progress.csv the catalogue (scope: focused + seeds)
|
|
20
|
+
<catalogue>/corpus/broad_hits.csv records matched only by broad terms
|
|
21
|
+
<catalogue>/corpus/before_window.csv seeds published before the window
|
|
22
|
+
<catalogue>/corpus/offtopic_seeds.csv local seeds outside the topic
|
|
23
|
+
<catalogue>/corpus/references.csv full citation fields for export
|
|
24
|
+
<catalogue>/changelog/added_<date>.csv, runs.csv
|
|
25
|
+
<catalogue>/summary/*.csv, stats.json counts
|
|
26
|
+
|
|
27
|
+
Usage (normally run by sourcelens update)
|
|
28
|
+
sourcelens __step catalogue
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import collections
|
|
34
|
+
import re
|
|
35
|
+
from pathlib import Path
|
|
36
|
+
|
|
37
|
+
from sourcelens.common.agelit import ( # noqa: E402
|
|
38
|
+
CORPUS,
|
|
39
|
+
FULLTEXT,
|
|
40
|
+
REPOS,
|
|
41
|
+
RESEARCH,
|
|
42
|
+
SEEDS,
|
|
43
|
+
WEBSITES,
|
|
44
|
+
config,
|
|
45
|
+
log,
|
|
46
|
+
query_window,
|
|
47
|
+
read_csv,
|
|
48
|
+
today,
|
|
49
|
+
write_csv,
|
|
50
|
+
write_json,
|
|
51
|
+
)
|
|
52
|
+
from sourcelens.common.store import Store # noqa: E402
|
|
53
|
+
|
|
54
|
+
PROGRESS = RESEARCH / "progress.csv"
|
|
55
|
+
CHANGELOG = RESEARCH / "changelog"
|
|
56
|
+
SUMMARY = RESEARCH / "summary"
|
|
57
|
+
|
|
58
|
+
COLUMNS = [
|
|
59
|
+
"added_on", "date", "year", "resource_type", "tier", "category", "modality",
|
|
60
|
+
"entities", "species", "title", "authors", "venue", "doi", "pmid", "pmcid",
|
|
61
|
+
"url", "code_links", "related", "open_access", "license", "fulltext_status",
|
|
62
|
+
"fulltext_pdf", "fulltext_md", "fulltext_txt", "metadata_file", "cited_by",
|
|
63
|
+
"details", "matched_groups", "found_by", "local_refs", "status", "uid", "notes", "user_tags",
|
|
64
|
+
]
|
|
65
|
+
USER_COLUMNS = {"notes", "user_tags"}
|
|
66
|
+
KEEP_ON_UPDATE = {"added_on"} | USER_COLUMNS
|
|
67
|
+
TIER_ORDER = {"landmark": 0, "core": 1, "related": 2}
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
# --------------------------------------------------------------------------
|
|
71
|
+
# classification
|
|
72
|
+
# --------------------------------------------------------------------------
|
|
73
|
+
|
|
74
|
+
class Classifier:
|
|
75
|
+
def __init__(self, cfg: dict):
|
|
76
|
+
f = re.I
|
|
77
|
+
self.modality = {k: re.compile(v, f) for k, v in (cfg.get("modality") or {}).items()}
|
|
78
|
+
self.species = {k: re.compile(v, f) for k, v in (cfg.get("species") or {}).items()}
|
|
79
|
+
self.category = []
|
|
80
|
+
for rule in cfg.get("category") or []:
|
|
81
|
+
self.category.append({
|
|
82
|
+
"name": rule["name"],
|
|
83
|
+
"pub_types": re.compile(rule["pub_types"], f) if rule.get("pub_types") else None,
|
|
84
|
+
"title": re.compile(rule["title"], f) if rule.get("title") else None,
|
|
85
|
+
"text": re.compile(rule["text"], f) if rule.get("text") else None,
|
|
86
|
+
})
|
|
87
|
+
self.core_title = re.compile(cfg["core_title_terms"], f) if cfg.get("core_title_terms") else None
|
|
88
|
+
# catalogues configured before 0.0.1 name registry roles clock_*
|
|
89
|
+
roles = set(cfg.get("landmark_roles") or [])
|
|
90
|
+
self.landmark_roles = roles | {r.replace("clock_", "registry_", 1) for r in roles}
|
|
91
|
+
self.default_category = cfg.get("default_category") or "application/association"
|
|
92
|
+
self.origin_category = cfg.get("origin_category") or "method development"
|
|
93
|
+
self.core_categories = list(cfg.get("core_categories") or
|
|
94
|
+
[self.origin_category, "benchmark/comparison", "review", "software/resource"])
|
|
95
|
+
entities = cfg.get("entities")
|
|
96
|
+
if entities is None:
|
|
97
|
+
entities = cfg.get("clock_names") or {} # name used before 0.0.1
|
|
98
|
+
self.entities = {k: re.compile(r"(?<![\w-])(?:" + v + r")(?![\w-])" if not v.startswith("(?i)")
|
|
99
|
+
else "(?i)(?<![\\w-])(?:" + v[4:] + ")(?![\\w-])")
|
|
100
|
+
for k, v in (entities or {}).items()}
|
|
101
|
+
|
|
102
|
+
def annotate(self, rec: dict) -> dict:
|
|
103
|
+
title = rec.get("title") or ""
|
|
104
|
+
text = " ".join(str(rec.get(k) or "") for k in ("title", "abstract", "keywords", "mesh"))
|
|
105
|
+
pubtypes = rec.get("pub_types") or ""
|
|
106
|
+
mods = [k for k, rx in self.modality.items() if rx.search(text)]
|
|
107
|
+
sp = [k for k, rx in self.species.items() if rx.search(text) and k != "human"]
|
|
108
|
+
if ("human" in self.species and self.species["human"].search(text)) or "Humans" in (rec.get("mesh") or ""):
|
|
109
|
+
sp = ["human"] + sp
|
|
110
|
+
cat = self.default_category
|
|
111
|
+
for rule in self.category:
|
|
112
|
+
if ((rule["pub_types"] and rule["pub_types"].search(pubtypes))
|
|
113
|
+
or (rule["title"] and rule["title"].search(title))
|
|
114
|
+
or (rule["text"] and rule["text"].search(text))):
|
|
115
|
+
cat = rule["name"]
|
|
116
|
+
break
|
|
117
|
+
ents = [k for k, rx in self.entities.items() if rx.search(text)]
|
|
118
|
+
return {"modality": "; ".join(mods), "species": "; ".join(sp), "category": cat,
|
|
119
|
+
"entities": "; ".join(ents),
|
|
120
|
+
"_core_title": bool(self.core_title and self.core_title.search(title))}
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
# --------------------------------------------------------------------------
|
|
124
|
+
# rows
|
|
125
|
+
# --------------------------------------------------------------------------
|
|
126
|
+
|
|
127
|
+
def resource_type(rec: dict, category: str) -> str:
|
|
128
|
+
pt = (rec.get("pub_types") or "").lower()
|
|
129
|
+
if rec.get("is_preprint") or rec.get("source_db") == "PPR" or "preprint" in pt:
|
|
130
|
+
return "preprint"
|
|
131
|
+
if category == "review":
|
|
132
|
+
return "review"
|
|
133
|
+
for key in ("dataset", "software", "book chapter", "conference paper", "thesis", "report"):
|
|
134
|
+
if key in pt:
|
|
135
|
+
return key
|
|
136
|
+
return "article"
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def best_date(rec: dict) -> str:
|
|
140
|
+
d = (rec.get("pub_date") or "").strip()
|
|
141
|
+
if re.fullmatch(r"\d{4}-\d{2}-\d{2}", d):
|
|
142
|
+
return d
|
|
143
|
+
y = (rec.get("pub_year") or d[:4]).strip()
|
|
144
|
+
return f"{y}-01-01" if re.fullmatch(r"\d{4}", y) else ""
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
SERVER_NAMES = {"biorxiv": "bioRxiv", "medrxiv": "medRxiv", "research square": "Research Square",
|
|
148
|
+
"preprints.org": "Preprints.org", "ssrn": "SSRN", "psyarxiv": "PsyArXiv"}
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def venue_name(v: str) -> str:
|
|
152
|
+
"""One spelling per preprint server ("bioRxiv : the preprint server for biology" -> "bioRxiv")."""
|
|
153
|
+
low = (v or "").strip().lower()
|
|
154
|
+
for key, name in SERVER_NAMES.items():
|
|
155
|
+
if low.startswith(key):
|
|
156
|
+
return name
|
|
157
|
+
return v
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def first_authors(a: str, n: int = 3) -> str:
|
|
161
|
+
parts = [p.strip() for p in (a or "").rstrip(".").split(",") if p.strip()]
|
|
162
|
+
return ", ".join(parts[:n]) + (" et al." if len(parts) > n else "")
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def article_row(rec, ann, groups, found_by, seeds, ft, links, cites) -> dict:
|
|
166
|
+
uid = rec["uid"]
|
|
167
|
+
doi = rec.get("doi") or ""
|
|
168
|
+
url = f"https://doi.org/{doi}" if doi else (
|
|
169
|
+
f"https://pubmed.ncbi.nlm.nih.gov/{rec['pmid']}/" if rec.get("pmid") else rec.get("url", ""))
|
|
170
|
+
f = ft.get(uid, {})
|
|
171
|
+
folder = f.get("folder", "")
|
|
172
|
+
|
|
173
|
+
def fp(name: str, flag: str) -> str:
|
|
174
|
+
return f"{folder}/{name}" if folder and str(f.get(flag)) == "True" else ""
|
|
175
|
+
|
|
176
|
+
seed_info = seeds.get(doi, {})
|
|
177
|
+
cited = max(int(rec.get("cited_by") or 0), int(cites.get(uid, 0) or 0))
|
|
178
|
+
oa = "yes" if (rec.get("is_open_access") or f.get("status") in ("ok", "partial")) else (
|
|
179
|
+
"unknown" if not f else "no")
|
|
180
|
+
date = best_date(rec)
|
|
181
|
+
return {
|
|
182
|
+
"date": date, "year": date[:4], "resource_type": resource_type(rec, ann["category"]),
|
|
183
|
+
"category": ann["category"], "modality": ann["modality"],
|
|
184
|
+
"entities": "; ".join(filter(None, [seed_info.get("registry_ids", ""), ann["entities"]])),
|
|
185
|
+
"species": ann["species"], "title": rec.get("title", ""),
|
|
186
|
+
"authors": first_authors(rec.get("authors", "")), "venue": venue_name(rec.get("journal", "")),
|
|
187
|
+
"doi": doi, "pmid": rec.get("pmid", ""), "pmcid": rec.get("pmcid", ""), "url": url,
|
|
188
|
+
"code_links": "; ".join(links.get(uid, [])), "related": "",
|
|
189
|
+
"open_access": oa, "license": f.get("license") or rec.get("license", ""),
|
|
190
|
+
"fulltext_status": f.get("status", ""),
|
|
191
|
+
"fulltext_pdf": fp("paper.pdf", "has_pdf"), "fulltext_md": fp("paper.md", "has_md"),
|
|
192
|
+
"fulltext_txt": fp("paper.txt", "has_txt"),
|
|
193
|
+
"metadata_file": f"{folder}/metadata.json" if folder else "",
|
|
194
|
+
"cited_by": cited or "", "matched_groups": "; ".join(sorted(groups)),
|
|
195
|
+
"found_by": "; ".join(sorted(found_by)), "local_refs": seed_info.get("refs", ""),
|
|
196
|
+
"uid": uid,
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def load_seeds() -> dict:
|
|
201
|
+
"""doi -> {roles, refs, registry_ids} from the reference folders."""
|
|
202
|
+
out: dict[str, dict] = {}
|
|
203
|
+
for s in read_csv(SEEDS / "local_seeds.csv"):
|
|
204
|
+
d = out.setdefault(s["doi"], {"roles": set(), "refs": set(), "ids": {}})
|
|
205
|
+
role = s["role"].replace("clock_", "registry_", 1) # seeds written before 0.0.1
|
|
206
|
+
d["roles"].add(role)
|
|
207
|
+
d["refs"].add(f"{s['source_repo']}:{role}")
|
|
208
|
+
rid = s.get("registry_id") or s.get("clock_name") or ""
|
|
209
|
+
if rid:
|
|
210
|
+
d["ids"].setdefault(s["source_repo"], set()).add(rid)
|
|
211
|
+
for d in out.values():
|
|
212
|
+
d["refs"] = "; ".join(sorted(d["refs"]))
|
|
213
|
+
d["registry_ids"] = "; ".join(f"{repo} registry: " + ", ".join(sorted(ids))
|
|
214
|
+
for repo, ids in sorted(d["ids"].items()))
|
|
215
|
+
return out
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def extra_rows(path: Path) -> list[dict]:
|
|
219
|
+
"""Rows produced by the repository / website harvesters, already in catalogue shape."""
|
|
220
|
+
return [r for r in read_csv(path) if r.get("uid")]
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
# --------------------------------------------------------------------------
|
|
224
|
+
# main
|
|
225
|
+
# --------------------------------------------------------------------------
|
|
226
|
+
|
|
227
|
+
def main() -> None:
|
|
228
|
+
cfg = config("classify")
|
|
229
|
+
clf = Classifier(cfg)
|
|
230
|
+
start, end = query_window()
|
|
231
|
+
store = Store()
|
|
232
|
+
log(f"store: {len(store)} records; window {start} .. {end}")
|
|
233
|
+
|
|
234
|
+
groups: dict[str, set] = collections.defaultdict(set)
|
|
235
|
+
scopes: dict[str, set] = collections.defaultdict(set)
|
|
236
|
+
found: dict[str, set] = collections.defaultdict(set)
|
|
237
|
+
for h in read_csv(CORPUS / "search_hits.csv"):
|
|
238
|
+
uid = store.find({"uid": h["uid"]}) or h["uid"]
|
|
239
|
+
groups[uid].add(h["group"])
|
|
240
|
+
scopes[uid].add(h["scope"])
|
|
241
|
+
found[uid].add(h["source"])
|
|
242
|
+
|
|
243
|
+
seeds = load_seeds()
|
|
244
|
+
ft = {r["uid"]: r for r in read_csv(FULLTEXT / "fulltext_index.csv")}
|
|
245
|
+
links: dict[str, list] = collections.defaultdict(list)
|
|
246
|
+
for r in read_csv(REPOS / "paper_links.csv"):
|
|
247
|
+
if r["url"] not in links[r["uid"]]:
|
|
248
|
+
links[r["uid"]].append(r["url"])
|
|
249
|
+
openalex = {r["uid"]: r for r in read_csv(CORPUS / "citations.csv")}
|
|
250
|
+
cites = {u: r.get("cited_by", 0) for u, r in openalex.items()}
|
|
251
|
+
# preprint <-> published version, both directions
|
|
252
|
+
pp_links: dict[str, str] = {}
|
|
253
|
+
for r in read_csv(CORPUS / "preprint_links.csv"):
|
|
254
|
+
if r.get("published_doi"):
|
|
255
|
+
pp_links[r["preprint_doi"]] = f"published as {r['published_doi']}"
|
|
256
|
+
pp_links[r["published_doi"]] = f"preprint {r['preprint_doi']}"
|
|
257
|
+
|
|
258
|
+
rows, broad, pre, offtopic = {}, [], [], []
|
|
259
|
+
for rec in store.values():
|
|
260
|
+
uid = rec["uid"]
|
|
261
|
+
ann = clf.annotate(rec)
|
|
262
|
+
doi = rec.get("doi", "")
|
|
263
|
+
sd = seeds.get(doi) if doi else None
|
|
264
|
+
roles = sd["roles"] if sd else set()
|
|
265
|
+
is_landmark = bool(roles & clf.landmark_roles)
|
|
266
|
+
focused = "focused" in scopes.get(uid, set())
|
|
267
|
+
fb = set(found.get(uid, set()))
|
|
268
|
+
if sd:
|
|
269
|
+
fb.add("local seed")
|
|
270
|
+
if "registry_origin" in roles:
|
|
271
|
+
# the paper that introduced a registry entry is development work by definition
|
|
272
|
+
ann = {**ann, "category": clf.origin_category}
|
|
273
|
+
row = article_row(rec, ann, groups.get(uid, set()), fb, seeds, ft, links, cites)
|
|
274
|
+
# the indexes give some records only a year (stored YYYY-01-01);
|
|
275
|
+
# OpenAlex's exact date replaces it when it falls in the same year
|
|
276
|
+
oa_date = openalex.get(uid, {}).get("publication_date", "")
|
|
277
|
+
if row["date"].endswith("-01-01") and oa_date[:4] == row["date"][:4] and oa_date != row["date"]:
|
|
278
|
+
row["date"] = oa_date
|
|
279
|
+
if doi in pp_links:
|
|
280
|
+
row["related"] = pp_links[doi]
|
|
281
|
+
elif rec.get("related_doi"): # arXiv entries that name their journal version
|
|
282
|
+
row["related"] = f"published as {rec['related_doi']}"
|
|
283
|
+
if not row["date"]:
|
|
284
|
+
row["status"] = "undated"
|
|
285
|
+
in_window = not row["date"] or start <= row["date"] <= end
|
|
286
|
+
if is_landmark:
|
|
287
|
+
row["tier"] = "landmark"
|
|
288
|
+
elif focused and (ann["_core_title"] or ann["category"] in clf.core_categories):
|
|
289
|
+
row["tier"] = "core"
|
|
290
|
+
elif focused:
|
|
291
|
+
row["tier"] = "related"
|
|
292
|
+
elif sd and ann["_core_title"]:
|
|
293
|
+
row["tier"] = "related"
|
|
294
|
+
else:
|
|
295
|
+
row["tier"] = ""
|
|
296
|
+
if row["tier"] and not in_window:
|
|
297
|
+
pre.append(row)
|
|
298
|
+
elif row["tier"]:
|
|
299
|
+
rows[uid] = row
|
|
300
|
+
elif sd:
|
|
301
|
+
offtopic.append(row)
|
|
302
|
+
elif "broad" in scopes.get(uid, set()):
|
|
303
|
+
broad.append(row)
|
|
304
|
+
|
|
305
|
+
for path in (REPOS / "repositories_catalogue.csv", WEBSITES / "websites_catalogue.csv"):
|
|
306
|
+
for r in extra_rows(path):
|
|
307
|
+
rows[r["uid"]] = {k: r.get(k, "") for k in COLUMNS if k not in KEEP_ON_UPDATE}
|
|
308
|
+
|
|
309
|
+
# ---- incremental merge with the existing progress.csv -----------------
|
|
310
|
+
existing = {r["uid"]: r for r in read_csv(PROGRESS)}
|
|
311
|
+
run = today()
|
|
312
|
+
added = []
|
|
313
|
+
merged = {}
|
|
314
|
+
for uid, new in rows.items():
|
|
315
|
+
old = existing.get(uid)
|
|
316
|
+
if old is None:
|
|
317
|
+
new = {**new, "added_on": run}
|
|
318
|
+
added.append(new)
|
|
319
|
+
else:
|
|
320
|
+
new = {**new, **{k: old.get(k, "") for k in KEEP_ON_UPDATE}}
|
|
321
|
+
new.setdefault("status", "")
|
|
322
|
+
merged[uid] = new
|
|
323
|
+
kept = 0
|
|
324
|
+
for uid, old in existing.items():
|
|
325
|
+
if uid not in merged:
|
|
326
|
+
old["status"] = "no longer matched by pipeline (kept)"
|
|
327
|
+
merged[uid] = old
|
|
328
|
+
kept += 1
|
|
329
|
+
|
|
330
|
+
def sort_key(r):
|
|
331
|
+
return (r.get("date") or "9999", TIER_ORDER.get(r.get("tier"), 9), r.get("title", "").lower())
|
|
332
|
+
|
|
333
|
+
final = sorted(merged.values(), key=sort_key)
|
|
334
|
+
write_csv(PROGRESS, final, COLUMNS)
|
|
335
|
+
CHANGELOG.mkdir(parents=True, exist_ok=True)
|
|
336
|
+
if added:
|
|
337
|
+
# update_all.sh builds several times a day; the day's file collects
|
|
338
|
+
# every pass's additions rather than keeping only the last pass
|
|
339
|
+
day_file = CHANGELOG / f"added_{run}.csv"
|
|
340
|
+
today_rows = {r["uid"]: r for r in read_csv(day_file)}
|
|
341
|
+
today_rows.update({r["uid"]: r for r in added})
|
|
342
|
+
write_csv(day_file, sorted(today_rows.values(), key=sort_key), COLUMNS)
|
|
343
|
+
runs = read_csv(CHANGELOG / "runs.csv")
|
|
344
|
+
runs.append({"run_date": run, "window_end": end, "rows_total": len(final), "rows_added": len(added),
|
|
345
|
+
"rows_kept_unmatched": kept, "broad_only": len(broad), "store_records": len(store)})
|
|
346
|
+
write_csv(CHANGELOG / "runs.csv", runs, list(runs[-1].keys()))
|
|
347
|
+
|
|
348
|
+
write_references(final, store)
|
|
349
|
+
write_csv(CORPUS / "broad_hits.csv", sorted(broad, key=sort_key), COLUMNS)
|
|
350
|
+
write_csv(CORPUS / "before_window.csv", sorted(pre, key=sort_key), COLUMNS)
|
|
351
|
+
write_csv(CORPUS / "offtopic_seeds.csv", sorted(offtopic, key=sort_key), COLUMNS)
|
|
352
|
+
summarise(final)
|
|
353
|
+
log(f"progress.csv: {len(final)} rows ({len(added)} added this run, {kept} kept though unmatched); "
|
|
354
|
+
f"broad-only {len(broad)}, before the window {len(pre)}, off-topic seeds {len(offtopic)}")
|
|
355
|
+
|
|
356
|
+
|
|
357
|
+
REF_COLUMNS = ["uid", "resource_type", "date", "year", "authors", "title", "venue", "volume", "issue",
|
|
358
|
+
"pages", "doi", "pmid", "pmcid", "url"]
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
def write_references(rows: list[dict], store: Store) -> None:
|
|
362
|
+
"""<catalogue>/corpus/references.csv: full citation fields for every catalogue row.
|
|
363
|
+
|
|
364
|
+
progress.csv keeps only the first three authors; reference export (both
|
|
365
|
+
sourcelens export) needs all of them plus volume / issue / pages, so they are
|
|
366
|
+
written here once per build instead of reading the record store.
|
|
367
|
+
"""
|
|
368
|
+
out = []
|
|
369
|
+
for r in rows:
|
|
370
|
+
rec = store.get(r["uid"]) or {}
|
|
371
|
+
out.append({
|
|
372
|
+
"uid": r["uid"], "resource_type": r.get("resource_type", ""), "date": r.get("date", ""),
|
|
373
|
+
"year": r.get("year", ""),
|
|
374
|
+
"authors": (rec.get("authors") or r.get("authors", "")).strip().rstrip("."),
|
|
375
|
+
"title": r.get("title", ""), "venue": r.get("venue", ""),
|
|
376
|
+
"volume": rec.get("volume", "") or "", "issue": rec.get("issue", "") or "",
|
|
377
|
+
"pages": rec.get("pages", "") or "", "doi": r.get("doi", ""), "pmid": r.get("pmid", ""),
|
|
378
|
+
"pmcid": r.get("pmcid", ""), "url": r.get("url", ""),
|
|
379
|
+
})
|
|
380
|
+
write_csv(CORPUS / "references.csv", out, REF_COLUMNS)
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
def summarise(rows: list[dict]) -> None:
|
|
384
|
+
SUMMARY.mkdir(parents=True, exist_ok=True)
|
|
385
|
+
by_year = collections.Counter((r["year"], r["resource_type"]) for r in rows)
|
|
386
|
+
years = sorted({y for y, _ in by_year})
|
|
387
|
+
types = sorted({t for _, t in by_year})
|
|
388
|
+
write_csv(SUMMARY / "by_year_type.csv",
|
|
389
|
+
[{"year": y, **{t: by_year.get((y, t), 0) for t in types}} for y in years], ["year"] + types)
|
|
390
|
+
mods = collections.Counter()
|
|
391
|
+
ymods = collections.Counter()
|
|
392
|
+
for r in rows:
|
|
393
|
+
for m in filter(None, (x.strip() for x in r.get("modality", "").split(";"))):
|
|
394
|
+
mods[m] += 1
|
|
395
|
+
ymods[(r["year"], m)] += 1
|
|
396
|
+
mlist = [m for m, _ in mods.most_common()]
|
|
397
|
+
write_csv(SUMMARY / "by_year_modality.csv",
|
|
398
|
+
[{"year": y, **{m: ymods.get((y, m), 0) for m in mlist}} for y in years], ["year"] + mlist)
|
|
399
|
+
cats = collections.Counter(r["category"] for r in rows)
|
|
400
|
+
tiers = collections.Counter(r["tier"] for r in rows)
|
|
401
|
+
ents = collections.Counter()
|
|
402
|
+
for r in rows:
|
|
403
|
+
for c in filter(None, (x.strip() for x in re.sub(r"^[^;]* registry: [^;]*;?", "", r.get("entities") or "").split(";"))):
|
|
404
|
+
ents[c] += 1
|
|
405
|
+
ft = collections.Counter(r.get("fulltext_status") or "not tried" for r in rows)
|
|
406
|
+
write_json(SUMMARY / "stats.json", {
|
|
407
|
+
"rows": len(rows), "tiers": dict(tiers), "categories": dict(cats.most_common()),
|
|
408
|
+
"modalities": dict(mods.most_common()), "entity_mentions": dict(ents.most_common(60)),
|
|
409
|
+
"fulltext": dict(ft), "resource_types": dict(collections.Counter(r["resource_type"] for r in rows)),
|
|
410
|
+
})
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
if __name__ == "__main__":
|
|
414
|
+
main()
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Render <catalogue>/summary/findings.md from <catalogue>/progress.csv.
|
|
3
|
+
|
|
4
|
+
Regenerated on every run so the numbers in it always match the catalogue:
|
|
5
|
+
growth per year, data layers per year, origin papers of registry entries, when
|
|
6
|
+
each named entity first appears in a catalogued title, the most-cited work
|
|
7
|
+
overall and per data layer, the most-starred repositories, websites, and
|
|
8
|
+
full-text coverage.
|
|
9
|
+
|
|
10
|
+
Usage (normally run by sourcelens update)
|
|
11
|
+
sourcelens __step summary
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import collections
|
|
17
|
+
import re
|
|
18
|
+
|
|
19
|
+
from sourcelens.buildcatalog.build_progress import Classifier
|
|
20
|
+
from sourcelens.common.agelit import RESEARCH, config, config_root, log, read_csv, today
|
|
21
|
+
from sourcelens.common.settings import default_topic
|
|
22
|
+
|
|
23
|
+
OUT = RESEARCH / "summary" / "findings.md"
|
|
24
|
+
PAPERS = {"article", "review", "preprint", "report", "book chapter", "conference paper", "thesis", "dataset"}
|
|
25
|
+
REGISTRY = re.compile(r"([^;]*) registry: ([^;]*)")
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def summary_modalities(classify: dict) -> list[str]:
|
|
29
|
+
"""classify.summary_modalities, else every modality rule, in order."""
|
|
30
|
+
return list(classify.get("summary_modalities") or []) or list((classify.get("modality") or {}).keys())
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def md_table(header: list[str], rows: list[list]) -> list[str]:
|
|
34
|
+
out = ["| " + " | ".join(header) + " |", "|" + "---|" * len(header)]
|
|
35
|
+
for r in rows:
|
|
36
|
+
out.append("| " + " | ".join(str(x).replace("|", "/").replace("\n", " ") for x in r) + " |")
|
|
37
|
+
return out + [""]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def cite(r: dict) -> str:
|
|
41
|
+
link = f"[{r['doi']}](https://doi.org/{r['doi']})" if r.get("doi") else r.get("url", "")
|
|
42
|
+
return f"{r['title'][:120]} ({r['authors'].split(',')[0] if r.get('authors') else ''}, {r['year']}) {link}"
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def main() -> None:
|
|
46
|
+
rows = read_csv(RESEARCH / "progress.csv")
|
|
47
|
+
papers = [r for r in rows if r["resource_type"] in PAPERS]
|
|
48
|
+
repos = [r for r in rows if r["resource_type"] in ("repository", "software package")]
|
|
49
|
+
sites = [r for r in rows if r["resource_type"] in ("website", "database", "web calculator")]
|
|
50
|
+
years = sorted({r["year"] for r in papers if r["year"]})
|
|
51
|
+
classify = config("classify")
|
|
52
|
+
modalities = summary_modalities(classify)
|
|
53
|
+
clf = Classifier(classify)
|
|
54
|
+
topic = config_root().get("topic") or default_topic()
|
|
55
|
+
L = [f"# Findings: {topic}", "",
|
|
56
|
+
f"_Generated {today()} by `sourcelens` (summary step) from `progress.csv` "
|
|
57
|
+
f"({len(rows)} rows: {len(papers)} papers/preprints/reviews, {len(repos)} repositories and packages, "
|
|
58
|
+
f"{len(sites)} websites and databases). Classification is rule-based (classify section of the configuration); "
|
|
59
|
+
f"treat counts as indicative._", ""]
|
|
60
|
+
|
|
61
|
+
# growth
|
|
62
|
+
L += ["## Publications per year", ""]
|
|
63
|
+
by = collections.Counter((r["year"], r["resource_type"]) for r in papers)
|
|
64
|
+
tiers = collections.Counter((r["year"], r["tier"]) for r in papers)
|
|
65
|
+
types = ["article", "review", "preprint"]
|
|
66
|
+
L += md_table(["year"] + types + ["landmark", "core", "related", "total"],
|
|
67
|
+
[[y] + [by.get((y, t), 0) for t in types] + [tiers.get((y, t), 0) for t in ("landmark", "core", "related")]
|
|
68
|
+
+ [sum(v for (yy, _), v in by.items() if yy == y)] for y in years])
|
|
69
|
+
|
|
70
|
+
# modality per year
|
|
71
|
+
if modalities:
|
|
72
|
+
L += ["## Data layers over time (core + landmark papers)", "",
|
|
73
|
+
"A paper can use several layers, so rows do not sum to the totals above.", ""]
|
|
74
|
+
cm = collections.Counter()
|
|
75
|
+
for r in papers:
|
|
76
|
+
if r["tier"] in ("core", "landmark"):
|
|
77
|
+
for m in (x.strip() for x in r["modality"].split(";")):
|
|
78
|
+
if m:
|
|
79
|
+
cm[(r["year"], m)] += 1
|
|
80
|
+
L += md_table(["year"] + modalities, [[y] + [cm.get((y, m), 0) for m in modalities] for y in years])
|
|
81
|
+
|
|
82
|
+
# origin papers of registry entries: the chronology of the field, which
|
|
83
|
+
# title mentions alone cannot give
|
|
84
|
+
origin = [r for r in rows if REGISTRY.search(r.get("entities") or "")]
|
|
85
|
+
if origin:
|
|
86
|
+
L += ["## Origin papers of registry entries, oldest first", "",
|
|
87
|
+
"Papers that introduced an entry of a reference registry; the ids are the registry's identifiers.", ""]
|
|
88
|
+
rows_o = []
|
|
89
|
+
for r in sorted(origin, key=lambda r: r["date"] or "9999"):
|
|
90
|
+
m = REGISTRY.search(r["entities"])
|
|
91
|
+
rows_o.append([r["date"], (m.group(1).strip() + ": " + m.group(2))[:90], cite(r)])
|
|
92
|
+
L += md_table(["date", "registry ids", "paper"], rows_o)
|
|
93
|
+
|
|
94
|
+
# first appearance of named entities in titles
|
|
95
|
+
entities = classify.get("entities")
|
|
96
|
+
if entities is None:
|
|
97
|
+
entities = classify.get("clock_names") or {}
|
|
98
|
+
first = []
|
|
99
|
+
for name, pat in entities.items():
|
|
100
|
+
flags = 0
|
|
101
|
+
if pat.startswith("(?i)"):
|
|
102
|
+
pat, flags = pat[4:], re.I
|
|
103
|
+
rx = re.compile(r"(?<![\w-])(?:" + pat + r")(?![\w-])", flags)
|
|
104
|
+
hits = sorted((r for r in papers if rx.search(r["title"])), key=lambda r: r["date"] or "9999")
|
|
105
|
+
if hits:
|
|
106
|
+
n = sum(1 for r in papers if name in (r.get("entities") or ""))
|
|
107
|
+
first.append([hits[0]["date"], name, n, cite(hits[0])])
|
|
108
|
+
first.sort()
|
|
109
|
+
if first:
|
|
110
|
+
L += ["## When each named entity first appears in a catalogued title", "",
|
|
111
|
+
"Earliest title mention within the catalogue window, not necessarily the origin paper; "
|
|
112
|
+
"`papers` counts every catalogued paper whose title, abstract or keywords mention it.", ""]
|
|
113
|
+
L += md_table(["first title mention", "entity", "papers", "earliest paper"], first)
|
|
114
|
+
|
|
115
|
+
# most cited
|
|
116
|
+
def num(r):
|
|
117
|
+
try:
|
|
118
|
+
return int(r.get("cited_by") or 0)
|
|
119
|
+
except ValueError:
|
|
120
|
+
return 0
|
|
121
|
+
L += ["## Most-cited papers", ""]
|
|
122
|
+
top = sorted(papers, key=num, reverse=True)[:30]
|
|
123
|
+
L += md_table(["cited by", "type", "category", "paper"], [[num(r), r["resource_type"], r["category"], cite(r)] for r in top])
|
|
124
|
+
if modalities:
|
|
125
|
+
L += [f"## Most-cited {clf.origin_category} papers per data layer", ""]
|
|
126
|
+
for m in modalities:
|
|
127
|
+
sub = [r for r in papers if m in r["modality"] and r["category"] == clf.origin_category]
|
|
128
|
+
if not sub:
|
|
129
|
+
continue
|
|
130
|
+
L += [f"**{m}** ({len(sub)} {clf.origin_category} papers)", ""]
|
|
131
|
+
L += md_table(["cited by", "paper"], [[num(r), cite(r)] for r in sorted(sub, key=num, reverse=True)[:5]])
|
|
132
|
+
|
|
133
|
+
# repositories
|
|
134
|
+
def stars(r):
|
|
135
|
+
m = re.search(r"stars=(\d+)", r.get("details", ""))
|
|
136
|
+
return int(m.group(1)) if m else 0
|
|
137
|
+
if repos:
|
|
138
|
+
L += ["## Most-starred repositories", ""]
|
|
139
|
+
L += md_table(["stars", "created", "repository"], [[stars(r), r["date"], f"[{r['title'][:100]}]({r['url']})"]
|
|
140
|
+
for r in sorted(repos, key=stars, reverse=True)[:30]])
|
|
141
|
+
if sites:
|
|
142
|
+
L += ["## Websites, databases and calculators", ""]
|
|
143
|
+
L += md_table(["first archived", "type", "site"], [[r["date"] or "n/a", r["resource_type"], f"[{r['title'][:90]}]({r['url']})"]
|
|
144
|
+
for r in sorted(sites, key=lambda r: r["date"] or "9999")])
|
|
145
|
+
|
|
146
|
+
# full text
|
|
147
|
+
ft = collections.Counter(r["fulltext_status"] or "not tried" for r in papers)
|
|
148
|
+
ftt = collections.Counter((r["tier"], r["fulltext_status"] or "not tried") for r in papers)
|
|
149
|
+
L += ["## Full-text coverage", ""]
|
|
150
|
+
L += md_table(["tier", "ok (md+txt)", "partial", "none (no open copy)", "not tried"],
|
|
151
|
+
[[t] + [ftt.get((t, s), 0) for s in ("ok", "partial", "none", "not tried")]
|
|
152
|
+
for t in ("landmark", "core", "related")])
|
|
153
|
+
L += [f"PDFs on disk: {sum(1 for r in papers if r['fulltext_pdf'])}; Markdown: "
|
|
154
|
+
f"{sum(1 for r in papers if r['fulltext_md'])}; plain text: {sum(1 for r in papers if r['fulltext_txt'])}.", ""]
|
|
155
|
+
|
|
156
|
+
OUT.parent.mkdir(parents=True, exist_ok=True)
|
|
157
|
+
OUT.write_text("\n".join(L), encoding="utf-8")
|
|
158
|
+
log(f"wrote {OUT} ({len(L)} lines); full text {dict(ft)}")
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
if __name__ == "__main__":
|
|
162
|
+
main()
|