sourcelens 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. sourcelens/__init__.py +7 -0
  2. sourcelens/__main__.py +7 -0
  3. sourcelens/buildcatalog/__init__.py +0 -0
  4. sourcelens/buildcatalog/build_progress.py +414 -0
  5. sourcelens/buildcatalog/summarise.py +162 -0
  6. sourcelens/buildcatalog/test_incremental.py +116 -0
  7. sourcelens/cli/__init__.py +0 -0
  8. sourcelens/cli/main.py +699 -0
  9. sourcelens/common/__init__.py +0 -0
  10. sourcelens/common/agelit.py +440 -0
  11. sourcelens/common/crossref.py +101 -0
  12. sourcelens/common/flags.py +47 -0
  13. sourcelens/common/hits.py +50 -0
  14. sourcelens/common/pubmed.py +135 -0
  15. sourcelens/common/settings.py +367 -0
  16. sourcelens/common/store.py +112 -0
  17. sourcelens/common/terms.py +217 -0
  18. sourcelens/defaults/default.yaml +733 -0
  19. sourcelens/localseeds/__init__.py +0 -0
  20. sourcelens/localseeds/extract_seeds.py +240 -0
  21. sourcelens/pullliturature/__init__.py +0 -0
  22. sourcelens/pullliturature/enrich_openalex.py +74 -0
  23. sourcelens/pullliturature/enrich_preprints.py +66 -0
  24. sourcelens/pullliturature/fetch_fulltext.py +662 -0
  25. sourcelens/pullliturature/jats.py +269 -0
  26. sourcelens/pullliturature/resolve_seeds.py +119 -0
  27. sourcelens/pullliturature/search_arxiv.py +141 -0
  28. sourcelens/pullliturature/search_europepmc.py +282 -0
  29. sourcelens/pullliturature/search_openalex.py +222 -0
  30. sourcelens/pullliturature/search_pubmed.py +111 -0
  31. sourcelens/pullrepos/__init__.py +0 -0
  32. sourcelens/pullrepos/build_repos.py +379 -0
  33. sourcelens/pullrepos/mine_links.py +146 -0
  34. sourcelens/pullrepos/search_github.py +108 -0
  35. sourcelens/query/__init__.py +0 -0
  36. sourcelens/query/catalog.py +191 -0
  37. sourcelens/query/export_refs.py +71 -0
  38. sourcelens/query/refs.py +566 -0
  39. sourcelens/query/search_catalog.py +53 -0
  40. sourcelens/websites/__init__.py +0 -0
  41. sourcelens/websites/build_websites.py +181 -0
  42. sourcelens-0.0.1.dist-info/METADATA +548 -0
  43. sourcelens-0.0.1.dist-info/RECORD +46 -0
  44. sourcelens-0.0.1.dist-info/WHEEL +4 -0
  45. sourcelens-0.0.1.dist-info/entry_points.txt +2 -0
  46. sourcelens-0.0.1.dist-info/licenses/LICENSE +674 -0
sourcelens/__init__.py ADDED
@@ -0,0 +1,7 @@
1
+ """sourcelens: collect the latest research on any topic.
2
+
3
+ Papers and preprints, open-access full texts, code repositories, software
4
+ packages and websites, in one chronological, append-only catalogue.
5
+ """
6
+
7
+ __version__ = "0.0.1"
sourcelens/__main__.py ADDED
@@ -0,0 +1,7 @@
1
+ """python -m sourcelens"""
2
+
3
+ import sys
4
+
5
+ from sourcelens.cli.main import main
6
+
7
+ sys.exit(main(sys.argv[1:]))
File without changes
@@ -0,0 +1,414 @@
1
+ #!/usr/bin/env python3
2
+ """Build <catalogue>/progress.csv: every resource, oldest first.
3
+
4
+ Merges the record store (articles and preprints from the literature searches
5
+ and the local seeds), repositories, packages and websites into one catalogue,
6
+ annotated with the classify section of the configuration.
7
+
8
+ INCREMENTAL BY DESIGN. progress.csv is never rebuilt from scratch:
9
+ * a row whose uid is already in the file keeps its `added_on` date and
10
+ its hand-edited columns (`notes`, `user_tags`) untouched;
11
+ * a resource not yet in the file is added with `added_on` = today;
12
+ * a row that no longer comes out of the pipeline (search terms changed,
13
+ source withdrew it) is kept and flagged in `status`, never deleted;
14
+ * the file is then re-sorted by date, so it stays chronological.
15
+ Every run writes <catalogue>/changelog/added_<date>.csv with only the rows it
16
+ added, and appends one line to <catalogue>/changelog/runs.csv.
17
+
18
+ Outputs
19
+ <catalogue>/progress.csv the catalogue (scope: focused + seeds)
20
+ <catalogue>/corpus/broad_hits.csv records matched only by broad terms
21
+ <catalogue>/corpus/before_window.csv seeds published before the window
22
+ <catalogue>/corpus/offtopic_seeds.csv local seeds outside the topic
23
+ <catalogue>/corpus/references.csv full citation fields for export
24
+ <catalogue>/changelog/added_<date>.csv, runs.csv
25
+ <catalogue>/summary/*.csv, stats.json counts
26
+
27
+ Usage (normally run by sourcelens update)
28
+ sourcelens __step catalogue
29
+ """
30
+
31
+ from __future__ import annotations
32
+
33
+ import collections
34
+ import re
35
+ from pathlib import Path
36
+
37
+ from sourcelens.common.agelit import ( # noqa: E402
38
+ CORPUS,
39
+ FULLTEXT,
40
+ REPOS,
41
+ RESEARCH,
42
+ SEEDS,
43
+ WEBSITES,
44
+ config,
45
+ log,
46
+ query_window,
47
+ read_csv,
48
+ today,
49
+ write_csv,
50
+ write_json,
51
+ )
52
+ from sourcelens.common.store import Store # noqa: E402
53
+
54
+ PROGRESS = RESEARCH / "progress.csv"
55
+ CHANGELOG = RESEARCH / "changelog"
56
+ SUMMARY = RESEARCH / "summary"
57
+
58
+ COLUMNS = [
59
+ "added_on", "date", "year", "resource_type", "tier", "category", "modality",
60
+ "entities", "species", "title", "authors", "venue", "doi", "pmid", "pmcid",
61
+ "url", "code_links", "related", "open_access", "license", "fulltext_status",
62
+ "fulltext_pdf", "fulltext_md", "fulltext_txt", "metadata_file", "cited_by",
63
+ "details", "matched_groups", "found_by", "local_refs", "status", "uid", "notes", "user_tags",
64
+ ]
65
+ USER_COLUMNS = {"notes", "user_tags"}
66
+ KEEP_ON_UPDATE = {"added_on"} | USER_COLUMNS
67
+ TIER_ORDER = {"landmark": 0, "core": 1, "related": 2}
68
+
69
+
70
+ # --------------------------------------------------------------------------
71
+ # classification
72
+ # --------------------------------------------------------------------------
73
+
74
+ class Classifier:
75
+ def __init__(self, cfg: dict):
76
+ f = re.I
77
+ self.modality = {k: re.compile(v, f) for k, v in (cfg.get("modality") or {}).items()}
78
+ self.species = {k: re.compile(v, f) for k, v in (cfg.get("species") or {}).items()}
79
+ self.category = []
80
+ for rule in cfg.get("category") or []:
81
+ self.category.append({
82
+ "name": rule["name"],
83
+ "pub_types": re.compile(rule["pub_types"], f) if rule.get("pub_types") else None,
84
+ "title": re.compile(rule["title"], f) if rule.get("title") else None,
85
+ "text": re.compile(rule["text"], f) if rule.get("text") else None,
86
+ })
87
+ self.core_title = re.compile(cfg["core_title_terms"], f) if cfg.get("core_title_terms") else None
88
+ # catalogues configured before 0.0.1 name registry roles clock_*
89
+ roles = set(cfg.get("landmark_roles") or [])
90
+ self.landmark_roles = roles | {r.replace("clock_", "registry_", 1) for r in roles}
91
+ self.default_category = cfg.get("default_category") or "application/association"
92
+ self.origin_category = cfg.get("origin_category") or "method development"
93
+ self.core_categories = list(cfg.get("core_categories") or
94
+ [self.origin_category, "benchmark/comparison", "review", "software/resource"])
95
+ entities = cfg.get("entities")
96
+ if entities is None:
97
+ entities = cfg.get("clock_names") or {} # name used before 0.0.1
98
+ self.entities = {k: re.compile(r"(?<![\w-])(?:" + v + r")(?![\w-])" if not v.startswith("(?i)")
99
+ else "(?i)(?<![\\w-])(?:" + v[4:] + ")(?![\\w-])")
100
+ for k, v in (entities or {}).items()}
101
+
102
+ def annotate(self, rec: dict) -> dict:
103
+ title = rec.get("title") or ""
104
+ text = " ".join(str(rec.get(k) or "") for k in ("title", "abstract", "keywords", "mesh"))
105
+ pubtypes = rec.get("pub_types") or ""
106
+ mods = [k for k, rx in self.modality.items() if rx.search(text)]
107
+ sp = [k for k, rx in self.species.items() if rx.search(text) and k != "human"]
108
+ if ("human" in self.species and self.species["human"].search(text)) or "Humans" in (rec.get("mesh") or ""):
109
+ sp = ["human"] + sp
110
+ cat = self.default_category
111
+ for rule in self.category:
112
+ if ((rule["pub_types"] and rule["pub_types"].search(pubtypes))
113
+ or (rule["title"] and rule["title"].search(title))
114
+ or (rule["text"] and rule["text"].search(text))):
115
+ cat = rule["name"]
116
+ break
117
+ ents = [k for k, rx in self.entities.items() if rx.search(text)]
118
+ return {"modality": "; ".join(mods), "species": "; ".join(sp), "category": cat,
119
+ "entities": "; ".join(ents),
120
+ "_core_title": bool(self.core_title and self.core_title.search(title))}
121
+
122
+
123
+ # --------------------------------------------------------------------------
124
+ # rows
125
+ # --------------------------------------------------------------------------
126
+
127
+ def resource_type(rec: dict, category: str) -> str:
128
+ pt = (rec.get("pub_types") or "").lower()
129
+ if rec.get("is_preprint") or rec.get("source_db") == "PPR" or "preprint" in pt:
130
+ return "preprint"
131
+ if category == "review":
132
+ return "review"
133
+ for key in ("dataset", "software", "book chapter", "conference paper", "thesis", "report"):
134
+ if key in pt:
135
+ return key
136
+ return "article"
137
+
138
+
139
+ def best_date(rec: dict) -> str:
140
+ d = (rec.get("pub_date") or "").strip()
141
+ if re.fullmatch(r"\d{4}-\d{2}-\d{2}", d):
142
+ return d
143
+ y = (rec.get("pub_year") or d[:4]).strip()
144
+ return f"{y}-01-01" if re.fullmatch(r"\d{4}", y) else ""
145
+
146
+
147
+ SERVER_NAMES = {"biorxiv": "bioRxiv", "medrxiv": "medRxiv", "research square": "Research Square",
148
+ "preprints.org": "Preprints.org", "ssrn": "SSRN", "psyarxiv": "PsyArXiv"}
149
+
150
+
151
+ def venue_name(v: str) -> str:
152
+ """One spelling per preprint server ("bioRxiv : the preprint server for biology" -> "bioRxiv")."""
153
+ low = (v or "").strip().lower()
154
+ for key, name in SERVER_NAMES.items():
155
+ if low.startswith(key):
156
+ return name
157
+ return v
158
+
159
+
160
+ def first_authors(a: str, n: int = 3) -> str:
161
+ parts = [p.strip() for p in (a or "").rstrip(".").split(",") if p.strip()]
162
+ return ", ".join(parts[:n]) + (" et al." if len(parts) > n else "")
163
+
164
+
165
+ def article_row(rec, ann, groups, found_by, seeds, ft, links, cites) -> dict:
166
+ uid = rec["uid"]
167
+ doi = rec.get("doi") or ""
168
+ url = f"https://doi.org/{doi}" if doi else (
169
+ f"https://pubmed.ncbi.nlm.nih.gov/{rec['pmid']}/" if rec.get("pmid") else rec.get("url", ""))
170
+ f = ft.get(uid, {})
171
+ folder = f.get("folder", "")
172
+
173
+ def fp(name: str, flag: str) -> str:
174
+ return f"{folder}/{name}" if folder and str(f.get(flag)) == "True" else ""
175
+
176
+ seed_info = seeds.get(doi, {})
177
+ cited = max(int(rec.get("cited_by") or 0), int(cites.get(uid, 0) or 0))
178
+ oa = "yes" if (rec.get("is_open_access") or f.get("status") in ("ok", "partial")) else (
179
+ "unknown" if not f else "no")
180
+ date = best_date(rec)
181
+ return {
182
+ "date": date, "year": date[:4], "resource_type": resource_type(rec, ann["category"]),
183
+ "category": ann["category"], "modality": ann["modality"],
184
+ "entities": "; ".join(filter(None, [seed_info.get("registry_ids", ""), ann["entities"]])),
185
+ "species": ann["species"], "title": rec.get("title", ""),
186
+ "authors": first_authors(rec.get("authors", "")), "venue": venue_name(rec.get("journal", "")),
187
+ "doi": doi, "pmid": rec.get("pmid", ""), "pmcid": rec.get("pmcid", ""), "url": url,
188
+ "code_links": "; ".join(links.get(uid, [])), "related": "",
189
+ "open_access": oa, "license": f.get("license") or rec.get("license", ""),
190
+ "fulltext_status": f.get("status", ""),
191
+ "fulltext_pdf": fp("paper.pdf", "has_pdf"), "fulltext_md": fp("paper.md", "has_md"),
192
+ "fulltext_txt": fp("paper.txt", "has_txt"),
193
+ "metadata_file": f"{folder}/metadata.json" if folder else "",
194
+ "cited_by": cited or "", "matched_groups": "; ".join(sorted(groups)),
195
+ "found_by": "; ".join(sorted(found_by)), "local_refs": seed_info.get("refs", ""),
196
+ "uid": uid,
197
+ }
198
+
199
+
200
+ def load_seeds() -> dict:
201
+ """doi -> {roles, refs, registry_ids} from the reference folders."""
202
+ out: dict[str, dict] = {}
203
+ for s in read_csv(SEEDS / "local_seeds.csv"):
204
+ d = out.setdefault(s["doi"], {"roles": set(), "refs": set(), "ids": {}})
205
+ role = s["role"].replace("clock_", "registry_", 1) # seeds written before 0.0.1
206
+ d["roles"].add(role)
207
+ d["refs"].add(f"{s['source_repo']}:{role}")
208
+ rid = s.get("registry_id") or s.get("clock_name") or ""
209
+ if rid:
210
+ d["ids"].setdefault(s["source_repo"], set()).add(rid)
211
+ for d in out.values():
212
+ d["refs"] = "; ".join(sorted(d["refs"]))
213
+ d["registry_ids"] = "; ".join(f"{repo} registry: " + ", ".join(sorted(ids))
214
+ for repo, ids in sorted(d["ids"].items()))
215
+ return out
216
+
217
+
218
+ def extra_rows(path: Path) -> list[dict]:
219
+ """Rows produced by the repository / website harvesters, already in catalogue shape."""
220
+ return [r for r in read_csv(path) if r.get("uid")]
221
+
222
+
223
+ # --------------------------------------------------------------------------
224
+ # main
225
+ # --------------------------------------------------------------------------
226
+
227
+ def main() -> None:
228
+ cfg = config("classify")
229
+ clf = Classifier(cfg)
230
+ start, end = query_window()
231
+ store = Store()
232
+ log(f"store: {len(store)} records; window {start} .. {end}")
233
+
234
+ groups: dict[str, set] = collections.defaultdict(set)
235
+ scopes: dict[str, set] = collections.defaultdict(set)
236
+ found: dict[str, set] = collections.defaultdict(set)
237
+ for h in read_csv(CORPUS / "search_hits.csv"):
238
+ uid = store.find({"uid": h["uid"]}) or h["uid"]
239
+ groups[uid].add(h["group"])
240
+ scopes[uid].add(h["scope"])
241
+ found[uid].add(h["source"])
242
+
243
+ seeds = load_seeds()
244
+ ft = {r["uid"]: r for r in read_csv(FULLTEXT / "fulltext_index.csv")}
245
+ links: dict[str, list] = collections.defaultdict(list)
246
+ for r in read_csv(REPOS / "paper_links.csv"):
247
+ if r["url"] not in links[r["uid"]]:
248
+ links[r["uid"]].append(r["url"])
249
+ openalex = {r["uid"]: r for r in read_csv(CORPUS / "citations.csv")}
250
+ cites = {u: r.get("cited_by", 0) for u, r in openalex.items()}
251
+ # preprint <-> published version, both directions
252
+ pp_links: dict[str, str] = {}
253
+ for r in read_csv(CORPUS / "preprint_links.csv"):
254
+ if r.get("published_doi"):
255
+ pp_links[r["preprint_doi"]] = f"published as {r['published_doi']}"
256
+ pp_links[r["published_doi"]] = f"preprint {r['preprint_doi']}"
257
+
258
+ rows, broad, pre, offtopic = {}, [], [], []
259
+ for rec in store.values():
260
+ uid = rec["uid"]
261
+ ann = clf.annotate(rec)
262
+ doi = rec.get("doi", "")
263
+ sd = seeds.get(doi) if doi else None
264
+ roles = sd["roles"] if sd else set()
265
+ is_landmark = bool(roles & clf.landmark_roles)
266
+ focused = "focused" in scopes.get(uid, set())
267
+ fb = set(found.get(uid, set()))
268
+ if sd:
269
+ fb.add("local seed")
270
+ if "registry_origin" in roles:
271
+ # the paper that introduced a registry entry is development work by definition
272
+ ann = {**ann, "category": clf.origin_category}
273
+ row = article_row(rec, ann, groups.get(uid, set()), fb, seeds, ft, links, cites)
274
+ # the indexes give some records only a year (stored YYYY-01-01);
275
+ # OpenAlex's exact date replaces it when it falls in the same year
276
+ oa_date = openalex.get(uid, {}).get("publication_date", "")
277
+ if row["date"].endswith("-01-01") and oa_date[:4] == row["date"][:4] and oa_date != row["date"]:
278
+ row["date"] = oa_date
279
+ if doi in pp_links:
280
+ row["related"] = pp_links[doi]
281
+ elif rec.get("related_doi"): # arXiv entries that name their journal version
282
+ row["related"] = f"published as {rec['related_doi']}"
283
+ if not row["date"]:
284
+ row["status"] = "undated"
285
+ in_window = not row["date"] or start <= row["date"] <= end
286
+ if is_landmark:
287
+ row["tier"] = "landmark"
288
+ elif focused and (ann["_core_title"] or ann["category"] in clf.core_categories):
289
+ row["tier"] = "core"
290
+ elif focused:
291
+ row["tier"] = "related"
292
+ elif sd and ann["_core_title"]:
293
+ row["tier"] = "related"
294
+ else:
295
+ row["tier"] = ""
296
+ if row["tier"] and not in_window:
297
+ pre.append(row)
298
+ elif row["tier"]:
299
+ rows[uid] = row
300
+ elif sd:
301
+ offtopic.append(row)
302
+ elif "broad" in scopes.get(uid, set()):
303
+ broad.append(row)
304
+
305
+ for path in (REPOS / "repositories_catalogue.csv", WEBSITES / "websites_catalogue.csv"):
306
+ for r in extra_rows(path):
307
+ rows[r["uid"]] = {k: r.get(k, "") for k in COLUMNS if k not in KEEP_ON_UPDATE}
308
+
309
+ # ---- incremental merge with the existing progress.csv -----------------
310
+ existing = {r["uid"]: r for r in read_csv(PROGRESS)}
311
+ run = today()
312
+ added = []
313
+ merged = {}
314
+ for uid, new in rows.items():
315
+ old = existing.get(uid)
316
+ if old is None:
317
+ new = {**new, "added_on": run}
318
+ added.append(new)
319
+ else:
320
+ new = {**new, **{k: old.get(k, "") for k in KEEP_ON_UPDATE}}
321
+ new.setdefault("status", "")
322
+ merged[uid] = new
323
+ kept = 0
324
+ for uid, old in existing.items():
325
+ if uid not in merged:
326
+ old["status"] = "no longer matched by pipeline (kept)"
327
+ merged[uid] = old
328
+ kept += 1
329
+
330
+ def sort_key(r):
331
+ return (r.get("date") or "9999", TIER_ORDER.get(r.get("tier"), 9), r.get("title", "").lower())
332
+
333
+ final = sorted(merged.values(), key=sort_key)
334
+ write_csv(PROGRESS, final, COLUMNS)
335
+ CHANGELOG.mkdir(parents=True, exist_ok=True)
336
+ if added:
337
+ # update_all.sh builds several times a day; the day's file collects
338
+ # every pass's additions rather than keeping only the last pass
339
+ day_file = CHANGELOG / f"added_{run}.csv"
340
+ today_rows = {r["uid"]: r for r in read_csv(day_file)}
341
+ today_rows.update({r["uid"]: r for r in added})
342
+ write_csv(day_file, sorted(today_rows.values(), key=sort_key), COLUMNS)
343
+ runs = read_csv(CHANGELOG / "runs.csv")
344
+ runs.append({"run_date": run, "window_end": end, "rows_total": len(final), "rows_added": len(added),
345
+ "rows_kept_unmatched": kept, "broad_only": len(broad), "store_records": len(store)})
346
+ write_csv(CHANGELOG / "runs.csv", runs, list(runs[-1].keys()))
347
+
348
+ write_references(final, store)
349
+ write_csv(CORPUS / "broad_hits.csv", sorted(broad, key=sort_key), COLUMNS)
350
+ write_csv(CORPUS / "before_window.csv", sorted(pre, key=sort_key), COLUMNS)
351
+ write_csv(CORPUS / "offtopic_seeds.csv", sorted(offtopic, key=sort_key), COLUMNS)
352
+ summarise(final)
353
+ log(f"progress.csv: {len(final)} rows ({len(added)} added this run, {kept} kept though unmatched); "
354
+ f"broad-only {len(broad)}, before the window {len(pre)}, off-topic seeds {len(offtopic)}")
355
+
356
+
357
+ REF_COLUMNS = ["uid", "resource_type", "date", "year", "authors", "title", "venue", "volume", "issue",
358
+ "pages", "doi", "pmid", "pmcid", "url"]
359
+
360
+
361
+ def write_references(rows: list[dict], store: Store) -> None:
362
+ """<catalogue>/corpus/references.csv: full citation fields for every catalogue row.
363
+
364
+ progress.csv keeps only the first three authors; reference export (both
365
+ sourcelens export) needs all of them plus volume / issue / pages, so they are
366
+ written here once per build instead of reading the record store.
367
+ """
368
+ out = []
369
+ for r in rows:
370
+ rec = store.get(r["uid"]) or {}
371
+ out.append({
372
+ "uid": r["uid"], "resource_type": r.get("resource_type", ""), "date": r.get("date", ""),
373
+ "year": r.get("year", ""),
374
+ "authors": (rec.get("authors") or r.get("authors", "")).strip().rstrip("."),
375
+ "title": r.get("title", ""), "venue": r.get("venue", ""),
376
+ "volume": rec.get("volume", "") or "", "issue": rec.get("issue", "") or "",
377
+ "pages": rec.get("pages", "") or "", "doi": r.get("doi", ""), "pmid": r.get("pmid", ""),
378
+ "pmcid": r.get("pmcid", ""), "url": r.get("url", ""),
379
+ })
380
+ write_csv(CORPUS / "references.csv", out, REF_COLUMNS)
381
+
382
+
383
+ def summarise(rows: list[dict]) -> None:
384
+ SUMMARY.mkdir(parents=True, exist_ok=True)
385
+ by_year = collections.Counter((r["year"], r["resource_type"]) for r in rows)
386
+ years = sorted({y for y, _ in by_year})
387
+ types = sorted({t for _, t in by_year})
388
+ write_csv(SUMMARY / "by_year_type.csv",
389
+ [{"year": y, **{t: by_year.get((y, t), 0) for t in types}} for y in years], ["year"] + types)
390
+ mods = collections.Counter()
391
+ ymods = collections.Counter()
392
+ for r in rows:
393
+ for m in filter(None, (x.strip() for x in r.get("modality", "").split(";"))):
394
+ mods[m] += 1
395
+ ymods[(r["year"], m)] += 1
396
+ mlist = [m for m, _ in mods.most_common()]
397
+ write_csv(SUMMARY / "by_year_modality.csv",
398
+ [{"year": y, **{m: ymods.get((y, m), 0) for m in mlist}} for y in years], ["year"] + mlist)
399
+ cats = collections.Counter(r["category"] for r in rows)
400
+ tiers = collections.Counter(r["tier"] for r in rows)
401
+ ents = collections.Counter()
402
+ for r in rows:
403
+ for c in filter(None, (x.strip() for x in re.sub(r"^[^;]* registry: [^;]*;?", "", r.get("entities") or "").split(";"))):
404
+ ents[c] += 1
405
+ ft = collections.Counter(r.get("fulltext_status") or "not tried" for r in rows)
406
+ write_json(SUMMARY / "stats.json", {
407
+ "rows": len(rows), "tiers": dict(tiers), "categories": dict(cats.most_common()),
408
+ "modalities": dict(mods.most_common()), "entity_mentions": dict(ents.most_common(60)),
409
+ "fulltext": dict(ft), "resource_types": dict(collections.Counter(r["resource_type"] for r in rows)),
410
+ })
411
+
412
+
413
+ if __name__ == "__main__":
414
+ main()
@@ -0,0 +1,162 @@
1
+ #!/usr/bin/env python3
2
+ """Render <catalogue>/summary/findings.md from <catalogue>/progress.csv.
3
+
4
+ Regenerated on every run so the numbers in it always match the catalogue:
5
+ growth per year, data layers per year, origin papers of registry entries, when
6
+ each named entity first appears in a catalogued title, the most-cited work
7
+ overall and per data layer, the most-starred repositories, websites, and
8
+ full-text coverage.
9
+
10
+ Usage (normally run by sourcelens update)
11
+ sourcelens __step summary
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import collections
17
+ import re
18
+
19
+ from sourcelens.buildcatalog.build_progress import Classifier
20
+ from sourcelens.common.agelit import RESEARCH, config, config_root, log, read_csv, today
21
+ from sourcelens.common.settings import default_topic
22
+
23
+ OUT = RESEARCH / "summary" / "findings.md"
24
+ PAPERS = {"article", "review", "preprint", "report", "book chapter", "conference paper", "thesis", "dataset"}
25
+ REGISTRY = re.compile(r"([^;]*) registry: ([^;]*)")
26
+
27
+
28
+ def summary_modalities(classify: dict) -> list[str]:
29
+ """classify.summary_modalities, else every modality rule, in order."""
30
+ return list(classify.get("summary_modalities") or []) or list((classify.get("modality") or {}).keys())
31
+
32
+
33
+ def md_table(header: list[str], rows: list[list]) -> list[str]:
34
+ out = ["| " + " | ".join(header) + " |", "|" + "---|" * len(header)]
35
+ for r in rows:
36
+ out.append("| " + " | ".join(str(x).replace("|", "/").replace("\n", " ") for x in r) + " |")
37
+ return out + [""]
38
+
39
+
40
+ def cite(r: dict) -> str:
41
+ link = f"[{r['doi']}](https://doi.org/{r['doi']})" if r.get("doi") else r.get("url", "")
42
+ return f"{r['title'][:120]} ({r['authors'].split(',')[0] if r.get('authors') else ''}, {r['year']}) {link}"
43
+
44
+
45
+ def main() -> None:
46
+ rows = read_csv(RESEARCH / "progress.csv")
47
+ papers = [r for r in rows if r["resource_type"] in PAPERS]
48
+ repos = [r for r in rows if r["resource_type"] in ("repository", "software package")]
49
+ sites = [r for r in rows if r["resource_type"] in ("website", "database", "web calculator")]
50
+ years = sorted({r["year"] for r in papers if r["year"]})
51
+ classify = config("classify")
52
+ modalities = summary_modalities(classify)
53
+ clf = Classifier(classify)
54
+ topic = config_root().get("topic") or default_topic()
55
+ L = [f"# Findings: {topic}", "",
56
+ f"_Generated {today()} by `sourcelens` (summary step) from `progress.csv` "
57
+ f"({len(rows)} rows: {len(papers)} papers/preprints/reviews, {len(repos)} repositories and packages, "
58
+ f"{len(sites)} websites and databases). Classification is rule-based (classify section of the configuration); "
59
+ f"treat counts as indicative._", ""]
60
+
61
+ # growth
62
+ L += ["## Publications per year", ""]
63
+ by = collections.Counter((r["year"], r["resource_type"]) for r in papers)
64
+ tiers = collections.Counter((r["year"], r["tier"]) for r in papers)
65
+ types = ["article", "review", "preprint"]
66
+ L += md_table(["year"] + types + ["landmark", "core", "related", "total"],
67
+ [[y] + [by.get((y, t), 0) for t in types] + [tiers.get((y, t), 0) for t in ("landmark", "core", "related")]
68
+ + [sum(v for (yy, _), v in by.items() if yy == y)] for y in years])
69
+
70
+ # modality per year
71
+ if modalities:
72
+ L += ["## Data layers over time (core + landmark papers)", "",
73
+ "A paper can use several layers, so rows do not sum to the totals above.", ""]
74
+ cm = collections.Counter()
75
+ for r in papers:
76
+ if r["tier"] in ("core", "landmark"):
77
+ for m in (x.strip() for x in r["modality"].split(";")):
78
+ if m:
79
+ cm[(r["year"], m)] += 1
80
+ L += md_table(["year"] + modalities, [[y] + [cm.get((y, m), 0) for m in modalities] for y in years])
81
+
82
+ # origin papers of registry entries: the chronology of the field, which
83
+ # title mentions alone cannot give
84
+ origin = [r for r in rows if REGISTRY.search(r.get("entities") or "")]
85
+ if origin:
86
+ L += ["## Origin papers of registry entries, oldest first", "",
87
+ "Papers that introduced an entry of a reference registry; the ids are the registry's identifiers.", ""]
88
+ rows_o = []
89
+ for r in sorted(origin, key=lambda r: r["date"] or "9999"):
90
+ m = REGISTRY.search(r["entities"])
91
+ rows_o.append([r["date"], (m.group(1).strip() + ": " + m.group(2))[:90], cite(r)])
92
+ L += md_table(["date", "registry ids", "paper"], rows_o)
93
+
94
+ # first appearance of named entities in titles
95
+ entities = classify.get("entities")
96
+ if entities is None:
97
+ entities = classify.get("clock_names") or {}
98
+ first = []
99
+ for name, pat in entities.items():
100
+ flags = 0
101
+ if pat.startswith("(?i)"):
102
+ pat, flags = pat[4:], re.I
103
+ rx = re.compile(r"(?<![\w-])(?:" + pat + r")(?![\w-])", flags)
104
+ hits = sorted((r for r in papers if rx.search(r["title"])), key=lambda r: r["date"] or "9999")
105
+ if hits:
106
+ n = sum(1 for r in papers if name in (r.get("entities") or ""))
107
+ first.append([hits[0]["date"], name, n, cite(hits[0])])
108
+ first.sort()
109
+ if first:
110
+ L += ["## When each named entity first appears in a catalogued title", "",
111
+ "Earliest title mention within the catalogue window, not necessarily the origin paper; "
112
+ "`papers` counts every catalogued paper whose title, abstract or keywords mention it.", ""]
113
+ L += md_table(["first title mention", "entity", "papers", "earliest paper"], first)
114
+
115
+ # most cited
116
+ def num(r):
117
+ try:
118
+ return int(r.get("cited_by") or 0)
119
+ except ValueError:
120
+ return 0
121
+ L += ["## Most-cited papers", ""]
122
+ top = sorted(papers, key=num, reverse=True)[:30]
123
+ L += md_table(["cited by", "type", "category", "paper"], [[num(r), r["resource_type"], r["category"], cite(r)] for r in top])
124
+ if modalities:
125
+ L += [f"## Most-cited {clf.origin_category} papers per data layer", ""]
126
+ for m in modalities:
127
+ sub = [r for r in papers if m in r["modality"] and r["category"] == clf.origin_category]
128
+ if not sub:
129
+ continue
130
+ L += [f"**{m}** ({len(sub)} {clf.origin_category} papers)", ""]
131
+ L += md_table(["cited by", "paper"], [[num(r), cite(r)] for r in sorted(sub, key=num, reverse=True)[:5]])
132
+
133
+ # repositories
134
+ def stars(r):
135
+ m = re.search(r"stars=(\d+)", r.get("details", ""))
136
+ return int(m.group(1)) if m else 0
137
+ if repos:
138
+ L += ["## Most-starred repositories", ""]
139
+ L += md_table(["stars", "created", "repository"], [[stars(r), r["date"], f"[{r['title'][:100]}]({r['url']})"]
140
+ for r in sorted(repos, key=stars, reverse=True)[:30]])
141
+ if sites:
142
+ L += ["## Websites, databases and calculators", ""]
143
+ L += md_table(["first archived", "type", "site"], [[r["date"] or "n/a", r["resource_type"], f"[{r['title'][:90]}]({r['url']})"]
144
+ for r in sorted(sites, key=lambda r: r["date"] or "9999")])
145
+
146
+ # full text
147
+ ft = collections.Counter(r["fulltext_status"] or "not tried" for r in papers)
148
+ ftt = collections.Counter((r["tier"], r["fulltext_status"] or "not tried") for r in papers)
149
+ L += ["## Full-text coverage", ""]
150
+ L += md_table(["tier", "ok (md+txt)", "partial", "none (no open copy)", "not tried"],
151
+ [[t] + [ftt.get((t, s), 0) for s in ("ok", "partial", "none", "not tried")]
152
+ for t in ("landmark", "core", "related")])
153
+ L += [f"PDFs on disk: {sum(1 for r in papers if r['fulltext_pdf'])}; Markdown: "
154
+ f"{sum(1 for r in papers if r['fulltext_md'])}; plain text: {sum(1 for r in papers if r['fulltext_txt'])}.", ""]
155
+
156
+ OUT.parent.mkdir(parents=True, exist_ok=True)
157
+ OUT.write_text("\n".join(L), encoding="utf-8")
158
+ log(f"wrote {OUT} ({len(L)} lines); full text {dict(ft)}")
159
+
160
+
161
+ if __name__ == "__main__":
162
+ main()