check-your-advisor 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- check_your_advisor-0.2.0/LICENSE +21 -0
- check_your_advisor-0.2.0/PKG-INFO +207 -0
- check_your_advisor-0.2.0/README.md +170 -0
- check_your_advisor-0.2.0/pyproject.toml +71 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/__init__.py +17 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/__main__.py +8 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/cache.py +160 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/citations.py +586 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/cli.py +1485 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/config.py +177 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/corpus.py +54 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/download_sources.py +267 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/engine.py +288 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/export.py +125 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/http_client.py +342 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/journals.py +1175 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/pdf_utils.py +229 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/profile/__init__.py +202 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/profile/caveats.py +426 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/profile/charts.py +791 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/profile/cohesion.py +178 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/profile/html_report.py +801 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/profile/impact.py +265 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/profile/metrics.py +643 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/profile/ranking.py +837 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/profile/report.py +2277 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/profile/roles.py +576 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/profile/scoring.py +759 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/profile/svg.py +285 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/pubmed_api.py +684 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/reports.py +101 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor/theses.py +1147 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor.egg-info/PKG-INFO +207 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor.egg-info/SOURCES.txt +56 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor.egg-info/dependency_links.txt +1 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor.egg-info/entry_points.txt +2 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor.egg-info/requires.txt +10 -0
- check_your_advisor-0.2.0/scripts/check_your_advisor.egg-info/top_level.txt +1 -0
- check_your_advisor-0.2.0/setup.cfg +4 -0
- check_your_advisor-0.2.0/tests/test_charts.py +660 -0
- check_your_advisor-0.2.0/tests/test_citations.py +621 -0
- check_your_advisor-0.2.0/tests/test_cli_profile.py +636 -0
- check_your_advisor-0.2.0/tests/test_cohesion.py +243 -0
- check_your_advisor-0.2.0/tests/test_html_report.py +588 -0
- check_your_advisor-0.2.0/tests/test_http_client.py +357 -0
- check_your_advisor-0.2.0/tests/test_identity_cli.py +156 -0
- check_your_advisor-0.2.0/tests/test_identity_filter.py +158 -0
- check_your_advisor-0.2.0/tests/test_impact.py +376 -0
- check_your_advisor-0.2.0/tests/test_journals.py +795 -0
- check_your_advisor-0.2.0/tests/test_pdf_validation.py +290 -0
- check_your_advisor-0.2.0/tests/test_profile.py +1118 -0
- check_your_advisor-0.2.0/tests/test_provenance.py +191 -0
- check_your_advisor-0.2.0/tests/test_pubmed_parse.py +284 -0
- check_your_advisor-0.2.0/tests/test_ranking.py +733 -0
- check_your_advisor-0.2.0/tests/test_scoring.py +469 -0
- check_your_advisor-0.2.0/tests/test_search_query.py +111 -0
- check_your_advisor-0.2.0/tests/test_stdio_encoding.py +106 -0
- check_your_advisor-0.2.0/tests/test_theses.py +613 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 AschoofAlpha
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,207 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: check-your-advisor
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Read what PubMed records about a researcher and report it back as facts with denominators — first-author slots, time to a first slot, turnover, byline position. 查导师:把发表记录读成带分母的事实,不替你下结论。
|
|
5
|
+
Author: AschoofAlpha
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/AschoofAlpha/check-your-advisor
|
|
8
|
+
Project-URL: Repository, https://github.com/AschoofAlpha/check-your-advisor
|
|
9
|
+
Project-URL: Issues, https://github.com/AschoofAlpha/check-your-advisor/issues
|
|
10
|
+
Project-URL: Chinese README, https://github.com/AschoofAlpha/check-your-advisor/blob/main/README.zh-CN.md
|
|
11
|
+
Keywords: pubmed,advisor,graduate-school,bibliometrics,research-integrity,scientometrics
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Environment :: Console
|
|
14
|
+
Classifier: Intended Audience :: Science/Research
|
|
15
|
+
Classifier: Intended Audience :: Education
|
|
16
|
+
Classifier: Natural Language :: Chinese (Simplified)
|
|
17
|
+
Classifier: Natural Language :: English
|
|
18
|
+
Classifier: Operating System :: OS Independent
|
|
19
|
+
Classifier: Programming Language :: Python :: 3
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
23
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
24
|
+
Classifier: Topic :: Scientific/Engineering :: Information Analysis
|
|
25
|
+
Classifier: Topic :: Text Processing :: Indexing
|
|
26
|
+
Requires-Python: >=3.10
|
|
27
|
+
Description-Content-Type: text/markdown
|
|
28
|
+
License-File: LICENSE
|
|
29
|
+
Provides-Extra: pdf
|
|
30
|
+
Requires-Dist: PyMuPDF>=1.23; extra == "pdf"
|
|
31
|
+
Provides-Extra: xlsx
|
|
32
|
+
Requires-Dist: openpyxl>=3.1; extra == "xlsx"
|
|
33
|
+
Provides-Extra: all
|
|
34
|
+
Requires-Dist: PyMuPDF>=1.23; extra == "all"
|
|
35
|
+
Requires-Dist: openpyxl>=3.1; extra == "all"
|
|
36
|
+
Dynamic: license-file
|
|
37
|
+
|
|
38
|
+
# Check Your Advisor
|
|
39
|
+
|
|
40
|
+
*[中文说明](README.zh-CN.md)*
|
|
41
|
+
|
|
42
|
+
You are choosing a PhD or master's advisor. You have a name, a lab page written
|
|
43
|
+
by the lab, and no way to check any of it. This reads what PubMed records about
|
|
44
|
+
that person and reports it back as facts with denominators.
|
|
45
|
+
|
|
46
|
+
**It does not tell you whether the advisor is good.** That judgement stays
|
|
47
|
+
yours. What it does is replace hearsay with counts that carry the population
|
|
48
|
+
they were counted over.
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
python scripts/run.py harvest --author "Wang Wei" --orcid 0000-0002-1825-0097 \
|
|
52
|
+
--years-back 10 --output-dir ./record --no-download
|
|
53
|
+
python scripts/run.py cite --output-dir ./record
|
|
54
|
+
python scripts/run.py profile --pi-name "Wang Wei" --output-dir ./record
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
Three commands, one HTML report. Nothing to install: the standard library only.
|
|
58
|
+
|
|
59
|
+
## What it answers
|
|
60
|
+
|
|
61
|
+
- who has appeared in this group over the window
|
|
62
|
+
- who the first-author slots went to, and how concentrated they are
|
|
63
|
+
- how long a newcomer waits before getting one
|
|
64
|
+
- how long people stay before they stop appearing
|
|
65
|
+
- where the PI sits in their own bylines — last, corresponding, or doing the work
|
|
66
|
+
- how much comes out per year, and whether it goes to the same few journals
|
|
67
|
+
- citation counts and an h-index, with the coverage they were computed over
|
|
68
|
+
- one composite score out of 100, with every component's raw input printed
|
|
69
|
+
|
|
70
|
+
And, once you supply a table it cannot fetch: what those journals are rated in
|
|
71
|
+
the edition you looked them up in, and **how many people finished a degree here
|
|
72
|
+
without a single indexed paper** — the one group every other number is blind to.
|
|
73
|
+
|
|
74
|
+
## What it refuses, and why the distinction matters
|
|
75
|
+
|
|
76
|
+
Four different things stand behind "it does not say whether the advisor is
|
|
77
|
+
good", and the report keeps them apart on the page:
|
|
78
|
+
|
|
79
|
+
| | Examples | Why |
|
|
80
|
+
|---|---|---|
|
|
81
|
+
| **Computed** | citation counts, h-index, composite score, rank, star band | measurements, printed with denominators |
|
|
82
|
+
| **Refused** | percentile, quantile, letter grade, fitted trend, any ordering of *people* | a decision — see below |
|
|
83
|
+
| **Waiting on a file** | impact factor, JCR quartile, CAS partition, graduate roster | licensed or defended databases; the tool ships schemas, not a crawler |
|
|
84
|
+
| **Never visible** | lab culture, whether the PI is decent, people who left before finishing | in no database at all |
|
|
85
|
+
|
|
86
|
+
A percentile is not withheld here, it is *uncomputable*: nothing in the
|
|
87
|
+
calculation ever holds more than the few corpora on one page, so there is no
|
|
88
|
+
reference population to take a position in. Letter grades are refused while
|
|
89
|
+
star bands are produced — that split between two coarsenings of one number is a
|
|
90
|
+
decision, and the report says so rather than leaving it looking like an
|
|
91
|
+
oversight.
|
|
92
|
+
|
|
93
|
+
## The corpus decides everything
|
|
94
|
+
|
|
95
|
+
Every number is computed over the papers `harvest` kept. If papers by a
|
|
96
|
+
different person with the same name get in, the roster, the time-to-first-author
|
|
97
|
+
and the turnover figures are all wrong — and wrong in a way that looks perfectly
|
|
98
|
+
normal on the page.
|
|
99
|
+
|
|
100
|
+
**Give `harvest` at least one unique identifier.** In descending order of
|
|
101
|
+
strength:
|
|
102
|
+
|
|
103
|
+
| Flag | Strength |
|
|
104
|
+
|---|---|
|
|
105
|
+
| `--orcid 0000-0002-...` | Strongest. One is worth all the rest |
|
|
106
|
+
| `--email-domain your-university.edu.cn` | The corresponding author's address. Repeatable |
|
|
107
|
+
| `--affiliation-keyword "..."` | Weakest — it fails on same-name colleagues inside one university system |
|
|
108
|
+
|
|
109
|
+
The report **refuses to render** if the harvest recorded no evidence at all
|
|
110
|
+
(gate G3). But the dangerous case passes that gate: weak evidence produces a
|
|
111
|
+
complete, normal-looking report about several people. A real run for one Chinese
|
|
112
|
+
surgeon, keyed on a province name rather than the full institution, returned 28
|
|
113
|
+
records spanning gastrointestinal surgery, analytical chemistry, structural
|
|
114
|
+
biology, soil microbiology and machine learning — and scored 78.2 out of 100.
|
|
115
|
+
|
|
116
|
+
**Section 19 is the check for that.** It removes the PI, who is on every record
|
|
117
|
+
by construction, and asks which records are still tied together by a shared
|
|
118
|
+
co-author. One person's output is held together by the people they work with;
|
|
119
|
+
two people sharing a name have no reason to share anyone else. It applies no
|
|
120
|
+
threshold — the measurements do not support one, and the README of a tool like
|
|
121
|
+
this should not pretend otherwise — it prints the clusters and their journals
|
|
122
|
+
and hands the reading to you. It is usually not subtle.
|
|
123
|
+
|
|
124
|
+
## Two tables you fill in by hand
|
|
125
|
+
|
|
126
|
+
Impact factor, JCR quartile and CAS partition live in licensed products with no
|
|
127
|
+
free redistributable source. Degree-thesis libraries defend against scraping.
|
|
128
|
+
So this ships the schema, the worklist and the join, and no crawler:
|
|
129
|
+
|
|
130
|
+
```bash
|
|
131
|
+
python scripts/run.py journal-worklist --output-dir ./record
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
That writes a CSV holding **only the journals this corpus actually uses** —
|
|
135
|
+
typically a couple of dozen, not the twenty thousand in the world — pre-filled
|
|
136
|
+
with ISSN and paper count, indicator columns blank. Fill it from whichever
|
|
137
|
+
source you have access to, then:
|
|
138
|
+
|
|
139
|
+
```bash
|
|
140
|
+
python scripts/run.py profile --output-dir ./record \
|
|
141
|
+
--journal-table ./record/journal_worklist_*.csv \
|
|
142
|
+
--thesis-roster ./record/theses.csv
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
Two columns are **required** in the journal table: `版本来源` (which edition —
|
|
146
|
+
official, rising-star, folk, or JCR) and `数据获取日期`. Without them a
|
|
147
|
+
partition number is unfalsifiable two years from now. Where a journal has rows
|
|
148
|
+
from two editions, both are shown and neither wins; disagreements are listed.
|
|
149
|
+
|
|
150
|
+
The graduate roster is the more important of the two. Export the advisor's
|
|
151
|
+
supervised theses from a degree library and it produces the number PubMed
|
|
152
|
+
structurally cannot: **how many graduates have no indexed paper at all.** Add a
|
|
153
|
+
romanised-name column or the Chinese roster will not match the English bylines.
|
|
154
|
+
Note that people who enrolled and left before finishing are in no library
|
|
155
|
+
either — the roster narrows the missing group, it does not close it.
|
|
156
|
+
|
|
157
|
+
## Install
|
|
158
|
+
|
|
159
|
+
As a Claude Code skill:
|
|
160
|
+
|
|
161
|
+
```bash
|
|
162
|
+
git clone https://github.com/AschoofAlpha/check-your-advisor.git \
|
|
163
|
+
~/.claude/skills/check-your-advisor
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
Or run it directly as a CLI from anywhere — `scripts/run.py` is the single
|
|
167
|
+
entry point and needs no installation.
|
|
168
|
+
|
|
169
|
+
## Optional extras
|
|
170
|
+
|
|
171
|
+
Both degrade with a tested fallback; neither is required.
|
|
172
|
+
|
|
173
|
+
- **PyMuPDF** (AGPL-3.0, so deliberately not a hard dependency of an MIT
|
|
174
|
+
project) lets PDF identity validation quarantine a wrong file. Without it
|
|
175
|
+
every download is still checked for the `%PDF-` magic.
|
|
176
|
+
- **openpyxl** enables the `.xlsx` export. Without it the same data is written
|
|
177
|
+
as a timestamped CSV.
|
|
178
|
+
|
|
179
|
+
## Tests
|
|
180
|
+
|
|
181
|
+
```bash
|
|
182
|
+
python tests/run_all.py
|
|
183
|
+
python tests/run_all.py --block-third-party
|
|
184
|
+
```
|
|
185
|
+
|
|
186
|
+
1925 assertions across 19 files. The second run installs an import hook that
|
|
187
|
+
blocks `requests`, `urllib3`, `pandas`, `numpy`, `matplotlib`, `fitz` and
|
|
188
|
+
`openpyxl` inside each test process. It is the only thing that keeps "no install
|
|
189
|
+
needed" true rather than merely claimed: exactly three assertions behave
|
|
190
|
+
differently without them, and all three are the cases that need a real PDF file.
|
|
191
|
+
|
|
192
|
+
## Known limitations
|
|
193
|
+
|
|
194
|
+
- The report is in English; the command-line logs are in Chinese. Not a decision
|
|
195
|
+
anyone would defend — it is where the tool grew up, and unifying it is open.
|
|
196
|
+
- `cite` has no cache of its own. `--max-age-days N` reuses counts from the
|
|
197
|
+
previous run's file, which is the workaround.
|
|
198
|
+
- Journal name matching falls back to a token heuristic for corpora harvested
|
|
199
|
+
before ISSN capture existed. Re-harvest to get the exact join.
|
|
200
|
+
|
|
201
|
+
## License
|
|
202
|
+
|
|
203
|
+
MIT. See [LICENSE](LICENSE).
|
|
204
|
+
|
|
205
|
+
Citation counts come from OpenAlex, Semantic Scholar and Europe PMC, none of
|
|
206
|
+
which requires a key. Bibliographic records come from NCBI E-utilities. This
|
|
207
|
+
project ships no licensed data.
|
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
# Check Your Advisor
|
|
2
|
+
|
|
3
|
+
*[中文说明](README.zh-CN.md)*
|
|
4
|
+
|
|
5
|
+
You are choosing a PhD or master's advisor. You have a name, a lab page written
|
|
6
|
+
by the lab, and no way to check any of it. This reads what PubMed records about
|
|
7
|
+
that person and reports it back as facts with denominators.
|
|
8
|
+
|
|
9
|
+
**It does not tell you whether the advisor is good.** That judgement stays
|
|
10
|
+
yours. What it does is replace hearsay with counts that carry the population
|
|
11
|
+
they were counted over.
|
|
12
|
+
|
|
13
|
+
```bash
|
|
14
|
+
python scripts/run.py harvest --author "Wang Wei" --orcid 0000-0002-1825-0097 \
|
|
15
|
+
--years-back 10 --output-dir ./record --no-download
|
|
16
|
+
python scripts/run.py cite --output-dir ./record
|
|
17
|
+
python scripts/run.py profile --pi-name "Wang Wei" --output-dir ./record
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
Three commands, one HTML report. Nothing to install: the standard library only.
|
|
21
|
+
|
|
22
|
+
## What it answers
|
|
23
|
+
|
|
24
|
+
- who has appeared in this group over the window
|
|
25
|
+
- who the first-author slots went to, and how concentrated they are
|
|
26
|
+
- how long a newcomer waits before getting one
|
|
27
|
+
- how long people stay before they stop appearing
|
|
28
|
+
- where the PI sits in their own bylines — last, corresponding, or doing the work
|
|
29
|
+
- how much comes out per year, and whether it goes to the same few journals
|
|
30
|
+
- citation counts and an h-index, with the coverage they were computed over
|
|
31
|
+
- one composite score out of 100, with every component's raw input printed
|
|
32
|
+
|
|
33
|
+
And, once you supply a table it cannot fetch: what those journals are rated in
|
|
34
|
+
the edition you looked them up in, and **how many people finished a degree here
|
|
35
|
+
without a single indexed paper** — the one group every other number is blind to.
|
|
36
|
+
|
|
37
|
+
## What it refuses, and why the distinction matters
|
|
38
|
+
|
|
39
|
+
Four different things stand behind "it does not say whether the advisor is
|
|
40
|
+
good", and the report keeps them apart on the page:
|
|
41
|
+
|
|
42
|
+
| | Examples | Why |
|
|
43
|
+
|---|---|---|
|
|
44
|
+
| **Computed** | citation counts, h-index, composite score, rank, star band | measurements, printed with denominators |
|
|
45
|
+
| **Refused** | percentile, quantile, letter grade, fitted trend, any ordering of *people* | a decision — see below |
|
|
46
|
+
| **Waiting on a file** | impact factor, JCR quartile, CAS partition, graduate roster | licensed or defended databases; the tool ships schemas, not a crawler |
|
|
47
|
+
| **Never visible** | lab culture, whether the PI is decent, people who left before finishing | in no database at all |
|
|
48
|
+
|
|
49
|
+
A percentile is not withheld here, it is *uncomputable*: nothing in the
|
|
50
|
+
calculation ever holds more than the few corpora on one page, so there is no
|
|
51
|
+
reference population to take a position in. Letter grades are refused while
|
|
52
|
+
star bands are produced — that split between two coarsenings of one number is a
|
|
53
|
+
decision, and the report says so rather than leaving it looking like an
|
|
54
|
+
oversight.
|
|
55
|
+
|
|
56
|
+
## The corpus decides everything
|
|
57
|
+
|
|
58
|
+
Every number is computed over the papers `harvest` kept. If papers by a
|
|
59
|
+
different person with the same name get in, the roster, the time-to-first-author
|
|
60
|
+
and the turnover figures are all wrong — and wrong in a way that looks perfectly
|
|
61
|
+
normal on the page.
|
|
62
|
+
|
|
63
|
+
**Give `harvest` at least one unique identifier.** In descending order of
|
|
64
|
+
strength:
|
|
65
|
+
|
|
66
|
+
| Flag | Strength |
|
|
67
|
+
|---|---|
|
|
68
|
+
| `--orcid 0000-0002-...` | Strongest. One is worth all the rest |
|
|
69
|
+
| `--email-domain your-university.edu.cn` | The corresponding author's address. Repeatable |
|
|
70
|
+
| `--affiliation-keyword "..."` | Weakest — it fails on same-name colleagues inside one university system |
|
|
71
|
+
|
|
72
|
+
The report **refuses to render** if the harvest recorded no evidence at all
|
|
73
|
+
(gate G3). But the dangerous case passes that gate: weak evidence produces a
|
|
74
|
+
complete, normal-looking report about several people. A real run for one Chinese
|
|
75
|
+
surgeon, keyed on a province name rather than the full institution, returned 28
|
|
76
|
+
records spanning gastrointestinal surgery, analytical chemistry, structural
|
|
77
|
+
biology, soil microbiology and machine learning — and scored 78.2 out of 100.
|
|
78
|
+
|
|
79
|
+
**Section 19 is the check for that.** It removes the PI, who is on every record
|
|
80
|
+
by construction, and asks which records are still tied together by a shared
|
|
81
|
+
co-author. One person's output is held together by the people they work with;
|
|
82
|
+
two people sharing a name have no reason to share anyone else. It applies no
|
|
83
|
+
threshold — the measurements do not support one, and the README of a tool like
|
|
84
|
+
this should not pretend otherwise — it prints the clusters and their journals
|
|
85
|
+
and hands the reading to you. It is usually not subtle.
|
|
86
|
+
|
|
87
|
+
## Two tables you fill in by hand
|
|
88
|
+
|
|
89
|
+
Impact factor, JCR quartile and CAS partition live in licensed products with no
|
|
90
|
+
free redistributable source. Degree-thesis libraries defend against scraping.
|
|
91
|
+
So this ships the schema, the worklist and the join, and no crawler:
|
|
92
|
+
|
|
93
|
+
```bash
|
|
94
|
+
python scripts/run.py journal-worklist --output-dir ./record
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
That writes a CSV holding **only the journals this corpus actually uses** —
|
|
98
|
+
typically a couple of dozen, not the twenty thousand in the world — pre-filled
|
|
99
|
+
with ISSN and paper count, indicator columns blank. Fill it from whichever
|
|
100
|
+
source you have access to, then:
|
|
101
|
+
|
|
102
|
+
```bash
|
|
103
|
+
python scripts/run.py profile --output-dir ./record \
|
|
104
|
+
--journal-table ./record/journal_worklist_*.csv \
|
|
105
|
+
--thesis-roster ./record/theses.csv
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
Two columns are **required** in the journal table: `版本来源` (which edition —
|
|
109
|
+
official, rising-star, folk, or JCR) and `数据获取日期`. Without them a
|
|
110
|
+
partition number is unfalsifiable two years from now. Where a journal has rows
|
|
111
|
+
from two editions, both are shown and neither wins; disagreements are listed.
|
|
112
|
+
|
|
113
|
+
The graduate roster is the more important of the two. Export the advisor's
|
|
114
|
+
supervised theses from a degree library and it produces the number PubMed
|
|
115
|
+
structurally cannot: **how many graduates have no indexed paper at all.** Add a
|
|
116
|
+
romanised-name column or the Chinese roster will not match the English bylines.
|
|
117
|
+
Note that people who enrolled and left before finishing are in no library
|
|
118
|
+
either — the roster narrows the missing group, it does not close it.
|
|
119
|
+
|
|
120
|
+
## Install
|
|
121
|
+
|
|
122
|
+
As a Claude Code skill:
|
|
123
|
+
|
|
124
|
+
```bash
|
|
125
|
+
git clone https://github.com/AschoofAlpha/check-your-advisor.git \
|
|
126
|
+
~/.claude/skills/check-your-advisor
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
Or run it directly as a CLI from anywhere — `scripts/run.py` is the single
|
|
130
|
+
entry point and needs no installation.
|
|
131
|
+
|
|
132
|
+
## Optional extras
|
|
133
|
+
|
|
134
|
+
Both degrade with a tested fallback; neither is required.
|
|
135
|
+
|
|
136
|
+
- **PyMuPDF** (AGPL-3.0, so deliberately not a hard dependency of an MIT
|
|
137
|
+
project) lets PDF identity validation quarantine a wrong file. Without it
|
|
138
|
+
every download is still checked for the `%PDF-` magic.
|
|
139
|
+
- **openpyxl** enables the `.xlsx` export. Without it the same data is written
|
|
140
|
+
as a timestamped CSV.
|
|
141
|
+
|
|
142
|
+
## Tests
|
|
143
|
+
|
|
144
|
+
```bash
|
|
145
|
+
python tests/run_all.py
|
|
146
|
+
python tests/run_all.py --block-third-party
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
1925 assertions across 19 files. The second run installs an import hook that
|
|
150
|
+
blocks `requests`, `urllib3`, `pandas`, `numpy`, `matplotlib`, `fitz` and
|
|
151
|
+
`openpyxl` inside each test process. It is the only thing that keeps "no install
|
|
152
|
+
needed" true rather than merely claimed: exactly three assertions behave
|
|
153
|
+
differently without them, and all three are the cases that need a real PDF file.
|
|
154
|
+
|
|
155
|
+
## Known limitations
|
|
156
|
+
|
|
157
|
+
- The report is in English; the command-line logs are in Chinese. Not a decision
|
|
158
|
+
anyone would defend — it is where the tool grew up, and unifying it is open.
|
|
159
|
+
- `cite` has no cache of its own. `--max-age-days N` reuses counts from the
|
|
160
|
+
previous run's file, which is the workaround.
|
|
161
|
+
- Journal name matching falls back to a token heuristic for corpora harvested
|
|
162
|
+
before ISSN capture existed. Re-harvest to get the exact join.
|
|
163
|
+
|
|
164
|
+
## License
|
|
165
|
+
|
|
166
|
+
MIT. See [LICENSE](LICENSE).
|
|
167
|
+
|
|
168
|
+
Citation counts come from OpenAlex, Semantic Scholar and Europe PMC, none of
|
|
169
|
+
which requires a key. Bibliographic records come from NCBI E-utilities. This
|
|
170
|
+
project ships no licensed data.
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "check-your-advisor"
|
|
7
|
+
version = "0.2.0"
|
|
8
|
+
description = "Read what PubMed records about a researcher and report it back as facts with denominators — first-author slots, time to a first slot, turnover, byline position. 查导师:把发表记录读成带分母的事实,不替你下结论。"
|
|
9
|
+
readme = { file = "README.md", content-type = "text/markdown" }
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [{ name = "AschoofAlpha" }]
|
|
14
|
+
keywords = [
|
|
15
|
+
"pubmed",
|
|
16
|
+
"advisor",
|
|
17
|
+
"graduate-school",
|
|
18
|
+
"bibliometrics",
|
|
19
|
+
"research-integrity",
|
|
20
|
+
"scientometrics",
|
|
21
|
+
]
|
|
22
|
+
classifiers = [
|
|
23
|
+
"Development Status :: 4 - Beta",
|
|
24
|
+
"Environment :: Console",
|
|
25
|
+
"Intended Audience :: Science/Research",
|
|
26
|
+
"Intended Audience :: Education",
|
|
27
|
+
"Natural Language :: Chinese (Simplified)",
|
|
28
|
+
"Natural Language :: English",
|
|
29
|
+
"Operating System :: OS Independent",
|
|
30
|
+
"Programming Language :: Python :: 3",
|
|
31
|
+
"Programming Language :: Python :: 3.10",
|
|
32
|
+
"Programming Language :: Python :: 3.11",
|
|
33
|
+
"Programming Language :: Python :: 3.12",
|
|
34
|
+
"Programming Language :: Python :: 3.13",
|
|
35
|
+
"Topic :: Scientific/Engineering :: Information Analysis",
|
|
36
|
+
"Topic :: Text Processing :: Indexing",
|
|
37
|
+
]
|
|
38
|
+
|
|
39
|
+
# No `dependencies` key, and that is the point rather than an omission. Every
|
|
40
|
+
# import in this package is from the standard library, and the test suite proves
|
|
41
|
+
# it: `python tests/run_all.py --block-third-party` installs an import hook that
|
|
42
|
+
# makes requests, urllib3, pandas, numpy, matplotlib, fitz and openpyxl
|
|
43
|
+
# unimportable, and 1922 of 1925 assertions still pass. Adding a dependency here
|
|
44
|
+
# would make that suite a claim nobody checks.
|
|
45
|
+
|
|
46
|
+
[project.optional-dependencies]
|
|
47
|
+
# Both are genuinely optional and both have a tested fallback path. PyMuPDF is
|
|
48
|
+
# AGPL-3.0 and this package is MIT, which is why it is not a hard dependency:
|
|
49
|
+
# without it a downloaded PDF is still checked for the `%PDF-` magic bytes, it
|
|
50
|
+
# just cannot be quarantined by content. openpyxl only decides whether the
|
|
51
|
+
# corpus export is .xlsx or a timestamped .csv.
|
|
52
|
+
pdf = ["PyMuPDF>=1.23"]
|
|
53
|
+
xlsx = ["openpyxl>=3.1"]
|
|
54
|
+
all = ["PyMuPDF>=1.23", "openpyxl>=3.1"]
|
|
55
|
+
|
|
56
|
+
[project.urls]
|
|
57
|
+
Homepage = "https://github.com/AschoofAlpha/check-your-advisor"
|
|
58
|
+
Repository = "https://github.com/AschoofAlpha/check-your-advisor"
|
|
59
|
+
Issues = "https://github.com/AschoofAlpha/check-your-advisor/issues"
|
|
60
|
+
"Chinese README" = "https://github.com/AschoofAlpha/check-your-advisor/blob/main/README.zh-CN.md"
|
|
61
|
+
|
|
62
|
+
[project.scripts]
|
|
63
|
+
check-your-advisor = "check_your_advisor.cli:main"
|
|
64
|
+
|
|
65
|
+
[tool.setuptools]
|
|
66
|
+
# The package lives under `scripts/` because this repository is also a Claude
|
|
67
|
+
# Code skill, where `SKILL.md` sits at the root and the code sits beside it.
|
|
68
|
+
# Moving the package to the root would break the skill layout; telling setuptools
|
|
69
|
+
# where to look costs one line and breaks nothing.
|
|
70
|
+
package-dir = { "" = "scripts" }
|
|
71
|
+
packages = ["check_your_advisor", "check_your_advisor.profile"]
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""
|
|
2
|
+
check-your-advisor — author-disambiguated PubMed harvesting and reference verification.
|
|
3
|
+
|
|
4
|
+
Two entry points over one shared HTTP and normalisation layer:
|
|
5
|
+
|
|
6
|
+
fetch/download Search PubMed, keep only the target researcher's papers via
|
|
7
|
+
ORCID + affiliation + email verification, race 8 open-access
|
|
8
|
+
sources for the PDF, then verify the downloaded file really
|
|
9
|
+
is the requested paper.
|
|
10
|
+
|
|
11
|
+
verify Check a bibliography against CrossRef and PubMed, including
|
|
12
|
+
bidirectional DOI <-> PMID resolution.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
__version__ = "0.3.1"
|
|
16
|
+
|
|
17
|
+
__all__ = ["__version__"]
|
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
"""
|
|
2
|
+
缓存模块(SQLite)
|
|
3
|
+
==================
|
|
4
|
+
替代原代码中仅通过 os.path.exists(pdf_path) 检查的简陋缓存。
|
|
5
|
+
|
|
6
|
+
改进:
|
|
7
|
+
- 记录每篇论文的元数据(DOI、PMID、下载状态、来源、时间)
|
|
8
|
+
- 支持过期清理
|
|
9
|
+
- 支持查询命中/未命中统计
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
import logging
|
|
13
|
+
import sqlite3
|
|
14
|
+
import threading
|
|
15
|
+
import time
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
logger = logging.getLogger("check_your_advisor.cache")
|
|
19
|
+
|
|
20
|
+
CREATE_TABLE_SQL = """
|
|
21
|
+
CREATE TABLE IF NOT EXISTS paper_cache (
|
|
22
|
+
doi TEXT,
|
|
23
|
+
pmid TEXT,
|
|
24
|
+
title TEXT,
|
|
25
|
+
pdf_path TEXT,
|
|
26
|
+
source TEXT,
|
|
27
|
+
status TEXT DEFAULT 'pending',
|
|
28
|
+
file_size_bytes INTEGER DEFAULT 0,
|
|
29
|
+
created_at REAL,
|
|
30
|
+
updated_at REAL,
|
|
31
|
+
PRIMARY KEY (pmid)
|
|
32
|
+
);
|
|
33
|
+
CREATE INDEX IF NOT EXISTS idx_doi ON paper_cache(doi);
|
|
34
|
+
CREATE INDEX IF NOT EXISTS idx_status ON paper_cache(status);
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class PaperCache:
|
|
39
|
+
def __init__(self, db_path: str = "paper_cache.db"):
|
|
40
|
+
self.db_path = db_path
|
|
41
|
+
db_parent = Path(db_path).parent
|
|
42
|
+
if str(db_parent) not in {"", "."}:
|
|
43
|
+
db_parent.mkdir(parents=True, exist_ok=True)
|
|
44
|
+
# check_same_thread=False 允许跨线程使用;加锁保证串行访问
|
|
45
|
+
self._conn = sqlite3.connect(db_path, check_same_thread=False)
|
|
46
|
+
self._conn.row_factory = sqlite3.Row
|
|
47
|
+
self._lock = threading.Lock()
|
|
48
|
+
self._init_db()
|
|
49
|
+
self.hits = 0
|
|
50
|
+
self.misses = 0
|
|
51
|
+
|
|
52
|
+
def _init_db(self):
|
|
53
|
+
with self._lock:
|
|
54
|
+
cursor = self._conn.cursor()
|
|
55
|
+
cursor.executescript(CREATE_TABLE_SQL)
|
|
56
|
+
self._conn.commit()
|
|
57
|
+
|
|
58
|
+
def lookup(self, pmid: str) -> dict | None:
|
|
59
|
+
"""查询缓存。返回 dict 或 None。"""
|
|
60
|
+
with self._lock:
|
|
61
|
+
cursor = self._conn.cursor()
|
|
62
|
+
cursor.execute("SELECT * FROM paper_cache WHERE pmid = ?", (pmid,))
|
|
63
|
+
row = cursor.fetchone()
|
|
64
|
+
|
|
65
|
+
if row and row["status"] == "downloaded" and row["pdf_path"]:
|
|
66
|
+
if Path(row["pdf_path"]).exists():
|
|
67
|
+
self.hits += 1
|
|
68
|
+
logger.debug(" 缓存命中: PMID %s", pmid)
|
|
69
|
+
return dict(row)
|
|
70
|
+
else:
|
|
71
|
+
logger.warning(" 缓存记录存在但文件丢失: %s", row["pdf_path"])
|
|
72
|
+
self.update(pmid, status="pending", pdf_path="")
|
|
73
|
+
self.misses += 1
|
|
74
|
+
return None
|
|
75
|
+
|
|
76
|
+
def update(self, pmid: str, **kwargs):
|
|
77
|
+
"""更新或插入缓存记录"""
|
|
78
|
+
now = time.time()
|
|
79
|
+
with self._lock:
|
|
80
|
+
existing = self._conn.execute(
|
|
81
|
+
"SELECT pmid FROM paper_cache WHERE pmid = ?", (pmid,)
|
|
82
|
+
).fetchone()
|
|
83
|
+
|
|
84
|
+
if existing:
|
|
85
|
+
sets = ", ".join(f"{k} = ?" for k in kwargs)
|
|
86
|
+
vals = list(kwargs.values()) + [now, pmid]
|
|
87
|
+
self._conn.execute(
|
|
88
|
+
f"UPDATE paper_cache SET {sets}, updated_at = ? WHERE pmid = ?",
|
|
89
|
+
vals,
|
|
90
|
+
)
|
|
91
|
+
else:
|
|
92
|
+
kwargs["pmid"] = pmid
|
|
93
|
+
kwargs["created_at"] = now
|
|
94
|
+
kwargs["updated_at"] = now
|
|
95
|
+
cols = ", ".join(kwargs.keys())
|
|
96
|
+
placeholders = ", ".join("?" for _ in kwargs)
|
|
97
|
+
self._conn.execute(
|
|
98
|
+
f"INSERT INTO paper_cache ({cols}) VALUES ({placeholders})",
|
|
99
|
+
list(kwargs.values()),
|
|
100
|
+
)
|
|
101
|
+
self._conn.commit()
|
|
102
|
+
|
|
103
|
+
def mark_downloaded(
|
|
104
|
+
self,
|
|
105
|
+
pmid: str,
|
|
106
|
+
pdf_path: str,
|
|
107
|
+
source: str,
|
|
108
|
+
file_size: int = 0,
|
|
109
|
+
doi: str = "",
|
|
110
|
+
title: str = "",
|
|
111
|
+
):
|
|
112
|
+
self.update(
|
|
113
|
+
pmid,
|
|
114
|
+
doi=doi,
|
|
115
|
+
title=title,
|
|
116
|
+
status="downloaded",
|
|
117
|
+
pdf_path=pdf_path,
|
|
118
|
+
source=source,
|
|
119
|
+
file_size_bytes=file_size,
|
|
120
|
+
)
|
|
121
|
+
|
|
122
|
+
def mark_failed(self, pmid: str, doi: str = "", title: str = ""):
|
|
123
|
+
self.update(pmid, doi=doi, title=title, status="all_sources_failed")
|
|
124
|
+
|
|
125
|
+
def cleanup_expired(self, max_age_days: int = 90) -> int:
|
|
126
|
+
"""清理超过指定天数的失败记录(下载成功的不清理)。返回删除条数。"""
|
|
127
|
+
cutoff = time.time() - max_age_days * 86400
|
|
128
|
+
with self._lock:
|
|
129
|
+
cursor = self._conn.cursor()
|
|
130
|
+
cursor.execute(
|
|
131
|
+
"DELETE FROM paper_cache WHERE status = 'all_sources_failed' AND updated_at < ?",
|
|
132
|
+
(cutoff,),
|
|
133
|
+
)
|
|
134
|
+
deleted = cursor.rowcount
|
|
135
|
+
self._conn.commit()
|
|
136
|
+
if deleted:
|
|
137
|
+
logger.info(" 清理了 %d 条过期失败记录", deleted)
|
|
138
|
+
return deleted
|
|
139
|
+
|
|
140
|
+
def stats(self) -> dict:
|
|
141
|
+
with self._lock:
|
|
142
|
+
cursor = self._conn.cursor()
|
|
143
|
+
total = cursor.execute("SELECT COUNT(*) FROM paper_cache").fetchone()[0]
|
|
144
|
+
downloaded = cursor.execute(
|
|
145
|
+
"SELECT COUNT(*) FROM paper_cache WHERE status = 'downloaded'"
|
|
146
|
+
).fetchone()[0]
|
|
147
|
+
failed = cursor.execute(
|
|
148
|
+
"SELECT COUNT(*) FROM paper_cache WHERE status = 'all_sources_failed'"
|
|
149
|
+
).fetchone()[0]
|
|
150
|
+
return {
|
|
151
|
+
"total": total,
|
|
152
|
+
"downloaded": downloaded,
|
|
153
|
+
"failed": failed,
|
|
154
|
+
"session_hits": self.hits,
|
|
155
|
+
"session_misses": self.misses,
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
def close(self):
|
|
159
|
+
with self._lock:
|
|
160
|
+
self._conn.close()
|