check-your-advisor 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. check_your_advisor-0.2.0/LICENSE +21 -0
  2. check_your_advisor-0.2.0/PKG-INFO +207 -0
  3. check_your_advisor-0.2.0/README.md +170 -0
  4. check_your_advisor-0.2.0/pyproject.toml +71 -0
  5. check_your_advisor-0.2.0/scripts/check_your_advisor/__init__.py +17 -0
  6. check_your_advisor-0.2.0/scripts/check_your_advisor/__main__.py +8 -0
  7. check_your_advisor-0.2.0/scripts/check_your_advisor/cache.py +160 -0
  8. check_your_advisor-0.2.0/scripts/check_your_advisor/citations.py +586 -0
  9. check_your_advisor-0.2.0/scripts/check_your_advisor/cli.py +1485 -0
  10. check_your_advisor-0.2.0/scripts/check_your_advisor/config.py +177 -0
  11. check_your_advisor-0.2.0/scripts/check_your_advisor/corpus.py +54 -0
  12. check_your_advisor-0.2.0/scripts/check_your_advisor/download_sources.py +267 -0
  13. check_your_advisor-0.2.0/scripts/check_your_advisor/engine.py +288 -0
  14. check_your_advisor-0.2.0/scripts/check_your_advisor/export.py +125 -0
  15. check_your_advisor-0.2.0/scripts/check_your_advisor/http_client.py +342 -0
  16. check_your_advisor-0.2.0/scripts/check_your_advisor/journals.py +1175 -0
  17. check_your_advisor-0.2.0/scripts/check_your_advisor/pdf_utils.py +229 -0
  18. check_your_advisor-0.2.0/scripts/check_your_advisor/profile/__init__.py +202 -0
  19. check_your_advisor-0.2.0/scripts/check_your_advisor/profile/caveats.py +426 -0
  20. check_your_advisor-0.2.0/scripts/check_your_advisor/profile/charts.py +791 -0
  21. check_your_advisor-0.2.0/scripts/check_your_advisor/profile/cohesion.py +178 -0
  22. check_your_advisor-0.2.0/scripts/check_your_advisor/profile/html_report.py +801 -0
  23. check_your_advisor-0.2.0/scripts/check_your_advisor/profile/impact.py +265 -0
  24. check_your_advisor-0.2.0/scripts/check_your_advisor/profile/metrics.py +643 -0
  25. check_your_advisor-0.2.0/scripts/check_your_advisor/profile/ranking.py +837 -0
  26. check_your_advisor-0.2.0/scripts/check_your_advisor/profile/report.py +2277 -0
  27. check_your_advisor-0.2.0/scripts/check_your_advisor/profile/roles.py +576 -0
  28. check_your_advisor-0.2.0/scripts/check_your_advisor/profile/scoring.py +759 -0
  29. check_your_advisor-0.2.0/scripts/check_your_advisor/profile/svg.py +285 -0
  30. check_your_advisor-0.2.0/scripts/check_your_advisor/pubmed_api.py +684 -0
  31. check_your_advisor-0.2.0/scripts/check_your_advisor/reports.py +101 -0
  32. check_your_advisor-0.2.0/scripts/check_your_advisor/theses.py +1147 -0
  33. check_your_advisor-0.2.0/scripts/check_your_advisor.egg-info/PKG-INFO +207 -0
  34. check_your_advisor-0.2.0/scripts/check_your_advisor.egg-info/SOURCES.txt +56 -0
  35. check_your_advisor-0.2.0/scripts/check_your_advisor.egg-info/dependency_links.txt +1 -0
  36. check_your_advisor-0.2.0/scripts/check_your_advisor.egg-info/entry_points.txt +2 -0
  37. check_your_advisor-0.2.0/scripts/check_your_advisor.egg-info/requires.txt +10 -0
  38. check_your_advisor-0.2.0/scripts/check_your_advisor.egg-info/top_level.txt +1 -0
  39. check_your_advisor-0.2.0/setup.cfg +4 -0
  40. check_your_advisor-0.2.0/tests/test_charts.py +660 -0
  41. check_your_advisor-0.2.0/tests/test_citations.py +621 -0
  42. check_your_advisor-0.2.0/tests/test_cli_profile.py +636 -0
  43. check_your_advisor-0.2.0/tests/test_cohesion.py +243 -0
  44. check_your_advisor-0.2.0/tests/test_html_report.py +588 -0
  45. check_your_advisor-0.2.0/tests/test_http_client.py +357 -0
  46. check_your_advisor-0.2.0/tests/test_identity_cli.py +156 -0
  47. check_your_advisor-0.2.0/tests/test_identity_filter.py +158 -0
  48. check_your_advisor-0.2.0/tests/test_impact.py +376 -0
  49. check_your_advisor-0.2.0/tests/test_journals.py +795 -0
  50. check_your_advisor-0.2.0/tests/test_pdf_validation.py +290 -0
  51. check_your_advisor-0.2.0/tests/test_profile.py +1118 -0
  52. check_your_advisor-0.2.0/tests/test_provenance.py +191 -0
  53. check_your_advisor-0.2.0/tests/test_pubmed_parse.py +284 -0
  54. check_your_advisor-0.2.0/tests/test_ranking.py +733 -0
  55. check_your_advisor-0.2.0/tests/test_scoring.py +469 -0
  56. check_your_advisor-0.2.0/tests/test_search_query.py +111 -0
  57. check_your_advisor-0.2.0/tests/test_stdio_encoding.py +106 -0
  58. check_your_advisor-0.2.0/tests/test_theses.py +613 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 AschoofAlpha
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,207 @@
1
+ Metadata-Version: 2.4
2
+ Name: check-your-advisor
3
+ Version: 0.2.0
4
+ Summary: Read what PubMed records about a researcher and report it back as facts with denominators — first-author slots, time to a first slot, turnover, byline position. 查导师:把发表记录读成带分母的事实,不替你下结论。
5
+ Author: AschoofAlpha
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/AschoofAlpha/check-your-advisor
8
+ Project-URL: Repository, https://github.com/AschoofAlpha/check-your-advisor
9
+ Project-URL: Issues, https://github.com/AschoofAlpha/check-your-advisor/issues
10
+ Project-URL: Chinese README, https://github.com/AschoofAlpha/check-your-advisor/blob/main/README.zh-CN.md
11
+ Keywords: pubmed,advisor,graduate-school,bibliometrics,research-integrity,scientometrics
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Environment :: Console
14
+ Classifier: Intended Audience :: Science/Research
15
+ Classifier: Intended Audience :: Education
16
+ Classifier: Natural Language :: Chinese (Simplified)
17
+ Classifier: Natural Language :: English
18
+ Classifier: Operating System :: OS Independent
19
+ Classifier: Programming Language :: Python :: 3
20
+ Classifier: Programming Language :: Python :: 3.10
21
+ Classifier: Programming Language :: Python :: 3.11
22
+ Classifier: Programming Language :: Python :: 3.12
23
+ Classifier: Programming Language :: Python :: 3.13
24
+ Classifier: Topic :: Scientific/Engineering :: Information Analysis
25
+ Classifier: Topic :: Text Processing :: Indexing
26
+ Requires-Python: >=3.10
27
+ Description-Content-Type: text/markdown
28
+ License-File: LICENSE
29
+ Provides-Extra: pdf
30
+ Requires-Dist: PyMuPDF>=1.23; extra == "pdf"
31
+ Provides-Extra: xlsx
32
+ Requires-Dist: openpyxl>=3.1; extra == "xlsx"
33
+ Provides-Extra: all
34
+ Requires-Dist: PyMuPDF>=1.23; extra == "all"
35
+ Requires-Dist: openpyxl>=3.1; extra == "all"
36
+ Dynamic: license-file
37
+
38
+ # Check Your Advisor
39
+
40
+ *[中文说明](README.zh-CN.md)*
41
+
42
+ You are choosing a PhD or master's advisor. You have a name, a lab page written
43
+ by the lab, and no way to check any of it. This reads what PubMed records about
44
+ that person and reports it back as facts with denominators.
45
+
46
+ **It does not tell you whether the advisor is good.** That judgement stays
47
+ yours. What it does is replace hearsay with counts that carry the population
48
+ they were counted over.
49
+
50
+ ```bash
51
+ python scripts/run.py harvest --author "Wang Wei" --orcid 0000-0002-1825-0097 \
52
+ --years-back 10 --output-dir ./record --no-download
53
+ python scripts/run.py cite --output-dir ./record
54
+ python scripts/run.py profile --pi-name "Wang Wei" --output-dir ./record
55
+ ```
56
+
57
+ Three commands, one HTML report. Nothing to install: the standard library only.
58
+
59
+ ## What it answers
60
+
61
+ - who has appeared in this group over the window
62
+ - who the first-author slots went to, and how concentrated they are
63
+ - how long a newcomer waits before getting one
64
+ - how long people stay before they stop appearing
65
+ - where the PI sits in their own bylines — last, corresponding, or doing the work
66
+ - how much comes out per year, and whether it goes to the same few journals
67
+ - citation counts and an h-index, with the coverage they were computed over
68
+ - one composite score out of 100, with every component's raw input printed
69
+
70
+ And, once you supply a table it cannot fetch: what those journals are rated in
71
+ the edition you looked them up in, and **how many people finished a degree here
72
+ without a single indexed paper** — the one group every other number is blind to.
73
+
74
+ ## What it refuses, and why the distinction matters
75
+
76
+ Four different things stand behind "it does not say whether the advisor is
77
+ good", and the report keeps them apart on the page:
78
+
79
+ | | Examples | Why |
80
+ |---|---|---|
81
+ | **Computed** | citation counts, h-index, composite score, rank, star band | measurements, printed with denominators |
82
+ | **Refused** | percentile, quantile, letter grade, fitted trend, any ordering of *people* | a decision — see below |
83
+ | **Waiting on a file** | impact factor, JCR quartile, CAS partition, graduate roster | licensed or defended databases; the tool ships schemas, not a crawler |
84
+ | **Never visible** | lab culture, whether the PI is decent, people who left before finishing | in no database at all |
85
+
86
+ A percentile is not withheld here, it is *uncomputable*: nothing in the
87
+ calculation ever holds more than the few corpora on one page, so there is no
88
+ reference population to take a position in. Letter grades are refused while
89
+ star bands are produced — that split between two coarsenings of one number is a
90
+ decision, and the report says so rather than leaving it looking like an
91
+ oversight.
92
+
93
+ ## The corpus decides everything
94
+
95
+ Every number is computed over the papers `harvest` kept. If papers by a
96
+ different person with the same name get in, the roster, the time-to-first-author
97
+ and the turnover figures are all wrong — and wrong in a way that looks perfectly
98
+ normal on the page.
99
+
100
+ **Give `harvest` at least one unique identifier.** In descending order of
101
+ strength:
102
+
103
+ | Flag | Strength |
104
+ |---|---|
105
+ | `--orcid 0000-0002-...` | Strongest. One is worth all the rest |
106
+ | `--email-domain your-university.edu.cn` | The corresponding author's address. Repeatable |
107
+ | `--affiliation-keyword "..."` | Weakest — it fails on same-name colleagues inside one university system |
108
+
109
+ The report **refuses to render** if the harvest recorded no evidence at all
110
+ (gate G3). But the dangerous case passes that gate: weak evidence produces a
111
+ complete, normal-looking report about several people. A real run for one Chinese
112
+ surgeon, keyed on a province name rather than the full institution, returned 28
113
+ records spanning gastrointestinal surgery, analytical chemistry, structural
114
+ biology, soil microbiology and machine learning — and scored 78.2 out of 100.
115
+
116
+ **Section 19 is the check for that.** It removes the PI, who is on every record
117
+ by construction, and asks which records are still tied together by a shared
118
+ co-author. One person's output is held together by the people they work with;
119
+ two people sharing a name have no reason to share anyone else. It applies no
120
+ threshold — the measurements do not support one, and the README of a tool like
121
+ this should not pretend otherwise — it prints the clusters and their journals
122
+ and hands the reading to you. It is usually not subtle.
123
+
124
+ ## Two tables you fill in by hand
125
+
126
+ Impact factor, JCR quartile and CAS partition live in licensed products with no
127
+ free redistributable source. Degree-thesis libraries defend against scraping.
128
+ So this ships the schema, the worklist and the join, and no crawler:
129
+
130
+ ```bash
131
+ python scripts/run.py journal-worklist --output-dir ./record
132
+ ```
133
+
134
+ That writes a CSV holding **only the journals this corpus actually uses** —
135
+ typically a couple of dozen, not the twenty thousand in the world — pre-filled
136
+ with ISSN and paper count, indicator columns blank. Fill it from whichever
137
+ source you have access to, then:
138
+
139
+ ```bash
140
+ python scripts/run.py profile --output-dir ./record \
141
+ --journal-table ./record/journal_worklist_*.csv \
142
+ --thesis-roster ./record/theses.csv
143
+ ```
144
+
145
+ Two columns are **required** in the journal table: `版本来源` (which edition —
146
+ official, rising-star, folk, or JCR) and `数据获取日期`. Without them a
147
+ partition number is unfalsifiable two years from now. Where a journal has rows
148
+ from two editions, both are shown and neither wins; disagreements are listed.
149
+
150
+ The graduate roster is the more important of the two. Export the advisor's
151
+ supervised theses from a degree library and it produces the number PubMed
152
+ structurally cannot: **how many graduates have no indexed paper at all.** Add a
153
+ romanised-name column or the Chinese roster will not match the English bylines.
154
+ Note that people who enrolled and left before finishing are in no library
155
+ either — the roster narrows the missing group, it does not close it.
156
+
157
+ ## Install
158
+
159
+ As a Claude Code skill:
160
+
161
+ ```bash
162
+ git clone https://github.com/AschoofAlpha/check-your-advisor.git \
163
+ ~/.claude/skills/check-your-advisor
164
+ ```
165
+
166
+ Or run it directly as a CLI from anywhere — `scripts/run.py` is the single
167
+ entry point and needs no installation.
168
+
169
+ ## Optional extras
170
+
171
+ Both degrade with a tested fallback; neither is required.
172
+
173
+ - **PyMuPDF** (AGPL-3.0, so deliberately not a hard dependency of an MIT
174
+ project) lets PDF identity validation quarantine a wrong file. Without it
175
+ every download is still checked for the `%PDF-` magic.
176
+ - **openpyxl** enables the `.xlsx` export. Without it the same data is written
177
+ as a timestamped CSV.
178
+
179
+ ## Tests
180
+
181
+ ```bash
182
+ python tests/run_all.py
183
+ python tests/run_all.py --block-third-party
184
+ ```
185
+
186
+ 1925 assertions across 19 files. The second run installs an import hook that
187
+ blocks `requests`, `urllib3`, `pandas`, `numpy`, `matplotlib`, `fitz` and
188
+ `openpyxl` inside each test process. It is the only thing that keeps "no install
189
+ needed" true rather than merely claimed: exactly three assertions behave
190
+ differently without them, and all three are the cases that need a real PDF file.
191
+
192
+ ## Known limitations
193
+
194
+ - The report is in English; the command-line logs are in Chinese. Not a decision
195
+ anyone would defend — it is where the tool grew up, and unifying it is open.
196
+ - `cite` has no cache of its own. `--max-age-days N` reuses counts from the
197
+ previous run's file, which is the workaround.
198
+ - Journal name matching falls back to a token heuristic for corpora harvested
199
+ before ISSN capture existed. Re-harvest to get the exact join.
200
+
201
+ ## License
202
+
203
+ MIT. See [LICENSE](LICENSE).
204
+
205
+ Citation counts come from OpenAlex, Semantic Scholar and Europe PMC, none of
206
+ which requires a key. Bibliographic records come from NCBI E-utilities. This
207
+ project ships no licensed data.
@@ -0,0 +1,170 @@
1
+ # Check Your Advisor
2
+
3
+ *[中文说明](README.zh-CN.md)*
4
+
5
+ You are choosing a PhD or master's advisor. You have a name, a lab page written
6
+ by the lab, and no way to check any of it. This reads what PubMed records about
7
+ that person and reports it back as facts with denominators.
8
+
9
+ **It does not tell you whether the advisor is good.** That judgement stays
10
+ yours. What it does is replace hearsay with counts that carry the population
11
+ they were counted over.
12
+
13
+ ```bash
14
+ python scripts/run.py harvest --author "Wang Wei" --orcid 0000-0002-1825-0097 \
15
+ --years-back 10 --output-dir ./record --no-download
16
+ python scripts/run.py cite --output-dir ./record
17
+ python scripts/run.py profile --pi-name "Wang Wei" --output-dir ./record
18
+ ```
19
+
20
+ Three commands, one HTML report. Nothing to install: the standard library only.
21
+
22
+ ## What it answers
23
+
24
+ - who has appeared in this group over the window
25
+ - who the first-author slots went to, and how concentrated they are
26
+ - how long a newcomer waits before getting one
27
+ - how long people stay before they stop appearing
28
+ - where the PI sits in their own bylines — last, corresponding, or doing the work
29
+ - how much comes out per year, and whether it goes to the same few journals
30
+ - citation counts and an h-index, with the coverage they were computed over
31
+ - one composite score out of 100, with every component's raw input printed
32
+
33
+ And, once you supply a table it cannot fetch: what those journals are rated in
34
+ the edition you looked them up in, and **how many people finished a degree here
35
+ without a single indexed paper** — the one group every other number is blind to.
36
+
37
+ ## What it refuses, and why the distinction matters
38
+
39
+ Four different things stand behind "it does not say whether the advisor is
40
+ good", and the report keeps them apart on the page:
41
+
42
+ | | Examples | Why |
43
+ |---|---|---|
44
+ | **Computed** | citation counts, h-index, composite score, rank, star band | measurements, printed with denominators |
45
+ | **Refused** | percentile, quantile, letter grade, fitted trend, any ordering of *people* | a decision — see below |
46
+ | **Waiting on a file** | impact factor, JCR quartile, CAS partition, graduate roster | licensed or defended databases; the tool ships schemas, not a crawler |
47
+ | **Never visible** | lab culture, whether the PI is decent, people who left before finishing | in no database at all |
48
+
49
+ A percentile is not withheld here, it is *uncomputable*: nothing in the
50
+ calculation ever holds more than the few corpora on one page, so there is no
51
+ reference population to take a position in. Letter grades are refused while
52
+ star bands are produced — that split between two coarsenings of one number is a
53
+ decision, and the report says so rather than leaving it looking like an
54
+ oversight.
55
+
56
+ ## The corpus decides everything
57
+
58
+ Every number is computed over the papers `harvest` kept. If papers by a
59
+ different person with the same name get in, the roster, the time-to-first-author
60
+ and the turnover figures are all wrong — and wrong in a way that looks perfectly
61
+ normal on the page.
62
+
63
+ **Give `harvest` at least one unique identifier.** In descending order of
64
+ strength:
65
+
66
+ | Flag | Strength |
67
+ |---|---|
68
+ | `--orcid 0000-0002-...` | Strongest. One is worth all the rest |
69
+ | `--email-domain your-university.edu.cn` | The corresponding author's address. Repeatable |
70
+ | `--affiliation-keyword "..."` | Weakest — it fails on same-name colleagues inside one university system |
71
+
72
+ The report **refuses to render** if the harvest recorded no evidence at all
73
+ (gate G3). But the dangerous case passes that gate: weak evidence produces a
74
+ complete, normal-looking report about several people. A real run for one Chinese
75
+ surgeon, keyed on a province name rather than the full institution, returned 28
76
+ records spanning gastrointestinal surgery, analytical chemistry, structural
77
+ biology, soil microbiology and machine learning — and scored 78.2 out of 100.
78
+
79
+ **Section 19 is the check for that.** It removes the PI, who is on every record
80
+ by construction, and asks which records are still tied together by a shared
81
+ co-author. One person's output is held together by the people they work with;
82
+ two people sharing a name have no reason to share anyone else. It applies no
83
+ threshold — the measurements do not support one, and the README of a tool like
84
+ this should not pretend otherwise — it prints the clusters and their journals
85
+ and hands the reading to you. It is usually not subtle.
86
+
87
+ ## Two tables you fill in by hand
88
+
89
+ Impact factor, JCR quartile and CAS partition live in licensed products with no
90
+ free redistributable source. Degree-thesis libraries defend against scraping.
91
+ So this ships the schema, the worklist and the join, and no crawler:
92
+
93
+ ```bash
94
+ python scripts/run.py journal-worklist --output-dir ./record
95
+ ```
96
+
97
+ That writes a CSV holding **only the journals this corpus actually uses** —
98
+ typically a couple of dozen, not the twenty thousand in the world — pre-filled
99
+ with ISSN and paper count, indicator columns blank. Fill it from whichever
100
+ source you have access to, then:
101
+
102
+ ```bash
103
+ python scripts/run.py profile --output-dir ./record \
104
+ --journal-table ./record/journal_worklist_*.csv \
105
+ --thesis-roster ./record/theses.csv
106
+ ```
107
+
108
+ Two columns are **required** in the journal table: `版本来源` (which edition —
109
+ official, rising-star, folk, or JCR) and `数据获取日期`. Without them a
110
+ partition number is unfalsifiable two years from now. Where a journal has rows
111
+ from two editions, both are shown and neither wins; disagreements are listed.
112
+
113
+ The graduate roster is the more important of the two. Export the advisor's
114
+ supervised theses from a degree library and it produces the number PubMed
115
+ structurally cannot: **how many graduates have no indexed paper at all.** Add a
116
+ romanised-name column or the Chinese roster will not match the English bylines.
117
+ Note that people who enrolled and left before finishing are in no library
118
+ either — the roster narrows the missing group, it does not close it.
119
+
120
+ ## Install
121
+
122
+ As a Claude Code skill:
123
+
124
+ ```bash
125
+ git clone https://github.com/AschoofAlpha/check-your-advisor.git \
126
+ ~/.claude/skills/check-your-advisor
127
+ ```
128
+
129
+ Or run it directly as a CLI from anywhere — `scripts/run.py` is the single
130
+ entry point and needs no installation.
131
+
132
+ ## Optional extras
133
+
134
+ Both degrade with a tested fallback; neither is required.
135
+
136
+ - **PyMuPDF** (AGPL-3.0, so deliberately not a hard dependency of an MIT
137
+ project) lets PDF identity validation quarantine a wrong file. Without it
138
+ every download is still checked for the `%PDF-` magic.
139
+ - **openpyxl** enables the `.xlsx` export. Without it the same data is written
140
+ as a timestamped CSV.
141
+
142
+ ## Tests
143
+
144
+ ```bash
145
+ python tests/run_all.py
146
+ python tests/run_all.py --block-third-party
147
+ ```
148
+
149
+ 1925 assertions across 19 files. The second run installs an import hook that
150
+ blocks `requests`, `urllib3`, `pandas`, `numpy`, `matplotlib`, `fitz` and
151
+ `openpyxl` inside each test process. It is the only thing that keeps "no install
152
+ needed" true rather than merely claimed: exactly three assertions behave
153
+ differently without them, and all three are the cases that need a real PDF file.
154
+
155
+ ## Known limitations
156
+
157
+ - The report is in English; the command-line logs are in Chinese. Not a decision
158
+ anyone would defend — it is where the tool grew up, and unifying it is open.
159
+ - `cite` has no cache of its own. `--max-age-days N` reuses counts from the
160
+ previous run's file, which is the workaround.
161
+ - Journal name matching falls back to a token heuristic for corpora harvested
162
+ before ISSN capture existed. Re-harvest to get the exact join.
163
+
164
+ ## License
165
+
166
+ MIT. See [LICENSE](LICENSE).
167
+
168
+ Citation counts come from OpenAlex, Semantic Scholar and Europe PMC, none of
169
+ which requires a key. Bibliographic records come from NCBI E-utilities. This
170
+ project ships no licensed data.
@@ -0,0 +1,71 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "check-your-advisor"
7
+ version = "0.2.0"
8
+ description = "Read what PubMed records about a researcher and report it back as facts with denominators — first-author slots, time to a first slot, turnover, byline position. 查导师:把发表记录读成带分母的事实,不替你下结论。"
9
+ readme = { file = "README.md", content-type = "text/markdown" }
10
+ requires-python = ">=3.10"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ authors = [{ name = "AschoofAlpha" }]
14
+ keywords = [
15
+ "pubmed",
16
+ "advisor",
17
+ "graduate-school",
18
+ "bibliometrics",
19
+ "research-integrity",
20
+ "scientometrics",
21
+ ]
22
+ classifiers = [
23
+ "Development Status :: 4 - Beta",
24
+ "Environment :: Console",
25
+ "Intended Audience :: Science/Research",
26
+ "Intended Audience :: Education",
27
+ "Natural Language :: Chinese (Simplified)",
28
+ "Natural Language :: English",
29
+ "Operating System :: OS Independent",
30
+ "Programming Language :: Python :: 3",
31
+ "Programming Language :: Python :: 3.10",
32
+ "Programming Language :: Python :: 3.11",
33
+ "Programming Language :: Python :: 3.12",
34
+ "Programming Language :: Python :: 3.13",
35
+ "Topic :: Scientific/Engineering :: Information Analysis",
36
+ "Topic :: Text Processing :: Indexing",
37
+ ]
38
+
39
+ # No `dependencies` key, and that is the point rather than an omission. Every
40
+ # import in this package is from the standard library, and the test suite proves
41
+ # it: `python tests/run_all.py --block-third-party` installs an import hook that
42
+ # makes requests, urllib3, pandas, numpy, matplotlib, fitz and openpyxl
43
+ # unimportable, and 1922 of 1925 assertions still pass. Adding a dependency here
44
+ # would make that suite a claim nobody checks.
45
+
46
+ [project.optional-dependencies]
47
+ # Both are genuinely optional and both have a tested fallback path. PyMuPDF is
48
+ # AGPL-3.0 and this package is MIT, which is why it is not a hard dependency:
49
+ # without it a downloaded PDF is still checked for the `%PDF-` magic bytes, it
50
+ # just cannot be quarantined by content. openpyxl only decides whether the
51
+ # corpus export is .xlsx or a timestamped .csv.
52
+ pdf = ["PyMuPDF>=1.23"]
53
+ xlsx = ["openpyxl>=3.1"]
54
+ all = ["PyMuPDF>=1.23", "openpyxl>=3.1"]
55
+
56
+ [project.urls]
57
+ Homepage = "https://github.com/AschoofAlpha/check-your-advisor"
58
+ Repository = "https://github.com/AschoofAlpha/check-your-advisor"
59
+ Issues = "https://github.com/AschoofAlpha/check-your-advisor/issues"
60
+ "Chinese README" = "https://github.com/AschoofAlpha/check-your-advisor/blob/main/README.zh-CN.md"
61
+
62
+ [project.scripts]
63
+ check-your-advisor = "check_your_advisor.cli:main"
64
+
65
+ [tool.setuptools]
66
+ # The package lives under `scripts/` because this repository is also a Claude
67
+ # Code skill, where `SKILL.md` sits at the root and the code sits beside it.
68
+ # Moving the package to the root would break the skill layout; telling setuptools
69
+ # where to look costs one line and breaks nothing.
70
+ package-dir = { "" = "scripts" }
71
+ packages = ["check_your_advisor", "check_your_advisor.profile"]
@@ -0,0 +1,17 @@
1
+ """
2
+ check-your-advisor — author-disambiguated PubMed harvesting and reference verification.
3
+
4
+ Two entry points over one shared HTTP and normalisation layer:
5
+
6
+ fetch/download Search PubMed, keep only the target researcher's papers via
7
+ ORCID + affiliation + email verification, race 8 open-access
8
+ sources for the PDF, then verify the downloaded file really
9
+ is the requested paper.
10
+
11
+ verify Check a bibliography against CrossRef and PubMed, including
12
+ bidirectional DOI <-> PMID resolution.
13
+ """
14
+
15
+ __version__ = "0.3.1"
16
+
17
+ __all__ = ["__version__"]
@@ -0,0 +1,8 @@
1
+ """Allow `python -m check_your_advisor ...`."""
2
+
3
+ import sys
4
+
5
+ from .cli import main
6
+
7
+ if __name__ == "__main__":
8
+ sys.exit(main() or 0)
@@ -0,0 +1,160 @@
1
+ """
2
+ 缓存模块(SQLite)
3
+ ==================
4
+ 替代原代码中仅通过 os.path.exists(pdf_path) 检查的简陋缓存。
5
+
6
+ 改进:
7
+ - 记录每篇论文的元数据(DOI、PMID、下载状态、来源、时间)
8
+ - 支持过期清理
9
+ - 支持查询命中/未命中统计
10
+ """
11
+
12
+ import logging
13
+ import sqlite3
14
+ import threading
15
+ import time
16
+ from pathlib import Path
17
+
18
+ logger = logging.getLogger("check_your_advisor.cache")
19
+
20
+ CREATE_TABLE_SQL = """
21
+ CREATE TABLE IF NOT EXISTS paper_cache (
22
+ doi TEXT,
23
+ pmid TEXT,
24
+ title TEXT,
25
+ pdf_path TEXT,
26
+ source TEXT,
27
+ status TEXT DEFAULT 'pending',
28
+ file_size_bytes INTEGER DEFAULT 0,
29
+ created_at REAL,
30
+ updated_at REAL,
31
+ PRIMARY KEY (pmid)
32
+ );
33
+ CREATE INDEX IF NOT EXISTS idx_doi ON paper_cache(doi);
34
+ CREATE INDEX IF NOT EXISTS idx_status ON paper_cache(status);
35
+ """
36
+
37
+
38
+ class PaperCache:
39
+ def __init__(self, db_path: str = "paper_cache.db"):
40
+ self.db_path = db_path
41
+ db_parent = Path(db_path).parent
42
+ if str(db_parent) not in {"", "."}:
43
+ db_parent.mkdir(parents=True, exist_ok=True)
44
+ # check_same_thread=False 允许跨线程使用;加锁保证串行访问
45
+ self._conn = sqlite3.connect(db_path, check_same_thread=False)
46
+ self._conn.row_factory = sqlite3.Row
47
+ self._lock = threading.Lock()
48
+ self._init_db()
49
+ self.hits = 0
50
+ self.misses = 0
51
+
52
+ def _init_db(self):
53
+ with self._lock:
54
+ cursor = self._conn.cursor()
55
+ cursor.executescript(CREATE_TABLE_SQL)
56
+ self._conn.commit()
57
+
58
+ def lookup(self, pmid: str) -> dict | None:
59
+ """查询缓存。返回 dict 或 None。"""
60
+ with self._lock:
61
+ cursor = self._conn.cursor()
62
+ cursor.execute("SELECT * FROM paper_cache WHERE pmid = ?", (pmid,))
63
+ row = cursor.fetchone()
64
+
65
+ if row and row["status"] == "downloaded" and row["pdf_path"]:
66
+ if Path(row["pdf_path"]).exists():
67
+ self.hits += 1
68
+ logger.debug(" 缓存命中: PMID %s", pmid)
69
+ return dict(row)
70
+ else:
71
+ logger.warning(" 缓存记录存在但文件丢失: %s", row["pdf_path"])
72
+ self.update(pmid, status="pending", pdf_path="")
73
+ self.misses += 1
74
+ return None
75
+
76
+ def update(self, pmid: str, **kwargs):
77
+ """更新或插入缓存记录"""
78
+ now = time.time()
79
+ with self._lock:
80
+ existing = self._conn.execute(
81
+ "SELECT pmid FROM paper_cache WHERE pmid = ?", (pmid,)
82
+ ).fetchone()
83
+
84
+ if existing:
85
+ sets = ", ".join(f"{k} = ?" for k in kwargs)
86
+ vals = list(kwargs.values()) + [now, pmid]
87
+ self._conn.execute(
88
+ f"UPDATE paper_cache SET {sets}, updated_at = ? WHERE pmid = ?",
89
+ vals,
90
+ )
91
+ else:
92
+ kwargs["pmid"] = pmid
93
+ kwargs["created_at"] = now
94
+ kwargs["updated_at"] = now
95
+ cols = ", ".join(kwargs.keys())
96
+ placeholders = ", ".join("?" for _ in kwargs)
97
+ self._conn.execute(
98
+ f"INSERT INTO paper_cache ({cols}) VALUES ({placeholders})",
99
+ list(kwargs.values()),
100
+ )
101
+ self._conn.commit()
102
+
103
+ def mark_downloaded(
104
+ self,
105
+ pmid: str,
106
+ pdf_path: str,
107
+ source: str,
108
+ file_size: int = 0,
109
+ doi: str = "",
110
+ title: str = "",
111
+ ):
112
+ self.update(
113
+ pmid,
114
+ doi=doi,
115
+ title=title,
116
+ status="downloaded",
117
+ pdf_path=pdf_path,
118
+ source=source,
119
+ file_size_bytes=file_size,
120
+ )
121
+
122
+ def mark_failed(self, pmid: str, doi: str = "", title: str = ""):
123
+ self.update(pmid, doi=doi, title=title, status="all_sources_failed")
124
+
125
+ def cleanup_expired(self, max_age_days: int = 90) -> int:
126
+ """清理超过指定天数的失败记录(下载成功的不清理)。返回删除条数。"""
127
+ cutoff = time.time() - max_age_days * 86400
128
+ with self._lock:
129
+ cursor = self._conn.cursor()
130
+ cursor.execute(
131
+ "DELETE FROM paper_cache WHERE status = 'all_sources_failed' AND updated_at < ?",
132
+ (cutoff,),
133
+ )
134
+ deleted = cursor.rowcount
135
+ self._conn.commit()
136
+ if deleted:
137
+ logger.info(" 清理了 %d 条过期失败记录", deleted)
138
+ return deleted
139
+
140
+ def stats(self) -> dict:
141
+ with self._lock:
142
+ cursor = self._conn.cursor()
143
+ total = cursor.execute("SELECT COUNT(*) FROM paper_cache").fetchone()[0]
144
+ downloaded = cursor.execute(
145
+ "SELECT COUNT(*) FROM paper_cache WHERE status = 'downloaded'"
146
+ ).fetchone()[0]
147
+ failed = cursor.execute(
148
+ "SELECT COUNT(*) FROM paper_cache WHERE status = 'all_sources_failed'"
149
+ ).fetchone()[0]
150
+ return {
151
+ "total": total,
152
+ "downloaded": downloaded,
153
+ "failed": failed,
154
+ "session_hits": self.hits,
155
+ "session_misses": self.misses,
156
+ }
157
+
158
+ def close(self):
159
+ with self._lock:
160
+ self._conn.close()