@cyanheads/pubmed-mcp-server 2.10.16 → 2.10.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +1 -1
- package/CLAUDE.md +1 -1
- package/README.md +6 -4
- package/changelog/2.10.x/2.10.17.md +19 -0
- package/dist/mcp-server/tools/definitions/_text.d.ts +20 -3
- package/dist/mcp-server/tools/definitions/_text.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/_text.js +35 -3
- package/dist/mcp-server/tools/definitions/_text.js.map +1 -1
- package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.d.ts +23 -31
- package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.js +349 -130
- package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.js.map +1 -1
- package/dist/services/error-contracts.d.ts +5 -3
- package/dist/services/error-contracts.d.ts.map +1 -1
- package/dist/services/error-contracts.js +5 -3
- package/dist/services/error-contracts.js.map +1 -1
- package/dist/services/ncbi/parsing/pmc-article-parser.d.ts +8 -0
- package/dist/services/ncbi/parsing/pmc-article-parser.d.ts.map +1 -1
- package/dist/services/ncbi/parsing/pmc-article-parser.js +159 -16
- package/dist/services/ncbi/parsing/pmc-article-parser.js.map +1 -1
- package/dist/services/unpaywall/types.d.ts +16 -2
- package/dist/services/unpaywall/types.d.ts.map +1 -1
- package/dist/services/unpaywall/unpaywall-service.d.ts +4 -3
- package/dist/services/unpaywall/unpaywall-service.d.ts.map +1 -1
- package/dist/services/unpaywall/unpaywall-service.js +15 -4
- package/dist/services/unpaywall/unpaywall-service.js.map +1 -1
- package/package.json +1 -1
- package/server.json +3 -3
|
@@ -3,39 +3,58 @@
|
|
|
3
3
|
* three-stage chain: NCBI PMC EFetch → Europe PMC `fullTextXML` → Unpaywall.
|
|
4
4
|
* Accepts three mutually-exclusive input shapes:
|
|
5
5
|
*
|
|
6
|
-
* - `pmcids` — fetch directly by PMC ID
|
|
7
|
-
*
|
|
6
|
+
* - `pmcids` — fetch directly by PMC ID, once per record however it is
|
|
7
|
+
* spelled (`PMC123`, `pmc123`, `123`, zero-padded `PMC0123`), and reported
|
|
8
|
+
* in `PMC<digits>` form. Articles not in PMC fall through to EPMC by PMC ID,
|
|
9
|
+
* then to Unpaywall when the DOI is available.
|
|
8
10
|
* - `pmids` — resolve PMID → PMCID via PMC ID Converter, then run the chain.
|
|
9
11
|
* A zero-padded PMID runs as the PMID it spells; `unavailable[]` and
|
|
10
12
|
* `deferred.ids` report it as the caller wrote it.
|
|
11
13
|
* - `dois` — resolve DOI → PMCID via the PMC ID Converter (mirroring `pmids`),
|
|
12
14
|
* then run the chain. DOIs with no PMC counterpart fall through to EPMC
|
|
13
15
|
* search-by-DOI → fullTextXML, then Unpaywall (EPMC-only OA, preprints).
|
|
16
|
+
* DOIs are case-insensitive: every casing of one DOI runs the chain once,
|
|
17
|
+
* and `unavailable[]` reports each casing as the caller wrote it.
|
|
14
18
|
*
|
|
15
19
|
* Output uses a discriminated union on `source` (`pmc` | `unpaywall`) with an
|
|
16
20
|
* extra `viaSource` discriminator that records which layer produced the
|
|
17
21
|
* content. EPMC's JATS reuses the `pmc` schema shape because it's the same
|
|
18
|
-
* DTD; `viaSource: 'europepmc'` distinguishes it from PMC EFetch output.
|
|
22
|
+
* DTD; `viaSource: 'europepmc'` distinguishes it from PMC EFetch output. An
|
|
23
|
+
* Unpaywall article takes its title from Unpaywall's record, else the Europe
|
|
24
|
+
* PMC record the chain searched, else (HTML only) the page itself.
|
|
25
|
+
*
|
|
26
|
+
* Europe PMC and Unpaywall failures are folded into each id's `triedTiers`
|
|
27
|
+
* rather than thrown, so the declared `errors[]` covers the NCBI ID routing
|
|
28
|
+
* alone.
|
|
19
29
|
*
|
|
20
30
|
* @module src/mcp-server/tools/definitions/fetch-fulltext.tool
|
|
21
31
|
*/
|
|
22
32
|
import { tool, z } from '@cyanheads/mcp-ts-core';
|
|
23
33
|
import { htmlExtractor, pdfParser } from '@cyanheads/mcp-ts-core/utils';
|
|
24
34
|
import { getServerConfig } from '../../../config/server-config.js';
|
|
25
|
-
import {
|
|
35
|
+
import { NCBI_SERVICE_ERRORS } from '../../../services/error-contracts.js';
|
|
26
36
|
import { getEuropePmcService, } from '../../../services/europe-pmc/europe-pmc-service.js';
|
|
27
37
|
import { getNcbiService } from '../../../services/ncbi/ncbi-service.js';
|
|
28
38
|
import { extractDoi, extractPmid } from '../../../services/ncbi/parsing/article-parser.js';
|
|
29
39
|
import { parsePmcArticle } from '../../../services/ncbi/parsing/pmc-article-parser.js';
|
|
30
40
|
import { findAll, findOne } from '../../../services/ncbi/parsing/pmc-xml-helpers.js';
|
|
41
|
+
import { toDisplayText } from '../../../services/ncbi/parsing/text-helpers.js';
|
|
31
42
|
import { ensureArray } from '../../../services/ncbi/parsing/xml-helpers.js';
|
|
32
43
|
import { getUnpaywallService, } from '../../../services/unpaywall/unpaywall-service.js';
|
|
33
44
|
import { fitWholeItems } from './_budget.js';
|
|
34
45
|
import { conceptMeta, EDAM_DATA_RETRIEVAL, SCHEMA_SCHOLARLY_ARTICLE } from './_concepts.js';
|
|
35
46
|
import { doiStringSchema, normalizePmid, pmcidStringSchema, pmidStringSchema } from './_schemas.js';
|
|
36
|
-
import { escapeMarkdownInline, escapeMarkdownTableCell,
|
|
47
|
+
import { escapeMarkdownInline, escapeMarkdownTableCell, sliceAtWordBoundary } from './_text.js';
|
|
48
|
+
/**
|
|
49
|
+
* Canonical digits of a PMC ID {@link pmcidStringSchema} accepted: the `PMC`
|
|
50
|
+
* prefix dropped in any case, and leading zeros stripped the way
|
|
51
|
+
* {@link normalizePmid} strips them. PMC EFetch reads `03531190` as PMC3531190
|
|
52
|
+
* and answers with that record, so every `pmcids` path keys on this form — and
|
|
53
|
+
* reports it back as `PMC<digits>`. An all-zero ID keeps `0`, which EFetch
|
|
54
|
+
* still answers with its empty-list envelope. (#170)
|
|
55
|
+
*/
|
|
37
56
|
function normalizePmcId(id) {
|
|
38
|
-
return id.replace(/^PMC/i, '');
|
|
57
|
+
return normalizePmid(id.replace(/^PMC/i, ''));
|
|
39
58
|
}
|
|
40
59
|
function withPmcPrefix(id) {
|
|
41
60
|
return id.startsWith('PMC') ? id : `PMC${id}`;
|
|
@@ -540,7 +559,18 @@ const UnpaywallArticleSchema = z
|
|
|
540
559
|
pubmedUrl: z.string().optional().describe('PubMed URL — present when `pmid` is set'),
|
|
541
560
|
doi: z.string().describe('DOI used to locate the open-access copy'),
|
|
542
561
|
sourceUrl: z.string().describe('URL the content was fetched from'),
|
|
543
|
-
title: z
|
|
562
|
+
title: z
|
|
563
|
+
.string()
|
|
564
|
+
.optional()
|
|
565
|
+
.describe("Article title, from the first source that carries one: Unpaywall's record for the DOI, then the Europe PMC record when the chain searched Europe PMC for this id, then — for `html-markdown` content only — the title detected on the page. Absent when none of them has a title."),
|
|
566
|
+
journalName: z
|
|
567
|
+
.string()
|
|
568
|
+
.optional()
|
|
569
|
+
.describe("Journal or repository name from Unpaywall's record for the DOI (e.g. `medRxiv` for a medRxiv preprint). Absent when Unpaywall has none."),
|
|
570
|
+
year: z
|
|
571
|
+
.number()
|
|
572
|
+
.optional()
|
|
573
|
+
.describe("Publication year from Unpaywall's record for the DOI. Absent when Unpaywall has none."),
|
|
544
574
|
content: z.string().describe('Full article text — Markdown or plain text per `contentFormat`'),
|
|
545
575
|
wordCount: z
|
|
546
576
|
.number()
|
|
@@ -619,18 +649,54 @@ const UnavailableSchema = z
|
|
|
619
649
|
})
|
|
620
650
|
.describe('One identifier that could not be returned, with the full chain it traversed');
|
|
621
651
|
// ─── Character-budget schemas ────────────────────────────────────────────────
|
|
652
|
+
/**
|
|
653
|
+
* Character accounting for one subsection. Its own type rather than a
|
|
654
|
+
* self-reference for the reason {@link SubsectionSchema} is inlined: a
|
|
655
|
+
* `z.lazy()` schema emits `$defs`/`$ref`. The article schema carries sections
|
|
656
|
+
* two levels deep, so the ledger does too.
|
|
657
|
+
*
|
|
658
|
+
* The chain `truncation` → `articles[]` → `sections[]` → `subsections[]` → a
|
|
659
|
+
* leaf is eight schema hops, the most the `format-parity` sentinel walker
|
|
660
|
+
* reaches. Re-run `bun run lint:mcp` after any change to this shape. (#143)
|
|
661
|
+
*/
|
|
662
|
+
const TruncatedSubsectionSchema = z
|
|
663
|
+
.object({
|
|
664
|
+
title: z.string().optional().describe('Subsection heading, when the subsection carries one'),
|
|
665
|
+
label: z
|
|
666
|
+
.string()
|
|
667
|
+
.optional()
|
|
668
|
+
.describe('Subsection label as printed (e.g. `2.1`), when the subsection carries one'),
|
|
669
|
+
originalCharacters: z
|
|
670
|
+
.number()
|
|
671
|
+
.describe('Body characters this subsection carried before the budget pass'),
|
|
672
|
+
returnedCharacters: z
|
|
673
|
+
.number()
|
|
674
|
+
.describe('Body characters this subsection carries in the response. Zero means it was dropped in `truncate` mode and counted in `omittedSections`, or kept as a heading-only entry in `outline` mode, marked as such in the rendered text.'),
|
|
675
|
+
truncated: z
|
|
676
|
+
.boolean()
|
|
677
|
+
.describe('True when the subsection returned fewer characters than it originally carried'),
|
|
678
|
+
})
|
|
679
|
+
.describe('Character accounting for one subsection of a shortened section');
|
|
622
680
|
const TruncatedSectionSchema = z
|
|
623
681
|
.object({
|
|
624
682
|
title: z.string().optional().describe('Section heading, when the section carries one'),
|
|
683
|
+
label: z
|
|
684
|
+
.string()
|
|
685
|
+
.optional()
|
|
686
|
+
.describe('Section label as printed (e.g. `2`), when the section carries one'),
|
|
625
687
|
originalCharacters: z
|
|
626
688
|
.number()
|
|
627
689
|
.describe('Body characters this section carried before the budget pass'),
|
|
628
690
|
returnedCharacters: z
|
|
629
691
|
.number()
|
|
630
|
-
.describe('Body characters this section carries in the response. Zero means the section was dropped in `truncate` mode, or kept as a heading-only entry in `outline` mode.'),
|
|
692
|
+
.describe('Body characters this section carries in the response. Zero means the section was dropped in `truncate` mode, or kept as a heading-only entry in `outline` mode, marked as such in the rendered text.'),
|
|
631
693
|
truncated: z
|
|
632
694
|
.boolean()
|
|
633
695
|
.describe('True when the section returned fewer characters than it originally carried'),
|
|
696
|
+
subsections: z
|
|
697
|
+
.array(TruncatedSubsectionSchema)
|
|
698
|
+
.optional()
|
|
699
|
+
.describe('Per-subsection accounting for a shortened section, in document order, including subsections dropped for budget — where inside the section the cut landed. Absent when the section was returned whole or carries no subsections.'),
|
|
634
700
|
})
|
|
635
701
|
.describe('Character accounting for one body section of a budgeted article');
|
|
636
702
|
const TruncatedArticleSchema = z
|
|
@@ -648,7 +714,7 @@ const TruncatedArticleSchema = z
|
|
|
648
714
|
sections: z
|
|
649
715
|
.array(TruncatedSectionSchema)
|
|
650
716
|
.optional()
|
|
651
|
-
.describe('Per-section accounting for `source: pmc` articles, in document order, including sections dropped for budget. Absent for `source: unpaywall`, whose body has no section structure.'),
|
|
717
|
+
.describe('Per-section accounting for `source: pmc` articles, in document order, including sections dropped for budget; a shortened section lists its subsections. Absent for `source: unpaywall`, whose body has no section structure.'),
|
|
652
718
|
omittedTables: z
|
|
653
719
|
.number()
|
|
654
720
|
.optional()
|
|
@@ -685,7 +751,7 @@ const TruncationSchema = z
|
|
|
685
751
|
.describe('Body characters the shortened articles carry in this response'),
|
|
686
752
|
omittedSections: z
|
|
687
753
|
.number()
|
|
688
|
-
.describe('Body sections dropped entirely because an article budget was exhausted before reaching them. Always 0 in `outline` mode, which keeps every heading.'),
|
|
754
|
+
.describe('Body sections and subsections dropped entirely because an article budget was exhausted before reaching them. A dropped section counts once, together with its subsections. Always 0 in `outline` mode, which keeps every heading.'),
|
|
689
755
|
omittedTables: z
|
|
690
756
|
.number()
|
|
691
757
|
.optional()
|
|
@@ -786,30 +852,81 @@ function fitWholeNamed(items, allowance, measure, name) {
|
|
|
786
852
|
};
|
|
787
853
|
}
|
|
788
854
|
/**
|
|
789
|
-
*
|
|
790
|
-
*
|
|
791
|
-
*
|
|
855
|
+
* True when the budget emptied a node that had text — which `truncate` mode
|
|
856
|
+
* drops, at any depth, and `outline` mode keeps as a heading-only entry. A node
|
|
857
|
+
* that never carried text is kept either way: there was nothing to cut.
|
|
858
|
+
*/
|
|
859
|
+
function isBudgetEmptied(entry) {
|
|
860
|
+
return entry.returnedCharacters === 0 && entry.originalCharacters > 0;
|
|
861
|
+
}
|
|
862
|
+
/**
|
|
863
|
+
* Sections and subsections `truncate` mode dropped, read off the ledger. A
|
|
864
|
+
* dropped node counts once and takes its subtree with it, so its subsections
|
|
865
|
+
* are not counted again. Shared by the budget pass and the deferral roll-back
|
|
866
|
+
* so both count by one rule. (#81, #143)
|
|
867
|
+
*/
|
|
868
|
+
function countDroppedSections(entries, mode) {
|
|
869
|
+
if (mode !== 'truncate')
|
|
870
|
+
return 0;
|
|
871
|
+
return entries.reduce((n, entry) => n + (isBudgetEmptied(entry) ? 1 : countDroppedSections(entry.subsections ?? [], mode)), 0);
|
|
872
|
+
}
|
|
873
|
+
/**
|
|
874
|
+
* Rebuild a section subtree from `fitted` — one entry per node, in the document
|
|
875
|
+
* order {@link sectionTextFields} produced them, `cursor` walking the flat list
|
|
876
|
+
* across the whole subtree — together with its ledger entry.
|
|
877
|
+
*
|
|
878
|
+
* In `truncate` mode a node the budget emptied is dropped (`kept` is absent) at
|
|
879
|
+
* every depth, where only a top-level section used to be: a subsection left as
|
|
880
|
+
* a heading over nothing is a stub, not content. `outline` mode keeps it, and
|
|
881
|
+
* `format()` marks it. The entry lists its subsections only when this node was
|
|
882
|
+
* shortened, which is where they say something a whole section's entry does
|
|
883
|
+
* not. (#81, #143)
|
|
792
884
|
*/
|
|
793
|
-
function
|
|
885
|
+
function fitSectionTree(section, fitted, cursor, mode) {
|
|
794
886
|
const text = fitted[cursor.i++] ?? '';
|
|
795
|
-
const
|
|
796
|
-
|
|
887
|
+
const children = (section.subsections ?? []).map((sub) => fitSectionTree(sub, fitted, cursor, mode));
|
|
888
|
+
const originalCharacters = sectionCharacters(section);
|
|
889
|
+
const returnedCharacters = children.reduce((n, child) => n + child.entry.returnedCharacters, text.length);
|
|
890
|
+
const truncated = returnedCharacters < originalCharacters;
|
|
891
|
+
const entry = {
|
|
892
|
+
...(section.title !== undefined && { title: section.title }),
|
|
893
|
+
...(section.label !== undefined && { label: section.label }),
|
|
894
|
+
originalCharacters,
|
|
895
|
+
returnedCharacters,
|
|
896
|
+
truncated,
|
|
897
|
+
...(truncated && children.length > 0 && { subsections: children.map((c) => c.entry) }),
|
|
898
|
+
};
|
|
899
|
+
if (mode === 'truncate' && isBudgetEmptied(entry))
|
|
900
|
+
return { entry };
|
|
901
|
+
const { subsections: _replaced, ...rest } = section;
|
|
902
|
+
const subsections = children.flatMap((child) => (child.kept ? [child.kept] : []));
|
|
903
|
+
return {
|
|
904
|
+
entry,
|
|
905
|
+
kept: { ...rest, text, ...(subsections.length > 0 && { subsections }) },
|
|
906
|
+
};
|
|
797
907
|
}
|
|
798
908
|
/**
|
|
799
909
|
* Shorten an ordered list of text fields so their combined length fits
|
|
800
|
-
* `allowance`. Fields are filled in order, so earlier fields survive whole
|
|
801
|
-
*
|
|
802
|
-
*
|
|
803
|
-
*
|
|
804
|
-
*
|
|
805
|
-
*
|
|
806
|
-
*
|
|
910
|
+
* `allowance`. Fields are filled in order, so earlier fields survive whole — the
|
|
911
|
+
* section's own text before its subsections — and the first field that does not
|
|
912
|
+
* fit is cut at the last word boundary inside what is left. Every field after
|
|
913
|
+
* that cut is past it and returns empty, the way a top-level section past the
|
|
914
|
+
* budget does, even when the cut left a few characters unspent: handing those
|
|
915
|
+
* on would open the next subsection with a fragment of its first word. (#143)
|
|
916
|
+
*
|
|
917
|
+
* No marker is appended, so the reported `returnedCharacters` is exact;
|
|
918
|
+
* `format()` carries the human-visible note. Counts are measured off the
|
|
919
|
+
* returned text, never off the allowance, which stays a ceiling. (#93)
|
|
807
920
|
*/
|
|
808
921
|
function fitFields(fields, allowance) {
|
|
809
922
|
let remaining = Math.max(allowance, 0);
|
|
810
923
|
return fields.map((text) => {
|
|
811
|
-
|
|
812
|
-
|
|
924
|
+
if (text.length <= remaining) {
|
|
925
|
+
remaining -= text.length;
|
|
926
|
+
return text;
|
|
927
|
+
}
|
|
928
|
+
const kept = sliceAtWordBoundary(text, remaining);
|
|
929
|
+
remaining = 0;
|
|
813
930
|
return kept;
|
|
814
931
|
});
|
|
815
932
|
}
|
|
@@ -882,10 +999,10 @@ function allotSectionBudgets(sizes, budget) {
|
|
|
882
999
|
* nothing bounds either list and every entry is kept. (#111, #130)
|
|
883
1000
|
*
|
|
884
1001
|
* Returns the article untouched (same object identity) when no budget was
|
|
885
|
-
* requested or nothing exceeded it. A section left with zero
|
|
886
|
-
* dropped in `truncate` mode and counted as omitted; `outline`
|
|
887
|
-
* heading-only entry. Dropped
|
|
888
|
-
* caller can see which headings exist. (#81)
|
|
1002
|
+
* requested or nothing exceeded it. A section or subsection left with zero
|
|
1003
|
+
* characters is dropped in `truncate` mode and counted as omitted; `outline`
|
|
1004
|
+
* keeps it as a heading-only entry. Dropped nodes still appear in the accounting
|
|
1005
|
+
* so the caller can see which headings exist. (#81, #143)
|
|
889
1006
|
*/
|
|
890
1007
|
function applyPmcBudget(article, budget) {
|
|
891
1008
|
const tables = article.tables ?? [];
|
|
@@ -901,25 +1018,16 @@ function applyPmcBudget(article, budget) {
|
|
|
901
1018
|
const allowances = allotSectionBudgets(sizes, budget);
|
|
902
1019
|
const kept = [];
|
|
903
1020
|
const sectionReports = [];
|
|
904
|
-
let omittedSections = 0;
|
|
905
1021
|
let returnedCharacters = 0;
|
|
906
1022
|
article.sections.forEach((section, i) => {
|
|
907
|
-
const original = sizes[i] ?? 0;
|
|
908
1023
|
const fitted = fitFields(sectionTextFields(section), allowances[i] ?? 0);
|
|
909
|
-
const
|
|
910
|
-
returnedCharacters +=
|
|
911
|
-
sectionReports.push(
|
|
912
|
-
|
|
913
|
-
|
|
914
|
-
returnedCharacters: returned,
|
|
915
|
-
truncated: returned < original,
|
|
916
|
-
});
|
|
917
|
-
if (returned === 0 && original > 0 && budget.overflowMode === 'truncate') {
|
|
918
|
-
omittedSections += 1;
|
|
919
|
-
return;
|
|
920
|
-
}
|
|
921
|
-
kept.push(withFittedTexts(section, fitted, { i: 0 }));
|
|
1024
|
+
const fit = fitSectionTree(section, fitted, { i: 0 }, budget.overflowMode);
|
|
1025
|
+
returnedCharacters += fit.entry.returnedCharacters;
|
|
1026
|
+
sectionReports.push(fit.entry);
|
|
1027
|
+
if (fit.kept)
|
|
1028
|
+
kept.push(fit.kept);
|
|
922
1029
|
});
|
|
1030
|
+
const omittedSections = countDroppedSections(sectionReports, budget.overflowMode);
|
|
923
1031
|
// Sections are served first, then tables, then assets — each spending whatever
|
|
924
1032
|
// `maxCharacters` has left. A bare per-section budget sets no total, so nothing
|
|
925
1033
|
// bounds either list.
|
|
@@ -960,13 +1068,13 @@ function applyPmcBudget(article, budget) {
|
|
|
960
1068
|
* Apply the character budget to an Unpaywall body. That body is one
|
|
961
1069
|
* unstructured blob — HTML-as-Markdown or PDF-as-text — so only `maxCharacters`
|
|
962
1070
|
* applies, and `outline` mode has no headings to preserve and behaves like
|
|
963
|
-
* `truncate`. (#81)
|
|
1071
|
+
* `truncate`. The cut ends at a word boundary, as a section cut does. (#81, #143)
|
|
964
1072
|
*/
|
|
965
1073
|
function applyContentBudget(content, budget) {
|
|
966
1074
|
const cap = budget.maxCharacters;
|
|
967
1075
|
if (cap === undefined || content.length <= cap)
|
|
968
1076
|
return { content };
|
|
969
|
-
const kept =
|
|
1077
|
+
const kept = sliceAtWordBoundary(content, cap);
|
|
970
1078
|
return {
|
|
971
1079
|
content: kept,
|
|
972
1080
|
truncation: { originalCharacters: content.length, returnedCharacters: kept.length },
|
|
@@ -1075,17 +1183,6 @@ function buildDeferralNotice(deferred) {
|
|
|
1075
1183
|
: `Response character budget reached: ${deferred.returnedCharacters} of ${deferred.maxResponseCharacters} characters returned.`;
|
|
1076
1184
|
return `${spent} ${deferred.deferredCount} resolved article(s) were deferred whole: ${deferred.ids.join(', ')}. Re-call pubmed_fetch_fulltext with those ids under \`${deferred.idType}s\` to retrieve them, or raise maxResponseCharacters to at least ${deferred.nextDeferredCharacters} — the size of the next deferred article.`;
|
|
1077
1185
|
}
|
|
1078
|
-
/**
|
|
1079
|
-
* Body sections an article's per-article budget dropped, derived from that
|
|
1080
|
-
* article's own accounting by the rule {@link applyPmcBudget} counts by:
|
|
1081
|
-
* `outline` mode keeps every heading, so it drops none. Used to take a deferred
|
|
1082
|
-
* article's contribution back out of the response-level roll-up. (#100)
|
|
1083
|
-
*/
|
|
1084
|
-
function countOmittedSections(entry, mode) {
|
|
1085
|
-
if (mode !== 'truncate')
|
|
1086
|
-
return 0;
|
|
1087
|
-
return (entry.sections ?? []).filter((s) => s.originalCharacters > 0 && s.returnedCharacters === 0).length;
|
|
1088
|
-
}
|
|
1089
1186
|
// ─── Tool Definition ─────────────────────────────────────────────────────────
|
|
1090
1187
|
/**
|
|
1091
1188
|
* Compose the tool description for the fallback tiers enabled in this
|
|
@@ -1132,11 +1229,11 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1132
1229
|
annotations: { readOnlyHint: true, openWorldHint: true },
|
|
1133
1230
|
_meta: conceptMeta([SCHEMA_SCHOLARLY_ARTICLE, EDAM_DATA_RETRIEVAL]),
|
|
1134
1231
|
sourceUrl: 'https://github.com/cyanheads/pubmed-mcp-server/blob/main/src/mcp-server/tools/definitions/fetch-fulltext.tool.ts',
|
|
1135
|
-
|
|
1136
|
-
|
|
1137
|
-
|
|
1138
|
-
|
|
1139
|
-
],
|
|
1232
|
+
// Only the ID routing's `idConvert` calls run unwrapped. Every Europe PMC and
|
|
1233
|
+
// Unpaywall call sits behind a catch that folds the failure into
|
|
1234
|
+
// `unavailable[].triedTiers`, so their service reasons never reach a caller
|
|
1235
|
+
// and are not declared here. (#168)
|
|
1236
|
+
errors: [...NCBI_SERVICE_ERRORS],
|
|
1140
1237
|
input: z
|
|
1141
1238
|
.object({
|
|
1142
1239
|
pmcids: z
|
|
@@ -1186,7 +1283,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1186
1283
|
.min(1)
|
|
1187
1284
|
.max(1_000_000)
|
|
1188
1285
|
.optional()
|
|
1189
|
-
.describe('Per-article budget for body text, in characters. Counts `source=pmc` section and subsection text — which carries the inline blocks the parser renders in place, such as lists, definition lists, block quotes, boxed text, preformatted blocks and displayed formulae — plus table label, caption, cell and footnote text and asset label, caption and `href` text; or the `source=unpaywall` `content` body. Titles, abstracts, identifiers, and references are never counted or shortened. The counted unit is that text alone — the Markdown grid `content[]` renders around the cells (pipes, padding, the divider row, headings) is scaffolding this budget does not measure, so a table renders longer than it costs here. Sections are served first, then tables, then assets, each spending what is left, in document order — admission stops at the first entry that does not fit, and every entry from there on is dropped whole rather than cut mid-row or returned with a shortened caption, counted in `truncation.omittedTables` / `truncation.omittedAssets` and named in `truncation.articles[].omittedTableNames` / `omittedAssetNames`. Applied after `sections`, `maxSections`, `includeReferences`, `includeTables`, and `includeAssets`, so semantic filtering is unaffected. This knob alone bounds only bodies: the response-wide ceiling it implies is this value times the number of articles returned, plus every uncounted field. Use `maxResponseCharacters` for a true whole-response ceiling. Omit for the full body.'),
|
|
1286
|
+
.describe('Per-article budget for body text, in characters. Counts `source=pmc` section and subsection text — which carries the inline blocks the parser renders in place, such as lists, definition lists, block quotes, boxed text, preformatted blocks and displayed formulae — plus table label, caption, cell and footnote text and asset label, caption and `href` text; or the `source=unpaywall` `content` body. Titles, abstracts, identifiers, and references are never counted or shortened. Shortened text ends at the last word boundary inside its allowance, so it can come back a few characters under it. The counted unit is that text alone — the Markdown grid `content[]` renders around the cells (pipes, padding, the divider row, headings) is scaffolding this budget does not measure, so a table renders longer than it costs here. Sections are served first, then tables, then assets, each spending what is left, in document order — admission stops at the first entry that does not fit, and every entry from there on is dropped whole rather than cut mid-row or returned with a shortened caption, counted in `truncation.omittedTables` / `truncation.omittedAssets` and named in `truncation.articles[].omittedTableNames` / `omittedAssetNames`. Applied after `sections`, `maxSections`, `includeReferences`, `includeTables`, and `includeAssets`, so semantic filtering is unaffected. This knob alone bounds only bodies: the response-wide ceiling it implies is this value times the number of articles returned, plus every uncounted field. Use `maxResponseCharacters` for a true whole-response ceiling. Omit for the full body.'),
|
|
1190
1287
|
maxCharactersPerSection: z
|
|
1191
1288
|
.number()
|
|
1192
1289
|
.int()
|
|
@@ -1204,7 +1301,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1204
1301
|
overflowMode: z
|
|
1205
1302
|
.enum(['truncate', 'outline'])
|
|
1206
1303
|
.default('truncate')
|
|
1207
|
-
.describe('How to spend `maxCharacters` across an article that exceeds it. truncate: fill sections in document order, so early sections stay whole and
|
|
1304
|
+
.describe('How to spend `maxCharacters` across an article that exceeds it. truncate: fill sections in document order, so early sections stay whole, the section the budget runs out in is cut, and every section or subsection past that point is dropped (counted in `truncation.omittedSections`). outline: split the budget evenly so every section and subsection keeps its heading, and an excerpt as far as the budget reaches — a heading the budget left empty is marked as such in the rendered text. Use it to survey what an article contains before requesting specific `sections`. Ignored when no budget is set, and identical for `source=unpaywall` bodies, which have no headings to preserve.'),
|
|
1208
1305
|
})
|
|
1209
1306
|
.refine((v) => [v.pmcids, v.pmids, v.dois].filter((b) => b !== undefined).length === 1, {
|
|
1210
1307
|
message: 'Provide exactly one of `pmcids`, `pmids`, or `dois` (not zero, not more).',
|
|
@@ -1289,10 +1386,18 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1289
1386
|
// carry — a `pmids` request recovers articles keyed by PMCID. (#100)
|
|
1290
1387
|
const inputIdByArticle = new Map();
|
|
1291
1388
|
// The caller's own spellings of each chain key, so `unavailable[]` and
|
|
1292
|
-
// `deferred.ids` report what was submitted.
|
|
1293
|
-
//
|
|
1294
|
-
//
|
|
1389
|
+
// `deferred.ids` report what was submitted. The `pmids` branch fills it — a
|
|
1390
|
+
// PMID's chain runs on its canonical form — and so does the `dois` branch,
|
|
1391
|
+
// whose chain runs once per DOI however many casings name it. Both also
|
|
1392
|
+
// fold in an input id that resolves to a PMC record another one already
|
|
1393
|
+
// claimed (see `routeToPmc`). Any other key is its own spelling. (#161, #166)
|
|
1295
1394
|
const callerIds = new Map();
|
|
1395
|
+
const addCallerId = (key, spelling) => {
|
|
1396
|
+
const spellings = callerIds.get(key) ?? [];
|
|
1397
|
+
if (!spellings.includes(spelling))
|
|
1398
|
+
spellings.push(spelling);
|
|
1399
|
+
callerIds.set(key, spellings);
|
|
1400
|
+
};
|
|
1296
1401
|
const budget = {
|
|
1297
1402
|
overflowMode: input.overflowMode,
|
|
1298
1403
|
...(input.maxCharacters !== undefined && { maxCharacters: input.maxCharacters }),
|
|
@@ -1306,17 +1411,34 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1306
1411
|
let pmidFallbackCandidates = [];
|
|
1307
1412
|
let pmcidFallbackCandidates = [];
|
|
1308
1413
|
let doiCandidates = [];
|
|
1414
|
+
// Send a converter-resolved PMCID to PMC EFetch under the input id that
|
|
1415
|
+
// named it. A second input id the converter places on the same PMC record
|
|
1416
|
+
// — two DOIs for one article — joins the first one's chain as another
|
|
1417
|
+
// spelling of it: the record is fetched once, and both ids are recovered,
|
|
1418
|
+
// or reported unavailable with that chain, rather than the later id taking
|
|
1419
|
+
// the PMCID over and leaving the earlier one with an empty chain.
|
|
1420
|
+
const routeToPmc = (inputId, pmcid) => {
|
|
1421
|
+
const normalized = normalizePmcId(pmcid);
|
|
1422
|
+
const prefixed = withPmcPrefix(normalized);
|
|
1423
|
+
const owner = pmcidToInputId.get(prefixed);
|
|
1424
|
+
if (owner === undefined) {
|
|
1425
|
+
pmcIds.push(normalized);
|
|
1426
|
+
pmcidToInputId.set(prefixed, inputId);
|
|
1427
|
+
return;
|
|
1428
|
+
}
|
|
1429
|
+
if (owner === inputId)
|
|
1430
|
+
return;
|
|
1431
|
+
for (const spelling of callerIds.get(inputId) ?? [inputId])
|
|
1432
|
+
addCallerId(owner, spelling);
|
|
1433
|
+
callerIds.delete(inputId);
|
|
1434
|
+
chainByInput.delete(inputId);
|
|
1435
|
+
};
|
|
1309
1436
|
if (input.pmids) {
|
|
1310
1437
|
// The ID Converter parses `00000001` as PMID 1 yet reports it "not found
|
|
1311
1438
|
// in PMC", and every later stage answers with NCBI's own PMID, so the chain
|
|
1312
1439
|
// runs once per distinct PMID in its canonical form. (#161)
|
|
1313
|
-
for (const id of input.pmids)
|
|
1314
|
-
|
|
1315
|
-
const spellings = callerIds.get(pmid) ?? [];
|
|
1316
|
-
if (!spellings.includes(id))
|
|
1317
|
-
spellings.push(id);
|
|
1318
|
-
callerIds.set(pmid, spellings);
|
|
1319
|
-
}
|
|
1440
|
+
for (const id of input.pmids)
|
|
1441
|
+
addCallerId(normalizePmid(id), id);
|
|
1320
1442
|
const pmids = [...callerIds.keys()];
|
|
1321
1443
|
for (const id of pmids)
|
|
1322
1444
|
chainByInput.set(id, []);
|
|
@@ -1328,9 +1450,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1328
1450
|
const pmid = String(r.pmid);
|
|
1329
1451
|
seen.add(pmid);
|
|
1330
1452
|
if (r.pmcid) {
|
|
1331
|
-
|
|
1332
|
-
pmcIds.push(normalized);
|
|
1333
|
-
pmcidToInputId.set(withPmcPrefix(normalized), pmid);
|
|
1453
|
+
routeToPmc(pmid, String(r.pmcid));
|
|
1334
1454
|
pmidContext.set(pmid, { pmid, ...(r.doi && { doi: r.doi }) });
|
|
1335
1455
|
}
|
|
1336
1456
|
else {
|
|
@@ -1354,31 +1474,42 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1354
1474
|
}
|
|
1355
1475
|
}
|
|
1356
1476
|
else if (input.pmcids) {
|
|
1357
|
-
|
|
1358
|
-
|
|
1359
|
-
pmcIds = input.pmcids.map(normalizePmcId);
|
|
1477
|
+
// `PMC123`, `pmc123`, `123` and `PMC0123` name one record; it runs the
|
|
1478
|
+
// chain once. (#170)
|
|
1479
|
+
pmcIds = [...new Set(input.pmcids.map(normalizePmcId))];
|
|
1480
|
+
for (const id of pmcIds)
|
|
1481
|
+
chainByInput.set(withPmcPrefix(id), []);
|
|
1360
1482
|
}
|
|
1361
1483
|
else if (input.dois) {
|
|
1362
1484
|
// Mirror the `pmids` branch: resolve DOI → PMCID via the PMC ID Converter
|
|
1363
1485
|
// so PMC-indexed DOIs reach PMC EFetch instead of going straight to the
|
|
1364
1486
|
// EPMC/Unpaywall fallback (which misses articles whose only OA copy is the
|
|
1365
1487
|
// PMC JATS). DOIs the converter can't place in PMC seed `doiCandidates`.
|
|
1366
|
-
|
|
1367
|
-
|
|
1488
|
+
//
|
|
1489
|
+
// DOIs are case-insensitive, and the converter echoes one casing in
|
|
1490
|
+
// `requested-id` for DOIs that differ only in case, so every casing of a
|
|
1491
|
+
// DOI shares one chain, keyed by the first spelling submitted, and
|
|
1492
|
+
// converter records are matched to it case-insensitively. (#166)
|
|
1493
|
+
const chainKeyByDoi = new Map();
|
|
1494
|
+
for (const doi of input.dois) {
|
|
1495
|
+
const key = chainKeyByDoi.get(doi.toLowerCase()) ?? doi;
|
|
1496
|
+
chainKeyByDoi.set(doi.toLowerCase(), key);
|
|
1497
|
+
addCallerId(key, doi);
|
|
1498
|
+
}
|
|
1499
|
+
const dois = [...callerIds.keys()];
|
|
1500
|
+
for (const doi of dois)
|
|
1368
1501
|
chainByInput.set(doi, []);
|
|
1369
|
-
const records = await getNcbiService().idConvert(
|
|
1502
|
+
const records = await getNcbiService().idConvert(dois, 'doi', ctx.signal ? { signal: ctx.signal } : undefined);
|
|
1370
1503
|
const seen = new Set();
|
|
1371
1504
|
for (const r of records) {
|
|
1372
|
-
//
|
|
1373
|
-
//
|
|
1374
|
-
const doi = String(r['requested-id']);
|
|
1375
|
-
if (
|
|
1505
|
+
// Match on the echoed `requested-id`, not `r.doi`: the record's own DOI
|
|
1506
|
+
// can be cased differently from anything the caller sent.
|
|
1507
|
+
const doi = chainKeyByDoi.get(String(r['requested-id']).toLowerCase());
|
|
1508
|
+
if (doi === undefined)
|
|
1376
1509
|
continue;
|
|
1377
1510
|
seen.add(doi);
|
|
1378
1511
|
if (r.pmcid) {
|
|
1379
|
-
|
|
1380
|
-
pmcIds.push(normalized);
|
|
1381
|
-
pmcidToInputId.set(withPmcPrefix(normalized), doi);
|
|
1512
|
+
routeToPmc(doi, String(r.pmcid));
|
|
1382
1513
|
}
|
|
1383
1514
|
else {
|
|
1384
1515
|
chainByInput.get(doi)?.push({
|
|
@@ -1389,7 +1520,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1389
1520
|
doiCandidates.push({ doi });
|
|
1390
1521
|
}
|
|
1391
1522
|
}
|
|
1392
|
-
for (const requested of
|
|
1523
|
+
for (const requested of dois) {
|
|
1393
1524
|
if (!seen.has(requested)) {
|
|
1394
1525
|
chainByInput.get(requested)?.push({
|
|
1395
1526
|
tier: 'pmc',
|
|
@@ -1662,7 +1793,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1662
1793
|
return {
|
|
1663
1794
|
pmcId,
|
|
1664
1795
|
result: candidate.doi
|
|
1665
|
-
? await resolveUnpaywall({ pmcId, doi: candidate.doi, budget }, unpaywall, ctx)
|
|
1796
|
+
? await resolveUnpaywall({ pmcId, doi: candidate.doi, epmcTitle: candidate.title, budget }, unpaywall, ctx)
|
|
1666
1797
|
: undefined,
|
|
1667
1798
|
};
|
|
1668
1799
|
}));
|
|
@@ -1734,7 +1865,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1734
1865
|
const outcomes = await Promise.all(pmidFallbackCandidates.map(async (candidate) => ({
|
|
1735
1866
|
candidate,
|
|
1736
1867
|
result: candidate.doi
|
|
1737
|
-
? await resolveUnpaywall({ pmid: candidate.pmid, doi: candidate.doi, budget }, unpaywall, ctx)
|
|
1868
|
+
? await resolveUnpaywall({ pmid: candidate.pmid, doi: candidate.doi, epmcTitle: candidate.title, budget }, unpaywall, ctx)
|
|
1738
1869
|
: undefined,
|
|
1739
1870
|
})));
|
|
1740
1871
|
for (const { candidate, result } of outcomes) {
|
|
@@ -1777,7 +1908,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1777
1908
|
// doesn't reject under normal operation.
|
|
1778
1909
|
const outcomes = await Promise.all(doiCandidates.map(async (c) => ({
|
|
1779
1910
|
doi: c.doi,
|
|
1780
|
-
result: await resolveUnpaywall({ doi: c.doi, budget }, unpaywall, ctx),
|
|
1911
|
+
result: await resolveUnpaywall({ doi: c.doi, epmcTitle: c.title, budget }, unpaywall, ctx),
|
|
1781
1912
|
})));
|
|
1782
1913
|
for (const { doi, result } of outcomes) {
|
|
1783
1914
|
if ('article' in result) {
|
|
@@ -1848,8 +1979,9 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1848
1979
|
if (index === -1)
|
|
1849
1980
|
continue;
|
|
1850
1981
|
const [dropped] = truncatedArticles.splice(index, 1);
|
|
1851
|
-
if (dropped)
|
|
1852
|
-
omittedSections -=
|
|
1982
|
+
if (dropped) {
|
|
1983
|
+
omittedSections -= countDroppedSections(dropped.sections ?? [], input.overflowMode);
|
|
1984
|
+
}
|
|
1853
1985
|
}
|
|
1854
1986
|
}
|
|
1855
1987
|
ctx.log.info('pubmed_fetch_fulltext completed', {
|
|
@@ -1957,13 +2089,29 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1957
2089
|
lines.push('');
|
|
1958
2090
|
const t = truncationById.get(articleDisplayId(a));
|
|
1959
2091
|
if (a.source === 'pmc')
|
|
1960
|
-
formatPmcArticle(a, lines, t);
|
|
2092
|
+
formatPmcArticle(a, lines, t, result.truncation?.mode);
|
|
1961
2093
|
else
|
|
1962
2094
|
formatUnpaywallArticle(a, lines, t);
|
|
1963
2095
|
}
|
|
1964
2096
|
return [{ type: 'text', text: lines.join('\n') }];
|
|
1965
2097
|
},
|
|
1966
2098
|
});
|
|
2099
|
+
/**
|
|
2100
|
+
* Merge what a Europe PMC hit carried onto a candidate bound for Unpaywall. A
|
|
2101
|
+
* DOI the candidate already holds wins — for `dois` input it is the caller's
|
|
2102
|
+
* own identifier.
|
|
2103
|
+
*/
|
|
2104
|
+
function carryEpmcHit(candidate, hit) {
|
|
2105
|
+
return {
|
|
2106
|
+
...candidate,
|
|
2107
|
+
...(hit.doi && !candidate.doi && { doi: hit.doi }),
|
|
2108
|
+
...(hit.title && { title: hit.title }),
|
|
2109
|
+
};
|
|
2110
|
+
}
|
|
2111
|
+
/** Display-ready plain text, or `undefined` when nothing is left once markup is stripped. */
|
|
2112
|
+
function nonEmptyDisplayText(raw) {
|
|
2113
|
+
return (raw && toDisplayText(raw)) || undefined;
|
|
2114
|
+
}
|
|
1967
2115
|
/**
|
|
1968
2116
|
* Run the Europe PMC step against everything that fell through PMC EFetch
|
|
1969
2117
|
* plus any direct DOI input. Each candidate goes through search-by-best-id →
|
|
@@ -1983,24 +2131,28 @@ async function runEpmcStage(epmc, args) {
|
|
|
1983
2131
|
}
|
|
1984
2132
|
if (search.kind === 'miss')
|
|
1985
2133
|
return { c, outcome: { kind: 'miss' } };
|
|
1986
|
-
const
|
|
2134
|
+
const title = nonEmptyDisplayText(search.hit.title);
|
|
2135
|
+
const hit = {
|
|
2136
|
+
...(search.hit.doi && { doi: search.hit.doi }),
|
|
2137
|
+
...(title && { title }),
|
|
2138
|
+
};
|
|
1987
2139
|
const fetched = await fetchEpmcArticle(epmc, search.hit, args, contextPmid);
|
|
1988
2140
|
if (fetched.kind === 'error') {
|
|
1989
|
-
return { c, ...
|
|
2141
|
+
return { c, ...hit, outcome: { kind: 'service-error', detail: fetched.detail } };
|
|
1990
2142
|
}
|
|
1991
2143
|
if (fetched.kind === 'no-fulltext') {
|
|
1992
2144
|
return {
|
|
1993
2145
|
c,
|
|
1994
|
-
...
|
|
2146
|
+
...hit,
|
|
1995
2147
|
outcome: { kind: 'no-fulltext', ...(fetched.detail && { detail: fetched.detail }) },
|
|
1996
2148
|
};
|
|
1997
2149
|
}
|
|
1998
2150
|
if (fetched.kind === 'no-body') {
|
|
1999
|
-
return { c, ...
|
|
2151
|
+
return { c, ...hit, outcome: { kind: 'no-body', detail: fetched.detail } };
|
|
2000
2152
|
}
|
|
2001
2153
|
return {
|
|
2002
2154
|
c,
|
|
2003
|
-
...
|
|
2155
|
+
...hit,
|
|
2004
2156
|
outcome: { kind: 'hit' },
|
|
2005
2157
|
article: fetched.article,
|
|
2006
2158
|
sectionFilterMiss: fetched.sectionFilterMiss,
|
|
@@ -2052,25 +2204,25 @@ async function runEpmcStage(epmc, args) {
|
|
|
2052
2204
|
pmidOutcomes.set(run.c.pmid, run.outcome);
|
|
2053
2205
|
if (run.article)
|
|
2054
2206
|
collectHit(run.c.pmid, { ...run, article: run.article });
|
|
2055
|
-
//
|
|
2056
|
-
// fetch outcome was
|
|
2057
|
-
//
|
|
2207
|
+
// What the EPMC hit carried is evidence the next stage needs, whatever the
|
|
2208
|
+
// fetch outcome was: its DOI spares Unpaywall a PubMed metadata round-trip
|
|
2209
|
+
// (#119), and its title backs up Unpaywall's own record (#144).
|
|
2058
2210
|
else
|
|
2059
|
-
remainingPmid.push(run.
|
|
2211
|
+
remainingPmid.push(carryEpmcHit(run.c, run));
|
|
2060
2212
|
}
|
|
2061
2213
|
for (const run of pmcidResults) {
|
|
2062
2214
|
pmcidOutcomes.set(run.c.normalized, run.outcome);
|
|
2063
2215
|
if (run.article)
|
|
2064
2216
|
collectHit(run.c.normalized, { ...run, article: run.article });
|
|
2065
2217
|
else
|
|
2066
|
-
remainingPmcid.push(run.
|
|
2218
|
+
remainingPmcid.push(carryEpmcHit(run.c.c, run));
|
|
2067
2219
|
}
|
|
2068
2220
|
for (const run of doiResults) {
|
|
2069
2221
|
doiOutcomes.set(run.c.doi, run.outcome);
|
|
2070
2222
|
if (run.article)
|
|
2071
2223
|
collectHit(run.c.doi, { ...run, article: run.article });
|
|
2072
2224
|
else
|
|
2073
|
-
remainingDoi.push(run.c);
|
|
2225
|
+
remainingDoi.push(carryEpmcHit(run.c, run));
|
|
2074
2226
|
}
|
|
2075
2227
|
return {
|
|
2076
2228
|
articles,
|
|
@@ -2220,9 +2372,15 @@ async function fetchPubmedDois(pmids, signal) {
|
|
|
2220
2372
|
* Resolve a DOI to an open-access article via Unpaywall. `pmcId` and `pmid`,
|
|
2221
2373
|
* when set, are stamped onto the resulting article so the branch that requested
|
|
2222
2374
|
* it carries its identifier through — Unpaywall itself only knows the DOI.
|
|
2375
|
+
*
|
|
2376
|
+
* The article's title is the first one found in Unpaywall's own record, then
|
|
2377
|
+
* `epmcTitle` — the Europe PMC record the chain searched for this id — then,
|
|
2378
|
+
* for HTML content only, the title the extractor detects on the page. None is
|
|
2379
|
+
* ever invented: with no source carrying one, the article has no title. The
|
|
2380
|
+
* journal name and year come from Unpaywall's record alone. (#144)
|
|
2223
2381
|
*/
|
|
2224
2382
|
async function resolveUnpaywall(args, service, ctx) {
|
|
2225
|
-
const { pmcId, pmid, doi, budget } = args;
|
|
2383
|
+
const { pmcId, pmid, doi, epmcTitle, budget } = args;
|
|
2226
2384
|
const requestedIds = { ...(pmcId && { pmcId }), ...(pmid && { pmid }) };
|
|
2227
2385
|
/** Budget the extracted body, then pair the article with its accounting. */
|
|
2228
2386
|
const budgeted = (build, content) => {
|
|
@@ -2260,6 +2418,15 @@ async function resolveUnpaywall(args, service, ctx) {
|
|
|
2260
2418
|
ctx.log.warning('Unpaywall content fetch failed', { doi, error: detail });
|
|
2261
2419
|
return { unavailable: { reason: 'fetch-failed', detail } };
|
|
2262
2420
|
}
|
|
2421
|
+
const recordTitle = nonEmptyDisplayText(resolution.title) ?? epmcTitle;
|
|
2422
|
+
const record = {
|
|
2423
|
+
...requestedIds,
|
|
2424
|
+
doi,
|
|
2425
|
+
sourceUrl: content.fetchedUrl,
|
|
2426
|
+
location: resolution.location,
|
|
2427
|
+
journalName: nonEmptyDisplayText(resolution.journalName),
|
|
2428
|
+
year: resolution.year,
|
|
2429
|
+
};
|
|
2263
2430
|
try {
|
|
2264
2431
|
if (content.kind === 'html') {
|
|
2265
2432
|
const extracted = await htmlExtractor.extract(content.body, {
|
|
@@ -2276,13 +2443,10 @@ async function resolveUnpaywall(args, service, ctx) {
|
|
|
2276
2443
|
};
|
|
2277
2444
|
}
|
|
2278
2445
|
return budgeted((text) => buildUnpaywallArticle({
|
|
2279
|
-
...
|
|
2280
|
-
doi,
|
|
2281
|
-
sourceUrl: content.fetchedUrl,
|
|
2282
|
-
location: resolution.location,
|
|
2446
|
+
...record,
|
|
2283
2447
|
contentFormat: 'html-markdown',
|
|
2284
2448
|
content: text,
|
|
2285
|
-
title: extracted.title,
|
|
2449
|
+
title: recordTitle ?? extracted.title,
|
|
2286
2450
|
wordCount: extracted.wordCount,
|
|
2287
2451
|
}), body);
|
|
2288
2452
|
}
|
|
@@ -2294,12 +2458,10 @@ async function resolveUnpaywall(args, service, ctx) {
|
|
|
2294
2458
|
};
|
|
2295
2459
|
}
|
|
2296
2460
|
return budgeted((body) => buildUnpaywallArticle({
|
|
2297
|
-
...
|
|
2298
|
-
doi,
|
|
2299
|
-
sourceUrl: content.fetchedUrl,
|
|
2300
|
-
location: resolution.location,
|
|
2461
|
+
...record,
|
|
2301
2462
|
contentFormat: 'pdf-text',
|
|
2302
2463
|
content: body,
|
|
2464
|
+
title: recordTitle,
|
|
2303
2465
|
totalPages: extracted.totalPages,
|
|
2304
2466
|
}), text);
|
|
2305
2467
|
}
|
|
@@ -2324,6 +2486,8 @@ function buildUnpaywallArticle(args) {
|
|
|
2324
2486
|
sourceUrl: args.sourceUrl,
|
|
2325
2487
|
content: args.content,
|
|
2326
2488
|
...(args.title && { title: args.title }),
|
|
2489
|
+
...(args.journalName && { journalName: args.journalName }),
|
|
2490
|
+
...(args.year !== undefined && { year: args.year }),
|
|
2327
2491
|
...(args.wordCount !== undefined && { wordCount: args.wordCount }),
|
|
2328
2492
|
...(args.totalPages !== undefined && { totalPages: args.totalPages }),
|
|
2329
2493
|
...(location.license && { license: location.license }),
|
|
@@ -2481,16 +2645,39 @@ function formatTruncation(t, lines) {
|
|
|
2481
2645
|
? ''
|
|
2482
2646
|
: `, ${a.omittedAssets} asset(s) dropped whole: ${(a.omittedAssetNames ?? []).join(', ')}`;
|
|
2483
2647
|
lines.push(`- ${a.id} (${a.source}): ${a.returnedCharacters} of ${a.originalCharacters} characters${tablesDropped}${assetsDropped}`);
|
|
2484
|
-
for (const s of a.sections ?? [])
|
|
2485
|
-
|
|
2486
|
-
}
|
|
2648
|
+
for (const s of a.sections ?? [])
|
|
2649
|
+
formatLedgerEntry(s, lines, t.mode, 1);
|
|
2487
2650
|
}
|
|
2488
2651
|
}
|
|
2652
|
+
/**
|
|
2653
|
+
* One section's ledger line, then its subsections' one level deeper. The line
|
|
2654
|
+
* names the section by {@link sectionHeading}, so it reads exactly as the
|
|
2655
|
+
* section's heading does in the body. An entry the budget emptied says what
|
|
2656
|
+
* became of it — dropped in `truncate` mode, kept as a bare heading in
|
|
2657
|
+
* `outline` mode. (#143, #148)
|
|
2658
|
+
*/
|
|
2659
|
+
function formatLedgerEntry(entry, lines, mode, depth) {
|
|
2660
|
+
const fate = isBudgetEmptied(entry)
|
|
2661
|
+
? mode === 'truncate'
|
|
2662
|
+
? ' — dropped'
|
|
2663
|
+
: ' — heading only'
|
|
2664
|
+
: '';
|
|
2665
|
+
lines.push(`${' '.repeat(depth)}- ${sectionHeading(entry)} — ${entry.returnedCharacters} of ${entry.originalCharacters} characters (truncated: ${entry.truncated})${fate}`);
|
|
2666
|
+
for (const sub of entry.subsections ?? [])
|
|
2667
|
+
formatLedgerEntry(sub, lines, mode, depth + 1);
|
|
2668
|
+
}
|
|
2489
2669
|
/** Per-article inline marker so a reader of one article's body knows it is partial. */
|
|
2490
2670
|
function truncationNote(t) {
|
|
2491
2671
|
return `\n> Body shortened to fit the requested character budget — ${t.returnedCharacters} of ${t.originalCharacters} characters returned. See \`truncation\` for per-section counts.`;
|
|
2492
2672
|
}
|
|
2493
|
-
|
|
2673
|
+
/**
|
|
2674
|
+
* The marker an `outline`-mode heading carries when the budget left it no text,
|
|
2675
|
+
* so it reads as withheld rather than as a heading over nothing. (#143)
|
|
2676
|
+
*/
|
|
2677
|
+
function emptiedSectionNote(originalCharacters) {
|
|
2678
|
+
return `> Text omitted to fit the requested character budget — 0 of ${originalCharacters} characters returned.`;
|
|
2679
|
+
}
|
|
2680
|
+
function formatPmcArticle(a, lines, truncation, mode) {
|
|
2494
2681
|
// Render-time only — `structuredContent.articles[].title` keeps the
|
|
2495
2682
|
// plain-text value the JATS parser produced. (#102)
|
|
2496
2683
|
lines.push(`### ${escapeMarkdownInline(a.title ?? articleDisplayId(a))}`);
|
|
@@ -2549,8 +2736,13 @@ function formatPmcArticle(a, lines, truncation) {
|
|
|
2549
2736
|
lines.push(truncationNote(truncation));
|
|
2550
2737
|
if (a.abstract)
|
|
2551
2738
|
lines.push(`\n#### Abstract\n${a.abstract}`);
|
|
2552
|
-
|
|
2553
|
-
|
|
2739
|
+
// `outline` mode keeps every section and subsection, so each one lines up
|
|
2740
|
+
// with its ledger entry by position. `truncate` mode drops the emptied ones —
|
|
2741
|
+
// positions no longer line up, and nothing left needs a marker.
|
|
2742
|
+
const ledger = mode === 'outline' ? truncation?.sections : undefined;
|
|
2743
|
+
a.sections.forEach((sec, i) => {
|
|
2744
|
+
formatSection(sec, lines, 4, ledger?.[i]);
|
|
2745
|
+
});
|
|
2554
2746
|
if (a.tables?.length)
|
|
2555
2747
|
formatTables(a.tables, lines);
|
|
2556
2748
|
if (a.assets?.length)
|
|
@@ -2677,6 +2869,10 @@ function formatUnpaywallArticle(a, lines, truncation) {
|
|
|
2677
2869
|
: 'Unpaywall (PDF → plain text)';
|
|
2678
2870
|
lines.push(`### ${escapeMarkdownInline(heading)}`);
|
|
2679
2871
|
lines.push(`**Source:** ${formatLabel}`);
|
|
2872
|
+
if (a.journalName)
|
|
2873
|
+
lines.push(`**Journal:** ${escapeMarkdownInline(a.journalName)}`);
|
|
2874
|
+
if (a.year !== undefined)
|
|
2875
|
+
lines.push(`**Year:** ${a.year}`);
|
|
2680
2876
|
if (a.pmcId)
|
|
2681
2877
|
lines.push(`**PMCID:** ${a.pmcId}`);
|
|
2682
2878
|
if (a.pmid)
|
|
@@ -2712,20 +2908,43 @@ function formatPmcAuthor(au) {
|
|
|
2712
2908
|
function formatHeading(label, title) {
|
|
2713
2909
|
return label ? `${label} ${title}` : title;
|
|
2714
2910
|
}
|
|
2911
|
+
/** How a section with no title and no label is named — in its body heading and its ledger line. */
|
|
2912
|
+
const UNTITLED_SECTION_LABEL = 'untitled section';
|
|
2913
|
+
/**
|
|
2914
|
+
* The name a section goes by in `content[]`: its label and title, else its
|
|
2915
|
+
* label alone, else {@link UNTITLED_SECTION_LABEL}. The body heading and the
|
|
2916
|
+
* truncation ledger line both take it from here, so the two always read the
|
|
2917
|
+
* same. Render-time escaped like every other upstream string interpolated into
|
|
2918
|
+
* a line, so a title's `*`, `_`, `` ` `` or `[` cannot restyle the heading;
|
|
2919
|
+
* `structuredContent` keeps the plain title. (#148, #169)
|
|
2920
|
+
*/
|
|
2921
|
+
function sectionHeading(section) {
|
|
2922
|
+
if (section.title)
|
|
2923
|
+
return escapeMarkdownInline(formatHeading(section.label, section.title));
|
|
2924
|
+
return section.label ? escapeMarkdownInline(section.label) : UNTITLED_SECTION_LABEL;
|
|
2925
|
+
}
|
|
2715
2926
|
/**
|
|
2716
2927
|
* Render one body section and everything nested under it, one markdown heading
|
|
2717
2928
|
* level per nesting level. Walks the full depth the output schema carries, so
|
|
2718
2929
|
* `content[]` shows every section `structuredContent` does. Headings stop
|
|
2719
2930
|
* deepening at `######`, the deepest markdown supports. (#112)
|
|
2931
|
+
*
|
|
2932
|
+
* Every section gets a heading — an untitled one included, under the name the
|
|
2933
|
+
* truncation ledger gives it — so its text never runs on from the block before
|
|
2934
|
+
* it. `ledger` is the section's `outline`-mode accounting, when there is one: a
|
|
2935
|
+
* section it shows the budget emptied carries a marker under its heading.
|
|
2936
|
+
* (#143, #148)
|
|
2720
2937
|
*/
|
|
2721
|
-
function formatSection(section, lines, depth) {
|
|
2722
|
-
|
|
2723
|
-
lines.push(`\n${'#'.repeat(Math.min(depth, 6))} ${formatHeading(section.label, section.title)}`);
|
|
2724
|
-
}
|
|
2938
|
+
function formatSection(section, lines, depth, ledger) {
|
|
2939
|
+
lines.push(`\n${'#'.repeat(Math.min(depth, 6))} ${sectionHeading(section)}`);
|
|
2725
2940
|
if (section.text)
|
|
2726
2941
|
lines.push(section.text);
|
|
2727
|
-
|
|
2728
|
-
|
|
2942
|
+
else if (ledger && isBudgetEmptied(ledger)) {
|
|
2943
|
+
lines.push(emptiedSectionNote(ledger.originalCharacters));
|
|
2944
|
+
}
|
|
2945
|
+
section.subsections?.forEach((sub, i) => {
|
|
2946
|
+
formatSection(sub, lines, depth + 1, ledger?.subsections?.[i]);
|
|
2947
|
+
});
|
|
2729
2948
|
}
|
|
2730
2949
|
/**
|
|
2731
2950
|
* Strip absolute URLs from chain detail strings. Upstream errors (e.g.
|