@cyanheads/pubmed-mcp-server 2.10.15 → 2.10.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +1 -1
- package/CLAUDE.md +1 -1
- package/README.md +26 -15
- package/changelog/2.10.x/2.10.16.md +26 -0
- package/changelog/2.10.x/2.10.17.md +19 -0
- package/dist/mcp-server/tools/definitions/_schemas.d.ts +11 -1
- package/dist/mcp-server/tools/definitions/_schemas.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/_schemas.js +13 -1
- package/dist/mcp-server/tools/definitions/_schemas.js.map +1 -1
- package/dist/mcp-server/tools/definitions/_text.d.ts +20 -3
- package/dist/mcp-server/tools/definitions/_text.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/_text.js +35 -3
- package/dist/mcp-server/tools/definitions/_text.js.map +1 -1
- package/dist/mcp-server/tools/definitions/convert-ids.tool.d.ts +3 -0
- package/dist/mcp-server/tools/definitions/convert-ids.tool.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/convert-ids.tool.js +41 -9
- package/dist/mcp-server/tools/definitions/convert-ids.tool.js.map +1 -1
- package/dist/mcp-server/tools/definitions/fetch-articles.tool.d.ts +3 -1
- package/dist/mcp-server/tools/definitions/fetch-articles.tool.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/fetch-articles.tool.js +12 -4
- package/dist/mcp-server/tools/definitions/fetch-articles.tool.js.map +1 -1
- package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.d.ts +25 -31
- package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.js +373 -131
- package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.js.map +1 -1
- package/dist/mcp-server/tools/definitions/find-related.tool.d.ts +7 -3
- package/dist/mcp-server/tools/definitions/find-related.tool.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/find-related.tool.js +38 -21
- package/dist/mcp-server/tools/definitions/find-related.tool.js.map +1 -1
- package/dist/mcp-server/tools/definitions/format-citations.tool.d.ts +3 -1
- package/dist/mcp-server/tools/definitions/format-citations.tool.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/format-citations.tool.js +12 -4
- package/dist/mcp-server/tools/definitions/format-citations.tool.js.map +1 -1
- package/dist/mcp-server/tools/definitions/lookup-citation.tool.d.ts +11 -2
- package/dist/mcp-server/tools/definitions/lookup-citation.tool.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/lookup-citation.tool.js +62 -41
- package/dist/mcp-server/tools/definitions/lookup-citation.tool.js.map +1 -1
- package/dist/mcp-server/tools/definitions/lookup-mesh.tool.d.ts +3 -3
- package/dist/mcp-server/tools/definitions/lookup-mesh.tool.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/lookup-mesh.tool.js +3 -1
- package/dist/mcp-server/tools/definitions/lookup-mesh.tool.js.map +1 -1
- package/dist/mcp-server/tools/definitions/pubmed-europepmc-search.tool.d.ts +5 -2
- package/dist/mcp-server/tools/definitions/pubmed-europepmc-search.tool.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/pubmed-europepmc-search.tool.js +19 -10
- package/dist/mcp-server/tools/definitions/pubmed-europepmc-search.tool.js.map +1 -1
- package/dist/mcp-server/tools/definitions/search-articles.tool.d.ts +9 -3
- package/dist/mcp-server/tools/definitions/search-articles.tool.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/search-articles.tool.js +71 -11
- package/dist/mcp-server/tools/definitions/search-articles.tool.js.map +1 -1
- package/dist/mcp-server/tools/definitions/spell-check.tool.d.ts +5 -4
- package/dist/mcp-server/tools/definitions/spell-check.tool.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/spell-check.tool.js +4 -3
- package/dist/mcp-server/tools/definitions/spell-check.tool.js.map +1 -1
- package/dist/services/error-contracts.d.ts +7 -5
- package/dist/services/error-contracts.d.ts.map +1 -1
- package/dist/services/error-contracts.js +7 -5
- package/dist/services/error-contracts.js.map +1 -1
- package/dist/services/europe-pmc/api-client.d.ts +14 -11
- package/dist/services/europe-pmc/api-client.d.ts.map +1 -1
- package/dist/services/europe-pmc/api-client.js +21 -18
- package/dist/services/europe-pmc/api-client.js.map +1 -1
- package/dist/services/europe-pmc/europe-pmc-service.d.ts +2 -0
- package/dist/services/europe-pmc/europe-pmc-service.d.ts.map +1 -1
- package/dist/services/europe-pmc/europe-pmc-service.js +6 -2
- package/dist/services/europe-pmc/europe-pmc-service.js.map +1 -1
- package/dist/services/ncbi/parsing/pmc-article-parser.d.ts +8 -0
- package/dist/services/ncbi/parsing/pmc-article-parser.d.ts.map +1 -1
- package/dist/services/ncbi/parsing/pmc-article-parser.js +159 -16
- package/dist/services/ncbi/parsing/pmc-article-parser.js.map +1 -1
- package/dist/services/unpaywall/types.d.ts +16 -2
- package/dist/services/unpaywall/types.d.ts.map +1 -1
- package/dist/services/unpaywall/unpaywall-service.d.ts +4 -3
- package/dist/services/unpaywall/unpaywall-service.d.ts.map +1 -1
- package/dist/services/unpaywall/unpaywall-service.js +15 -4
- package/dist/services/unpaywall/unpaywall-service.js.map +1 -1
- package/package.json +1 -1
- package/server.json +3 -3
|
@@ -3,37 +3,58 @@
|
|
|
3
3
|
* three-stage chain: NCBI PMC EFetch → Europe PMC `fullTextXML` → Unpaywall.
|
|
4
4
|
* Accepts three mutually-exclusive input shapes:
|
|
5
5
|
*
|
|
6
|
-
* - `pmcids` — fetch directly by PMC ID
|
|
7
|
-
*
|
|
6
|
+
* - `pmcids` — fetch directly by PMC ID, once per record however it is
|
|
7
|
+
* spelled (`PMC123`, `pmc123`, `123`, zero-padded `PMC0123`), and reported
|
|
8
|
+
* in `PMC<digits>` form. Articles not in PMC fall through to EPMC by PMC ID,
|
|
9
|
+
* then to Unpaywall when the DOI is available.
|
|
8
10
|
* - `pmids` — resolve PMID → PMCID via PMC ID Converter, then run the chain.
|
|
11
|
+
* A zero-padded PMID runs as the PMID it spells; `unavailable[]` and
|
|
12
|
+
* `deferred.ids` report it as the caller wrote it.
|
|
9
13
|
* - `dois` — resolve DOI → PMCID via the PMC ID Converter (mirroring `pmids`),
|
|
10
14
|
* then run the chain. DOIs with no PMC counterpart fall through to EPMC
|
|
11
15
|
* search-by-DOI → fullTextXML, then Unpaywall (EPMC-only OA, preprints).
|
|
16
|
+
* DOIs are case-insensitive: every casing of one DOI runs the chain once,
|
|
17
|
+
* and `unavailable[]` reports each casing as the caller wrote it.
|
|
12
18
|
*
|
|
13
19
|
* Output uses a discriminated union on `source` (`pmc` | `unpaywall`) with an
|
|
14
20
|
* extra `viaSource` discriminator that records which layer produced the
|
|
15
21
|
* content. EPMC's JATS reuses the `pmc` schema shape because it's the same
|
|
16
|
-
* DTD; `viaSource: 'europepmc'` distinguishes it from PMC EFetch output.
|
|
22
|
+
* DTD; `viaSource: 'europepmc'` distinguishes it from PMC EFetch output. An
|
|
23
|
+
* Unpaywall article takes its title from Unpaywall's record, else the Europe
|
|
24
|
+
* PMC record the chain searched, else (HTML only) the page itself.
|
|
25
|
+
*
|
|
26
|
+
* Europe PMC and Unpaywall failures are folded into each id's `triedTiers`
|
|
27
|
+
* rather than thrown, so the declared `errors[]` covers the NCBI ID routing
|
|
28
|
+
* alone.
|
|
17
29
|
*
|
|
18
30
|
* @module src/mcp-server/tools/definitions/fetch-fulltext.tool
|
|
19
31
|
*/
|
|
20
32
|
import { tool, z } from '@cyanheads/mcp-ts-core';
|
|
21
33
|
import { htmlExtractor, pdfParser } from '@cyanheads/mcp-ts-core/utils';
|
|
22
34
|
import { getServerConfig } from '../../../config/server-config.js';
|
|
23
|
-
import {
|
|
35
|
+
import { NCBI_SERVICE_ERRORS } from '../../../services/error-contracts.js';
|
|
24
36
|
import { getEuropePmcService, } from '../../../services/europe-pmc/europe-pmc-service.js';
|
|
25
37
|
import { getNcbiService } from '../../../services/ncbi/ncbi-service.js';
|
|
26
38
|
import { extractDoi, extractPmid } from '../../../services/ncbi/parsing/article-parser.js';
|
|
27
39
|
import { parsePmcArticle } from '../../../services/ncbi/parsing/pmc-article-parser.js';
|
|
28
40
|
import { findAll, findOne } from '../../../services/ncbi/parsing/pmc-xml-helpers.js';
|
|
41
|
+
import { toDisplayText } from '../../../services/ncbi/parsing/text-helpers.js';
|
|
29
42
|
import { ensureArray } from '../../../services/ncbi/parsing/xml-helpers.js';
|
|
30
43
|
import { getUnpaywallService, } from '../../../services/unpaywall/unpaywall-service.js';
|
|
31
44
|
import { fitWholeItems } from './_budget.js';
|
|
32
45
|
import { conceptMeta, EDAM_DATA_RETRIEVAL, SCHEMA_SCHOLARLY_ARTICLE } from './_concepts.js';
|
|
33
|
-
import { doiStringSchema, pmcidStringSchema, pmidStringSchema } from './_schemas.js';
|
|
34
|
-
import { escapeMarkdownInline, escapeMarkdownTableCell,
|
|
46
|
+
import { doiStringSchema, normalizePmid, pmcidStringSchema, pmidStringSchema } from './_schemas.js';
|
|
47
|
+
import { escapeMarkdownInline, escapeMarkdownTableCell, sliceAtWordBoundary } from './_text.js';
|
|
48
|
+
/**
|
|
49
|
+
* Canonical digits of a PMC ID {@link pmcidStringSchema} accepted: the `PMC`
|
|
50
|
+
* prefix dropped in any case, and leading zeros stripped the way
|
|
51
|
+
* {@link normalizePmid} strips them. PMC EFetch reads `03531190` as PMC3531190
|
|
52
|
+
* and answers with that record, so every `pmcids` path keys on this form — and
|
|
53
|
+
* reports it back as `PMC<digits>`. An all-zero ID keeps `0`, which EFetch
|
|
54
|
+
* still answers with its empty-list envelope. (#170)
|
|
55
|
+
*/
|
|
35
56
|
function normalizePmcId(id) {
|
|
36
|
-
return id.replace(/^PMC/i, '');
|
|
57
|
+
return normalizePmid(id.replace(/^PMC/i, ''));
|
|
37
58
|
}
|
|
38
59
|
function withPmcPrefix(id) {
|
|
39
60
|
return id.startsWith('PMC') ? id : `PMC${id}`;
|
|
@@ -538,7 +559,18 @@ const UnpaywallArticleSchema = z
|
|
|
538
559
|
pubmedUrl: z.string().optional().describe('PubMed URL — present when `pmid` is set'),
|
|
539
560
|
doi: z.string().describe('DOI used to locate the open-access copy'),
|
|
540
561
|
sourceUrl: z.string().describe('URL the content was fetched from'),
|
|
541
|
-
title: z
|
|
562
|
+
title: z
|
|
563
|
+
.string()
|
|
564
|
+
.optional()
|
|
565
|
+
.describe("Article title, from the first source that carries one: Unpaywall's record for the DOI, then the Europe PMC record when the chain searched Europe PMC for this id, then — for `html-markdown` content only — the title detected on the page. Absent when none of them has a title."),
|
|
566
|
+
journalName: z
|
|
567
|
+
.string()
|
|
568
|
+
.optional()
|
|
569
|
+
.describe("Journal or repository name from Unpaywall's record for the DOI (e.g. `medRxiv` for a medRxiv preprint). Absent when Unpaywall has none."),
|
|
570
|
+
year: z
|
|
571
|
+
.number()
|
|
572
|
+
.optional()
|
|
573
|
+
.describe("Publication year from Unpaywall's record for the DOI. Absent when Unpaywall has none."),
|
|
542
574
|
content: z.string().describe('Full article text — Markdown or plain text per `contentFormat`'),
|
|
543
575
|
wordCount: z
|
|
544
576
|
.number()
|
|
@@ -617,18 +649,54 @@ const UnavailableSchema = z
|
|
|
617
649
|
})
|
|
618
650
|
.describe('One identifier that could not be returned, with the full chain it traversed');
|
|
619
651
|
// ─── Character-budget schemas ────────────────────────────────────────────────
|
|
652
|
+
/**
|
|
653
|
+
* Character accounting for one subsection. Its own type rather than a
|
|
654
|
+
* self-reference for the reason {@link SubsectionSchema} is inlined: a
|
|
655
|
+
* `z.lazy()` schema emits `$defs`/`$ref`. The article schema carries sections
|
|
656
|
+
* two levels deep, so the ledger does too.
|
|
657
|
+
*
|
|
658
|
+
* The chain `truncation` → `articles[]` → `sections[]` → `subsections[]` → a
|
|
659
|
+
* leaf is eight schema hops, the most the `format-parity` sentinel walker
|
|
660
|
+
* reaches. Re-run `bun run lint:mcp` after any change to this shape. (#143)
|
|
661
|
+
*/
|
|
662
|
+
const TruncatedSubsectionSchema = z
|
|
663
|
+
.object({
|
|
664
|
+
title: z.string().optional().describe('Subsection heading, when the subsection carries one'),
|
|
665
|
+
label: z
|
|
666
|
+
.string()
|
|
667
|
+
.optional()
|
|
668
|
+
.describe('Subsection label as printed (e.g. `2.1`), when the subsection carries one'),
|
|
669
|
+
originalCharacters: z
|
|
670
|
+
.number()
|
|
671
|
+
.describe('Body characters this subsection carried before the budget pass'),
|
|
672
|
+
returnedCharacters: z
|
|
673
|
+
.number()
|
|
674
|
+
.describe('Body characters this subsection carries in the response. Zero means it was dropped in `truncate` mode and counted in `omittedSections`, or kept as a heading-only entry in `outline` mode, marked as such in the rendered text.'),
|
|
675
|
+
truncated: z
|
|
676
|
+
.boolean()
|
|
677
|
+
.describe('True when the subsection returned fewer characters than it originally carried'),
|
|
678
|
+
})
|
|
679
|
+
.describe('Character accounting for one subsection of a shortened section');
|
|
620
680
|
const TruncatedSectionSchema = z
|
|
621
681
|
.object({
|
|
622
682
|
title: z.string().optional().describe('Section heading, when the section carries one'),
|
|
683
|
+
label: z
|
|
684
|
+
.string()
|
|
685
|
+
.optional()
|
|
686
|
+
.describe('Section label as printed (e.g. `2`), when the section carries one'),
|
|
623
687
|
originalCharacters: z
|
|
624
688
|
.number()
|
|
625
689
|
.describe('Body characters this section carried before the budget pass'),
|
|
626
690
|
returnedCharacters: z
|
|
627
691
|
.number()
|
|
628
|
-
.describe('Body characters this section carries in the response. Zero means the section was dropped in `truncate` mode, or kept as a heading-only entry in `outline` mode.'),
|
|
692
|
+
.describe('Body characters this section carries in the response. Zero means the section was dropped in `truncate` mode, or kept as a heading-only entry in `outline` mode, marked as such in the rendered text.'),
|
|
629
693
|
truncated: z
|
|
630
694
|
.boolean()
|
|
631
695
|
.describe('True when the section returned fewer characters than it originally carried'),
|
|
696
|
+
subsections: z
|
|
697
|
+
.array(TruncatedSubsectionSchema)
|
|
698
|
+
.optional()
|
|
699
|
+
.describe('Per-subsection accounting for a shortened section, in document order, including subsections dropped for budget — where inside the section the cut landed. Absent when the section was returned whole or carries no subsections.'),
|
|
632
700
|
})
|
|
633
701
|
.describe('Character accounting for one body section of a budgeted article');
|
|
634
702
|
const TruncatedArticleSchema = z
|
|
@@ -646,7 +714,7 @@ const TruncatedArticleSchema = z
|
|
|
646
714
|
sections: z
|
|
647
715
|
.array(TruncatedSectionSchema)
|
|
648
716
|
.optional()
|
|
649
|
-
.describe('Per-section accounting for `source: pmc` articles, in document order, including sections dropped for budget. Absent for `source: unpaywall`, whose body has no section structure.'),
|
|
717
|
+
.describe('Per-section accounting for `source: pmc` articles, in document order, including sections dropped for budget; a shortened section lists its subsections. Absent for `source: unpaywall`, whose body has no section structure.'),
|
|
650
718
|
omittedTables: z
|
|
651
719
|
.number()
|
|
652
720
|
.optional()
|
|
@@ -683,7 +751,7 @@ const TruncationSchema = z
|
|
|
683
751
|
.describe('Body characters the shortened articles carry in this response'),
|
|
684
752
|
omittedSections: z
|
|
685
753
|
.number()
|
|
686
|
-
.describe('Body sections dropped entirely because an article budget was exhausted before reaching them. Always 0 in `outline` mode, which keeps every heading.'),
|
|
754
|
+
.describe('Body sections and subsections dropped entirely because an article budget was exhausted before reaching them. A dropped section counts once, together with its subsections. Always 0 in `outline` mode, which keeps every heading.'),
|
|
687
755
|
omittedTables: z
|
|
688
756
|
.number()
|
|
689
757
|
.optional()
|
|
@@ -784,30 +852,81 @@ function fitWholeNamed(items, allowance, measure, name) {
|
|
|
784
852
|
};
|
|
785
853
|
}
|
|
786
854
|
/**
|
|
787
|
-
*
|
|
788
|
-
*
|
|
789
|
-
*
|
|
855
|
+
* True when the budget emptied a node that had text — which `truncate` mode
|
|
856
|
+
* drops, at any depth, and `outline` mode keeps as a heading-only entry. A node
|
|
857
|
+
* that never carried text is kept either way: there was nothing to cut.
|
|
858
|
+
*/
|
|
859
|
+
function isBudgetEmptied(entry) {
|
|
860
|
+
return entry.returnedCharacters === 0 && entry.originalCharacters > 0;
|
|
861
|
+
}
|
|
862
|
+
/**
|
|
863
|
+
* Sections and subsections `truncate` mode dropped, read off the ledger. A
|
|
864
|
+
* dropped node counts once and takes its subtree with it, so its subsections
|
|
865
|
+
* are not counted again. Shared by the budget pass and the deferral roll-back
|
|
866
|
+
* so both count by one rule. (#81, #143)
|
|
790
867
|
*/
|
|
791
|
-
function
|
|
868
|
+
function countDroppedSections(entries, mode) {
|
|
869
|
+
if (mode !== 'truncate')
|
|
870
|
+
return 0;
|
|
871
|
+
return entries.reduce((n, entry) => n + (isBudgetEmptied(entry) ? 1 : countDroppedSections(entry.subsections ?? [], mode)), 0);
|
|
872
|
+
}
|
|
873
|
+
/**
|
|
874
|
+
* Rebuild a section subtree from `fitted` — one entry per node, in the document
|
|
875
|
+
* order {@link sectionTextFields} produced them, `cursor` walking the flat list
|
|
876
|
+
* across the whole subtree — together with its ledger entry.
|
|
877
|
+
*
|
|
878
|
+
* In `truncate` mode a node the budget emptied is dropped (`kept` is absent) at
|
|
879
|
+
* every depth, where only a top-level section used to be: a subsection left as
|
|
880
|
+
* a heading over nothing is a stub, not content. `outline` mode keeps it, and
|
|
881
|
+
* `format()` marks it. The entry lists its subsections only when this node was
|
|
882
|
+
* shortened, which is where they say something a whole section's entry does
|
|
883
|
+
* not. (#81, #143)
|
|
884
|
+
*/
|
|
885
|
+
function fitSectionTree(section, fitted, cursor, mode) {
|
|
792
886
|
const text = fitted[cursor.i++] ?? '';
|
|
793
|
-
const
|
|
794
|
-
|
|
887
|
+
const children = (section.subsections ?? []).map((sub) => fitSectionTree(sub, fitted, cursor, mode));
|
|
888
|
+
const originalCharacters = sectionCharacters(section);
|
|
889
|
+
const returnedCharacters = children.reduce((n, child) => n + child.entry.returnedCharacters, text.length);
|
|
890
|
+
const truncated = returnedCharacters < originalCharacters;
|
|
891
|
+
const entry = {
|
|
892
|
+
...(section.title !== undefined && { title: section.title }),
|
|
893
|
+
...(section.label !== undefined && { label: section.label }),
|
|
894
|
+
originalCharacters,
|
|
895
|
+
returnedCharacters,
|
|
896
|
+
truncated,
|
|
897
|
+
...(truncated && children.length > 0 && { subsections: children.map((c) => c.entry) }),
|
|
898
|
+
};
|
|
899
|
+
if (mode === 'truncate' && isBudgetEmptied(entry))
|
|
900
|
+
return { entry };
|
|
901
|
+
const { subsections: _replaced, ...rest } = section;
|
|
902
|
+
const subsections = children.flatMap((child) => (child.kept ? [child.kept] : []));
|
|
903
|
+
return {
|
|
904
|
+
entry,
|
|
905
|
+
kept: { ...rest, text, ...(subsections.length > 0 && { subsections }) },
|
|
906
|
+
};
|
|
795
907
|
}
|
|
796
908
|
/**
|
|
797
909
|
* Shorten an ordered list of text fields so their combined length fits
|
|
798
|
-
* `allowance`. Fields are filled in order, so earlier fields survive whole
|
|
799
|
-
*
|
|
800
|
-
*
|
|
801
|
-
*
|
|
802
|
-
*
|
|
803
|
-
*
|
|
804
|
-
*
|
|
910
|
+
* `allowance`. Fields are filled in order, so earlier fields survive whole — the
|
|
911
|
+
* section's own text before its subsections — and the first field that does not
|
|
912
|
+
* fit is cut at the last word boundary inside what is left. Every field after
|
|
913
|
+
* that cut is past it and returns empty, the way a top-level section past the
|
|
914
|
+
* budget does, even when the cut left a few characters unspent: handing those
|
|
915
|
+
* on would open the next subsection with a fragment of its first word. (#143)
|
|
916
|
+
*
|
|
917
|
+
* No marker is appended, so the reported `returnedCharacters` is exact;
|
|
918
|
+
* `format()` carries the human-visible note. Counts are measured off the
|
|
919
|
+
* returned text, never off the allowance, which stays a ceiling. (#93)
|
|
805
920
|
*/
|
|
806
921
|
function fitFields(fields, allowance) {
|
|
807
922
|
let remaining = Math.max(allowance, 0);
|
|
808
923
|
return fields.map((text) => {
|
|
809
|
-
|
|
810
|
-
|
|
924
|
+
if (text.length <= remaining) {
|
|
925
|
+
remaining -= text.length;
|
|
926
|
+
return text;
|
|
927
|
+
}
|
|
928
|
+
const kept = sliceAtWordBoundary(text, remaining);
|
|
929
|
+
remaining = 0;
|
|
811
930
|
return kept;
|
|
812
931
|
});
|
|
813
932
|
}
|
|
@@ -880,10 +999,10 @@ function allotSectionBudgets(sizes, budget) {
|
|
|
880
999
|
* nothing bounds either list and every entry is kept. (#111, #130)
|
|
881
1000
|
*
|
|
882
1001
|
* Returns the article untouched (same object identity) when no budget was
|
|
883
|
-
* requested or nothing exceeded it. A section left with zero
|
|
884
|
-
* dropped in `truncate` mode and counted as omitted; `outline`
|
|
885
|
-
* heading-only entry. Dropped
|
|
886
|
-
* caller can see which headings exist. (#81)
|
|
1002
|
+
* requested or nothing exceeded it. A section or subsection left with zero
|
|
1003
|
+
* characters is dropped in `truncate` mode and counted as omitted; `outline`
|
|
1004
|
+
* keeps it as a heading-only entry. Dropped nodes still appear in the accounting
|
|
1005
|
+
* so the caller can see which headings exist. (#81, #143)
|
|
887
1006
|
*/
|
|
888
1007
|
function applyPmcBudget(article, budget) {
|
|
889
1008
|
const tables = article.tables ?? [];
|
|
@@ -899,25 +1018,16 @@ function applyPmcBudget(article, budget) {
|
|
|
899
1018
|
const allowances = allotSectionBudgets(sizes, budget);
|
|
900
1019
|
const kept = [];
|
|
901
1020
|
const sectionReports = [];
|
|
902
|
-
let omittedSections = 0;
|
|
903
1021
|
let returnedCharacters = 0;
|
|
904
1022
|
article.sections.forEach((section, i) => {
|
|
905
|
-
const original = sizes[i] ?? 0;
|
|
906
1023
|
const fitted = fitFields(sectionTextFields(section), allowances[i] ?? 0);
|
|
907
|
-
const
|
|
908
|
-
returnedCharacters +=
|
|
909
|
-
sectionReports.push(
|
|
910
|
-
|
|
911
|
-
|
|
912
|
-
returnedCharacters: returned,
|
|
913
|
-
truncated: returned < original,
|
|
914
|
-
});
|
|
915
|
-
if (returned === 0 && original > 0 && budget.overflowMode === 'truncate') {
|
|
916
|
-
omittedSections += 1;
|
|
917
|
-
return;
|
|
918
|
-
}
|
|
919
|
-
kept.push(withFittedTexts(section, fitted, { i: 0 }));
|
|
1024
|
+
const fit = fitSectionTree(section, fitted, { i: 0 }, budget.overflowMode);
|
|
1025
|
+
returnedCharacters += fit.entry.returnedCharacters;
|
|
1026
|
+
sectionReports.push(fit.entry);
|
|
1027
|
+
if (fit.kept)
|
|
1028
|
+
kept.push(fit.kept);
|
|
920
1029
|
});
|
|
1030
|
+
const omittedSections = countDroppedSections(sectionReports, budget.overflowMode);
|
|
921
1031
|
// Sections are served first, then tables, then assets — each spending whatever
|
|
922
1032
|
// `maxCharacters` has left. A bare per-section budget sets no total, so nothing
|
|
923
1033
|
// bounds either list.
|
|
@@ -958,13 +1068,13 @@ function applyPmcBudget(article, budget) {
|
|
|
958
1068
|
* Apply the character budget to an Unpaywall body. That body is one
|
|
959
1069
|
* unstructured blob — HTML-as-Markdown or PDF-as-text — so only `maxCharacters`
|
|
960
1070
|
* applies, and `outline` mode has no headings to preserve and behaves like
|
|
961
|
-
* `truncate`. (#81)
|
|
1071
|
+
* `truncate`. The cut ends at a word boundary, as a section cut does. (#81, #143)
|
|
962
1072
|
*/
|
|
963
1073
|
function applyContentBudget(content, budget) {
|
|
964
1074
|
const cap = budget.maxCharacters;
|
|
965
1075
|
if (cap === undefined || content.length <= cap)
|
|
966
1076
|
return { content };
|
|
967
|
-
const kept =
|
|
1077
|
+
const kept = sliceAtWordBoundary(content, cap);
|
|
968
1078
|
return {
|
|
969
1079
|
content: kept,
|
|
970
1080
|
truncation: { originalCharacters: content.length, returnedCharacters: kept.length },
|
|
@@ -1073,17 +1183,6 @@ function buildDeferralNotice(deferred) {
|
|
|
1073
1183
|
: `Response character budget reached: ${deferred.returnedCharacters} of ${deferred.maxResponseCharacters} characters returned.`;
|
|
1074
1184
|
return `${spent} ${deferred.deferredCount} resolved article(s) were deferred whole: ${deferred.ids.join(', ')}. Re-call pubmed_fetch_fulltext with those ids under \`${deferred.idType}s\` to retrieve them, or raise maxResponseCharacters to at least ${deferred.nextDeferredCharacters} — the size of the next deferred article.`;
|
|
1075
1185
|
}
|
|
1076
|
-
/**
|
|
1077
|
-
* Body sections an article's per-article budget dropped, derived from that
|
|
1078
|
-
* article's own accounting by the rule {@link applyPmcBudget} counts by:
|
|
1079
|
-
* `outline` mode keeps every heading, so it drops none. Used to take a deferred
|
|
1080
|
-
* article's contribution back out of the response-level roll-up. (#100)
|
|
1081
|
-
*/
|
|
1082
|
-
function countOmittedSections(entry, mode) {
|
|
1083
|
-
if (mode !== 'truncate')
|
|
1084
|
-
return 0;
|
|
1085
|
-
return (entry.sections ?? []).filter((s) => s.originalCharacters > 0 && s.returnedCharacters === 0).length;
|
|
1086
|
-
}
|
|
1087
1186
|
// ─── Tool Definition ─────────────────────────────────────────────────────────
|
|
1088
1187
|
/**
|
|
1089
1188
|
* Compose the tool description for the fallback tiers enabled in this
|
|
@@ -1130,11 +1229,11 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1130
1229
|
annotations: { readOnlyHint: true, openWorldHint: true },
|
|
1131
1230
|
_meta: conceptMeta([SCHEMA_SCHOLARLY_ARTICLE, EDAM_DATA_RETRIEVAL]),
|
|
1132
1231
|
sourceUrl: 'https://github.com/cyanheads/pubmed-mcp-server/blob/main/src/mcp-server/tools/definitions/fetch-fulltext.tool.ts',
|
|
1133
|
-
|
|
1134
|
-
|
|
1135
|
-
|
|
1136
|
-
|
|
1137
|
-
],
|
|
1232
|
+
// Only the ID routing's `idConvert` calls run unwrapped. Every Europe PMC and
|
|
1233
|
+
// Unpaywall call sits behind a catch that folds the failure into
|
|
1234
|
+
// `unavailable[].triedTiers`, so their service reasons never reach a caller
|
|
1235
|
+
// and are not declared here. (#168)
|
|
1236
|
+
errors: [...NCBI_SERVICE_ERRORS],
|
|
1138
1237
|
input: z
|
|
1139
1238
|
.object({
|
|
1140
1239
|
pmcids: z
|
|
@@ -1184,7 +1283,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1184
1283
|
.min(1)
|
|
1185
1284
|
.max(1_000_000)
|
|
1186
1285
|
.optional()
|
|
1187
|
-
.describe('Per-article budget for body text, in characters. Counts `source=pmc` section and subsection text — which carries the inline blocks the parser renders in place, such as lists, definition lists, block quotes, boxed text, preformatted blocks and displayed formulae — plus table label, caption, cell and footnote text and asset label, caption and `href` text; or the `source=unpaywall` `content` body. Titles, abstracts, identifiers, and references are never counted or shortened. The counted unit is that text alone — the Markdown grid `content[]` renders around the cells (pipes, padding, the divider row, headings) is scaffolding this budget does not measure, so a table renders longer than it costs here. Sections are served first, then tables, then assets, each spending what is left, in document order — admission stops at the first entry that does not fit, and every entry from there on is dropped whole rather than cut mid-row or returned with a shortened caption, counted in `truncation.omittedTables` / `truncation.omittedAssets` and named in `truncation.articles[].omittedTableNames` / `omittedAssetNames`. Applied after `sections`, `maxSections`, `includeReferences`, `includeTables`, and `includeAssets`, so semantic filtering is unaffected. This knob alone bounds only bodies: the response-wide ceiling it implies is this value times the number of articles returned, plus every uncounted field. Use `maxResponseCharacters` for a true whole-response ceiling. Omit for the full body.'),
|
|
1286
|
+
.describe('Per-article budget for body text, in characters. Counts `source=pmc` section and subsection text — which carries the inline blocks the parser renders in place, such as lists, definition lists, block quotes, boxed text, preformatted blocks and displayed formulae — plus table label, caption, cell and footnote text and asset label, caption and `href` text; or the `source=unpaywall` `content` body. Titles, abstracts, identifiers, and references are never counted or shortened. Shortened text ends at the last word boundary inside its allowance, so it can come back a few characters under it. The counted unit is that text alone — the Markdown grid `content[]` renders around the cells (pipes, padding, the divider row, headings) is scaffolding this budget does not measure, so a table renders longer than it costs here. Sections are served first, then tables, then assets, each spending what is left, in document order — admission stops at the first entry that does not fit, and every entry from there on is dropped whole rather than cut mid-row or returned with a shortened caption, counted in `truncation.omittedTables` / `truncation.omittedAssets` and named in `truncation.articles[].omittedTableNames` / `omittedAssetNames`. Applied after `sections`, `maxSections`, `includeReferences`, `includeTables`, and `includeAssets`, so semantic filtering is unaffected. This knob alone bounds only bodies: the response-wide ceiling it implies is this value times the number of articles returned, plus every uncounted field. Use `maxResponseCharacters` for a true whole-response ceiling. Omit for the full body.'),
|
|
1188
1287
|
maxCharactersPerSection: z
|
|
1189
1288
|
.number()
|
|
1190
1289
|
.int()
|
|
@@ -1202,7 +1301,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1202
1301
|
overflowMode: z
|
|
1203
1302
|
.enum(['truncate', 'outline'])
|
|
1204
1303
|
.default('truncate')
|
|
1205
|
-
.describe('How to spend `maxCharacters` across an article that exceeds it. truncate: fill sections in document order, so early sections stay whole and
|
|
1304
|
+
.describe('How to spend `maxCharacters` across an article that exceeds it. truncate: fill sections in document order, so early sections stay whole, the section the budget runs out in is cut, and every section or subsection past that point is dropped (counted in `truncation.omittedSections`). outline: split the budget evenly so every section and subsection keeps its heading, and an excerpt as far as the budget reaches — a heading the budget left empty is marked as such in the rendered text. Use it to survey what an article contains before requesting specific `sections`. Ignored when no budget is set, and identical for `source=unpaywall` bodies, which have no headings to preserve.'),
|
|
1206
1305
|
})
|
|
1207
1306
|
.refine((v) => [v.pmcids, v.pmids, v.dois].filter((b) => b !== undefined).length === 1, {
|
|
1208
1307
|
message: 'Provide exactly one of `pmcids`, `pmids`, or `dois` (not zero, not more).',
|
|
@@ -1286,6 +1385,19 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1286
1385
|
// caller can re-submit rather than whatever id the article happens to
|
|
1287
1386
|
// carry — a `pmids` request recovers articles keyed by PMCID. (#100)
|
|
1288
1387
|
const inputIdByArticle = new Map();
|
|
1388
|
+
// The caller's own spellings of each chain key, so `unavailable[]` and
|
|
1389
|
+
// `deferred.ids` report what was submitted. The `pmids` branch fills it — a
|
|
1390
|
+
// PMID's chain runs on its canonical form — and so does the `dois` branch,
|
|
1391
|
+
// whose chain runs once per DOI however many casings name it. Both also
|
|
1392
|
+
// fold in an input id that resolves to a PMC record another one already
|
|
1393
|
+
// claimed (see `routeToPmc`). Any other key is its own spelling. (#161, #166)
|
|
1394
|
+
const callerIds = new Map();
|
|
1395
|
+
const addCallerId = (key, spelling) => {
|
|
1396
|
+
const spellings = callerIds.get(key) ?? [];
|
|
1397
|
+
if (!spellings.includes(spelling))
|
|
1398
|
+
spellings.push(spelling);
|
|
1399
|
+
callerIds.set(key, spellings);
|
|
1400
|
+
};
|
|
1289
1401
|
const budget = {
|
|
1290
1402
|
overflowMode: input.overflowMode,
|
|
1291
1403
|
...(input.maxCharacters !== undefined && { maxCharacters: input.maxCharacters }),
|
|
@@ -1299,10 +1411,38 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1299
1411
|
let pmidFallbackCandidates = [];
|
|
1300
1412
|
let pmcidFallbackCandidates = [];
|
|
1301
1413
|
let doiCandidates = [];
|
|
1414
|
+
// Send a converter-resolved PMCID to PMC EFetch under the input id that
|
|
1415
|
+
// named it. A second input id the converter places on the same PMC record
|
|
1416
|
+
// — two DOIs for one article — joins the first one's chain as another
|
|
1417
|
+
// spelling of it: the record is fetched once, and both ids are recovered,
|
|
1418
|
+
// or reported unavailable with that chain, rather than the later id taking
|
|
1419
|
+
// the PMCID over and leaving the earlier one with an empty chain.
|
|
1420
|
+
const routeToPmc = (inputId, pmcid) => {
|
|
1421
|
+
const normalized = normalizePmcId(pmcid);
|
|
1422
|
+
const prefixed = withPmcPrefix(normalized);
|
|
1423
|
+
const owner = pmcidToInputId.get(prefixed);
|
|
1424
|
+
if (owner === undefined) {
|
|
1425
|
+
pmcIds.push(normalized);
|
|
1426
|
+
pmcidToInputId.set(prefixed, inputId);
|
|
1427
|
+
return;
|
|
1428
|
+
}
|
|
1429
|
+
if (owner === inputId)
|
|
1430
|
+
return;
|
|
1431
|
+
for (const spelling of callerIds.get(inputId) ?? [inputId])
|
|
1432
|
+
addCallerId(owner, spelling);
|
|
1433
|
+
callerIds.delete(inputId);
|
|
1434
|
+
chainByInput.delete(inputId);
|
|
1435
|
+
};
|
|
1302
1436
|
if (input.pmids) {
|
|
1437
|
+
// The ID Converter parses `00000001` as PMID 1 yet reports it "not found
|
|
1438
|
+
// in PMC", and every later stage answers with NCBI's own PMID, so the chain
|
|
1439
|
+
// runs once per distinct PMID in its canonical form. (#161)
|
|
1303
1440
|
for (const id of input.pmids)
|
|
1441
|
+
addCallerId(normalizePmid(id), id);
|
|
1442
|
+
const pmids = [...callerIds.keys()];
|
|
1443
|
+
for (const id of pmids)
|
|
1304
1444
|
chainByInput.set(id, []);
|
|
1305
|
-
const records = await getNcbiService().idConvert(
|
|
1445
|
+
const records = await getNcbiService().idConvert(pmids, 'pmid', ctx.signal ? { signal: ctx.signal } : undefined);
|
|
1306
1446
|
const seen = new Set();
|
|
1307
1447
|
for (const r of records) {
|
|
1308
1448
|
if (r.pmid === undefined)
|
|
@@ -1310,9 +1450,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1310
1450
|
const pmid = String(r.pmid);
|
|
1311
1451
|
seen.add(pmid);
|
|
1312
1452
|
if (r.pmcid) {
|
|
1313
|
-
|
|
1314
|
-
pmcIds.push(normalized);
|
|
1315
|
-
pmcidToInputId.set(withPmcPrefix(normalized), pmid);
|
|
1453
|
+
routeToPmc(pmid, String(r.pmcid));
|
|
1316
1454
|
pmidContext.set(pmid, { pmid, ...(r.doi && { doi: r.doi }) });
|
|
1317
1455
|
}
|
|
1318
1456
|
else {
|
|
@@ -1324,7 +1462,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1324
1462
|
pmidFallbackCandidates.push({ pmid, ...(r.doi && { doi: r.doi }) });
|
|
1325
1463
|
}
|
|
1326
1464
|
}
|
|
1327
|
-
for (const requested of
|
|
1465
|
+
for (const requested of pmids) {
|
|
1328
1466
|
if (!seen.has(requested)) {
|
|
1329
1467
|
chainByInput.get(requested)?.push({
|
|
1330
1468
|
tier: 'pmc',
|
|
@@ -1336,31 +1474,42 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1336
1474
|
}
|
|
1337
1475
|
}
|
|
1338
1476
|
else if (input.pmcids) {
|
|
1339
|
-
|
|
1340
|
-
|
|
1341
|
-
pmcIds = input.pmcids.map(normalizePmcId);
|
|
1477
|
+
// `PMC123`, `pmc123`, `123` and `PMC0123` name one record; it runs the
|
|
1478
|
+
// chain once. (#170)
|
|
1479
|
+
pmcIds = [...new Set(input.pmcids.map(normalizePmcId))];
|
|
1480
|
+
for (const id of pmcIds)
|
|
1481
|
+
chainByInput.set(withPmcPrefix(id), []);
|
|
1342
1482
|
}
|
|
1343
1483
|
else if (input.dois) {
|
|
1344
1484
|
// Mirror the `pmids` branch: resolve DOI → PMCID via the PMC ID Converter
|
|
1345
1485
|
// so PMC-indexed DOIs reach PMC EFetch instead of going straight to the
|
|
1346
1486
|
// EPMC/Unpaywall fallback (which misses articles whose only OA copy is the
|
|
1347
1487
|
// PMC JATS). DOIs the converter can't place in PMC seed `doiCandidates`.
|
|
1348
|
-
|
|
1349
|
-
|
|
1488
|
+
//
|
|
1489
|
+
// DOIs are case-insensitive, and the converter echoes one casing in
|
|
1490
|
+
// `requested-id` for DOIs that differ only in case, so every casing of a
|
|
1491
|
+
// DOI shares one chain, keyed by the first spelling submitted, and
|
|
1492
|
+
// converter records are matched to it case-insensitively. (#166)
|
|
1493
|
+
const chainKeyByDoi = new Map();
|
|
1494
|
+
for (const doi of input.dois) {
|
|
1495
|
+
const key = chainKeyByDoi.get(doi.toLowerCase()) ?? doi;
|
|
1496
|
+
chainKeyByDoi.set(doi.toLowerCase(), key);
|
|
1497
|
+
addCallerId(key, doi);
|
|
1498
|
+
}
|
|
1499
|
+
const dois = [...callerIds.keys()];
|
|
1500
|
+
for (const doi of dois)
|
|
1350
1501
|
chainByInput.set(doi, []);
|
|
1351
|
-
const records = await getNcbiService().idConvert(
|
|
1502
|
+
const records = await getNcbiService().idConvert(dois, 'doi', ctx.signal ? { signal: ctx.signal } : undefined);
|
|
1352
1503
|
const seen = new Set();
|
|
1353
1504
|
for (const r of records) {
|
|
1354
|
-
//
|
|
1355
|
-
//
|
|
1356
|
-
const doi = String(r['requested-id']);
|
|
1357
|
-
if (
|
|
1505
|
+
// Match on the echoed `requested-id`, not `r.doi`: the record's own DOI
|
|
1506
|
+
// can be cased differently from anything the caller sent.
|
|
1507
|
+
const doi = chainKeyByDoi.get(String(r['requested-id']).toLowerCase());
|
|
1508
|
+
if (doi === undefined)
|
|
1358
1509
|
continue;
|
|
1359
1510
|
seen.add(doi);
|
|
1360
1511
|
if (r.pmcid) {
|
|
1361
|
-
|
|
1362
|
-
pmcIds.push(normalized);
|
|
1363
|
-
pmcidToInputId.set(withPmcPrefix(normalized), doi);
|
|
1512
|
+
routeToPmc(doi, String(r.pmcid));
|
|
1364
1513
|
}
|
|
1365
1514
|
else {
|
|
1366
1515
|
chainByInput.get(doi)?.push({
|
|
@@ -1371,7 +1520,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1371
1520
|
doiCandidates.push({ doi });
|
|
1372
1521
|
}
|
|
1373
1522
|
}
|
|
1374
|
-
for (const requested of
|
|
1523
|
+
for (const requested of dois) {
|
|
1375
1524
|
if (!seen.has(requested)) {
|
|
1376
1525
|
chainByInput.get(requested)?.push({
|
|
1377
1526
|
tier: 'pmc',
|
|
@@ -1644,7 +1793,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1644
1793
|
return {
|
|
1645
1794
|
pmcId,
|
|
1646
1795
|
result: candidate.doi
|
|
1647
|
-
? await resolveUnpaywall({ pmcId, doi: candidate.doi, budget }, unpaywall, ctx)
|
|
1796
|
+
? await resolveUnpaywall({ pmcId, doi: candidate.doi, epmcTitle: candidate.title, budget }, unpaywall, ctx)
|
|
1648
1797
|
: undefined,
|
|
1649
1798
|
};
|
|
1650
1799
|
}));
|
|
@@ -1716,7 +1865,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1716
1865
|
const outcomes = await Promise.all(pmidFallbackCandidates.map(async (candidate) => ({
|
|
1717
1866
|
candidate,
|
|
1718
1867
|
result: candidate.doi
|
|
1719
|
-
? await resolveUnpaywall({ pmid: candidate.pmid, doi: candidate.doi, budget }, unpaywall, ctx)
|
|
1868
|
+
? await resolveUnpaywall({ pmid: candidate.pmid, doi: candidate.doi, epmcTitle: candidate.title, budget }, unpaywall, ctx)
|
|
1720
1869
|
: undefined,
|
|
1721
1870
|
})));
|
|
1722
1871
|
for (const { candidate, result } of outcomes) {
|
|
@@ -1759,7 +1908,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1759
1908
|
// doesn't reject under normal operation.
|
|
1760
1909
|
const outcomes = await Promise.all(doiCandidates.map(async (c) => ({
|
|
1761
1910
|
doi: c.doi,
|
|
1762
|
-
result: await resolveUnpaywall({ doi: c.doi, budget }, unpaywall, ctx),
|
|
1911
|
+
result: await resolveUnpaywall({ doi: c.doi, epmcTitle: c.title, budget }, unpaywall, ctx),
|
|
1763
1912
|
})));
|
|
1764
1913
|
for (const { doi, result } of outcomes) {
|
|
1765
1914
|
if ('article' in result) {
|
|
@@ -1786,13 +1935,15 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1786
1935
|
if (recoveredIds.has(id))
|
|
1787
1936
|
continue;
|
|
1788
1937
|
const unqueried = unqueriedByInput.get(id);
|
|
1789
|
-
|
|
1790
|
-
|
|
1791
|
-
|
|
1792
|
-
|
|
1793
|
-
|
|
1794
|
-
|
|
1795
|
-
|
|
1938
|
+
for (const callerId of callerIds.get(id) ?? [id]) {
|
|
1939
|
+
unavailable.push({
|
|
1940
|
+
id: callerId,
|
|
1941
|
+
idType,
|
|
1942
|
+
reason: reasonFromChain(chain),
|
|
1943
|
+
triedTiers: chain,
|
|
1944
|
+
...(unqueried?.size && { unqueriedTiers: [...unqueried] }),
|
|
1945
|
+
});
|
|
1946
|
+
}
|
|
1796
1947
|
}
|
|
1797
1948
|
// Whole-response budget: fill with complete records in response order and
|
|
1798
1949
|
// hand the rest back as identifiers the caller can re-submit. One ledger for
|
|
@@ -1812,7 +1963,10 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1812
1963
|
idType,
|
|
1813
1964
|
// Every recovery site records the input id; `articleDisplayId` is
|
|
1814
1965
|
// the total-function fallback, not an expected path.
|
|
1815
|
-
ids: fit.deferred.map((a) =>
|
|
1966
|
+
ids: fit.deferred.map((a) => {
|
|
1967
|
+
const id = inputIdByArticle.get(a) ?? articleDisplayId(a);
|
|
1968
|
+
return callerIds.get(id)?.[0] ?? id;
|
|
1969
|
+
}),
|
|
1816
1970
|
nextDeferredCharacters,
|
|
1817
1971
|
}
|
|
1818
1972
|
: undefined;
|
|
@@ -1825,8 +1979,9 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1825
1979
|
if (index === -1)
|
|
1826
1980
|
continue;
|
|
1827
1981
|
const [dropped] = truncatedArticles.splice(index, 1);
|
|
1828
|
-
if (dropped)
|
|
1829
|
-
omittedSections -=
|
|
1982
|
+
if (dropped) {
|
|
1983
|
+
omittedSections -= countDroppedSections(dropped.sections ?? [], input.overflowMode);
|
|
1984
|
+
}
|
|
1830
1985
|
}
|
|
1831
1986
|
}
|
|
1832
1987
|
ctx.log.info('pubmed_fetch_fulltext completed', {
|
|
@@ -1934,13 +2089,29 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
|
|
|
1934
2089
|
lines.push('');
|
|
1935
2090
|
const t = truncationById.get(articleDisplayId(a));
|
|
1936
2091
|
if (a.source === 'pmc')
|
|
1937
|
-
formatPmcArticle(a, lines, t);
|
|
2092
|
+
formatPmcArticle(a, lines, t, result.truncation?.mode);
|
|
1938
2093
|
else
|
|
1939
2094
|
formatUnpaywallArticle(a, lines, t);
|
|
1940
2095
|
}
|
|
1941
2096
|
return [{ type: 'text', text: lines.join('\n') }];
|
|
1942
2097
|
},
|
|
1943
2098
|
});
|
|
2099
|
+
/**
|
|
2100
|
+
* Merge what a Europe PMC hit carried onto a candidate bound for Unpaywall. A
|
|
2101
|
+
* DOI the candidate already holds wins — for `dois` input it is the caller's
|
|
2102
|
+
* own identifier.
|
|
2103
|
+
*/
|
|
2104
|
+
function carryEpmcHit(candidate, hit) {
|
|
2105
|
+
return {
|
|
2106
|
+
...candidate,
|
|
2107
|
+
...(hit.doi && !candidate.doi && { doi: hit.doi }),
|
|
2108
|
+
...(hit.title && { title: hit.title }),
|
|
2109
|
+
};
|
|
2110
|
+
}
|
|
2111
|
+
/** Display-ready plain text, or `undefined` when nothing is left once markup is stripped. */
|
|
2112
|
+
function nonEmptyDisplayText(raw) {
|
|
2113
|
+
return (raw && toDisplayText(raw)) || undefined;
|
|
2114
|
+
}
|
|
1944
2115
|
/**
|
|
1945
2116
|
* Run the Europe PMC step against everything that fell through PMC EFetch
|
|
1946
2117
|
* plus any direct DOI input. Each candidate goes through search-by-best-id →
|
|
@@ -1960,24 +2131,28 @@ async function runEpmcStage(epmc, args) {
|
|
|
1960
2131
|
}
|
|
1961
2132
|
if (search.kind === 'miss')
|
|
1962
2133
|
return { c, outcome: { kind: 'miss' } };
|
|
1963
|
-
const
|
|
2134
|
+
const title = nonEmptyDisplayText(search.hit.title);
|
|
2135
|
+
const hit = {
|
|
2136
|
+
...(search.hit.doi && { doi: search.hit.doi }),
|
|
2137
|
+
...(title && { title }),
|
|
2138
|
+
};
|
|
1964
2139
|
const fetched = await fetchEpmcArticle(epmc, search.hit, args, contextPmid);
|
|
1965
2140
|
if (fetched.kind === 'error') {
|
|
1966
|
-
return { c, ...
|
|
2141
|
+
return { c, ...hit, outcome: { kind: 'service-error', detail: fetched.detail } };
|
|
1967
2142
|
}
|
|
1968
2143
|
if (fetched.kind === 'no-fulltext') {
|
|
1969
2144
|
return {
|
|
1970
2145
|
c,
|
|
1971
|
-
...
|
|
2146
|
+
...hit,
|
|
1972
2147
|
outcome: { kind: 'no-fulltext', ...(fetched.detail && { detail: fetched.detail }) },
|
|
1973
2148
|
};
|
|
1974
2149
|
}
|
|
1975
2150
|
if (fetched.kind === 'no-body') {
|
|
1976
|
-
return { c, ...
|
|
2151
|
+
return { c, ...hit, outcome: { kind: 'no-body', detail: fetched.detail } };
|
|
1977
2152
|
}
|
|
1978
2153
|
return {
|
|
1979
2154
|
c,
|
|
1980
|
-
...
|
|
2155
|
+
...hit,
|
|
1981
2156
|
outcome: { kind: 'hit' },
|
|
1982
2157
|
article: fetched.article,
|
|
1983
2158
|
sectionFilterMiss: fetched.sectionFilterMiss,
|
|
@@ -2029,25 +2204,25 @@ async function runEpmcStage(epmc, args) {
|
|
|
2029
2204
|
pmidOutcomes.set(run.c.pmid, run.outcome);
|
|
2030
2205
|
if (run.article)
|
|
2031
2206
|
collectHit(run.c.pmid, { ...run, article: run.article });
|
|
2032
|
-
//
|
|
2033
|
-
// fetch outcome was
|
|
2034
|
-
//
|
|
2207
|
+
// What the EPMC hit carried is evidence the next stage needs, whatever the
|
|
2208
|
+
// fetch outcome was: its DOI spares Unpaywall a PubMed metadata round-trip
|
|
2209
|
+
// (#119), and its title backs up Unpaywall's own record (#144).
|
|
2035
2210
|
else
|
|
2036
|
-
remainingPmid.push(run.
|
|
2211
|
+
remainingPmid.push(carryEpmcHit(run.c, run));
|
|
2037
2212
|
}
|
|
2038
2213
|
for (const run of pmcidResults) {
|
|
2039
2214
|
pmcidOutcomes.set(run.c.normalized, run.outcome);
|
|
2040
2215
|
if (run.article)
|
|
2041
2216
|
collectHit(run.c.normalized, { ...run, article: run.article });
|
|
2042
2217
|
else
|
|
2043
|
-
remainingPmcid.push(run.
|
|
2218
|
+
remainingPmcid.push(carryEpmcHit(run.c.c, run));
|
|
2044
2219
|
}
|
|
2045
2220
|
for (const run of doiResults) {
|
|
2046
2221
|
doiOutcomes.set(run.c.doi, run.outcome);
|
|
2047
2222
|
if (run.article)
|
|
2048
2223
|
collectHit(run.c.doi, { ...run, article: run.article });
|
|
2049
2224
|
else
|
|
2050
|
-
remainingDoi.push(run.c);
|
|
2225
|
+
remainingDoi.push(carryEpmcHit(run.c, run));
|
|
2051
2226
|
}
|
|
2052
2227
|
return {
|
|
2053
2228
|
articles,
|
|
@@ -2197,9 +2372,15 @@ async function fetchPubmedDois(pmids, signal) {
|
|
|
2197
2372
|
* Resolve a DOI to an open-access article via Unpaywall. `pmcId` and `pmid`,
|
|
2198
2373
|
* when set, are stamped onto the resulting article so the branch that requested
|
|
2199
2374
|
* it carries its identifier through — Unpaywall itself only knows the DOI.
|
|
2375
|
+
*
|
|
2376
|
+
* The article's title is the first one found in Unpaywall's own record, then
|
|
2377
|
+
* `epmcTitle` — the Europe PMC record the chain searched for this id — then,
|
|
2378
|
+
* for HTML content only, the title the extractor detects on the page. None is
|
|
2379
|
+
* ever invented: with no source carrying one, the article has no title. The
|
|
2380
|
+
* journal name and year come from Unpaywall's record alone. (#144)
|
|
2200
2381
|
*/
|
|
2201
2382
|
async function resolveUnpaywall(args, service, ctx) {
|
|
2202
|
-
const { pmcId, pmid, doi, budget } = args;
|
|
2383
|
+
const { pmcId, pmid, doi, epmcTitle, budget } = args;
|
|
2203
2384
|
const requestedIds = { ...(pmcId && { pmcId }), ...(pmid && { pmid }) };
|
|
2204
2385
|
/** Budget the extracted body, then pair the article with its accounting. */
|
|
2205
2386
|
const budgeted = (build, content) => {
|
|
@@ -2237,6 +2418,15 @@ async function resolveUnpaywall(args, service, ctx) {
|
|
|
2237
2418
|
ctx.log.warning('Unpaywall content fetch failed', { doi, error: detail });
|
|
2238
2419
|
return { unavailable: { reason: 'fetch-failed', detail } };
|
|
2239
2420
|
}
|
|
2421
|
+
const recordTitle = nonEmptyDisplayText(resolution.title) ?? epmcTitle;
|
|
2422
|
+
const record = {
|
|
2423
|
+
...requestedIds,
|
|
2424
|
+
doi,
|
|
2425
|
+
sourceUrl: content.fetchedUrl,
|
|
2426
|
+
location: resolution.location,
|
|
2427
|
+
journalName: nonEmptyDisplayText(resolution.journalName),
|
|
2428
|
+
year: resolution.year,
|
|
2429
|
+
};
|
|
2240
2430
|
try {
|
|
2241
2431
|
if (content.kind === 'html') {
|
|
2242
2432
|
const extracted = await htmlExtractor.extract(content.body, {
|
|
@@ -2253,13 +2443,10 @@ async function resolveUnpaywall(args, service, ctx) {
|
|
|
2253
2443
|
};
|
|
2254
2444
|
}
|
|
2255
2445
|
return budgeted((text) => buildUnpaywallArticle({
|
|
2256
|
-
...
|
|
2257
|
-
doi,
|
|
2258
|
-
sourceUrl: content.fetchedUrl,
|
|
2259
|
-
location: resolution.location,
|
|
2446
|
+
...record,
|
|
2260
2447
|
contentFormat: 'html-markdown',
|
|
2261
2448
|
content: text,
|
|
2262
|
-
title: extracted.title,
|
|
2449
|
+
title: recordTitle ?? extracted.title,
|
|
2263
2450
|
wordCount: extracted.wordCount,
|
|
2264
2451
|
}), body);
|
|
2265
2452
|
}
|
|
@@ -2271,12 +2458,10 @@ async function resolveUnpaywall(args, service, ctx) {
|
|
|
2271
2458
|
};
|
|
2272
2459
|
}
|
|
2273
2460
|
return budgeted((body) => buildUnpaywallArticle({
|
|
2274
|
-
...
|
|
2275
|
-
doi,
|
|
2276
|
-
sourceUrl: content.fetchedUrl,
|
|
2277
|
-
location: resolution.location,
|
|
2461
|
+
...record,
|
|
2278
2462
|
contentFormat: 'pdf-text',
|
|
2279
2463
|
content: body,
|
|
2464
|
+
title: recordTitle,
|
|
2280
2465
|
totalPages: extracted.totalPages,
|
|
2281
2466
|
}), text);
|
|
2282
2467
|
}
|
|
@@ -2301,6 +2486,8 @@ function buildUnpaywallArticle(args) {
|
|
|
2301
2486
|
sourceUrl: args.sourceUrl,
|
|
2302
2487
|
content: args.content,
|
|
2303
2488
|
...(args.title && { title: args.title }),
|
|
2489
|
+
...(args.journalName && { journalName: args.journalName }),
|
|
2490
|
+
...(args.year !== undefined && { year: args.year }),
|
|
2304
2491
|
...(args.wordCount !== undefined && { wordCount: args.wordCount }),
|
|
2305
2492
|
...(args.totalPages !== undefined && { totalPages: args.totalPages }),
|
|
2306
2493
|
...(location.license && { license: location.license }),
|
|
@@ -2458,16 +2645,39 @@ function formatTruncation(t, lines) {
|
|
|
2458
2645
|
? ''
|
|
2459
2646
|
: `, ${a.omittedAssets} asset(s) dropped whole: ${(a.omittedAssetNames ?? []).join(', ')}`;
|
|
2460
2647
|
lines.push(`- ${a.id} (${a.source}): ${a.returnedCharacters} of ${a.originalCharacters} characters${tablesDropped}${assetsDropped}`);
|
|
2461
|
-
for (const s of a.sections ?? [])
|
|
2462
|
-
|
|
2463
|
-
}
|
|
2648
|
+
for (const s of a.sections ?? [])
|
|
2649
|
+
formatLedgerEntry(s, lines, t.mode, 1);
|
|
2464
2650
|
}
|
|
2465
2651
|
}
|
|
2652
|
+
/**
|
|
2653
|
+
* One section's ledger line, then its subsections' one level deeper. The line
|
|
2654
|
+
* names the section by {@link sectionHeading}, so it reads exactly as the
|
|
2655
|
+
* section's heading does in the body. An entry the budget emptied says what
|
|
2656
|
+
* became of it — dropped in `truncate` mode, kept as a bare heading in
|
|
2657
|
+
* `outline` mode. (#143, #148)
|
|
2658
|
+
*/
|
|
2659
|
+
function formatLedgerEntry(entry, lines, mode, depth) {
|
|
2660
|
+
const fate = isBudgetEmptied(entry)
|
|
2661
|
+
? mode === 'truncate'
|
|
2662
|
+
? ' — dropped'
|
|
2663
|
+
: ' — heading only'
|
|
2664
|
+
: '';
|
|
2665
|
+
lines.push(`${' '.repeat(depth)}- ${sectionHeading(entry)} — ${entry.returnedCharacters} of ${entry.originalCharacters} characters (truncated: ${entry.truncated})${fate}`);
|
|
2666
|
+
for (const sub of entry.subsections ?? [])
|
|
2667
|
+
formatLedgerEntry(sub, lines, mode, depth + 1);
|
|
2668
|
+
}
|
|
2466
2669
|
/** Per-article inline marker so a reader of one article's body knows it is partial. */
|
|
2467
2670
|
function truncationNote(t) {
|
|
2468
2671
|
return `\n> Body shortened to fit the requested character budget — ${t.returnedCharacters} of ${t.originalCharacters} characters returned. See \`truncation\` for per-section counts.`;
|
|
2469
2672
|
}
|
|
2470
|
-
|
|
2673
|
+
/**
|
|
2674
|
+
* The marker an `outline`-mode heading carries when the budget left it no text,
|
|
2675
|
+
* so it reads as withheld rather than as a heading over nothing. (#143)
|
|
2676
|
+
*/
|
|
2677
|
+
function emptiedSectionNote(originalCharacters) {
|
|
2678
|
+
return `> Text omitted to fit the requested character budget — 0 of ${originalCharacters} characters returned.`;
|
|
2679
|
+
}
|
|
2680
|
+
function formatPmcArticle(a, lines, truncation, mode) {
|
|
2471
2681
|
// Render-time only — `structuredContent.articles[].title` keeps the
|
|
2472
2682
|
// plain-text value the JATS parser produced. (#102)
|
|
2473
2683
|
lines.push(`### ${escapeMarkdownInline(a.title ?? articleDisplayId(a))}`);
|
|
@@ -2526,8 +2736,13 @@ function formatPmcArticle(a, lines, truncation) {
|
|
|
2526
2736
|
lines.push(truncationNote(truncation));
|
|
2527
2737
|
if (a.abstract)
|
|
2528
2738
|
lines.push(`\n#### Abstract\n${a.abstract}`);
|
|
2529
|
-
|
|
2530
|
-
|
|
2739
|
+
// `outline` mode keeps every section and subsection, so each one lines up
|
|
2740
|
+
// with its ledger entry by position. `truncate` mode drops the emptied ones —
|
|
2741
|
+
// positions no longer line up, and nothing left needs a marker.
|
|
2742
|
+
const ledger = mode === 'outline' ? truncation?.sections : undefined;
|
|
2743
|
+
a.sections.forEach((sec, i) => {
|
|
2744
|
+
formatSection(sec, lines, 4, ledger?.[i]);
|
|
2745
|
+
});
|
|
2531
2746
|
if (a.tables?.length)
|
|
2532
2747
|
formatTables(a.tables, lines);
|
|
2533
2748
|
if (a.assets?.length)
|
|
@@ -2654,6 +2869,10 @@ function formatUnpaywallArticle(a, lines, truncation) {
|
|
|
2654
2869
|
: 'Unpaywall (PDF → plain text)';
|
|
2655
2870
|
lines.push(`### ${escapeMarkdownInline(heading)}`);
|
|
2656
2871
|
lines.push(`**Source:** ${formatLabel}`);
|
|
2872
|
+
if (a.journalName)
|
|
2873
|
+
lines.push(`**Journal:** ${escapeMarkdownInline(a.journalName)}`);
|
|
2874
|
+
if (a.year !== undefined)
|
|
2875
|
+
lines.push(`**Year:** ${a.year}`);
|
|
2657
2876
|
if (a.pmcId)
|
|
2658
2877
|
lines.push(`**PMCID:** ${a.pmcId}`);
|
|
2659
2878
|
if (a.pmid)
|
|
@@ -2689,20 +2908,43 @@ function formatPmcAuthor(au) {
|
|
|
2689
2908
|
function formatHeading(label, title) {
|
|
2690
2909
|
return label ? `${label} ${title}` : title;
|
|
2691
2910
|
}
|
|
2911
|
+
/** How a section with no title and no label is named — in its body heading and its ledger line. */
|
|
2912
|
+
const UNTITLED_SECTION_LABEL = 'untitled section';
|
|
2913
|
+
/**
|
|
2914
|
+
* The name a section goes by in `content[]`: its label and title, else its
|
|
2915
|
+
* label alone, else {@link UNTITLED_SECTION_LABEL}. The body heading and the
|
|
2916
|
+
* truncation ledger line both take it from here, so the two always read the
|
|
2917
|
+
* same. Render-time escaped like every other upstream string interpolated into
|
|
2918
|
+
* a line, so a title's `*`, `_`, `` ` `` or `[` cannot restyle the heading;
|
|
2919
|
+
* `structuredContent` keeps the plain title. (#148, #169)
|
|
2920
|
+
*/
|
|
2921
|
+
function sectionHeading(section) {
|
|
2922
|
+
if (section.title)
|
|
2923
|
+
return escapeMarkdownInline(formatHeading(section.label, section.title));
|
|
2924
|
+
return section.label ? escapeMarkdownInline(section.label) : UNTITLED_SECTION_LABEL;
|
|
2925
|
+
}
|
|
2692
2926
|
/**
|
|
2693
2927
|
* Render one body section and everything nested under it, one markdown heading
|
|
2694
2928
|
* level per nesting level. Walks the full depth the output schema carries, so
|
|
2695
2929
|
* `content[]` shows every section `structuredContent` does. Headings stop
|
|
2696
2930
|
* deepening at `######`, the deepest markdown supports. (#112)
|
|
2931
|
+
*
|
|
2932
|
+
* Every section gets a heading — an untitled one included, under the name the
|
|
2933
|
+
* truncation ledger gives it — so its text never runs on from the block before
|
|
2934
|
+
* it. `ledger` is the section's `outline`-mode accounting, when there is one: a
|
|
2935
|
+
* section it shows the budget emptied carries a marker under its heading.
|
|
2936
|
+
* (#143, #148)
|
|
2697
2937
|
*/
|
|
2698
|
-
function formatSection(section, lines, depth) {
|
|
2699
|
-
|
|
2700
|
-
lines.push(`\n${'#'.repeat(Math.min(depth, 6))} ${formatHeading(section.label, section.title)}`);
|
|
2701
|
-
}
|
|
2938
|
+
function formatSection(section, lines, depth, ledger) {
|
|
2939
|
+
lines.push(`\n${'#'.repeat(Math.min(depth, 6))} ${sectionHeading(section)}`);
|
|
2702
2940
|
if (section.text)
|
|
2703
2941
|
lines.push(section.text);
|
|
2704
|
-
|
|
2705
|
-
|
|
2942
|
+
else if (ledger && isBudgetEmptied(ledger)) {
|
|
2943
|
+
lines.push(emptiedSectionNote(ledger.originalCharacters));
|
|
2944
|
+
}
|
|
2945
|
+
section.subsections?.forEach((sub, i) => {
|
|
2946
|
+
formatSection(sub, lines, depth + 1, ledger?.subsections?.[i]);
|
|
2947
|
+
});
|
|
2706
2948
|
}
|
|
2707
2949
|
/**
|
|
2708
2950
|
* Strip absolute URLs from chain detail strings. Upstream errors (e.g.
|