@cyanheads/pubmed-mcp-server 2.10.15 → 2.10.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. package/AGENTS.md +1 -1
  2. package/CLAUDE.md +1 -1
  3. package/README.md +26 -15
  4. package/changelog/2.10.x/2.10.16.md +26 -0
  5. package/changelog/2.10.x/2.10.17.md +19 -0
  6. package/dist/mcp-server/tools/definitions/_schemas.d.ts +11 -1
  7. package/dist/mcp-server/tools/definitions/_schemas.d.ts.map +1 -1
  8. package/dist/mcp-server/tools/definitions/_schemas.js +13 -1
  9. package/dist/mcp-server/tools/definitions/_schemas.js.map +1 -1
  10. package/dist/mcp-server/tools/definitions/_text.d.ts +20 -3
  11. package/dist/mcp-server/tools/definitions/_text.d.ts.map +1 -1
  12. package/dist/mcp-server/tools/definitions/_text.js +35 -3
  13. package/dist/mcp-server/tools/definitions/_text.js.map +1 -1
  14. package/dist/mcp-server/tools/definitions/convert-ids.tool.d.ts +3 -0
  15. package/dist/mcp-server/tools/definitions/convert-ids.tool.d.ts.map +1 -1
  16. package/dist/mcp-server/tools/definitions/convert-ids.tool.js +41 -9
  17. package/dist/mcp-server/tools/definitions/convert-ids.tool.js.map +1 -1
  18. package/dist/mcp-server/tools/definitions/fetch-articles.tool.d.ts +3 -1
  19. package/dist/mcp-server/tools/definitions/fetch-articles.tool.d.ts.map +1 -1
  20. package/dist/mcp-server/tools/definitions/fetch-articles.tool.js +12 -4
  21. package/dist/mcp-server/tools/definitions/fetch-articles.tool.js.map +1 -1
  22. package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.d.ts +25 -31
  23. package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.d.ts.map +1 -1
  24. package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.js +373 -131
  25. package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.js.map +1 -1
  26. package/dist/mcp-server/tools/definitions/find-related.tool.d.ts +7 -3
  27. package/dist/mcp-server/tools/definitions/find-related.tool.d.ts.map +1 -1
  28. package/dist/mcp-server/tools/definitions/find-related.tool.js +38 -21
  29. package/dist/mcp-server/tools/definitions/find-related.tool.js.map +1 -1
  30. package/dist/mcp-server/tools/definitions/format-citations.tool.d.ts +3 -1
  31. package/dist/mcp-server/tools/definitions/format-citations.tool.d.ts.map +1 -1
  32. package/dist/mcp-server/tools/definitions/format-citations.tool.js +12 -4
  33. package/dist/mcp-server/tools/definitions/format-citations.tool.js.map +1 -1
  34. package/dist/mcp-server/tools/definitions/lookup-citation.tool.d.ts +11 -2
  35. package/dist/mcp-server/tools/definitions/lookup-citation.tool.d.ts.map +1 -1
  36. package/dist/mcp-server/tools/definitions/lookup-citation.tool.js +62 -41
  37. package/dist/mcp-server/tools/definitions/lookup-citation.tool.js.map +1 -1
  38. package/dist/mcp-server/tools/definitions/lookup-mesh.tool.d.ts +3 -3
  39. package/dist/mcp-server/tools/definitions/lookup-mesh.tool.d.ts.map +1 -1
  40. package/dist/mcp-server/tools/definitions/lookup-mesh.tool.js +3 -1
  41. package/dist/mcp-server/tools/definitions/lookup-mesh.tool.js.map +1 -1
  42. package/dist/mcp-server/tools/definitions/pubmed-europepmc-search.tool.d.ts +5 -2
  43. package/dist/mcp-server/tools/definitions/pubmed-europepmc-search.tool.d.ts.map +1 -1
  44. package/dist/mcp-server/tools/definitions/pubmed-europepmc-search.tool.js +19 -10
  45. package/dist/mcp-server/tools/definitions/pubmed-europepmc-search.tool.js.map +1 -1
  46. package/dist/mcp-server/tools/definitions/search-articles.tool.d.ts +9 -3
  47. package/dist/mcp-server/tools/definitions/search-articles.tool.d.ts.map +1 -1
  48. package/dist/mcp-server/tools/definitions/search-articles.tool.js +71 -11
  49. package/dist/mcp-server/tools/definitions/search-articles.tool.js.map +1 -1
  50. package/dist/mcp-server/tools/definitions/spell-check.tool.d.ts +5 -4
  51. package/dist/mcp-server/tools/definitions/spell-check.tool.d.ts.map +1 -1
  52. package/dist/mcp-server/tools/definitions/spell-check.tool.js +4 -3
  53. package/dist/mcp-server/tools/definitions/spell-check.tool.js.map +1 -1
  54. package/dist/services/error-contracts.d.ts +7 -5
  55. package/dist/services/error-contracts.d.ts.map +1 -1
  56. package/dist/services/error-contracts.js +7 -5
  57. package/dist/services/error-contracts.js.map +1 -1
  58. package/dist/services/europe-pmc/api-client.d.ts +14 -11
  59. package/dist/services/europe-pmc/api-client.d.ts.map +1 -1
  60. package/dist/services/europe-pmc/api-client.js +21 -18
  61. package/dist/services/europe-pmc/api-client.js.map +1 -1
  62. package/dist/services/europe-pmc/europe-pmc-service.d.ts +2 -0
  63. package/dist/services/europe-pmc/europe-pmc-service.d.ts.map +1 -1
  64. package/dist/services/europe-pmc/europe-pmc-service.js +6 -2
  65. package/dist/services/europe-pmc/europe-pmc-service.js.map +1 -1
  66. package/dist/services/ncbi/parsing/pmc-article-parser.d.ts +8 -0
  67. package/dist/services/ncbi/parsing/pmc-article-parser.d.ts.map +1 -1
  68. package/dist/services/ncbi/parsing/pmc-article-parser.js +159 -16
  69. package/dist/services/ncbi/parsing/pmc-article-parser.js.map +1 -1
  70. package/dist/services/unpaywall/types.d.ts +16 -2
  71. package/dist/services/unpaywall/types.d.ts.map +1 -1
  72. package/dist/services/unpaywall/unpaywall-service.d.ts +4 -3
  73. package/dist/services/unpaywall/unpaywall-service.d.ts.map +1 -1
  74. package/dist/services/unpaywall/unpaywall-service.js +15 -4
  75. package/dist/services/unpaywall/unpaywall-service.js.map +1 -1
  76. package/package.json +1 -1
  77. package/server.json +3 -3
@@ -3,37 +3,58 @@
3
3
  * three-stage chain: NCBI PMC EFetch → Europe PMC `fullTextXML` → Unpaywall.
4
4
  * Accepts three mutually-exclusive input shapes:
5
5
  *
6
- * - `pmcids` — fetch directly by PMC ID. Articles not in PMC fall through to
7
- * EPMC by PMC ID, then to Unpaywall when the DOI is available.
6
+ * - `pmcids` — fetch directly by PMC ID, once per record however it is
7
+ * spelled (`PMC123`, `pmc123`, `123`, zero-padded `PMC0123`), and reported
8
+ * in `PMC<digits>` form. Articles not in PMC fall through to EPMC by PMC ID,
9
+ * then to Unpaywall when the DOI is available.
8
10
  * - `pmids` — resolve PMID → PMCID via PMC ID Converter, then run the chain.
11
+ * A zero-padded PMID runs as the PMID it spells; `unavailable[]` and
12
+ * `deferred.ids` report it as the caller wrote it.
9
13
  * - `dois` — resolve DOI → PMCID via the PMC ID Converter (mirroring `pmids`),
10
14
  * then run the chain. DOIs with no PMC counterpart fall through to EPMC
11
15
  * search-by-DOI → fullTextXML, then Unpaywall (EPMC-only OA, preprints).
16
+ * DOIs are case-insensitive: every casing of one DOI runs the chain once,
17
+ * and `unavailable[]` reports each casing as the caller wrote it.
12
18
  *
13
19
  * Output uses a discriminated union on `source` (`pmc` | `unpaywall`) with an
14
20
  * extra `viaSource` discriminator that records which layer produced the
15
21
  * content. EPMC's JATS reuses the `pmc` schema shape because it's the same
16
- * DTD; `viaSource: 'europepmc'` distinguishes it from PMC EFetch output.
22
+ * DTD; `viaSource: 'europepmc'` distinguishes it from PMC EFetch output. An
23
+ * Unpaywall article takes its title from Unpaywall's record, else the Europe
24
+ * PMC record the chain searched, else (HTML only) the page itself.
25
+ *
26
+ * Europe PMC and Unpaywall failures are folded into each id's `triedTiers`
27
+ * rather than thrown, so the declared `errors[]` covers the NCBI ID routing
28
+ * alone.
17
29
  *
18
30
  * @module src/mcp-server/tools/definitions/fetch-fulltext.tool
19
31
  */
20
32
  import { tool, z } from '@cyanheads/mcp-ts-core';
21
33
  import { htmlExtractor, pdfParser } from '@cyanheads/mcp-ts-core/utils';
22
34
  import { getServerConfig } from '../../../config/server-config.js';
23
- import { EUROPEPMC_SERVICE_ERRORS, NCBI_SERVICE_ERRORS, UNPAYWALL_SERVICE_ERRORS, } from '../../../services/error-contracts.js';
35
+ import { NCBI_SERVICE_ERRORS } from '../../../services/error-contracts.js';
24
36
  import { getEuropePmcService, } from '../../../services/europe-pmc/europe-pmc-service.js';
25
37
  import { getNcbiService } from '../../../services/ncbi/ncbi-service.js';
26
38
  import { extractDoi, extractPmid } from '../../../services/ncbi/parsing/article-parser.js';
27
39
  import { parsePmcArticle } from '../../../services/ncbi/parsing/pmc-article-parser.js';
28
40
  import { findAll, findOne } from '../../../services/ncbi/parsing/pmc-xml-helpers.js';
41
+ import { toDisplayText } from '../../../services/ncbi/parsing/text-helpers.js';
29
42
  import { ensureArray } from '../../../services/ncbi/parsing/xml-helpers.js';
30
43
  import { getUnpaywallService, } from '../../../services/unpaywall/unpaywall-service.js';
31
44
  import { fitWholeItems } from './_budget.js';
32
45
  import { conceptMeta, EDAM_DATA_RETRIEVAL, SCHEMA_SCHOLARLY_ARTICLE } from './_concepts.js';
33
- import { doiStringSchema, pmcidStringSchema, pmidStringSchema } from './_schemas.js';
34
- import { escapeMarkdownInline, escapeMarkdownTableCell, sliceCodeUnits } from './_text.js';
46
+ import { doiStringSchema, normalizePmid, pmcidStringSchema, pmidStringSchema } from './_schemas.js';
47
+ import { escapeMarkdownInline, escapeMarkdownTableCell, sliceAtWordBoundary } from './_text.js';
48
+ /**
49
+ * Canonical digits of a PMC ID {@link pmcidStringSchema} accepted: the `PMC`
50
+ * prefix dropped in any case, and leading zeros stripped the way
51
+ * {@link normalizePmid} strips them. PMC EFetch reads `03531190` as PMC3531190
52
+ * and answers with that record, so every `pmcids` path keys on this form — and
53
+ * reports it back as `PMC<digits>`. An all-zero ID keeps `0`, which EFetch
54
+ * still answers with its empty-list envelope. (#170)
55
+ */
35
56
  function normalizePmcId(id) {
36
- return id.replace(/^PMC/i, '');
57
+ return normalizePmid(id.replace(/^PMC/i, ''));
37
58
  }
38
59
  function withPmcPrefix(id) {
39
60
  return id.startsWith('PMC') ? id : `PMC${id}`;
@@ -538,7 +559,18 @@ const UnpaywallArticleSchema = z
538
559
  pubmedUrl: z.string().optional().describe('PubMed URL — present when `pmid` is set'),
539
560
  doi: z.string().describe('DOI used to locate the open-access copy'),
540
561
  sourceUrl: z.string().describe('URL the content was fetched from'),
541
- title: z.string().optional().describe('Detected article title when present'),
562
+ title: z
563
+ .string()
564
+ .optional()
565
+ .describe("Article title, from the first source that carries one: Unpaywall's record for the DOI, then the Europe PMC record when the chain searched Europe PMC for this id, then — for `html-markdown` content only — the title detected on the page. Absent when none of them has a title."),
566
+ journalName: z
567
+ .string()
568
+ .optional()
569
+ .describe("Journal or repository name from Unpaywall's record for the DOI (e.g. `medRxiv` for a medRxiv preprint). Absent when Unpaywall has none."),
570
+ year: z
571
+ .number()
572
+ .optional()
573
+ .describe("Publication year from Unpaywall's record for the DOI. Absent when Unpaywall has none."),
542
574
  content: z.string().describe('Full article text — Markdown or plain text per `contentFormat`'),
543
575
  wordCount: z
544
576
  .number()
@@ -617,18 +649,54 @@ const UnavailableSchema = z
617
649
  })
618
650
  .describe('One identifier that could not be returned, with the full chain it traversed');
619
651
  // ─── Character-budget schemas ────────────────────────────────────────────────
652
+ /**
653
+ * Character accounting for one subsection. Its own type rather than a
654
+ * self-reference for the reason {@link SubsectionSchema} is inlined: a
655
+ * `z.lazy()` schema emits `$defs`/`$ref`. The article schema carries sections
656
+ * two levels deep, so the ledger does too.
657
+ *
658
+ * The chain `truncation` → `articles[]` → `sections[]` → `subsections[]` → a
659
+ * leaf is eight schema hops, the most the `format-parity` sentinel walker
660
+ * reaches. Re-run `bun run lint:mcp` after any change to this shape. (#143)
661
+ */
662
+ const TruncatedSubsectionSchema = z
663
+ .object({
664
+ title: z.string().optional().describe('Subsection heading, when the subsection carries one'),
665
+ label: z
666
+ .string()
667
+ .optional()
668
+ .describe('Subsection label as printed (e.g. `2.1`), when the subsection carries one'),
669
+ originalCharacters: z
670
+ .number()
671
+ .describe('Body characters this subsection carried before the budget pass'),
672
+ returnedCharacters: z
673
+ .number()
674
+ .describe('Body characters this subsection carries in the response. Zero means it was dropped in `truncate` mode and counted in `omittedSections`, or kept as a heading-only entry in `outline` mode, marked as such in the rendered text.'),
675
+ truncated: z
676
+ .boolean()
677
+ .describe('True when the subsection returned fewer characters than it originally carried'),
678
+ })
679
+ .describe('Character accounting for one subsection of a shortened section');
620
680
  const TruncatedSectionSchema = z
621
681
  .object({
622
682
  title: z.string().optional().describe('Section heading, when the section carries one'),
683
+ label: z
684
+ .string()
685
+ .optional()
686
+ .describe('Section label as printed (e.g. `2`), when the section carries one'),
623
687
  originalCharacters: z
624
688
  .number()
625
689
  .describe('Body characters this section carried before the budget pass'),
626
690
  returnedCharacters: z
627
691
  .number()
628
- .describe('Body characters this section carries in the response. Zero means the section was dropped in `truncate` mode, or kept as a heading-only entry in `outline` mode.'),
692
+ .describe('Body characters this section carries in the response. Zero means the section was dropped in `truncate` mode, or kept as a heading-only entry in `outline` mode, marked as such in the rendered text.'),
629
693
  truncated: z
630
694
  .boolean()
631
695
  .describe('True when the section returned fewer characters than it originally carried'),
696
+ subsections: z
697
+ .array(TruncatedSubsectionSchema)
698
+ .optional()
699
+ .describe('Per-subsection accounting for a shortened section, in document order, including subsections dropped for budget — where inside the section the cut landed. Absent when the section was returned whole or carries no subsections.'),
632
700
  })
633
701
  .describe('Character accounting for one body section of a budgeted article');
634
702
  const TruncatedArticleSchema = z
@@ -646,7 +714,7 @@ const TruncatedArticleSchema = z
646
714
  sections: z
647
715
  .array(TruncatedSectionSchema)
648
716
  .optional()
649
- .describe('Per-section accounting for `source: pmc` articles, in document order, including sections dropped for budget. Absent for `source: unpaywall`, whose body has no section structure.'),
717
+ .describe('Per-section accounting for `source: pmc` articles, in document order, including sections dropped for budget; a shortened section lists its subsections. Absent for `source: unpaywall`, whose body has no section structure.'),
650
718
  omittedTables: z
651
719
  .number()
652
720
  .optional()
@@ -683,7 +751,7 @@ const TruncationSchema = z
683
751
  .describe('Body characters the shortened articles carry in this response'),
684
752
  omittedSections: z
685
753
  .number()
686
- .describe('Body sections dropped entirely because an article budget was exhausted before reaching them. Always 0 in `outline` mode, which keeps every heading.'),
754
+ .describe('Body sections and subsections dropped entirely because an article budget was exhausted before reaching them. A dropped section counts once, together with its subsections. Always 0 in `outline` mode, which keeps every heading.'),
687
755
  omittedTables: z
688
756
  .number()
689
757
  .optional()
@@ -784,30 +852,81 @@ function fitWholeNamed(items, allowance, measure, name) {
784
852
  };
785
853
  }
786
854
  /**
787
- * Rebuild a section subtree from `fitted`, consuming one entry per node in the
788
- * same document order {@link sectionTextFields} produced them. `cursor` walks
789
- * the flat list across the whole subtree.
855
+ * True when the budget emptied a node that had text — which `truncate` mode
856
+ * drops, at any depth, and `outline` mode keeps as a heading-only entry. A node
857
+ * that never carried text is kept either way: there was nothing to cut.
858
+ */
859
+ function isBudgetEmptied(entry) {
860
+ return entry.returnedCharacters === 0 && entry.originalCharacters > 0;
861
+ }
862
+ /**
863
+ * Sections and subsections `truncate` mode dropped, read off the ledger. A
864
+ * dropped node counts once and takes its subtree with it, so its subsections
865
+ * are not counted again. Shared by the budget pass and the deferral roll-back
866
+ * so both count by one rule. (#81, #143)
790
867
  */
791
- function withFittedTexts(section, fitted, cursor) {
868
+ function countDroppedSections(entries, mode) {
869
+ if (mode !== 'truncate')
870
+ return 0;
871
+ return entries.reduce((n, entry) => n + (isBudgetEmptied(entry) ? 1 : countDroppedSections(entry.subsections ?? [], mode)), 0);
872
+ }
873
+ /**
874
+ * Rebuild a section subtree from `fitted` — one entry per node, in the document
875
+ * order {@link sectionTextFields} produced them, `cursor` walking the flat list
876
+ * across the whole subtree — together with its ledger entry.
877
+ *
878
+ * In `truncate` mode a node the budget emptied is dropped (`kept` is absent) at
879
+ * every depth, where only a top-level section used to be: a subsection left as
880
+ * a heading over nothing is a stub, not content. `outline` mode keeps it, and
881
+ * `format()` marks it. The entry lists its subsections only when this node was
882
+ * shortened, which is where they say something a whole section's entry does
883
+ * not. (#81, #143)
884
+ */
885
+ function fitSectionTree(section, fitted, cursor, mode) {
792
886
  const text = fitted[cursor.i++] ?? '';
793
- const subsections = section.subsections?.map((sub) => withFittedTexts(sub, fitted, cursor));
794
- return { ...section, text, ...(subsections && { subsections }) };
887
+ const children = (section.subsections ?? []).map((sub) => fitSectionTree(sub, fitted, cursor, mode));
888
+ const originalCharacters = sectionCharacters(section);
889
+ const returnedCharacters = children.reduce((n, child) => n + child.entry.returnedCharacters, text.length);
890
+ const truncated = returnedCharacters < originalCharacters;
891
+ const entry = {
892
+ ...(section.title !== undefined && { title: section.title }),
893
+ ...(section.label !== undefined && { label: section.label }),
894
+ originalCharacters,
895
+ returnedCharacters,
896
+ truncated,
897
+ ...(truncated && children.length > 0 && { subsections: children.map((c) => c.entry) }),
898
+ };
899
+ if (mode === 'truncate' && isBudgetEmptied(entry))
900
+ return { entry };
901
+ const { subsections: _replaced, ...rest } = section;
902
+ const subsections = children.flatMap((child) => (child.kept ? [child.kept] : []));
903
+ return {
904
+ entry,
905
+ kept: { ...rest, text, ...(subsections.length > 0 && { subsections }) },
906
+ };
795
907
  }
796
908
  /**
797
909
  * Shorten an ordered list of text fields so their combined length fits
798
- * `allowance`. Fields are filled in order, so earlier fields survive whole and
799
- * later ones absorb the shortfall — the section's own text before its
800
- * subsections. Cuts at the character boundary with no appended marker so the
801
- * reported `returnedCharacters` is exact; `format()` carries the human-visible
802
- * note. A cut that would split a surrogate pair backs off a code unit, so a
803
- * field can return one character under its share — counts are measured off the
804
- * returned text, never off the allowance. (#93)
910
+ * `allowance`. Fields are filled in order, so earlier fields survive whole — the
911
+ * section's own text before its subsections — and the first field that does not
912
+ * fit is cut at the last word boundary inside what is left. Every field after
913
+ * that cut is past it and returns empty, the way a top-level section past the
914
+ * budget does, even when the cut left a few characters unspent: handing those
915
+ * on would open the next subsection with a fragment of its first word. (#143)
916
+ *
917
+ * No marker is appended, so the reported `returnedCharacters` is exact;
918
+ * `format()` carries the human-visible note. Counts are measured off the
919
+ * returned text, never off the allowance, which stays a ceiling. (#93)
805
920
  */
806
921
  function fitFields(fields, allowance) {
807
922
  let remaining = Math.max(allowance, 0);
808
923
  return fields.map((text) => {
809
- const kept = sliceCodeUnits(text, remaining);
810
- remaining -= kept.length;
924
+ if (text.length <= remaining) {
925
+ remaining -= text.length;
926
+ return text;
927
+ }
928
+ const kept = sliceAtWordBoundary(text, remaining);
929
+ remaining = 0;
811
930
  return kept;
812
931
  });
813
932
  }
@@ -880,10 +999,10 @@ function allotSectionBudgets(sizes, budget) {
880
999
  * nothing bounds either list and every entry is kept. (#111, #130)
881
1000
  *
882
1001
  * Returns the article untouched (same object identity) when no budget was
883
- * requested or nothing exceeded it. A section left with zero characters is
884
- * dropped in `truncate` mode and counted as omitted; `outline` keeps it as a
885
- * heading-only entry. Dropped sections still appear in the accounting so the
886
- * caller can see which headings exist. (#81)
1002
+ * requested or nothing exceeded it. A section or subsection left with zero
1003
+ * characters is dropped in `truncate` mode and counted as omitted; `outline`
1004
+ * keeps it as a heading-only entry. Dropped nodes still appear in the accounting
1005
+ * so the caller can see which headings exist. (#81, #143)
887
1006
  */
888
1007
  function applyPmcBudget(article, budget) {
889
1008
  const tables = article.tables ?? [];
@@ -899,25 +1018,16 @@ function applyPmcBudget(article, budget) {
899
1018
  const allowances = allotSectionBudgets(sizes, budget);
900
1019
  const kept = [];
901
1020
  const sectionReports = [];
902
- let omittedSections = 0;
903
1021
  let returnedCharacters = 0;
904
1022
  article.sections.forEach((section, i) => {
905
- const original = sizes[i] ?? 0;
906
1023
  const fitted = fitFields(sectionTextFields(section), allowances[i] ?? 0);
907
- const returned = totalLength(fitted);
908
- returnedCharacters += returned;
909
- sectionReports.push({
910
- ...(section.title !== undefined && { title: section.title }),
911
- originalCharacters: original,
912
- returnedCharacters: returned,
913
- truncated: returned < original,
914
- });
915
- if (returned === 0 && original > 0 && budget.overflowMode === 'truncate') {
916
- omittedSections += 1;
917
- return;
918
- }
919
- kept.push(withFittedTexts(section, fitted, { i: 0 }));
1024
+ const fit = fitSectionTree(section, fitted, { i: 0 }, budget.overflowMode);
1025
+ returnedCharacters += fit.entry.returnedCharacters;
1026
+ sectionReports.push(fit.entry);
1027
+ if (fit.kept)
1028
+ kept.push(fit.kept);
920
1029
  });
1030
+ const omittedSections = countDroppedSections(sectionReports, budget.overflowMode);
921
1031
  // Sections are served first, then tables, then assets — each spending whatever
922
1032
  // `maxCharacters` has left. A bare per-section budget sets no total, so nothing
923
1033
  // bounds either list.
@@ -958,13 +1068,13 @@ function applyPmcBudget(article, budget) {
958
1068
  * Apply the character budget to an Unpaywall body. That body is one
959
1069
  * unstructured blob — HTML-as-Markdown or PDF-as-text — so only `maxCharacters`
960
1070
  * applies, and `outline` mode has no headings to preserve and behaves like
961
- * `truncate`. (#81)
1071
+ * `truncate`. The cut ends at a word boundary, as a section cut does. (#81, #143)
962
1072
  */
963
1073
  function applyContentBudget(content, budget) {
964
1074
  const cap = budget.maxCharacters;
965
1075
  if (cap === undefined || content.length <= cap)
966
1076
  return { content };
967
- const kept = sliceCodeUnits(content, cap);
1077
+ const kept = sliceAtWordBoundary(content, cap);
968
1078
  return {
969
1079
  content: kept,
970
1080
  truncation: { originalCharacters: content.length, returnedCharacters: kept.length },
@@ -1073,17 +1183,6 @@ function buildDeferralNotice(deferred) {
1073
1183
  : `Response character budget reached: ${deferred.returnedCharacters} of ${deferred.maxResponseCharacters} characters returned.`;
1074
1184
  return `${spent} ${deferred.deferredCount} resolved article(s) were deferred whole: ${deferred.ids.join(', ')}. Re-call pubmed_fetch_fulltext with those ids under \`${deferred.idType}s\` to retrieve them, or raise maxResponseCharacters to at least ${deferred.nextDeferredCharacters} — the size of the next deferred article.`;
1075
1185
  }
1076
- /**
1077
- * Body sections an article's per-article budget dropped, derived from that
1078
- * article's own accounting by the rule {@link applyPmcBudget} counts by:
1079
- * `outline` mode keeps every heading, so it drops none. Used to take a deferred
1080
- * article's contribution back out of the response-level roll-up. (#100)
1081
- */
1082
- function countOmittedSections(entry, mode) {
1083
- if (mode !== 'truncate')
1084
- return 0;
1085
- return (entry.sections ?? []).filter((s) => s.originalCharacters > 0 && s.returnedCharacters === 0).length;
1086
- }
1087
1186
  // ─── Tool Definition ─────────────────────────────────────────────────────────
1088
1187
  /**
1089
1188
  * Compose the tool description for the fallback tiers enabled in this
@@ -1130,11 +1229,11 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1130
1229
  annotations: { readOnlyHint: true, openWorldHint: true },
1131
1230
  _meta: conceptMeta([SCHEMA_SCHOLARLY_ARTICLE, EDAM_DATA_RETRIEVAL]),
1132
1231
  sourceUrl: 'https://github.com/cyanheads/pubmed-mcp-server/blob/main/src/mcp-server/tools/definitions/fetch-fulltext.tool.ts',
1133
- errors: [
1134
- ...NCBI_SERVICE_ERRORS,
1135
- ...UNPAYWALL_SERVICE_ERRORS,
1136
- ...EUROPEPMC_SERVICE_ERRORS,
1137
- ],
1232
+ // Only the ID routing's `idConvert` calls run unwrapped. Every Europe PMC and
1233
+ // Unpaywall call sits behind a catch that folds the failure into
1234
+ // `unavailable[].triedTiers`, so their service reasons never reach a caller
1235
+ // and are not declared here. (#168)
1236
+ errors: [...NCBI_SERVICE_ERRORS],
1138
1237
  input: z
1139
1238
  .object({
1140
1239
  pmcids: z
@@ -1184,7 +1283,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1184
1283
  .min(1)
1185
1284
  .max(1_000_000)
1186
1285
  .optional()
1187
- .describe('Per-article budget for body text, in characters. Counts `source=pmc` section and subsection text — which carries the inline blocks the parser renders in place, such as lists, definition lists, block quotes, boxed text, preformatted blocks and displayed formulae — plus table label, caption, cell and footnote text and asset label, caption and `href` text; or the `source=unpaywall` `content` body. Titles, abstracts, identifiers, and references are never counted or shortened. The counted unit is that text alone — the Markdown grid `content[]` renders around the cells (pipes, padding, the divider row, headings) is scaffolding this budget does not measure, so a table renders longer than it costs here. Sections are served first, then tables, then assets, each spending what is left, in document order — admission stops at the first entry that does not fit, and every entry from there on is dropped whole rather than cut mid-row or returned with a shortened caption, counted in `truncation.omittedTables` / `truncation.omittedAssets` and named in `truncation.articles[].omittedTableNames` / `omittedAssetNames`. Applied after `sections`, `maxSections`, `includeReferences`, `includeTables`, and `includeAssets`, so semantic filtering is unaffected. This knob alone bounds only bodies: the response-wide ceiling it implies is this value times the number of articles returned, plus every uncounted field. Use `maxResponseCharacters` for a true whole-response ceiling. Omit for the full body.'),
1286
+ .describe('Per-article budget for body text, in characters. Counts `source=pmc` section and subsection text — which carries the inline blocks the parser renders in place, such as lists, definition lists, block quotes, boxed text, preformatted blocks and displayed formulae — plus table label, caption, cell and footnote text and asset label, caption and `href` text; or the `source=unpaywall` `content` body. Titles, abstracts, identifiers, and references are never counted or shortened. Shortened text ends at the last word boundary inside its allowance, so it can come back a few characters under it. The counted unit is that text alone — the Markdown grid `content[]` renders around the cells (pipes, padding, the divider row, headings) is scaffolding this budget does not measure, so a table renders longer than it costs here. Sections are served first, then tables, then assets, each spending what is left, in document order — admission stops at the first entry that does not fit, and every entry from there on is dropped whole rather than cut mid-row or returned with a shortened caption, counted in `truncation.omittedTables` / `truncation.omittedAssets` and named in `truncation.articles[].omittedTableNames` / `omittedAssetNames`. Applied after `sections`, `maxSections`, `includeReferences`, `includeTables`, and `includeAssets`, so semantic filtering is unaffected. This knob alone bounds only bodies: the response-wide ceiling it implies is this value times the number of articles returned, plus every uncounted field. Use `maxResponseCharacters` for a true whole-response ceiling. Omit for the full body.'),
1188
1287
  maxCharactersPerSection: z
1189
1288
  .number()
1190
1289
  .int()
@@ -1202,7 +1301,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1202
1301
  overflowMode: z
1203
1302
  .enum(['truncate', 'outline'])
1204
1303
  .default('truncate')
1205
- .describe('How to spend `maxCharacters` across an article that exceeds it. truncate: fill sections in document order, so early sections stay whole and sections past the budget are dropped (counted in `truncation.omittedSections`). outline: split the budget evenly so every section keeps its heading, and an excerpt as far as the budget reaches — use it to survey what an article contains before requesting specific `sections`. Ignored when no budget is set, and identical for `source=unpaywall` bodies, which have no headings to preserve.'),
1304
+ .describe('How to spend `maxCharacters` across an article that exceeds it. truncate: fill sections in document order, so early sections stay whole, the section the budget runs out in is cut, and every section or subsection past that point is dropped (counted in `truncation.omittedSections`). outline: split the budget evenly so every section and subsection keeps its heading, and an excerpt as far as the budget reaches — a heading the budget left empty is marked as such in the rendered text. Use it to survey what an article contains before requesting specific `sections`. Ignored when no budget is set, and identical for `source=unpaywall` bodies, which have no headings to preserve.'),
1206
1305
  })
1207
1306
  .refine((v) => [v.pmcids, v.pmids, v.dois].filter((b) => b !== undefined).length === 1, {
1208
1307
  message: 'Provide exactly one of `pmcids`, `pmids`, or `dois` (not zero, not more).',
@@ -1286,6 +1385,19 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1286
1385
  // caller can re-submit rather than whatever id the article happens to
1287
1386
  // carry — a `pmids` request recovers articles keyed by PMCID. (#100)
1288
1387
  const inputIdByArticle = new Map();
1388
+ // The caller's own spellings of each chain key, so `unavailable[]` and
1389
+ // `deferred.ids` report what was submitted. The `pmids` branch fills it — a
1390
+ // PMID's chain runs on its canonical form — and so does the `dois` branch,
1391
+ // whose chain runs once per DOI however many casings name it. Both also
1392
+ // fold in an input id that resolves to a PMC record another one already
1393
+ // claimed (see `routeToPmc`). Any other key is its own spelling. (#161, #166)
1394
+ const callerIds = new Map();
1395
+ const addCallerId = (key, spelling) => {
1396
+ const spellings = callerIds.get(key) ?? [];
1397
+ if (!spellings.includes(spelling))
1398
+ spellings.push(spelling);
1399
+ callerIds.set(key, spellings);
1400
+ };
1289
1401
  const budget = {
1290
1402
  overflowMode: input.overflowMode,
1291
1403
  ...(input.maxCharacters !== undefined && { maxCharacters: input.maxCharacters }),
@@ -1299,10 +1411,38 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1299
1411
  let pmidFallbackCandidates = [];
1300
1412
  let pmcidFallbackCandidates = [];
1301
1413
  let doiCandidates = [];
1414
+ // Send a converter-resolved PMCID to PMC EFetch under the input id that
1415
+ // named it. A second input id the converter places on the same PMC record
1416
+ // — two DOIs for one article — joins the first one's chain as another
1417
+ // spelling of it: the record is fetched once, and both ids are recovered,
1418
+ // or reported unavailable with that chain, rather than the later id taking
1419
+ // the PMCID over and leaving the earlier one with an empty chain.
1420
+ const routeToPmc = (inputId, pmcid) => {
1421
+ const normalized = normalizePmcId(pmcid);
1422
+ const prefixed = withPmcPrefix(normalized);
1423
+ const owner = pmcidToInputId.get(prefixed);
1424
+ if (owner === undefined) {
1425
+ pmcIds.push(normalized);
1426
+ pmcidToInputId.set(prefixed, inputId);
1427
+ return;
1428
+ }
1429
+ if (owner === inputId)
1430
+ return;
1431
+ for (const spelling of callerIds.get(inputId) ?? [inputId])
1432
+ addCallerId(owner, spelling);
1433
+ callerIds.delete(inputId);
1434
+ chainByInput.delete(inputId);
1435
+ };
1302
1436
  if (input.pmids) {
1437
+ // The ID Converter parses `00000001` as PMID 1 yet reports it "not found
1438
+ // in PMC", and every later stage answers with NCBI's own PMID, so the chain
1439
+ // runs once per distinct PMID in its canonical form. (#161)
1303
1440
  for (const id of input.pmids)
1441
+ addCallerId(normalizePmid(id), id);
1442
+ const pmids = [...callerIds.keys()];
1443
+ for (const id of pmids)
1304
1444
  chainByInput.set(id, []);
1305
- const records = await getNcbiService().idConvert(input.pmids, 'pmid', ctx.signal ? { signal: ctx.signal } : undefined);
1445
+ const records = await getNcbiService().idConvert(pmids, 'pmid', ctx.signal ? { signal: ctx.signal } : undefined);
1306
1446
  const seen = new Set();
1307
1447
  for (const r of records) {
1308
1448
  if (r.pmid === undefined)
@@ -1310,9 +1450,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1310
1450
  const pmid = String(r.pmid);
1311
1451
  seen.add(pmid);
1312
1452
  if (r.pmcid) {
1313
- const normalized = normalizePmcId(String(r.pmcid));
1314
- pmcIds.push(normalized);
1315
- pmcidToInputId.set(withPmcPrefix(normalized), pmid);
1453
+ routeToPmc(pmid, String(r.pmcid));
1316
1454
  pmidContext.set(pmid, { pmid, ...(r.doi && { doi: r.doi }) });
1317
1455
  }
1318
1456
  else {
@@ -1324,7 +1462,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1324
1462
  pmidFallbackCandidates.push({ pmid, ...(r.doi && { doi: r.doi }) });
1325
1463
  }
1326
1464
  }
1327
- for (const requested of input.pmids) {
1465
+ for (const requested of pmids) {
1328
1466
  if (!seen.has(requested)) {
1329
1467
  chainByInput.get(requested)?.push({
1330
1468
  tier: 'pmc',
@@ -1336,31 +1474,42 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1336
1474
  }
1337
1475
  }
1338
1476
  else if (input.pmcids) {
1339
- for (const id of input.pmcids)
1340
- chainByInput.set(withPmcPrefix(normalizePmcId(id)), []);
1341
- pmcIds = input.pmcids.map(normalizePmcId);
1477
+ // `PMC123`, `pmc123`, `123` and `PMC0123` name one record; it runs the
1478
+ // chain once. (#170)
1479
+ pmcIds = [...new Set(input.pmcids.map(normalizePmcId))];
1480
+ for (const id of pmcIds)
1481
+ chainByInput.set(withPmcPrefix(id), []);
1342
1482
  }
1343
1483
  else if (input.dois) {
1344
1484
  // Mirror the `pmids` branch: resolve DOI → PMCID via the PMC ID Converter
1345
1485
  // so PMC-indexed DOIs reach PMC EFetch instead of going straight to the
1346
1486
  // EPMC/Unpaywall fallback (which misses articles whose only OA copy is the
1347
1487
  // PMC JATS). DOIs the converter can't place in PMC seed `doiCandidates`.
1348
- const requestedDois = new Set(input.dois);
1349
- for (const doi of input.dois)
1488
+ //
1489
+ // DOIs are case-insensitive, and the converter echoes one casing in
1490
+ // `requested-id` for DOIs that differ only in case, so every casing of a
1491
+ // DOI shares one chain, keyed by the first spelling submitted, and
1492
+ // converter records are matched to it case-insensitively. (#166)
1493
+ const chainKeyByDoi = new Map();
1494
+ for (const doi of input.dois) {
1495
+ const key = chainKeyByDoi.get(doi.toLowerCase()) ?? doi;
1496
+ chainKeyByDoi.set(doi.toLowerCase(), key);
1497
+ addCallerId(key, doi);
1498
+ }
1499
+ const dois = [...callerIds.keys()];
1500
+ for (const doi of dois)
1350
1501
  chainByInput.set(doi, []);
1351
- const records = await getNcbiService().idConvert(input.dois, 'doi', ctx.signal ? { signal: ctx.signal } : undefined);
1502
+ const records = await getNcbiService().idConvert(dois, 'doi', ctx.signal ? { signal: ctx.signal } : undefined);
1352
1503
  const seen = new Set();
1353
1504
  for (const r of records) {
1354
- // The converter echoes the submitted id verbatim in `requested-id`;
1355
- // match on it, not `r.doi` (DOIs aren't case-stable across the API).
1356
- const doi = String(r['requested-id']);
1357
- if (!requestedDois.has(doi))
1505
+ // Match on the echoed `requested-id`, not `r.doi`: the record's own DOI
1506
+ // can be cased differently from anything the caller sent.
1507
+ const doi = chainKeyByDoi.get(String(r['requested-id']).toLowerCase());
1508
+ if (doi === undefined)
1358
1509
  continue;
1359
1510
  seen.add(doi);
1360
1511
  if (r.pmcid) {
1361
- const normalized = normalizePmcId(String(r.pmcid));
1362
- pmcIds.push(normalized);
1363
- pmcidToInputId.set(withPmcPrefix(normalized), doi);
1512
+ routeToPmc(doi, String(r.pmcid));
1364
1513
  }
1365
1514
  else {
1366
1515
  chainByInput.get(doi)?.push({
@@ -1371,7 +1520,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1371
1520
  doiCandidates.push({ doi });
1372
1521
  }
1373
1522
  }
1374
- for (const requested of input.dois) {
1523
+ for (const requested of dois) {
1375
1524
  if (!seen.has(requested)) {
1376
1525
  chainByInput.get(requested)?.push({
1377
1526
  tier: 'pmc',
@@ -1644,7 +1793,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1644
1793
  return {
1645
1794
  pmcId,
1646
1795
  result: candidate.doi
1647
- ? await resolveUnpaywall({ pmcId, doi: candidate.doi, budget }, unpaywall, ctx)
1796
+ ? await resolveUnpaywall({ pmcId, doi: candidate.doi, epmcTitle: candidate.title, budget }, unpaywall, ctx)
1648
1797
  : undefined,
1649
1798
  };
1650
1799
  }));
@@ -1716,7 +1865,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1716
1865
  const outcomes = await Promise.all(pmidFallbackCandidates.map(async (candidate) => ({
1717
1866
  candidate,
1718
1867
  result: candidate.doi
1719
- ? await resolveUnpaywall({ pmid: candidate.pmid, doi: candidate.doi, budget }, unpaywall, ctx)
1868
+ ? await resolveUnpaywall({ pmid: candidate.pmid, doi: candidate.doi, epmcTitle: candidate.title, budget }, unpaywall, ctx)
1720
1869
  : undefined,
1721
1870
  })));
1722
1871
  for (const { candidate, result } of outcomes) {
@@ -1759,7 +1908,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1759
1908
  // doesn't reject under normal operation.
1760
1909
  const outcomes = await Promise.all(doiCandidates.map(async (c) => ({
1761
1910
  doi: c.doi,
1762
- result: await resolveUnpaywall({ doi: c.doi, budget }, unpaywall, ctx),
1911
+ result: await resolveUnpaywall({ doi: c.doi, epmcTitle: c.title, budget }, unpaywall, ctx),
1763
1912
  })));
1764
1913
  for (const { doi, result } of outcomes) {
1765
1914
  if ('article' in result) {
@@ -1786,13 +1935,15 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1786
1935
  if (recoveredIds.has(id))
1787
1936
  continue;
1788
1937
  const unqueried = unqueriedByInput.get(id);
1789
- unavailable.push({
1790
- id,
1791
- idType,
1792
- reason: reasonFromChain(chain),
1793
- triedTiers: chain,
1794
- ...(unqueried?.size && { unqueriedTiers: [...unqueried] }),
1795
- });
1938
+ for (const callerId of callerIds.get(id) ?? [id]) {
1939
+ unavailable.push({
1940
+ id: callerId,
1941
+ idType,
1942
+ reason: reasonFromChain(chain),
1943
+ triedTiers: chain,
1944
+ ...(unqueried?.size && { unqueriedTiers: [...unqueried] }),
1945
+ });
1946
+ }
1796
1947
  }
1797
1948
  // Whole-response budget: fill with complete records in response order and
1798
1949
  // hand the rest back as identifiers the caller can re-submit. One ledger for
@@ -1812,7 +1963,10 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1812
1963
  idType,
1813
1964
  // Every recovery site records the input id; `articleDisplayId` is
1814
1965
  // the total-function fallback, not an expected path.
1815
- ids: fit.deferred.map((a) => inputIdByArticle.get(a) ?? articleDisplayId(a)),
1966
+ ids: fit.deferred.map((a) => {
1967
+ const id = inputIdByArticle.get(a) ?? articleDisplayId(a);
1968
+ return callerIds.get(id)?.[0] ?? id;
1969
+ }),
1816
1970
  nextDeferredCharacters,
1817
1971
  }
1818
1972
  : undefined;
@@ -1825,8 +1979,9 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1825
1979
  if (index === -1)
1826
1980
  continue;
1827
1981
  const [dropped] = truncatedArticles.splice(index, 1);
1828
- if (dropped)
1829
- omittedSections -= countOmittedSections(dropped, input.overflowMode);
1982
+ if (dropped) {
1983
+ omittedSections -= countDroppedSections(dropped.sections ?? [], input.overflowMode);
1984
+ }
1830
1985
  }
1831
1986
  }
1832
1987
  ctx.log.info('pubmed_fetch_fulltext completed', {
@@ -1934,13 +2089,29 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1934
2089
  lines.push('');
1935
2090
  const t = truncationById.get(articleDisplayId(a));
1936
2091
  if (a.source === 'pmc')
1937
- formatPmcArticle(a, lines, t);
2092
+ formatPmcArticle(a, lines, t, result.truncation?.mode);
1938
2093
  else
1939
2094
  formatUnpaywallArticle(a, lines, t);
1940
2095
  }
1941
2096
  return [{ type: 'text', text: lines.join('\n') }];
1942
2097
  },
1943
2098
  });
2099
+ /**
2100
+ * Merge what a Europe PMC hit carried onto a candidate bound for Unpaywall. A
2101
+ * DOI the candidate already holds wins — for `dois` input it is the caller's
2102
+ * own identifier.
2103
+ */
2104
+ function carryEpmcHit(candidate, hit) {
2105
+ return {
2106
+ ...candidate,
2107
+ ...(hit.doi && !candidate.doi && { doi: hit.doi }),
2108
+ ...(hit.title && { title: hit.title }),
2109
+ };
2110
+ }
2111
+ /** Display-ready plain text, or `undefined` when nothing is left once markup is stripped. */
2112
+ function nonEmptyDisplayText(raw) {
2113
+ return (raw && toDisplayText(raw)) || undefined;
2114
+ }
1944
2115
  /**
1945
2116
  * Run the Europe PMC step against everything that fell through PMC EFetch
1946
2117
  * plus any direct DOI input. Each candidate goes through search-by-best-id →
@@ -1960,24 +2131,28 @@ async function runEpmcStage(epmc, args) {
1960
2131
  }
1961
2132
  if (search.kind === 'miss')
1962
2133
  return { c, outcome: { kind: 'miss' } };
1963
- const doi = search.hit.doi ? { doi: search.hit.doi } : {};
2134
+ const title = nonEmptyDisplayText(search.hit.title);
2135
+ const hit = {
2136
+ ...(search.hit.doi && { doi: search.hit.doi }),
2137
+ ...(title && { title }),
2138
+ };
1964
2139
  const fetched = await fetchEpmcArticle(epmc, search.hit, args, contextPmid);
1965
2140
  if (fetched.kind === 'error') {
1966
- return { c, ...doi, outcome: { kind: 'service-error', detail: fetched.detail } };
2141
+ return { c, ...hit, outcome: { kind: 'service-error', detail: fetched.detail } };
1967
2142
  }
1968
2143
  if (fetched.kind === 'no-fulltext') {
1969
2144
  return {
1970
2145
  c,
1971
- ...doi,
2146
+ ...hit,
1972
2147
  outcome: { kind: 'no-fulltext', ...(fetched.detail && { detail: fetched.detail }) },
1973
2148
  };
1974
2149
  }
1975
2150
  if (fetched.kind === 'no-body') {
1976
- return { c, ...doi, outcome: { kind: 'no-body', detail: fetched.detail } };
2151
+ return { c, ...hit, outcome: { kind: 'no-body', detail: fetched.detail } };
1977
2152
  }
1978
2153
  return {
1979
2154
  c,
1980
- ...doi,
2155
+ ...hit,
1981
2156
  outcome: { kind: 'hit' },
1982
2157
  article: fetched.article,
1983
2158
  sectionFilterMiss: fetched.sectionFilterMiss,
@@ -2029,25 +2204,25 @@ async function runEpmcStage(epmc, args) {
2029
2204
  pmidOutcomes.set(run.c.pmid, run.outcome);
2030
2205
  if (run.article)
2031
2206
  collectHit(run.c.pmid, { ...run, article: run.article });
2032
- // A DOI the EPMC hit carried is evidence the next stage needs, whatever the
2033
- // fetch outcome was — merge it in like the pmcid branch below, so Unpaywall
2034
- // gets it without a redundant PubMed metadata round-trip. (#119)
2207
+ // What the EPMC hit carried is evidence the next stage needs, whatever the
2208
+ // fetch outcome was: its DOI spares Unpaywall a PubMed metadata round-trip
2209
+ // (#119), and its title backs up Unpaywall's own record (#144).
2035
2210
  else
2036
- remainingPmid.push(run.doi && !run.c.doi ? { ...run.c, doi: run.doi } : run.c);
2211
+ remainingPmid.push(carryEpmcHit(run.c, run));
2037
2212
  }
2038
2213
  for (const run of pmcidResults) {
2039
2214
  pmcidOutcomes.set(run.c.normalized, run.outcome);
2040
2215
  if (run.article)
2041
2216
  collectHit(run.c.normalized, { ...run, article: run.article });
2042
2217
  else
2043
- remainingPmcid.push(run.doi && !run.c.c.doi ? { ...run.c.c, doi: run.doi } : run.c.c);
2218
+ remainingPmcid.push(carryEpmcHit(run.c.c, run));
2044
2219
  }
2045
2220
  for (const run of doiResults) {
2046
2221
  doiOutcomes.set(run.c.doi, run.outcome);
2047
2222
  if (run.article)
2048
2223
  collectHit(run.c.doi, { ...run, article: run.article });
2049
2224
  else
2050
- remainingDoi.push(run.c);
2225
+ remainingDoi.push(carryEpmcHit(run.c, run));
2051
2226
  }
2052
2227
  return {
2053
2228
  articles,
@@ -2197,9 +2372,15 @@ async function fetchPubmedDois(pmids, signal) {
2197
2372
  * Resolve a DOI to an open-access article via Unpaywall. `pmcId` and `pmid`,
2198
2373
  * when set, are stamped onto the resulting article so the branch that requested
2199
2374
  * it carries its identifier through — Unpaywall itself only knows the DOI.
2375
+ *
2376
+ * The article's title is the first one found in Unpaywall's own record, then
2377
+ * `epmcTitle` — the Europe PMC record the chain searched for this id — then,
2378
+ * for HTML content only, the title the extractor detects on the page. None is
2379
+ * ever invented: with no source carrying one, the article has no title. The
2380
+ * journal name and year come from Unpaywall's record alone. (#144)
2200
2381
  */
2201
2382
  async function resolveUnpaywall(args, service, ctx) {
2202
- const { pmcId, pmid, doi, budget } = args;
2383
+ const { pmcId, pmid, doi, epmcTitle, budget } = args;
2203
2384
  const requestedIds = { ...(pmcId && { pmcId }), ...(pmid && { pmid }) };
2204
2385
  /** Budget the extracted body, then pair the article with its accounting. */
2205
2386
  const budgeted = (build, content) => {
@@ -2237,6 +2418,15 @@ async function resolveUnpaywall(args, service, ctx) {
2237
2418
  ctx.log.warning('Unpaywall content fetch failed', { doi, error: detail });
2238
2419
  return { unavailable: { reason: 'fetch-failed', detail } };
2239
2420
  }
2421
+ const recordTitle = nonEmptyDisplayText(resolution.title) ?? epmcTitle;
2422
+ const record = {
2423
+ ...requestedIds,
2424
+ doi,
2425
+ sourceUrl: content.fetchedUrl,
2426
+ location: resolution.location,
2427
+ journalName: nonEmptyDisplayText(resolution.journalName),
2428
+ year: resolution.year,
2429
+ };
2240
2430
  try {
2241
2431
  if (content.kind === 'html') {
2242
2432
  const extracted = await htmlExtractor.extract(content.body, {
@@ -2253,13 +2443,10 @@ async function resolveUnpaywall(args, service, ctx) {
2253
2443
  };
2254
2444
  }
2255
2445
  return budgeted((text) => buildUnpaywallArticle({
2256
- ...requestedIds,
2257
- doi,
2258
- sourceUrl: content.fetchedUrl,
2259
- location: resolution.location,
2446
+ ...record,
2260
2447
  contentFormat: 'html-markdown',
2261
2448
  content: text,
2262
- title: extracted.title,
2449
+ title: recordTitle ?? extracted.title,
2263
2450
  wordCount: extracted.wordCount,
2264
2451
  }), body);
2265
2452
  }
@@ -2271,12 +2458,10 @@ async function resolveUnpaywall(args, service, ctx) {
2271
2458
  };
2272
2459
  }
2273
2460
  return budgeted((body) => buildUnpaywallArticle({
2274
- ...requestedIds,
2275
- doi,
2276
- sourceUrl: content.fetchedUrl,
2277
- location: resolution.location,
2461
+ ...record,
2278
2462
  contentFormat: 'pdf-text',
2279
2463
  content: body,
2464
+ title: recordTitle,
2280
2465
  totalPages: extracted.totalPages,
2281
2466
  }), text);
2282
2467
  }
@@ -2301,6 +2486,8 @@ function buildUnpaywallArticle(args) {
2301
2486
  sourceUrl: args.sourceUrl,
2302
2487
  content: args.content,
2303
2488
  ...(args.title && { title: args.title }),
2489
+ ...(args.journalName && { journalName: args.journalName }),
2490
+ ...(args.year !== undefined && { year: args.year }),
2304
2491
  ...(args.wordCount !== undefined && { wordCount: args.wordCount }),
2305
2492
  ...(args.totalPages !== undefined && { totalPages: args.totalPages }),
2306
2493
  ...(location.license && { license: location.license }),
@@ -2458,16 +2645,39 @@ function formatTruncation(t, lines) {
2458
2645
  ? ''
2459
2646
  : `, ${a.omittedAssets} asset(s) dropped whole: ${(a.omittedAssetNames ?? []).join(', ')}`;
2460
2647
  lines.push(`- ${a.id} (${a.source}): ${a.returnedCharacters} of ${a.originalCharacters} characters${tablesDropped}${assetsDropped}`);
2461
- for (const s of a.sections ?? []) {
2462
- lines.push(` - ${s.title ?? 'untitled section'} — ${s.returnedCharacters} of ${s.originalCharacters} characters (truncated: ${s.truncated})`);
2463
- }
2648
+ for (const s of a.sections ?? [])
2649
+ formatLedgerEntry(s, lines, t.mode, 1);
2464
2650
  }
2465
2651
  }
2652
+ /**
2653
+ * One section's ledger line, then its subsections' one level deeper. The line
2654
+ * names the section by {@link sectionHeading}, so it reads exactly as the
2655
+ * section's heading does in the body. An entry the budget emptied says what
2656
+ * became of it — dropped in `truncate` mode, kept as a bare heading in
2657
+ * `outline` mode. (#143, #148)
2658
+ */
2659
+ function formatLedgerEntry(entry, lines, mode, depth) {
2660
+ const fate = isBudgetEmptied(entry)
2661
+ ? mode === 'truncate'
2662
+ ? ' — dropped'
2663
+ : ' — heading only'
2664
+ : '';
2665
+ lines.push(`${' '.repeat(depth)}- ${sectionHeading(entry)} — ${entry.returnedCharacters} of ${entry.originalCharacters} characters (truncated: ${entry.truncated})${fate}`);
2666
+ for (const sub of entry.subsections ?? [])
2667
+ formatLedgerEntry(sub, lines, mode, depth + 1);
2668
+ }
2466
2669
  /** Per-article inline marker so a reader of one article's body knows it is partial. */
2467
2670
  function truncationNote(t) {
2468
2671
  return `\n> Body shortened to fit the requested character budget — ${t.returnedCharacters} of ${t.originalCharacters} characters returned. See \`truncation\` for per-section counts.`;
2469
2672
  }
2470
- function formatPmcArticle(a, lines, truncation) {
2673
+ /**
2674
+ * The marker an `outline`-mode heading carries when the budget left it no text,
2675
+ * so it reads as withheld rather than as a heading over nothing. (#143)
2676
+ */
2677
+ function emptiedSectionNote(originalCharacters) {
2678
+ return `> Text omitted to fit the requested character budget — 0 of ${originalCharacters} characters returned.`;
2679
+ }
2680
+ function formatPmcArticle(a, lines, truncation, mode) {
2471
2681
  // Render-time only — `structuredContent.articles[].title` keeps the
2472
2682
  // plain-text value the JATS parser produced. (#102)
2473
2683
  lines.push(`### ${escapeMarkdownInline(a.title ?? articleDisplayId(a))}`);
@@ -2526,8 +2736,13 @@ function formatPmcArticle(a, lines, truncation) {
2526
2736
  lines.push(truncationNote(truncation));
2527
2737
  if (a.abstract)
2528
2738
  lines.push(`\n#### Abstract\n${a.abstract}`);
2529
- for (const sec of a.sections)
2530
- formatSection(sec, lines, 4);
2739
+ // `outline` mode keeps every section and subsection, so each one lines up
2740
+ // with its ledger entry by position. `truncate` mode drops the emptied ones —
2741
+ // positions no longer line up, and nothing left needs a marker.
2742
+ const ledger = mode === 'outline' ? truncation?.sections : undefined;
2743
+ a.sections.forEach((sec, i) => {
2744
+ formatSection(sec, lines, 4, ledger?.[i]);
2745
+ });
2531
2746
  if (a.tables?.length)
2532
2747
  formatTables(a.tables, lines);
2533
2748
  if (a.assets?.length)
@@ -2654,6 +2869,10 @@ function formatUnpaywallArticle(a, lines, truncation) {
2654
2869
  : 'Unpaywall (PDF → plain text)';
2655
2870
  lines.push(`### ${escapeMarkdownInline(heading)}`);
2656
2871
  lines.push(`**Source:** ${formatLabel}`);
2872
+ if (a.journalName)
2873
+ lines.push(`**Journal:** ${escapeMarkdownInline(a.journalName)}`);
2874
+ if (a.year !== undefined)
2875
+ lines.push(`**Year:** ${a.year}`);
2657
2876
  if (a.pmcId)
2658
2877
  lines.push(`**PMCID:** ${a.pmcId}`);
2659
2878
  if (a.pmid)
@@ -2689,20 +2908,43 @@ function formatPmcAuthor(au) {
2689
2908
  function formatHeading(label, title) {
2690
2909
  return label ? `${label} ${title}` : title;
2691
2910
  }
2911
+ /** How a section with no title and no label is named — in its body heading and its ledger line. */
2912
+ const UNTITLED_SECTION_LABEL = 'untitled section';
2913
+ /**
2914
+ * The name a section goes by in `content[]`: its label and title, else its
2915
+ * label alone, else {@link UNTITLED_SECTION_LABEL}. The body heading and the
2916
+ * truncation ledger line both take it from here, so the two always read the
2917
+ * same. Render-time escaped like every other upstream string interpolated into
2918
+ * a line, so a title's `*`, `_`, `` ` `` or `[` cannot restyle the heading;
2919
+ * `structuredContent` keeps the plain title. (#148, #169)
2920
+ */
2921
+ function sectionHeading(section) {
2922
+ if (section.title)
2923
+ return escapeMarkdownInline(formatHeading(section.label, section.title));
2924
+ return section.label ? escapeMarkdownInline(section.label) : UNTITLED_SECTION_LABEL;
2925
+ }
2692
2926
  /**
2693
2927
  * Render one body section and everything nested under it, one markdown heading
2694
2928
  * level per nesting level. Walks the full depth the output schema carries, so
2695
2929
  * `content[]` shows every section `structuredContent` does. Headings stop
2696
2930
  * deepening at `######`, the deepest markdown supports. (#112)
2931
+ *
2932
+ * Every section gets a heading — an untitled one included, under the name the
2933
+ * truncation ledger gives it — so its text never runs on from the block before
2934
+ * it. `ledger` is the section's `outline`-mode accounting, when there is one: a
2935
+ * section it shows the budget emptied carries a marker under its heading.
2936
+ * (#143, #148)
2697
2937
  */
2698
- function formatSection(section, lines, depth) {
2699
- if (section.title) {
2700
- lines.push(`\n${'#'.repeat(Math.min(depth, 6))} ${formatHeading(section.label, section.title)}`);
2701
- }
2938
+ function formatSection(section, lines, depth, ledger) {
2939
+ lines.push(`\n${'#'.repeat(Math.min(depth, 6))} ${sectionHeading(section)}`);
2702
2940
  if (section.text)
2703
2941
  lines.push(section.text);
2704
- for (const sub of section.subsections ?? [])
2705
- formatSection(sub, lines, depth + 1);
2942
+ else if (ledger && isBudgetEmptied(ledger)) {
2943
+ lines.push(emptiedSectionNote(ledger.originalCharacters));
2944
+ }
2945
+ section.subsections?.forEach((sub, i) => {
2946
+ formatSection(sub, lines, depth + 1, ledger?.subsections?.[i]);
2947
+ });
2706
2948
  }
2707
2949
  /**
2708
2950
  * Strip absolute URLs from chain detail strings. Upstream errors (e.g.