@cyanheads/pubmed-mcp-server 2.10.16 → 2.10.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (28) hide show
  1. package/AGENTS.md +1 -1
  2. package/CLAUDE.md +1 -1
  3. package/README.md +6 -4
  4. package/changelog/2.10.x/2.10.17.md +19 -0
  5. package/dist/mcp-server/tools/definitions/_text.d.ts +20 -3
  6. package/dist/mcp-server/tools/definitions/_text.d.ts.map +1 -1
  7. package/dist/mcp-server/tools/definitions/_text.js +35 -3
  8. package/dist/mcp-server/tools/definitions/_text.js.map +1 -1
  9. package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.d.ts +23 -31
  10. package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.d.ts.map +1 -1
  11. package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.js +349 -130
  12. package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.js.map +1 -1
  13. package/dist/services/error-contracts.d.ts +5 -3
  14. package/dist/services/error-contracts.d.ts.map +1 -1
  15. package/dist/services/error-contracts.js +5 -3
  16. package/dist/services/error-contracts.js.map +1 -1
  17. package/dist/services/ncbi/parsing/pmc-article-parser.d.ts +8 -0
  18. package/dist/services/ncbi/parsing/pmc-article-parser.d.ts.map +1 -1
  19. package/dist/services/ncbi/parsing/pmc-article-parser.js +159 -16
  20. package/dist/services/ncbi/parsing/pmc-article-parser.js.map +1 -1
  21. package/dist/services/unpaywall/types.d.ts +16 -2
  22. package/dist/services/unpaywall/types.d.ts.map +1 -1
  23. package/dist/services/unpaywall/unpaywall-service.d.ts +4 -3
  24. package/dist/services/unpaywall/unpaywall-service.d.ts.map +1 -1
  25. package/dist/services/unpaywall/unpaywall-service.js +15 -4
  26. package/dist/services/unpaywall/unpaywall-service.js.map +1 -1
  27. package/package.json +1 -1
  28. package/server.json +3 -3
@@ -3,39 +3,58 @@
3
3
  * three-stage chain: NCBI PMC EFetch → Europe PMC `fullTextXML` → Unpaywall.
4
4
  * Accepts three mutually-exclusive input shapes:
5
5
  *
6
- * - `pmcids` — fetch directly by PMC ID. Articles not in PMC fall through to
7
- * EPMC by PMC ID, then to Unpaywall when the DOI is available.
6
+ * - `pmcids` — fetch directly by PMC ID, once per record however it is
7
+ * spelled (`PMC123`, `pmc123`, `123`, zero-padded `PMC0123`), and reported
8
+ * in `PMC<digits>` form. Articles not in PMC fall through to EPMC by PMC ID,
9
+ * then to Unpaywall when the DOI is available.
8
10
  * - `pmids` — resolve PMID → PMCID via PMC ID Converter, then run the chain.
9
11
  * A zero-padded PMID runs as the PMID it spells; `unavailable[]` and
10
12
  * `deferred.ids` report it as the caller wrote it.
11
13
  * - `dois` — resolve DOI → PMCID via the PMC ID Converter (mirroring `pmids`),
12
14
  * then run the chain. DOIs with no PMC counterpart fall through to EPMC
13
15
  * search-by-DOI → fullTextXML, then Unpaywall (EPMC-only OA, preprints).
16
+ * DOIs are case-insensitive: every casing of one DOI runs the chain once,
17
+ * and `unavailable[]` reports each casing as the caller wrote it.
14
18
  *
15
19
  * Output uses a discriminated union on `source` (`pmc` | `unpaywall`) with an
16
20
  * extra `viaSource` discriminator that records which layer produced the
17
21
  * content. EPMC's JATS reuses the `pmc` schema shape because it's the same
18
- * DTD; `viaSource: 'europepmc'` distinguishes it from PMC EFetch output.
22
+ * DTD; `viaSource: 'europepmc'` distinguishes it from PMC EFetch output. An
23
+ * Unpaywall article takes its title from Unpaywall's record, else the Europe
24
+ * PMC record the chain searched, else (HTML only) the page itself.
25
+ *
26
+ * Europe PMC and Unpaywall failures are folded into each id's `triedTiers`
27
+ * rather than thrown, so the declared `errors[]` covers the NCBI ID routing
28
+ * alone.
19
29
  *
20
30
  * @module src/mcp-server/tools/definitions/fetch-fulltext.tool
21
31
  */
22
32
  import { tool, z } from '@cyanheads/mcp-ts-core';
23
33
  import { htmlExtractor, pdfParser } from '@cyanheads/mcp-ts-core/utils';
24
34
  import { getServerConfig } from '../../../config/server-config.js';
25
- import { EUROPEPMC_SERVICE_ERRORS, NCBI_SERVICE_ERRORS, UNPAYWALL_SERVICE_ERRORS, } from '../../../services/error-contracts.js';
35
+ import { NCBI_SERVICE_ERRORS } from '../../../services/error-contracts.js';
26
36
  import { getEuropePmcService, } from '../../../services/europe-pmc/europe-pmc-service.js';
27
37
  import { getNcbiService } from '../../../services/ncbi/ncbi-service.js';
28
38
  import { extractDoi, extractPmid } from '../../../services/ncbi/parsing/article-parser.js';
29
39
  import { parsePmcArticle } from '../../../services/ncbi/parsing/pmc-article-parser.js';
30
40
  import { findAll, findOne } from '../../../services/ncbi/parsing/pmc-xml-helpers.js';
41
+ import { toDisplayText } from '../../../services/ncbi/parsing/text-helpers.js';
31
42
  import { ensureArray } from '../../../services/ncbi/parsing/xml-helpers.js';
32
43
  import { getUnpaywallService, } from '../../../services/unpaywall/unpaywall-service.js';
33
44
  import { fitWholeItems } from './_budget.js';
34
45
  import { conceptMeta, EDAM_DATA_RETRIEVAL, SCHEMA_SCHOLARLY_ARTICLE } from './_concepts.js';
35
46
  import { doiStringSchema, normalizePmid, pmcidStringSchema, pmidStringSchema } from './_schemas.js';
36
- import { escapeMarkdownInline, escapeMarkdownTableCell, sliceCodeUnits } from './_text.js';
47
+ import { escapeMarkdownInline, escapeMarkdownTableCell, sliceAtWordBoundary } from './_text.js';
48
+ /**
49
+ * Canonical digits of a PMC ID {@link pmcidStringSchema} accepted: the `PMC`
50
+ * prefix dropped in any case, and leading zeros stripped the way
51
+ * {@link normalizePmid} strips them. PMC EFetch reads `03531190` as PMC3531190
52
+ * and answers with that record, so every `pmcids` path keys on this form — and
53
+ * reports it back as `PMC<digits>`. An all-zero ID keeps `0`, which EFetch
54
+ * still answers with its empty-list envelope. (#170)
55
+ */
37
56
  function normalizePmcId(id) {
38
- return id.replace(/^PMC/i, '');
57
+ return normalizePmid(id.replace(/^PMC/i, ''));
39
58
  }
40
59
  function withPmcPrefix(id) {
41
60
  return id.startsWith('PMC') ? id : `PMC${id}`;
@@ -540,7 +559,18 @@ const UnpaywallArticleSchema = z
540
559
  pubmedUrl: z.string().optional().describe('PubMed URL — present when `pmid` is set'),
541
560
  doi: z.string().describe('DOI used to locate the open-access copy'),
542
561
  sourceUrl: z.string().describe('URL the content was fetched from'),
543
- title: z.string().optional().describe('Detected article title when present'),
562
+ title: z
563
+ .string()
564
+ .optional()
565
+ .describe("Article title, from the first source that carries one: Unpaywall's record for the DOI, then the Europe PMC record when the chain searched Europe PMC for this id, then — for `html-markdown` content only — the title detected on the page. Absent when none of them has a title."),
566
+ journalName: z
567
+ .string()
568
+ .optional()
569
+ .describe("Journal or repository name from Unpaywall's record for the DOI (e.g. `medRxiv` for a medRxiv preprint). Absent when Unpaywall has none."),
570
+ year: z
571
+ .number()
572
+ .optional()
573
+ .describe("Publication year from Unpaywall's record for the DOI. Absent when Unpaywall has none."),
544
574
  content: z.string().describe('Full article text — Markdown or plain text per `contentFormat`'),
545
575
  wordCount: z
546
576
  .number()
@@ -619,18 +649,54 @@ const UnavailableSchema = z
619
649
  })
620
650
  .describe('One identifier that could not be returned, with the full chain it traversed');
621
651
  // ─── Character-budget schemas ────────────────────────────────────────────────
652
+ /**
653
+ * Character accounting for one subsection. Its own type rather than a
654
+ * self-reference for the reason {@link SubsectionSchema} is inlined: a
655
+ * `z.lazy()` schema emits `$defs`/`$ref`. The article schema carries sections
656
+ * two levels deep, so the ledger does too.
657
+ *
658
+ * The chain `truncation` → `articles[]` → `sections[]` → `subsections[]` → a
659
+ * leaf is eight schema hops, the most the `format-parity` sentinel walker
660
+ * reaches. Re-run `bun run lint:mcp` after any change to this shape. (#143)
661
+ */
662
+ const TruncatedSubsectionSchema = z
663
+ .object({
664
+ title: z.string().optional().describe('Subsection heading, when the subsection carries one'),
665
+ label: z
666
+ .string()
667
+ .optional()
668
+ .describe('Subsection label as printed (e.g. `2.1`), when the subsection carries one'),
669
+ originalCharacters: z
670
+ .number()
671
+ .describe('Body characters this subsection carried before the budget pass'),
672
+ returnedCharacters: z
673
+ .number()
674
+ .describe('Body characters this subsection carries in the response. Zero means it was dropped in `truncate` mode and counted in `omittedSections`, or kept as a heading-only entry in `outline` mode, marked as such in the rendered text.'),
675
+ truncated: z
676
+ .boolean()
677
+ .describe('True when the subsection returned fewer characters than it originally carried'),
678
+ })
679
+ .describe('Character accounting for one subsection of a shortened section');
622
680
  const TruncatedSectionSchema = z
623
681
  .object({
624
682
  title: z.string().optional().describe('Section heading, when the section carries one'),
683
+ label: z
684
+ .string()
685
+ .optional()
686
+ .describe('Section label as printed (e.g. `2`), when the section carries one'),
625
687
  originalCharacters: z
626
688
  .number()
627
689
  .describe('Body characters this section carried before the budget pass'),
628
690
  returnedCharacters: z
629
691
  .number()
630
- .describe('Body characters this section carries in the response. Zero means the section was dropped in `truncate` mode, or kept as a heading-only entry in `outline` mode.'),
692
+ .describe('Body characters this section carries in the response. Zero means the section was dropped in `truncate` mode, or kept as a heading-only entry in `outline` mode, marked as such in the rendered text.'),
631
693
  truncated: z
632
694
  .boolean()
633
695
  .describe('True when the section returned fewer characters than it originally carried'),
696
+ subsections: z
697
+ .array(TruncatedSubsectionSchema)
698
+ .optional()
699
+ .describe('Per-subsection accounting for a shortened section, in document order, including subsections dropped for budget — where inside the section the cut landed. Absent when the section was returned whole or carries no subsections.'),
634
700
  })
635
701
  .describe('Character accounting for one body section of a budgeted article');
636
702
  const TruncatedArticleSchema = z
@@ -648,7 +714,7 @@ const TruncatedArticleSchema = z
648
714
  sections: z
649
715
  .array(TruncatedSectionSchema)
650
716
  .optional()
651
- .describe('Per-section accounting for `source: pmc` articles, in document order, including sections dropped for budget. Absent for `source: unpaywall`, whose body has no section structure.'),
717
+ .describe('Per-section accounting for `source: pmc` articles, in document order, including sections dropped for budget; a shortened section lists its subsections. Absent for `source: unpaywall`, whose body has no section structure.'),
652
718
  omittedTables: z
653
719
  .number()
654
720
  .optional()
@@ -685,7 +751,7 @@ const TruncationSchema = z
685
751
  .describe('Body characters the shortened articles carry in this response'),
686
752
  omittedSections: z
687
753
  .number()
688
- .describe('Body sections dropped entirely because an article budget was exhausted before reaching them. Always 0 in `outline` mode, which keeps every heading.'),
754
+ .describe('Body sections and subsections dropped entirely because an article budget was exhausted before reaching them. A dropped section counts once, together with its subsections. Always 0 in `outline` mode, which keeps every heading.'),
689
755
  omittedTables: z
690
756
  .number()
691
757
  .optional()
@@ -786,30 +852,81 @@ function fitWholeNamed(items, allowance, measure, name) {
786
852
  };
787
853
  }
788
854
  /**
789
- * Rebuild a section subtree from `fitted`, consuming one entry per node in the
790
- * same document order {@link sectionTextFields} produced them. `cursor` walks
791
- * the flat list across the whole subtree.
855
+ * True when the budget emptied a node that had text — which `truncate` mode
856
+ * drops, at any depth, and `outline` mode keeps as a heading-only entry. A node
857
+ * that never carried text is kept either way: there was nothing to cut.
858
+ */
859
+ function isBudgetEmptied(entry) {
860
+ return entry.returnedCharacters === 0 && entry.originalCharacters > 0;
861
+ }
862
+ /**
863
+ * Sections and subsections `truncate` mode dropped, read off the ledger. A
864
+ * dropped node counts once and takes its subtree with it, so its subsections
865
+ * are not counted again. Shared by the budget pass and the deferral roll-back
866
+ * so both count by one rule. (#81, #143)
867
+ */
868
+ function countDroppedSections(entries, mode) {
869
+ if (mode !== 'truncate')
870
+ return 0;
871
+ return entries.reduce((n, entry) => n + (isBudgetEmptied(entry) ? 1 : countDroppedSections(entry.subsections ?? [], mode)), 0);
872
+ }
873
+ /**
874
+ * Rebuild a section subtree from `fitted` — one entry per node, in the document
875
+ * order {@link sectionTextFields} produced them, `cursor` walking the flat list
876
+ * across the whole subtree — together with its ledger entry.
877
+ *
878
+ * In `truncate` mode a node the budget emptied is dropped (`kept` is absent) at
879
+ * every depth, where only a top-level section used to be: a subsection left as
880
+ * a heading over nothing is a stub, not content. `outline` mode keeps it, and
881
+ * `format()` marks it. The entry lists its subsections only when this node was
882
+ * shortened, which is where they say something a whole section's entry does
883
+ * not. (#81, #143)
792
884
  */
793
- function withFittedTexts(section, fitted, cursor) {
885
+ function fitSectionTree(section, fitted, cursor, mode) {
794
886
  const text = fitted[cursor.i++] ?? '';
795
- const subsections = section.subsections?.map((sub) => withFittedTexts(sub, fitted, cursor));
796
- return { ...section, text, ...(subsections && { subsections }) };
887
+ const children = (section.subsections ?? []).map((sub) => fitSectionTree(sub, fitted, cursor, mode));
888
+ const originalCharacters = sectionCharacters(section);
889
+ const returnedCharacters = children.reduce((n, child) => n + child.entry.returnedCharacters, text.length);
890
+ const truncated = returnedCharacters < originalCharacters;
891
+ const entry = {
892
+ ...(section.title !== undefined && { title: section.title }),
893
+ ...(section.label !== undefined && { label: section.label }),
894
+ originalCharacters,
895
+ returnedCharacters,
896
+ truncated,
897
+ ...(truncated && children.length > 0 && { subsections: children.map((c) => c.entry) }),
898
+ };
899
+ if (mode === 'truncate' && isBudgetEmptied(entry))
900
+ return { entry };
901
+ const { subsections: _replaced, ...rest } = section;
902
+ const subsections = children.flatMap((child) => (child.kept ? [child.kept] : []));
903
+ return {
904
+ entry,
905
+ kept: { ...rest, text, ...(subsections.length > 0 && { subsections }) },
906
+ };
797
907
  }
798
908
  /**
799
909
  * Shorten an ordered list of text fields so their combined length fits
800
- * `allowance`. Fields are filled in order, so earlier fields survive whole and
801
- * later ones absorb the shortfall — the section's own text before its
802
- * subsections. Cuts at the character boundary with no appended marker so the
803
- * reported `returnedCharacters` is exact; `format()` carries the human-visible
804
- * note. A cut that would split a surrogate pair backs off a code unit, so a
805
- * field can return one character under its share — counts are measured off the
806
- * returned text, never off the allowance. (#93)
910
+ * `allowance`. Fields are filled in order, so earlier fields survive whole — the
911
+ * section's own text before its subsections — and the first field that does not
912
+ * fit is cut at the last word boundary inside what is left. Every field after
913
+ * that cut is past it and returns empty, the way a top-level section past the
914
+ * budget does, even when the cut left a few characters unspent: handing those
915
+ * on would open the next subsection with a fragment of its first word. (#143)
916
+ *
917
+ * No marker is appended, so the reported `returnedCharacters` is exact;
918
+ * `format()` carries the human-visible note. Counts are measured off the
919
+ * returned text, never off the allowance, which stays a ceiling. (#93)
807
920
  */
808
921
  function fitFields(fields, allowance) {
809
922
  let remaining = Math.max(allowance, 0);
810
923
  return fields.map((text) => {
811
- const kept = sliceCodeUnits(text, remaining);
812
- remaining -= kept.length;
924
+ if (text.length <= remaining) {
925
+ remaining -= text.length;
926
+ return text;
927
+ }
928
+ const kept = sliceAtWordBoundary(text, remaining);
929
+ remaining = 0;
813
930
  return kept;
814
931
  });
815
932
  }
@@ -882,10 +999,10 @@ function allotSectionBudgets(sizes, budget) {
882
999
  * nothing bounds either list and every entry is kept. (#111, #130)
883
1000
  *
884
1001
  * Returns the article untouched (same object identity) when no budget was
885
- * requested or nothing exceeded it. A section left with zero characters is
886
- * dropped in `truncate` mode and counted as omitted; `outline` keeps it as a
887
- * heading-only entry. Dropped sections still appear in the accounting so the
888
- * caller can see which headings exist. (#81)
1002
+ * requested or nothing exceeded it. A section or subsection left with zero
1003
+ * characters is dropped in `truncate` mode and counted as omitted; `outline`
1004
+ * keeps it as a heading-only entry. Dropped nodes still appear in the accounting
1005
+ * so the caller can see which headings exist. (#81, #143)
889
1006
  */
890
1007
  function applyPmcBudget(article, budget) {
891
1008
  const tables = article.tables ?? [];
@@ -901,25 +1018,16 @@ function applyPmcBudget(article, budget) {
901
1018
  const allowances = allotSectionBudgets(sizes, budget);
902
1019
  const kept = [];
903
1020
  const sectionReports = [];
904
- let omittedSections = 0;
905
1021
  let returnedCharacters = 0;
906
1022
  article.sections.forEach((section, i) => {
907
- const original = sizes[i] ?? 0;
908
1023
  const fitted = fitFields(sectionTextFields(section), allowances[i] ?? 0);
909
- const returned = totalLength(fitted);
910
- returnedCharacters += returned;
911
- sectionReports.push({
912
- ...(section.title !== undefined && { title: section.title }),
913
- originalCharacters: original,
914
- returnedCharacters: returned,
915
- truncated: returned < original,
916
- });
917
- if (returned === 0 && original > 0 && budget.overflowMode === 'truncate') {
918
- omittedSections += 1;
919
- return;
920
- }
921
- kept.push(withFittedTexts(section, fitted, { i: 0 }));
1024
+ const fit = fitSectionTree(section, fitted, { i: 0 }, budget.overflowMode);
1025
+ returnedCharacters += fit.entry.returnedCharacters;
1026
+ sectionReports.push(fit.entry);
1027
+ if (fit.kept)
1028
+ kept.push(fit.kept);
922
1029
  });
1030
+ const omittedSections = countDroppedSections(sectionReports, budget.overflowMode);
923
1031
  // Sections are served first, then tables, then assets — each spending whatever
924
1032
  // `maxCharacters` has left. A bare per-section budget sets no total, so nothing
925
1033
  // bounds either list.
@@ -960,13 +1068,13 @@ function applyPmcBudget(article, budget) {
960
1068
  * Apply the character budget to an Unpaywall body. That body is one
961
1069
  * unstructured blob — HTML-as-Markdown or PDF-as-text — so only `maxCharacters`
962
1070
  * applies, and `outline` mode has no headings to preserve and behaves like
963
- * `truncate`. (#81)
1071
+ * `truncate`. The cut ends at a word boundary, as a section cut does. (#81, #143)
964
1072
  */
965
1073
  function applyContentBudget(content, budget) {
966
1074
  const cap = budget.maxCharacters;
967
1075
  if (cap === undefined || content.length <= cap)
968
1076
  return { content };
969
- const kept = sliceCodeUnits(content, cap);
1077
+ const kept = sliceAtWordBoundary(content, cap);
970
1078
  return {
971
1079
  content: kept,
972
1080
  truncation: { originalCharacters: content.length, returnedCharacters: kept.length },
@@ -1075,17 +1183,6 @@ function buildDeferralNotice(deferred) {
1075
1183
  : `Response character budget reached: ${deferred.returnedCharacters} of ${deferred.maxResponseCharacters} characters returned.`;
1076
1184
  return `${spent} ${deferred.deferredCount} resolved article(s) were deferred whole: ${deferred.ids.join(', ')}. Re-call pubmed_fetch_fulltext with those ids under \`${deferred.idType}s\` to retrieve them, or raise maxResponseCharacters to at least ${deferred.nextDeferredCharacters} — the size of the next deferred article.`;
1077
1185
  }
1078
- /**
1079
- * Body sections an article's per-article budget dropped, derived from that
1080
- * article's own accounting by the rule {@link applyPmcBudget} counts by:
1081
- * `outline` mode keeps every heading, so it drops none. Used to take a deferred
1082
- * article's contribution back out of the response-level roll-up. (#100)
1083
- */
1084
- function countOmittedSections(entry, mode) {
1085
- if (mode !== 'truncate')
1086
- return 0;
1087
- return (entry.sections ?? []).filter((s) => s.originalCharacters > 0 && s.returnedCharacters === 0).length;
1088
- }
1089
1186
  // ─── Tool Definition ─────────────────────────────────────────────────────────
1090
1187
  /**
1091
1188
  * Compose the tool description for the fallback tiers enabled in this
@@ -1132,11 +1229,11 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1132
1229
  annotations: { readOnlyHint: true, openWorldHint: true },
1133
1230
  _meta: conceptMeta([SCHEMA_SCHOLARLY_ARTICLE, EDAM_DATA_RETRIEVAL]),
1134
1231
  sourceUrl: 'https://github.com/cyanheads/pubmed-mcp-server/blob/main/src/mcp-server/tools/definitions/fetch-fulltext.tool.ts',
1135
- errors: [
1136
- ...NCBI_SERVICE_ERRORS,
1137
- ...UNPAYWALL_SERVICE_ERRORS,
1138
- ...EUROPEPMC_SERVICE_ERRORS,
1139
- ],
1232
+ // Only the ID routing's `idConvert` calls run unwrapped. Every Europe PMC and
1233
+ // Unpaywall call sits behind a catch that folds the failure into
1234
+ // `unavailable[].triedTiers`, so their service reasons never reach a caller
1235
+ // and are not declared here. (#168)
1236
+ errors: [...NCBI_SERVICE_ERRORS],
1140
1237
  input: z
1141
1238
  .object({
1142
1239
  pmcids: z
@@ -1186,7 +1283,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1186
1283
  .min(1)
1187
1284
  .max(1_000_000)
1188
1285
  .optional()
1189
- .describe('Per-article budget for body text, in characters. Counts `source=pmc` section and subsection text — which carries the inline blocks the parser renders in place, such as lists, definition lists, block quotes, boxed text, preformatted blocks and displayed formulae — plus table label, caption, cell and footnote text and asset label, caption and `href` text; or the `source=unpaywall` `content` body. Titles, abstracts, identifiers, and references are never counted or shortened. The counted unit is that text alone — the Markdown grid `content[]` renders around the cells (pipes, padding, the divider row, headings) is scaffolding this budget does not measure, so a table renders longer than it costs here. Sections are served first, then tables, then assets, each spending what is left, in document order — admission stops at the first entry that does not fit, and every entry from there on is dropped whole rather than cut mid-row or returned with a shortened caption, counted in `truncation.omittedTables` / `truncation.omittedAssets` and named in `truncation.articles[].omittedTableNames` / `omittedAssetNames`. Applied after `sections`, `maxSections`, `includeReferences`, `includeTables`, and `includeAssets`, so semantic filtering is unaffected. This knob alone bounds only bodies: the response-wide ceiling it implies is this value times the number of articles returned, plus every uncounted field. Use `maxResponseCharacters` for a true whole-response ceiling. Omit for the full body.'),
1286
+ .describe('Per-article budget for body text, in characters. Counts `source=pmc` section and subsection text — which carries the inline blocks the parser renders in place, such as lists, definition lists, block quotes, boxed text, preformatted blocks and displayed formulae — plus table label, caption, cell and footnote text and asset label, caption and `href` text; or the `source=unpaywall` `content` body. Titles, abstracts, identifiers, and references are never counted or shortened. Shortened text ends at the last word boundary inside its allowance, so it can come back a few characters under it. The counted unit is that text alone — the Markdown grid `content[]` renders around the cells (pipes, padding, the divider row, headings) is scaffolding this budget does not measure, so a table renders longer than it costs here. Sections are served first, then tables, then assets, each spending what is left, in document order — admission stops at the first entry that does not fit, and every entry from there on is dropped whole rather than cut mid-row or returned with a shortened caption, counted in `truncation.omittedTables` / `truncation.omittedAssets` and named in `truncation.articles[].omittedTableNames` / `omittedAssetNames`. Applied after `sections`, `maxSections`, `includeReferences`, `includeTables`, and `includeAssets`, so semantic filtering is unaffected. This knob alone bounds only bodies: the response-wide ceiling it implies is this value times the number of articles returned, plus every uncounted field. Use `maxResponseCharacters` for a true whole-response ceiling. Omit for the full body.'),
1190
1287
  maxCharactersPerSection: z
1191
1288
  .number()
1192
1289
  .int()
@@ -1204,7 +1301,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1204
1301
  overflowMode: z
1205
1302
  .enum(['truncate', 'outline'])
1206
1303
  .default('truncate')
1207
- .describe('How to spend `maxCharacters` across an article that exceeds it. truncate: fill sections in document order, so early sections stay whole and sections past the budget are dropped (counted in `truncation.omittedSections`). outline: split the budget evenly so every section keeps its heading, and an excerpt as far as the budget reaches — use it to survey what an article contains before requesting specific `sections`. Ignored when no budget is set, and identical for `source=unpaywall` bodies, which have no headings to preserve.'),
1304
+ .describe('How to spend `maxCharacters` across an article that exceeds it. truncate: fill sections in document order, so early sections stay whole, the section the budget runs out in is cut, and every section or subsection past that point is dropped (counted in `truncation.omittedSections`). outline: split the budget evenly so every section and subsection keeps its heading, and an excerpt as far as the budget reaches — a heading the budget left empty is marked as such in the rendered text. Use it to survey what an article contains before requesting specific `sections`. Ignored when no budget is set, and identical for `source=unpaywall` bodies, which have no headings to preserve.'),
1208
1305
  })
1209
1306
  .refine((v) => [v.pmcids, v.pmids, v.dois].filter((b) => b !== undefined).length === 1, {
1210
1307
  message: 'Provide exactly one of `pmcids`, `pmids`, or `dois` (not zero, not more).',
@@ -1289,10 +1386,18 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1289
1386
  // carry — a `pmids` request recovers articles keyed by PMCID. (#100)
1290
1387
  const inputIdByArticle = new Map();
1291
1388
  // The caller's own spellings of each chain key, so `unavailable[]` and
1292
- // `deferred.ids` report what was submitted. Only the `pmids` branch fills
1293
- // it — a PMID's chain runs on its canonical form — and any other key is its
1294
- // own spelling. (#161)
1389
+ // `deferred.ids` report what was submitted. The `pmids` branch fills it — a
1390
+ // PMID's chain runs on its canonical form — and so does the `dois` branch,
1391
+ // whose chain runs once per DOI however many casings name it. Both also
1392
+ // fold in an input id that resolves to a PMC record another one already
1393
+ // claimed (see `routeToPmc`). Any other key is its own spelling. (#161, #166)
1295
1394
  const callerIds = new Map();
1395
+ const addCallerId = (key, spelling) => {
1396
+ const spellings = callerIds.get(key) ?? [];
1397
+ if (!spellings.includes(spelling))
1398
+ spellings.push(spelling);
1399
+ callerIds.set(key, spellings);
1400
+ };
1296
1401
  const budget = {
1297
1402
  overflowMode: input.overflowMode,
1298
1403
  ...(input.maxCharacters !== undefined && { maxCharacters: input.maxCharacters }),
@@ -1306,17 +1411,34 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1306
1411
  let pmidFallbackCandidates = [];
1307
1412
  let pmcidFallbackCandidates = [];
1308
1413
  let doiCandidates = [];
1414
+ // Send a converter-resolved PMCID to PMC EFetch under the input id that
1415
+ // named it. A second input id the converter places on the same PMC record
1416
+ // — two DOIs for one article — joins the first one's chain as another
1417
+ // spelling of it: the record is fetched once, and both ids are recovered,
1418
+ // or reported unavailable with that chain, rather than the later id taking
1419
+ // the PMCID over and leaving the earlier one with an empty chain.
1420
+ const routeToPmc = (inputId, pmcid) => {
1421
+ const normalized = normalizePmcId(pmcid);
1422
+ const prefixed = withPmcPrefix(normalized);
1423
+ const owner = pmcidToInputId.get(prefixed);
1424
+ if (owner === undefined) {
1425
+ pmcIds.push(normalized);
1426
+ pmcidToInputId.set(prefixed, inputId);
1427
+ return;
1428
+ }
1429
+ if (owner === inputId)
1430
+ return;
1431
+ for (const spelling of callerIds.get(inputId) ?? [inputId])
1432
+ addCallerId(owner, spelling);
1433
+ callerIds.delete(inputId);
1434
+ chainByInput.delete(inputId);
1435
+ };
1309
1436
  if (input.pmids) {
1310
1437
  // The ID Converter parses `00000001` as PMID 1 yet reports it "not found
1311
1438
  // in PMC", and every later stage answers with NCBI's own PMID, so the chain
1312
1439
  // runs once per distinct PMID in its canonical form. (#161)
1313
- for (const id of input.pmids) {
1314
- const pmid = normalizePmid(id);
1315
- const spellings = callerIds.get(pmid) ?? [];
1316
- if (!spellings.includes(id))
1317
- spellings.push(id);
1318
- callerIds.set(pmid, spellings);
1319
- }
1440
+ for (const id of input.pmids)
1441
+ addCallerId(normalizePmid(id), id);
1320
1442
  const pmids = [...callerIds.keys()];
1321
1443
  for (const id of pmids)
1322
1444
  chainByInput.set(id, []);
@@ -1328,9 +1450,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1328
1450
  const pmid = String(r.pmid);
1329
1451
  seen.add(pmid);
1330
1452
  if (r.pmcid) {
1331
- const normalized = normalizePmcId(String(r.pmcid));
1332
- pmcIds.push(normalized);
1333
- pmcidToInputId.set(withPmcPrefix(normalized), pmid);
1453
+ routeToPmc(pmid, String(r.pmcid));
1334
1454
  pmidContext.set(pmid, { pmid, ...(r.doi && { doi: r.doi }) });
1335
1455
  }
1336
1456
  else {
@@ -1354,31 +1474,42 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1354
1474
  }
1355
1475
  }
1356
1476
  else if (input.pmcids) {
1357
- for (const id of input.pmcids)
1358
- chainByInput.set(withPmcPrefix(normalizePmcId(id)), []);
1359
- pmcIds = input.pmcids.map(normalizePmcId);
1477
+ // `PMC123`, `pmc123`, `123` and `PMC0123` name one record; it runs the
1478
+ // chain once. (#170)
1479
+ pmcIds = [...new Set(input.pmcids.map(normalizePmcId))];
1480
+ for (const id of pmcIds)
1481
+ chainByInput.set(withPmcPrefix(id), []);
1360
1482
  }
1361
1483
  else if (input.dois) {
1362
1484
  // Mirror the `pmids` branch: resolve DOI → PMCID via the PMC ID Converter
1363
1485
  // so PMC-indexed DOIs reach PMC EFetch instead of going straight to the
1364
1486
  // EPMC/Unpaywall fallback (which misses articles whose only OA copy is the
1365
1487
  // PMC JATS). DOIs the converter can't place in PMC seed `doiCandidates`.
1366
- const requestedDois = new Set(input.dois);
1367
- for (const doi of input.dois)
1488
+ //
1489
+ // DOIs are case-insensitive, and the converter echoes one casing in
1490
+ // `requested-id` for DOIs that differ only in case, so every casing of a
1491
+ // DOI shares one chain, keyed by the first spelling submitted, and
1492
+ // converter records are matched to it case-insensitively. (#166)
1493
+ const chainKeyByDoi = new Map();
1494
+ for (const doi of input.dois) {
1495
+ const key = chainKeyByDoi.get(doi.toLowerCase()) ?? doi;
1496
+ chainKeyByDoi.set(doi.toLowerCase(), key);
1497
+ addCallerId(key, doi);
1498
+ }
1499
+ const dois = [...callerIds.keys()];
1500
+ for (const doi of dois)
1368
1501
  chainByInput.set(doi, []);
1369
- const records = await getNcbiService().idConvert(input.dois, 'doi', ctx.signal ? { signal: ctx.signal } : undefined);
1502
+ const records = await getNcbiService().idConvert(dois, 'doi', ctx.signal ? { signal: ctx.signal } : undefined);
1370
1503
  const seen = new Set();
1371
1504
  for (const r of records) {
1372
- // The converter echoes the submitted id verbatim in `requested-id`;
1373
- // match on it, not `r.doi` (DOIs aren't case-stable across the API).
1374
- const doi = String(r['requested-id']);
1375
- if (!requestedDois.has(doi))
1505
+ // Match on the echoed `requested-id`, not `r.doi`: the record's own DOI
1506
+ // can be cased differently from anything the caller sent.
1507
+ const doi = chainKeyByDoi.get(String(r['requested-id']).toLowerCase());
1508
+ if (doi === undefined)
1376
1509
  continue;
1377
1510
  seen.add(doi);
1378
1511
  if (r.pmcid) {
1379
- const normalized = normalizePmcId(String(r.pmcid));
1380
- pmcIds.push(normalized);
1381
- pmcidToInputId.set(withPmcPrefix(normalized), doi);
1512
+ routeToPmc(doi, String(r.pmcid));
1382
1513
  }
1383
1514
  else {
1384
1515
  chainByInput.get(doi)?.push({
@@ -1389,7 +1520,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1389
1520
  doiCandidates.push({ doi });
1390
1521
  }
1391
1522
  }
1392
- for (const requested of input.dois) {
1523
+ for (const requested of dois) {
1393
1524
  if (!seen.has(requested)) {
1394
1525
  chainByInput.get(requested)?.push({
1395
1526
  tier: 'pmc',
@@ -1662,7 +1793,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1662
1793
  return {
1663
1794
  pmcId,
1664
1795
  result: candidate.doi
1665
- ? await resolveUnpaywall({ pmcId, doi: candidate.doi, budget }, unpaywall, ctx)
1796
+ ? await resolveUnpaywall({ pmcId, doi: candidate.doi, epmcTitle: candidate.title, budget }, unpaywall, ctx)
1666
1797
  : undefined,
1667
1798
  };
1668
1799
  }));
@@ -1734,7 +1865,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1734
1865
  const outcomes = await Promise.all(pmidFallbackCandidates.map(async (candidate) => ({
1735
1866
  candidate,
1736
1867
  result: candidate.doi
1737
- ? await resolveUnpaywall({ pmid: candidate.pmid, doi: candidate.doi, budget }, unpaywall, ctx)
1868
+ ? await resolveUnpaywall({ pmid: candidate.pmid, doi: candidate.doi, epmcTitle: candidate.title, budget }, unpaywall, ctx)
1738
1869
  : undefined,
1739
1870
  })));
1740
1871
  for (const { candidate, result } of outcomes) {
@@ -1777,7 +1908,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1777
1908
  // doesn't reject under normal operation.
1778
1909
  const outcomes = await Promise.all(doiCandidates.map(async (c) => ({
1779
1910
  doi: c.doi,
1780
- result: await resolveUnpaywall({ doi: c.doi, budget }, unpaywall, ctx),
1911
+ result: await resolveUnpaywall({ doi: c.doi, epmcTitle: c.title, budget }, unpaywall, ctx),
1781
1912
  })));
1782
1913
  for (const { doi, result } of outcomes) {
1783
1914
  if ('article' in result) {
@@ -1848,8 +1979,9 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1848
1979
  if (index === -1)
1849
1980
  continue;
1850
1981
  const [dropped] = truncatedArticles.splice(index, 1);
1851
- if (dropped)
1852
- omittedSections -= countOmittedSections(dropped, input.overflowMode);
1982
+ if (dropped) {
1983
+ omittedSections -= countDroppedSections(dropped.sections ?? [], input.overflowMode);
1984
+ }
1853
1985
  }
1854
1986
  }
1855
1987
  ctx.log.info('pubmed_fetch_fulltext completed', {
@@ -1957,13 +2089,29 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1957
2089
  lines.push('');
1958
2090
  const t = truncationById.get(articleDisplayId(a));
1959
2091
  if (a.source === 'pmc')
1960
- formatPmcArticle(a, lines, t);
2092
+ formatPmcArticle(a, lines, t, result.truncation?.mode);
1961
2093
  else
1962
2094
  formatUnpaywallArticle(a, lines, t);
1963
2095
  }
1964
2096
  return [{ type: 'text', text: lines.join('\n') }];
1965
2097
  },
1966
2098
  });
2099
+ /**
2100
+ * Merge what a Europe PMC hit carried onto a candidate bound for Unpaywall. A
2101
+ * DOI the candidate already holds wins — for `dois` input it is the caller's
2102
+ * own identifier.
2103
+ */
2104
+ function carryEpmcHit(candidate, hit) {
2105
+ return {
2106
+ ...candidate,
2107
+ ...(hit.doi && !candidate.doi && { doi: hit.doi }),
2108
+ ...(hit.title && { title: hit.title }),
2109
+ };
2110
+ }
2111
+ /** Display-ready plain text, or `undefined` when nothing is left once markup is stripped. */
2112
+ function nonEmptyDisplayText(raw) {
2113
+ return (raw && toDisplayText(raw)) || undefined;
2114
+ }
1967
2115
  /**
1968
2116
  * Run the Europe PMC step against everything that fell through PMC EFetch
1969
2117
  * plus any direct DOI input. Each candidate goes through search-by-best-id →
@@ -1983,24 +2131,28 @@ async function runEpmcStage(epmc, args) {
1983
2131
  }
1984
2132
  if (search.kind === 'miss')
1985
2133
  return { c, outcome: { kind: 'miss' } };
1986
- const doi = search.hit.doi ? { doi: search.hit.doi } : {};
2134
+ const title = nonEmptyDisplayText(search.hit.title);
2135
+ const hit = {
2136
+ ...(search.hit.doi && { doi: search.hit.doi }),
2137
+ ...(title && { title }),
2138
+ };
1987
2139
  const fetched = await fetchEpmcArticle(epmc, search.hit, args, contextPmid);
1988
2140
  if (fetched.kind === 'error') {
1989
- return { c, ...doi, outcome: { kind: 'service-error', detail: fetched.detail } };
2141
+ return { c, ...hit, outcome: { kind: 'service-error', detail: fetched.detail } };
1990
2142
  }
1991
2143
  if (fetched.kind === 'no-fulltext') {
1992
2144
  return {
1993
2145
  c,
1994
- ...doi,
2146
+ ...hit,
1995
2147
  outcome: { kind: 'no-fulltext', ...(fetched.detail && { detail: fetched.detail }) },
1996
2148
  };
1997
2149
  }
1998
2150
  if (fetched.kind === 'no-body') {
1999
- return { c, ...doi, outcome: { kind: 'no-body', detail: fetched.detail } };
2151
+ return { c, ...hit, outcome: { kind: 'no-body', detail: fetched.detail } };
2000
2152
  }
2001
2153
  return {
2002
2154
  c,
2003
- ...doi,
2155
+ ...hit,
2004
2156
  outcome: { kind: 'hit' },
2005
2157
  article: fetched.article,
2006
2158
  sectionFilterMiss: fetched.sectionFilterMiss,
@@ -2052,25 +2204,25 @@ async function runEpmcStage(epmc, args) {
2052
2204
  pmidOutcomes.set(run.c.pmid, run.outcome);
2053
2205
  if (run.article)
2054
2206
  collectHit(run.c.pmid, { ...run, article: run.article });
2055
- // A DOI the EPMC hit carried is evidence the next stage needs, whatever the
2056
- // fetch outcome was — merge it in like the pmcid branch below, so Unpaywall
2057
- // gets it without a redundant PubMed metadata round-trip. (#119)
2207
+ // What the EPMC hit carried is evidence the next stage needs, whatever the
2208
+ // fetch outcome was: its DOI spares Unpaywall a PubMed metadata round-trip
2209
+ // (#119), and its title backs up Unpaywall's own record (#144).
2058
2210
  else
2059
- remainingPmid.push(run.doi && !run.c.doi ? { ...run.c, doi: run.doi } : run.c);
2211
+ remainingPmid.push(carryEpmcHit(run.c, run));
2060
2212
  }
2061
2213
  for (const run of pmcidResults) {
2062
2214
  pmcidOutcomes.set(run.c.normalized, run.outcome);
2063
2215
  if (run.article)
2064
2216
  collectHit(run.c.normalized, { ...run, article: run.article });
2065
2217
  else
2066
- remainingPmcid.push(run.doi && !run.c.c.doi ? { ...run.c.c, doi: run.doi } : run.c.c);
2218
+ remainingPmcid.push(carryEpmcHit(run.c.c, run));
2067
2219
  }
2068
2220
  for (const run of doiResults) {
2069
2221
  doiOutcomes.set(run.c.doi, run.outcome);
2070
2222
  if (run.article)
2071
2223
  collectHit(run.c.doi, { ...run, article: run.article });
2072
2224
  else
2073
- remainingDoi.push(run.c);
2225
+ remainingDoi.push(carryEpmcHit(run.c, run));
2074
2226
  }
2075
2227
  return {
2076
2228
  articles,
@@ -2220,9 +2372,15 @@ async function fetchPubmedDois(pmids, signal) {
2220
2372
  * Resolve a DOI to an open-access article via Unpaywall. `pmcId` and `pmid`,
2221
2373
  * when set, are stamped onto the resulting article so the branch that requested
2222
2374
  * it carries its identifier through — Unpaywall itself only knows the DOI.
2375
+ *
2376
+ * The article's title is the first one found in Unpaywall's own record, then
2377
+ * `epmcTitle` — the Europe PMC record the chain searched for this id — then,
2378
+ * for HTML content only, the title the extractor detects on the page. None is
2379
+ * ever invented: with no source carrying one, the article has no title. The
2380
+ * journal name and year come from Unpaywall's record alone. (#144)
2223
2381
  */
2224
2382
  async function resolveUnpaywall(args, service, ctx) {
2225
- const { pmcId, pmid, doi, budget } = args;
2383
+ const { pmcId, pmid, doi, epmcTitle, budget } = args;
2226
2384
  const requestedIds = { ...(pmcId && { pmcId }), ...(pmid && { pmid }) };
2227
2385
  /** Budget the extracted body, then pair the article with its accounting. */
2228
2386
  const budgeted = (build, content) => {
@@ -2260,6 +2418,15 @@ async function resolveUnpaywall(args, service, ctx) {
2260
2418
  ctx.log.warning('Unpaywall content fetch failed', { doi, error: detail });
2261
2419
  return { unavailable: { reason: 'fetch-failed', detail } };
2262
2420
  }
2421
+ const recordTitle = nonEmptyDisplayText(resolution.title) ?? epmcTitle;
2422
+ const record = {
2423
+ ...requestedIds,
2424
+ doi,
2425
+ sourceUrl: content.fetchedUrl,
2426
+ location: resolution.location,
2427
+ journalName: nonEmptyDisplayText(resolution.journalName),
2428
+ year: resolution.year,
2429
+ };
2263
2430
  try {
2264
2431
  if (content.kind === 'html') {
2265
2432
  const extracted = await htmlExtractor.extract(content.body, {
@@ -2276,13 +2443,10 @@ async function resolveUnpaywall(args, service, ctx) {
2276
2443
  };
2277
2444
  }
2278
2445
  return budgeted((text) => buildUnpaywallArticle({
2279
- ...requestedIds,
2280
- doi,
2281
- sourceUrl: content.fetchedUrl,
2282
- location: resolution.location,
2446
+ ...record,
2283
2447
  contentFormat: 'html-markdown',
2284
2448
  content: text,
2285
- title: extracted.title,
2449
+ title: recordTitle ?? extracted.title,
2286
2450
  wordCount: extracted.wordCount,
2287
2451
  }), body);
2288
2452
  }
@@ -2294,12 +2458,10 @@ async function resolveUnpaywall(args, service, ctx) {
2294
2458
  };
2295
2459
  }
2296
2460
  return budgeted((body) => buildUnpaywallArticle({
2297
- ...requestedIds,
2298
- doi,
2299
- sourceUrl: content.fetchedUrl,
2300
- location: resolution.location,
2461
+ ...record,
2301
2462
  contentFormat: 'pdf-text',
2302
2463
  content: body,
2464
+ title: recordTitle,
2303
2465
  totalPages: extracted.totalPages,
2304
2466
  }), text);
2305
2467
  }
@@ -2324,6 +2486,8 @@ function buildUnpaywallArticle(args) {
2324
2486
  sourceUrl: args.sourceUrl,
2325
2487
  content: args.content,
2326
2488
  ...(args.title && { title: args.title }),
2489
+ ...(args.journalName && { journalName: args.journalName }),
2490
+ ...(args.year !== undefined && { year: args.year }),
2327
2491
  ...(args.wordCount !== undefined && { wordCount: args.wordCount }),
2328
2492
  ...(args.totalPages !== undefined && { totalPages: args.totalPages }),
2329
2493
  ...(location.license && { license: location.license }),
@@ -2481,16 +2645,39 @@ function formatTruncation(t, lines) {
2481
2645
  ? ''
2482
2646
  : `, ${a.omittedAssets} asset(s) dropped whole: ${(a.omittedAssetNames ?? []).join(', ')}`;
2483
2647
  lines.push(`- ${a.id} (${a.source}): ${a.returnedCharacters} of ${a.originalCharacters} characters${tablesDropped}${assetsDropped}`);
2484
- for (const s of a.sections ?? []) {
2485
- lines.push(` - ${s.title ?? 'untitled section'} — ${s.returnedCharacters} of ${s.originalCharacters} characters (truncated: ${s.truncated})`);
2486
- }
2648
+ for (const s of a.sections ?? [])
2649
+ formatLedgerEntry(s, lines, t.mode, 1);
2487
2650
  }
2488
2651
  }
2652
+ /**
2653
+ * One section's ledger line, then its subsections' one level deeper. The line
2654
+ * names the section by {@link sectionHeading}, so it reads exactly as the
2655
+ * section's heading does in the body. An entry the budget emptied says what
2656
+ * became of it — dropped in `truncate` mode, kept as a bare heading in
2657
+ * `outline` mode. (#143, #148)
2658
+ */
2659
+ function formatLedgerEntry(entry, lines, mode, depth) {
2660
+ const fate = isBudgetEmptied(entry)
2661
+ ? mode === 'truncate'
2662
+ ? ' — dropped'
2663
+ : ' — heading only'
2664
+ : '';
2665
+ lines.push(`${' '.repeat(depth)}- ${sectionHeading(entry)} — ${entry.returnedCharacters} of ${entry.originalCharacters} characters (truncated: ${entry.truncated})${fate}`);
2666
+ for (const sub of entry.subsections ?? [])
2667
+ formatLedgerEntry(sub, lines, mode, depth + 1);
2668
+ }
2489
2669
  /** Per-article inline marker so a reader of one article's body knows it is partial. */
2490
2670
  function truncationNote(t) {
2491
2671
  return `\n> Body shortened to fit the requested character budget — ${t.returnedCharacters} of ${t.originalCharacters} characters returned. See \`truncation\` for per-section counts.`;
2492
2672
  }
2493
- function formatPmcArticle(a, lines, truncation) {
2673
+ /**
2674
+ * The marker an `outline`-mode heading carries when the budget left it no text,
2675
+ * so it reads as withheld rather than as a heading over nothing. (#143)
2676
+ */
2677
+ function emptiedSectionNote(originalCharacters) {
2678
+ return `> Text omitted to fit the requested character budget — 0 of ${originalCharacters} characters returned.`;
2679
+ }
2680
+ function formatPmcArticle(a, lines, truncation, mode) {
2494
2681
  // Render-time only — `structuredContent.articles[].title` keeps the
2495
2682
  // plain-text value the JATS parser produced. (#102)
2496
2683
  lines.push(`### ${escapeMarkdownInline(a.title ?? articleDisplayId(a))}`);
@@ -2549,8 +2736,13 @@ function formatPmcArticle(a, lines, truncation) {
2549
2736
  lines.push(truncationNote(truncation));
2550
2737
  if (a.abstract)
2551
2738
  lines.push(`\n#### Abstract\n${a.abstract}`);
2552
- for (const sec of a.sections)
2553
- formatSection(sec, lines, 4);
2739
+ // `outline` mode keeps every section and subsection, so each one lines up
2740
+ // with its ledger entry by position. `truncate` mode drops the emptied ones —
2741
+ // positions no longer line up, and nothing left needs a marker.
2742
+ const ledger = mode === 'outline' ? truncation?.sections : undefined;
2743
+ a.sections.forEach((sec, i) => {
2744
+ formatSection(sec, lines, 4, ledger?.[i]);
2745
+ });
2554
2746
  if (a.tables?.length)
2555
2747
  formatTables(a.tables, lines);
2556
2748
  if (a.assets?.length)
@@ -2677,6 +2869,10 @@ function formatUnpaywallArticle(a, lines, truncation) {
2677
2869
  : 'Unpaywall (PDF → plain text)';
2678
2870
  lines.push(`### ${escapeMarkdownInline(heading)}`);
2679
2871
  lines.push(`**Source:** ${formatLabel}`);
2872
+ if (a.journalName)
2873
+ lines.push(`**Journal:** ${escapeMarkdownInline(a.journalName)}`);
2874
+ if (a.year !== undefined)
2875
+ lines.push(`**Year:** ${a.year}`);
2680
2876
  if (a.pmcId)
2681
2877
  lines.push(`**PMCID:** ${a.pmcId}`);
2682
2878
  if (a.pmid)
@@ -2712,20 +2908,43 @@ function formatPmcAuthor(au) {
2712
2908
  function formatHeading(label, title) {
2713
2909
  return label ? `${label} ${title}` : title;
2714
2910
  }
2911
+ /** How a section with no title and no label is named — in its body heading and its ledger line. */
2912
+ const UNTITLED_SECTION_LABEL = 'untitled section';
2913
+ /**
2914
+ * The name a section goes by in `content[]`: its label and title, else its
2915
+ * label alone, else {@link UNTITLED_SECTION_LABEL}. The body heading and the
2916
+ * truncation ledger line both take it from here, so the two always read the
2917
+ * same. Render-time escaped like every other upstream string interpolated into
2918
+ * a line, so a title's `*`, `_`, `` ` `` or `[` cannot restyle the heading;
2919
+ * `structuredContent` keeps the plain title. (#148, #169)
2920
+ */
2921
+ function sectionHeading(section) {
2922
+ if (section.title)
2923
+ return escapeMarkdownInline(formatHeading(section.label, section.title));
2924
+ return section.label ? escapeMarkdownInline(section.label) : UNTITLED_SECTION_LABEL;
2925
+ }
2715
2926
  /**
2716
2927
  * Render one body section and everything nested under it, one markdown heading
2717
2928
  * level per nesting level. Walks the full depth the output schema carries, so
2718
2929
  * `content[]` shows every section `structuredContent` does. Headings stop
2719
2930
  * deepening at `######`, the deepest markdown supports. (#112)
2931
+ *
2932
+ * Every section gets a heading — an untitled one included, under the name the
2933
+ * truncation ledger gives it — so its text never runs on from the block before
2934
+ * it. `ledger` is the section's `outline`-mode accounting, when there is one: a
2935
+ * section it shows the budget emptied carries a marker under its heading.
2936
+ * (#143, #148)
2720
2937
  */
2721
- function formatSection(section, lines, depth) {
2722
- if (section.title) {
2723
- lines.push(`\n${'#'.repeat(Math.min(depth, 6))} ${formatHeading(section.label, section.title)}`);
2724
- }
2938
+ function formatSection(section, lines, depth, ledger) {
2939
+ lines.push(`\n${'#'.repeat(Math.min(depth, 6))} ${sectionHeading(section)}`);
2725
2940
  if (section.text)
2726
2941
  lines.push(section.text);
2727
- for (const sub of section.subsections ?? [])
2728
- formatSection(sub, lines, depth + 1);
2942
+ else if (ledger && isBudgetEmptied(ledger)) {
2943
+ lines.push(emptiedSectionNote(ledger.originalCharacters));
2944
+ }
2945
+ section.subsections?.forEach((sub, i) => {
2946
+ formatSection(sub, lines, depth + 1, ledger?.subsections?.[i]);
2947
+ });
2729
2948
  }
2730
2949
  /**
2731
2950
  * Strip absolute URLs from chain detail strings. Upstream errors (e.g.