@cyanheads/pubmed-mcp-server 2.10.9 → 2.10.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. package/AGENTS.md +1 -1
  2. package/CLAUDE.md +1 -1
  3. package/README.md +13 -5
  4. package/dist/mcp-server/tools/definitions/_schemas.d.ts +16 -0
  5. package/dist/mcp-server/tools/definitions/_schemas.d.ts.map +1 -1
  6. package/dist/mcp-server/tools/definitions/_schemas.js +20 -0
  7. package/dist/mcp-server/tools/definitions/_schemas.js.map +1 -1
  8. package/dist/mcp-server/tools/definitions/convert-ids.tool.d.ts +6 -0
  9. package/dist/mcp-server/tools/definitions/convert-ids.tool.d.ts.map +1 -1
  10. package/dist/mcp-server/tools/definitions/convert-ids.tool.js +28 -3
  11. package/dist/mcp-server/tools/definitions/convert-ids.tool.js.map +1 -1
  12. package/dist/mcp-server/tools/definitions/fetch-articles.tool.d.ts +27 -0
  13. package/dist/mcp-server/tools/definitions/fetch-articles.tool.d.ts.map +1 -1
  14. package/dist/mcp-server/tools/definitions/fetch-articles.tool.js +166 -24
  15. package/dist/mcp-server/tools/definitions/fetch-articles.tool.js.map +1 -1
  16. package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.d.ts +18 -0
  17. package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.d.ts.map +1 -1
  18. package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.js +307 -68
  19. package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.js.map +1 -1
  20. package/dist/mcp-server/tools/definitions/find-related.tool.d.ts +12 -0
  21. package/dist/mcp-server/tools/definitions/find-related.tool.d.ts.map +1 -1
  22. package/dist/mcp-server/tools/definitions/find-related.tool.js +171 -27
  23. package/dist/mcp-server/tools/definitions/find-related.tool.js.map +1 -1
  24. package/dist/mcp-server/tools/definitions/format-citations.tool.d.ts.map +1 -1
  25. package/dist/mcp-server/tools/definitions/format-citations.tool.js +11 -13
  26. package/dist/mcp-server/tools/definitions/format-citations.tool.js.map +1 -1
  27. package/dist/mcp-server/tools/definitions/lookup-citation.tool.d.ts.map +1 -1
  28. package/dist/mcp-server/tools/definitions/lookup-citation.tool.js +42 -8
  29. package/dist/mcp-server/tools/definitions/lookup-citation.tool.js.map +1 -1
  30. package/dist/mcp-server/tools/definitions/lookup-mesh.tool.d.ts +6 -0
  31. package/dist/mcp-server/tools/definitions/lookup-mesh.tool.d.ts.map +1 -1
  32. package/dist/mcp-server/tools/definitions/lookup-mesh.tool.js +14 -3
  33. package/dist/mcp-server/tools/definitions/lookup-mesh.tool.js.map +1 -1
  34. package/dist/mcp-server/tools/definitions/search-articles.tool.d.ts +10 -0
  35. package/dist/mcp-server/tools/definitions/search-articles.tool.d.ts.map +1 -1
  36. package/dist/mcp-server/tools/definitions/search-articles.tool.js +52 -5
  37. package/dist/mcp-server/tools/definitions/search-articles.tool.js.map +1 -1
  38. package/dist/mcp-server/tools/definitions/spell-check.tool.d.ts +6 -0
  39. package/dist/mcp-server/tools/definitions/spell-check.tool.d.ts.map +1 -1
  40. package/dist/mcp-server/tools/definitions/spell-check.tool.js +15 -3
  41. package/dist/mcp-server/tools/definitions/spell-check.tool.js.map +1 -1
  42. package/dist/services/error-contracts.d.ts +39 -2
  43. package/dist/services/error-contracts.d.ts.map +1 -1
  44. package/dist/services/error-contracts.js +43 -2
  45. package/dist/services/error-contracts.js.map +1 -1
  46. package/dist/services/ncbi/formatting/citation-formatter.d.ts +0 -29
  47. package/dist/services/ncbi/formatting/citation-formatter.d.ts.map +1 -1
  48. package/dist/services/ncbi/formatting/citation-formatter.js +381 -30
  49. package/dist/services/ncbi/formatting/citation-formatter.js.map +1 -1
  50. package/dist/services/ncbi/ncbi-service.d.ts +3 -2
  51. package/dist/services/ncbi/ncbi-service.d.ts.map +1 -1
  52. package/dist/services/ncbi/ncbi-service.js +22 -10
  53. package/dist/services/ncbi/ncbi-service.js.map +1 -1
  54. package/dist/services/ncbi/parsing/article-parser.d.ts +45 -2
  55. package/dist/services/ncbi/parsing/article-parser.d.ts.map +1 -1
  56. package/dist/services/ncbi/parsing/article-parser.js +208 -1
  57. package/dist/services/ncbi/parsing/article-parser.js.map +1 -1
  58. package/dist/services/ncbi/parsing/esummary-parser.d.ts.map +1 -1
  59. package/dist/services/ncbi/parsing/esummary-parser.js +32 -1
  60. package/dist/services/ncbi/parsing/esummary-parser.js.map +1 -1
  61. package/dist/services/ncbi/parsing/pmc-article-parser.d.ts +24 -4
  62. package/dist/services/ncbi/parsing/pmc-article-parser.d.ts.map +1 -1
  63. package/dist/services/ncbi/parsing/pmc-article-parser.js +505 -57
  64. package/dist/services/ncbi/parsing/pmc-article-parser.js.map +1 -1
  65. package/dist/services/ncbi/response-handler.d.ts.map +1 -1
  66. package/dist/services/ncbi/response-handler.js +42 -1
  67. package/dist/services/ncbi/response-handler.js.map +1 -1
  68. package/dist/services/ncbi/types.d.ts +215 -2
  69. package/dist/services/ncbi/types.d.ts.map +1 -1
  70. package/dist/services/openalex/api-client.d.ts +13 -4
  71. package/dist/services/openalex/api-client.d.ts.map +1 -1
  72. package/dist/services/openalex/api-client.js +19 -8
  73. package/dist/services/openalex/api-client.js.map +1 -1
  74. package/dist/services/openalex/openalex-service.d.ts +34 -17
  75. package/dist/services/openalex/openalex-service.d.ts.map +1 -1
  76. package/dist/services/openalex/openalex-service.js +119 -35
  77. package/dist/services/openalex/openalex-service.js.map +1 -1
  78. package/dist/services/openalex/types.d.ts +34 -1
  79. package/dist/services/openalex/types.d.ts.map +1 -1
  80. package/dist/services/openalex/types.js +20 -0
  81. package/dist/services/openalex/types.js.map +1 -1
  82. package/package.json +1 -1
  83. package/server.json +3 -3
@@ -9,31 +9,90 @@
9
9
  */
10
10
  import { attrOf, childrenOf, findAll, findAllDescendants, findOne, isTextNode, rawTextContent, tagNameOf, textContent, textContentExcluding, textOf, } from './pmc-xml-helpers.js';
11
11
  /**
12
- * Block content extracted into its own field and therefore excluded from the
13
- * prose a paragraph contributes. A `<table-wrap>` nested inside a `<p>` would
14
- * otherwise be flattened into the surrounding sentence, concatenating adjacent
15
- * cell values into numbers that never existed. (#111)
12
+ * Block content extracted into a field of its own and therefore left out of the
13
+ * prose a section contributes. Each one nested inside a `<p>` would otherwise be
14
+ * flattened into the surrounding sentence — a `<table-wrap>` concatenating
15
+ * adjacent cell values into numbers that never existed, a `<fig>` gluing its
16
+ * label and caption onto the sentence terminator before it. (#111, #130)
16
17
  */
17
- const LIFTED_BLOCK_TAGS = new Set(['table-wrap']);
18
- /** Prose of one `<p>`, with lifted block content left out. */
19
- function paragraphText(paragraph) {
20
- return textContentExcluding(paragraph, LIFTED_BLOCK_TAGS);
21
- }
18
+ const LIFTED_BLOCK_TAGS = new Set([
19
+ 'table-wrap',
20
+ 'fig',
21
+ 'supplementary-material',
22
+ ]);
23
+ /**
24
+ * JATS block-level elements, which interrupt prose rather than reading inside
25
+ * it. Membership decides *placement*, not representation: every one of these
26
+ * flushes the prose run in progress and contributes at block position, whether
27
+ * it renders (a `<list>`), lifts to its own field leaving a marker (a `<fig>`),
28
+ * or contributes nothing (a `<table-wrap>`, already in `tables[]`). Anything not
29
+ * named here is inline markup — `<italic>`, `<xref>`, `<sup>`,
30
+ * `<inline-formula>` — and reads inside the sentence as it should.
31
+ *
32
+ * The set is the JATS block-display class rather than the tags observed in any
33
+ * one draw: an element the walk does not enumerate degrades to flattened text at
34
+ * block position, and getting it there is what stops the fusion. (#130)
35
+ *
36
+ * Membership governs only an element nested inside another. Every caller that
37
+ * walks a container renders its children one at a time, so a direct `<sec>` or
38
+ * `<body>` child already contributes at block position whether or not it is
39
+ * named here; the live question per tag is what a paragraph-internal occurrence
40
+ * should do. `address`, `related-article` and `related-object` are in the JATS
41
+ * model both as display blocks and inline within a `<p>`, and they stay: a
42
+ * wrong split costs a paragraph break with every character still present and in
43
+ * order, while a wrong fusion fabricates adjacency the source never had, which
44
+ * is the defect this walk exists to prevent — when a tag reads both ways,
45
+ * splitting is the recoverable error.
46
+ *
47
+ * `<alternatives>` is the one element deliberately absent. It is a container
48
+ * for equivalent renderings of a single object and never a block in its own
49
+ * right, so its placement is whatever holds it: naming it here broke the
50
+ * standard `<inline-formula><alternatives><tex-math/><mml:math/></alternatives>`
51
+ * deposit out of the sentence it belonged to and split that sentence in two.
52
+ * `<disp-formula>`, `<table-wrap>` and `<fig>` resolve their own
53
+ * `<alternatives>` children, so nothing depends on it flushing the run. (#130)
54
+ */
55
+ const BLOCK_TAGS = new Set([
56
+ ...LIFTED_BLOCK_TAGS,
57
+ 'address',
58
+ 'array',
59
+ 'boxed-text',
60
+ 'chem-struct-wrap',
61
+ 'code',
62
+ 'def-list',
63
+ 'disp-formula',
64
+ 'disp-formula-group',
65
+ 'disp-quote',
66
+ 'fig-group',
67
+ 'graphic',
68
+ 'list',
69
+ 'media',
70
+ 'preformat',
71
+ 'ref-list',
72
+ 'related-article',
73
+ 'related-object',
74
+ 'speech',
75
+ 'statement',
76
+ 'table-wrap-group',
77
+ 'verse-group',
78
+ ]);
22
79
  /** True when a lifted block sits anywhere in this subtree. */
23
80
  function containsLiftedBlock(node) {
24
81
  return childrenOf(node).some((child) => LIFTED_BLOCK_TAGS.has(tagNameOf(child) ?? '') || containsLiftedBlock(child));
25
82
  }
26
83
  /**
27
84
  * True when a `<sec>` carried block content that was extracted into its own
28
- * field — a `<table-wrap>` sitting beside its paragraphs, or nested inside one.
85
+ * field — a `<table-wrap>` or `<fig>` sitting beside its paragraphs, or nested
86
+ * inside one.
29
87
  *
30
88
  * This is what separates a section emptied by the lift from one that never had
31
89
  * readable prose. The first is a real heading a reader needs in order to place
32
- * the table that names it; the second is a structural wrapper — a `<sec>` around
33
- * a `<ref-list>`, say — which would arrive as a stray empty "References" entry
34
- * if every empty section survived. Only this section's own children are
35
- * considered: a `<table-wrap>` deeper down belongs to the nested `<sec>` that
36
- * holds it, and that section survives on its own account. (#111, #116)
90
+ * the table or figure that names it; the second is a structural wrapper — a
91
+ * `<sec>` around a `<ref-list>`, say — which would arrive as a stray empty
92
+ * "References" entry if every empty section survived. Only this section's own
93
+ * children are considered: a `<table-wrap>` deeper down belongs to the nested
94
+ * `<sec>` that holds it, and that section survives on its own account.
95
+ * (#111, #116, #130)
37
96
  */
38
97
  function hasLiftedBlockContent(sec) {
39
98
  return childrenOf(sec).some((child) => {
@@ -109,7 +168,12 @@ function extractJournal(journalMeta, articleMeta) {
109
168
  const fpage = articleMeta ? textContent(findOne(articleMeta, 'fpage')) : '';
110
169
  const lpage = articleMeta ? textContent(findOne(articleMeta, 'lpage')) : '';
111
170
  const pages = fpage && lpage ? `${fpage}-${lpage}` : fpage || undefined;
112
- if (!title && !issn && !volume && !issue && !pages)
171
+ // Article number for the roughly half of PMC deposits that carry
172
+ // <elocation-id> and no <fpage>. Distinct from pages, never a stand-in for it.
173
+ const elocationId = articleMeta
174
+ ? textContent(findOne(articleMeta, 'elocation-id')) || undefined
175
+ : undefined;
176
+ if (!title && !issn && !volume && !issue && !pages && !elocationId)
113
177
  return;
114
178
  return {
115
179
  ...(title && { title }),
@@ -117,6 +181,7 @@ function extractJournal(journalMeta, articleMeta) {
117
181
  ...(volume && { volume }),
118
182
  ...(issue && { issue }),
119
183
  ...(pages && { pages }),
184
+ ...(elocationId && { elocationId }),
120
185
  };
121
186
  }
122
187
  function extractPubDate(articleMeta) {
@@ -143,10 +208,28 @@ function extractPubDate(articleMeta) {
143
208
  };
144
209
  }
145
210
  // ─── Abstract & Keywords ────────────────────────────────────────────────────
211
+ /**
212
+ * Read the article's own abstract.
213
+ *
214
+ * JATS permits several `<abstract>` elements under `<article-meta>`,
215
+ * distinguished by `@abstract-type`, and publishers routinely deposit a
216
+ * graphical abstract, author highlights or an executive summary alongside the
217
+ * real one — 21 of 68 records in a validation draw carry more than one, and in
218
+ * 4 of those the first is not the untyped element. The untyped one is the
219
+ * article's own abstract, so it is preferred and a typed one used only when the
220
+ * record deposits nothing else, mirroring the ladder {@link extractPubDate}
221
+ * applies to `<pub-date>`. (#134)
222
+ *
223
+ * Content is read through the same flow walk the body uses, so a `<fig>` in an
224
+ * abstract leaves its marker and its caption reaches `assets[]` alone —
225
+ * {@link extractPmcAssets} already walks `<front>` — instead of concatenating
226
+ * into the prose, and a `<list>` renders as list lines rather than a token run.
227
+ */
146
228
  function extractAbstract(articleMeta) {
147
229
  if (!articleMeta)
148
230
  return;
149
- const abstractNode = findOne(articleMeta, 'abstract');
231
+ const abstracts = findAll(articleMeta, 'abstract');
232
+ const abstractNode = abstracts.find((node) => !attrOf(node, 'abstract-type')) ?? abstracts[0];
150
233
  if (!abstractNode)
151
234
  return;
152
235
  const sections = findAll(abstractNode, 'sec');
@@ -154,10 +237,7 @@ function extractAbstract(articleMeta) {
154
237
  const parts = [];
155
238
  for (const sec of sections) {
156
239
  const title = textContent(findOne(sec, 'title'));
157
- const text = findAll(sec, 'p')
158
- .map((p) => textContent(p))
159
- .filter(Boolean)
160
- .join(' ');
240
+ const text = abstractProse(sec);
161
241
  if (title && text)
162
242
  parts.push(`${title}: ${text}`);
163
243
  else if (text)
@@ -165,14 +245,20 @@ function extractAbstract(articleMeta) {
165
245
  }
166
246
  return parts.join('\n\n').trim() || undefined;
167
247
  }
168
- const paragraphs = findAll(abstractNode, 'p');
169
- if (paragraphs.length > 0) {
170
- return (paragraphs
171
- .map((p) => textContent(p))
172
- .filter(Boolean)
173
- .join(' ') || undefined);
174
- }
175
- return textContent(abstractNode) || undefined;
248
+ return abstractProse(abstractNode) || undefined;
249
+ }
250
+ /**
251
+ * The prose of an `<abstract>` or one of its `<sec>`s: every child but the
252
+ * heading the caller reports separately, space-joined. The single space is the
253
+ * spacing the paragraph-only read this replaces already produced, so a record
254
+ * depositing one untyped abstract of plain `<p>`s comes back byte-identical.
255
+ */
256
+ function abstractProse(node) {
257
+ const prose = childrenOf(node).filter((child) => {
258
+ const tag = tagNameOf(child);
259
+ return tag !== 'title' && tag !== 'label';
260
+ });
261
+ return flowBlocks(prose, STATEMENT_BLOCK_TAGS).join(' ');
176
262
  }
177
263
  function extractKeywords(articleMeta) {
178
264
  if (!articleMeta)
@@ -187,48 +273,348 @@ function extractKeywords(articleMeta) {
187
273
  }
188
274
  return keywords;
189
275
  }
276
+ // ─── Block Rendering ────────────────────────────────────────────────────────
277
+ /** Indent one nesting level of a `<list>`/`<def-list>` nested in a `<list-item>`. */
278
+ const NESTED_LIST_INDENT = ' ';
279
+ /** Close the prose run in progress, discarding it when it holds only whitespace. */
280
+ function flushRun(flow) {
281
+ const text = flow.run.replace(/\s+/g, ' ').trim();
282
+ if (text)
283
+ flow.out.push(text);
284
+ flow.run = '';
285
+ }
286
+ /**
287
+ * Walk a mixed-content child list, accumulating inline text into a prose run and
288
+ * emitting every block-level element at its own document position instead.
289
+ *
290
+ * The recursion through non-block elements is what makes the split reliable: a
291
+ * `<fig>` nested two elements deep inside a `<p>` still flushes the sentence
292
+ * before it rather than being flattened into the middle of one. Inline markup —
293
+ * `<italic>`, `<xref>`, `<sup>` — is transparent here and reads in place, which
294
+ * is the behavior {@link textContent} already gave paragraphs. (#130)
295
+ *
296
+ * `blockTags` names what interrupts the run: {@link BLOCK_TAGS} for article
297
+ * prose, {@link STATEMENT_BLOCK_TAGS} for a container whose `<title>` and `<p>`
298
+ * children are separate statements rather than one continuous sentence.
299
+ */
300
+ function walkFlow(nodes, flow, blockTags) {
301
+ for (const node of nodes) {
302
+ if (isTextNode(node)) {
303
+ flow.run += textOf(node);
304
+ continue;
305
+ }
306
+ const tag = tagNameOf(node) ?? '';
307
+ if (blockTags.has(tag)) {
308
+ flushRun(flow);
309
+ const rendered = renderBlock(node);
310
+ if (rendered)
311
+ flow.out.push(rendered);
312
+ continue;
313
+ }
314
+ walkFlow(childrenOf(node), flow, blockTags);
315
+ }
316
+ }
317
+ /** Ordered text blocks contributed by one `<p>` — or any other flow container. */
318
+ function flowBlocks(nodes, blockTags = BLOCK_TAGS) {
319
+ const flow = { out: [], run: '' };
320
+ walkFlow(nodes, flow, blockTags);
321
+ flushRun(flow);
322
+ return flow.out;
323
+ }
324
+ /**
325
+ * Block tags for a container whose `<title>` and `<p>` children each carry a
326
+ * statement of their own — a `<caption>`, an `<abstract>`, an abstract `<sec>`.
327
+ * Their children carry no punctuation between them, so the concatenating read
328
+ * ran a caption's title straight into its first sentence
329
+ * (`…observational constraints6.Cloud susceptibilities…`, 60 of 342 captions in
330
+ * a 68-record draw) and would run two sibling paragraphs together. Only a
331
+ * `<title>` or `<p>` boundary separates: inline markup between two text runs
332
+ * stays transparent, so `Expression of <italic>NF1</italic> across 12 tissues.`
333
+ * still reads as one sentence. (#111, #130, #134)
334
+ */
335
+ const STATEMENT_BLOCK_TAGS = new Set([...BLOCK_TAGS, 'title', 'p']);
336
+ /** A `<caption>`'s title and paragraphs, each rendered as section prose is, space-joined. */
337
+ function renderCaption(caption) {
338
+ if (!caption)
339
+ return;
340
+ return flowBlocks(childrenOf(caption), STATEMENT_BLOCK_TAGS).join(' ') || undefined;
341
+ }
342
+ /**
343
+ * Render one block element as the text it contributes to the enclosing section.
344
+ *
345
+ * The default arm is the point of the dispatch: an element this parser does not
346
+ * enumerate degrades to its flattened text at block position rather than to
347
+ * silence, and never lands inside a neighbouring sentence. `<table-wrap>`,
348
+ * `<fig>` and `<supplementary-material>` are carried by `tables[]` / `assets[]`,
349
+ * so the first contributes nothing and the other two leave a marker where they
350
+ * sat; `<ref-list>` is carried by `references[]` and likewise contributes
351
+ * nothing, which is what keeps a `<sec>` that only wraps one from surviving as
352
+ * an empty "References" heading. (#116, #130)
353
+ */
354
+ function renderBlock(node) {
355
+ switch (tagNameOf(node)) {
356
+ case 'table-wrap':
357
+ case 'ref-list':
358
+ return '';
359
+ case 'fig':
360
+ return assetMarker(node, 'Figure');
361
+ case 'supplementary-material':
362
+ return assetMarker(node, 'Supplementary');
363
+ case 'list':
364
+ return renderList(node, 0);
365
+ case 'def-list':
366
+ return renderDefList(node, 0);
367
+ case 'disp-quote':
368
+ return renderDispQuote(node);
369
+ case 'boxed-text':
370
+ return renderBoxedText(node);
371
+ case 'preformat':
372
+ return renderPreformat(node);
373
+ case 'disp-formula':
374
+ return renderDispFormula(node);
375
+ default:
376
+ return flowBlocks(childrenOf(node)).join('\n\n');
377
+ }
378
+ }
379
+ /**
380
+ * The `<graphic>`/`<media>` pointer an asset hangs its file on, and the element
381
+ * a deposit may hang the label and caption on instead of on the asset itself.
382
+ */
383
+ function assetPointer(node) {
384
+ return findOne(node, 'graphic') ?? findOne(node, 'media');
385
+ }
386
+ /**
387
+ * An asset's display label: its own `<label>`, else one the deposit hung on the
388
+ * pointer. Shared by {@link assetMarker} and {@link parseAsset} so the marker
389
+ * left in the section text and the `label` reported in `assets[]` cannot
390
+ * disagree — the tool layer removes the marker by rebuilding it from that label,
391
+ * so a label resolved one way here and another way there strands the marker in
392
+ * the prose under `includeAssets: false`. (#130)
393
+ */
394
+ function assetLabel(node) {
395
+ return (textContent(findOne(node, 'label')) ||
396
+ textContent(findOne(assetPointer(node), 'label')) ||
397
+ undefined);
398
+ }
399
+ /**
400
+ * The marker a lifted asset leaves at the position it occupied. Prose refers to
401
+ * figures and supplements by label far more often than to tables, so removing
402
+ * the anchor while leaving every `<xref>` pointing at it would cost more than
403
+ * the ~18 characters the marker spends. (#130)
404
+ */
405
+ function assetMarker(node, kind) {
406
+ const label = assetLabel(node);
407
+ return label ? `[${kind}: ${label}]` : `[${kind}]`;
408
+ }
409
+ /**
410
+ * Render a `<list>`: `@list-type` picks the item marker (`order` numbers,
411
+ * `simple` leaves the item bare, everything else bullets), the optional
412
+ * `<title>` takes a line of its own above the items, and a `<list>` or
413
+ * `<def-list>` nested inside a `<list-item>` — both in the JATS content model —
414
+ * indents one level per depth. (#130)
415
+ */
416
+ function renderList(list, depth) {
417
+ const pad = NESTED_LIST_INDENT.repeat(depth);
418
+ const listType = attrOf(list, 'list-type');
419
+ const lines = [];
420
+ const title = textContent(findOne(list, 'title'));
421
+ if (title)
422
+ lines.push(`${pad}${title}`);
423
+ let ordinal = 0;
424
+ for (const item of findAll(list, 'list-item')) {
425
+ ordinal += 1;
426
+ const marker = listType === 'simple' ? '' : listType === 'order' ? `${ordinal}. ` : '- ';
427
+ const parts = [];
428
+ const nested = [];
429
+ for (const child of childrenOf(item)) {
430
+ const tag = tagNameOf(child);
431
+ if (tag === 'label')
432
+ continue;
433
+ if (tag === 'list')
434
+ nested.push(renderList(child, depth + 1));
435
+ else if (tag === 'def-list')
436
+ nested.push(renderDefList(child, depth + 1));
437
+ else
438
+ parts.push(...flowBlocks([child]));
439
+ }
440
+ const text = parts.join(' ');
441
+ if (text)
442
+ lines.push(`${pad}${marker}${text}`);
443
+ for (const block of nested)
444
+ if (block)
445
+ lines.push(block);
446
+ }
447
+ return lines.join('\n');
448
+ }
449
+ /** Render a `<def-list>`: its title on a line of its own, then `- term — definition` per item. */
450
+ function renderDefList(defList, depth) {
451
+ const pad = NESTED_LIST_INDENT.repeat(depth);
452
+ const lines = [];
453
+ const title = textContent(findOne(defList, 'title'));
454
+ if (title)
455
+ lines.push(`${pad}${title}`);
456
+ for (const item of findAll(defList, 'def-item')) {
457
+ const term = textContent(findOne(item, 'term'));
458
+ const definition = textContent(findOne(item, 'def'));
459
+ const entry = [term, definition].filter(Boolean).join(' — ');
460
+ if (entry)
461
+ lines.push(`${pad}- ${entry}`);
462
+ }
463
+ return lines.join('\n');
464
+ }
465
+ /** Render a `<disp-quote>`: every line quoted, its `<attrib>` as a trailing attribution line. */
466
+ function renderDispQuote(quote) {
467
+ const lines = [];
468
+ for (const child of childrenOf(quote)) {
469
+ if (tagNameOf(child) === 'attrib')
470
+ continue;
471
+ for (const block of flowBlocks([child])) {
472
+ for (const line of block.split('\n'))
473
+ lines.push(`> ${line}`);
474
+ }
475
+ }
476
+ const attrib = textContent(findOne(quote, 'attrib'));
477
+ if (attrib)
478
+ lines.push(`> — ${attrib}`);
479
+ return lines.join('\n');
480
+ }
481
+ /**
482
+ * Render a `<boxed-text>` by flattening it. Almost every one is a section
483
+ * container rather than a captioned box — 13 of the 14 in a 68-record draw carry
484
+ * `<sec>` children and nothing else — so rendering it as a caption plus
485
+ * paragraphs would drop the nested headings entirely. (#130)
486
+ */
487
+ function renderBoxedText(boxedText) {
488
+ const blocks = [];
489
+ for (const child of childrenOf(boxedText)) {
490
+ if (tagNameOf(child) === 'sec')
491
+ blocks.push(...flattenSec(child));
492
+ else
493
+ blocks.push(...flowBlocks([child]));
494
+ }
495
+ return blocks.join('\n\n');
496
+ }
497
+ /**
498
+ * A `<sec>` subtree as flat text blocks: each heading on its own line above its
499
+ * prose, descendants following in document order. Used where a section has no
500
+ * node of its own to live in — inside a `<boxed-text>` — mirroring how the tool
501
+ * layer flattens sections past the depth its schema carries.
502
+ */
503
+ function flattenSec(sec) {
504
+ const title = textContent(findOne(sec, 'title'));
505
+ const label = textContent(findOne(sec, 'label'));
506
+ const heading = title ? (label ? `${label} ${title}` : title) : '';
507
+ const blocks = [];
508
+ const nested = [];
509
+ for (const child of childrenOf(sec)) {
510
+ const tag = tagNameOf(child);
511
+ if (tag === 'title' || tag === 'label')
512
+ continue;
513
+ if (tag === 'sec')
514
+ nested.push(...flattenSec(child));
515
+ else
516
+ blocks.push(...flowBlocks([child]));
517
+ }
518
+ const head = [heading, blocks.join('\n\n')].filter(Boolean).join('\n');
519
+ return [...(head ? [head] : []), ...nested];
520
+ }
521
+ /**
522
+ * Render a `<preformat>` as a fenced block, read raw. It carries
523
+ * `xml:space="preserve"` and, in legacy deposits, the whole article as OCR text
524
+ * whose meaning lives in its line breaks and column spacing — {@link textContent}
525
+ * collapses both. Only the surrounding whitespace is trimmed, which is the XML
526
+ * indentation the element was serialized with rather than content. (#130)
527
+ */
528
+ function renderPreformat(preformat) {
529
+ const raw = rawTextContent(preformat).trim();
530
+ return raw ? `\`\`\`\n${raw}\n\`\`\`` : '';
531
+ }
532
+ /** Children of a `<disp-formula>` that are not its fallback text. */
533
+ const DISP_FORMULA_NON_BODY = new Set(['label', 'graphic', 'media']);
534
+ /**
535
+ * Render a `<disp-formula>`: its label, then a `<tex-math>` where the deposit
536
+ * carries one (directly or under `<alternatives>`) and the flattened content
537
+ * otherwise — usually `<mml:math>`, which is 69 of the 78 formulae in a
538
+ * 68-record draw against 3 for `<tex-math>`. A graphic-only deposit has no
539
+ * fallback text and contributes nothing at all rather than a bare label on an
540
+ * otherwise empty line. (#130)
541
+ */
542
+ function renderDispFormula(formula) {
543
+ const texMath = findOne(formula, 'tex-math') ?? findOne(findOne(formula, 'alternatives'), 'tex-math');
544
+ const body = textContent(texMath) || textContentExcluding(formula, DISP_FORMULA_NON_BODY);
545
+ if (!body)
546
+ return '';
547
+ const label = textContent(findOne(formula, 'label'));
548
+ return [label, body].filter(Boolean).join(' ');
549
+ }
190
550
  // ─── Body Sections ──────────────────────────────────────────────────────────
191
551
  /**
192
552
  * Extract body sections from a `<body>` node, walking children in document order.
193
- * Consecutive bare `<p>` siblings are collected into an untitled section so
194
- * articles with mixed structure (direct paragraphs + trailing `<sec>`, common
195
- * in manuscript-submitted PMC deposits) preserve their main text.
553
+ * Consecutive bare `<p>` siblings — and any block element sitting directly under
554
+ * `<body>` — are collected into an untitled section so articles with mixed
555
+ * structure (direct paragraphs + trailing `<sec>`, common in
556
+ * manuscript-submitted PMC deposits) preserve their main text.
557
+ *
558
+ * A `<body>` whose whole content is one block and no `<sec>` therefore yields a
559
+ * section carrying that block's text, rather than the empty list the tool layer
560
+ * reads as an article with no body at all. Legacy
561
+ * `<preformat preformat-type="pmc-ocr-text">` deposits, which put the entire
562
+ * article in one such element, are the case that matters. (#130)
196
563
  */
197
564
  export function extractBodySections(body) {
198
565
  if (!body)
199
566
  return [];
200
567
  const sections = [];
201
- let pendingParagraphs = [];
568
+ let pendingBlocks = [];
202
569
  const flushPending = () => {
203
- if (pendingParagraphs.length > 0) {
204
- sections.push({ text: pendingParagraphs.join('\n\n') });
205
- pendingParagraphs = [];
570
+ if (pendingBlocks.length > 0) {
571
+ sections.push({ text: pendingBlocks.join('\n\n') });
572
+ pendingBlocks = [];
206
573
  }
207
574
  };
208
575
  for (const child of childrenOf(body)) {
209
- const tag = tagNameOf(child);
210
- if (tag === 'p') {
211
- const text = paragraphText(child);
212
- if (text)
213
- pendingParagraphs.push(text);
214
- }
215
- else if (tag === 'sec') {
576
+ if (tagNameOf(child) === 'sec') {
216
577
  flushPending();
217
578
  const section = extractSection(child);
218
579
  if (section)
219
580
  sections.push(section);
581
+ continue;
220
582
  }
583
+ pendingBlocks.push(...flowBlocks([child]));
221
584
  }
222
585
  flushPending();
223
586
  return sections;
224
587
  }
588
+ /**
589
+ * Read one `<sec>`, walking every child in document order. The first `<title>`
590
+ * and `<label>` are the section's own metadata, a `<sec>` is a subsection, and
591
+ * everything else — `<p>` and block elements alike — contributes text at the
592
+ * position it occupies. Reading `<p>` and `<sec>` alone is what dropped a
593
+ * section's lists, figures, formulae and boxed text outright. (#130)
594
+ */
225
595
  function extractSection(sec) {
226
- const title = textContent(findOne(sec, 'title')) || undefined;
227
- const label = textContent(findOne(sec, 'label')) || undefined;
228
- const textParts = findAll(sec, 'p').map(paragraphText).filter(Boolean);
229
- const subsections = findAll(sec, 'sec')
230
- .map(extractSection)
231
- .filter((s) => s !== null);
596
+ let title;
597
+ let label;
598
+ const textParts = [];
599
+ const subsections = [];
600
+ for (const child of childrenOf(sec)) {
601
+ const tag = tagNameOf(child);
602
+ if (tag === 'title') {
603
+ title ??= textContent(child) || undefined;
604
+ continue;
605
+ }
606
+ if (tag === 'label') {
607
+ label ??= textContent(child) || undefined;
608
+ continue;
609
+ }
610
+ if (tag === 'sec') {
611
+ const subsection = extractSection(child);
612
+ if (subsection)
613
+ subsections.push(subsection);
614
+ continue;
615
+ }
616
+ textParts.push(...flowBlocks([child]));
617
+ }
232
618
  const text = textParts.join('\n\n');
233
619
  // A section left empty *because* its block content was lifted into `tables[]`
234
620
  // survives as a heading-only entry: the heading is what places the table for a
@@ -267,28 +653,37 @@ export function extractPmcTables(root) {
267
653
  if (!root)
268
654
  return [];
269
655
  const tables = [];
270
- collectTables(root, undefined, tables);
656
+ collectSectioned(root, undefined, tables, (child, tag, sectionTitle) => tag === 'table-wrap' ? parseTableWrap(child, sectionTitle) : undefined);
271
657
  return tables;
272
658
  }
273
- function collectTables(node, sectionTitle, out) {
659
+ /**
660
+ * Walk a subtree in document order, collecting whatever `take` recognizes and
661
+ * carrying the innermost enclosing `<sec>` title down to it. An untitled `<sec>`
662
+ * keeps its parent's title rather than dropping the reader's only positional
663
+ * cue, and an element inside no `<sec>` at all gets none. A recognized element
664
+ * is not descended into — it parses its own subtree.
665
+ *
666
+ * Shared by the table and asset walks so the section-title rule has one
667
+ * definition and cannot drift between them.
668
+ */
669
+ function collectSectioned(node, sectionTitle, out, take) {
274
670
  for (const child of childrenOf(node)) {
275
671
  const tag = tagNameOf(child);
276
672
  if (!tag)
277
673
  continue;
278
- if (tag === 'table-wrap') {
279
- out.push(parseTableWrap(child, sectionTitle));
674
+ const collected = take(child, tag, sectionTitle);
675
+ if (collected) {
676
+ out.push(collected);
280
677
  continue;
281
678
  }
282
- // An untitled <sec> keeps its parent's title rather than dropping the
283
- // reader's only positional cue.
284
679
  const nested = tag === 'sec' ? textContent(findOne(child, 'title')) || sectionTitle : sectionTitle;
285
- collectTables(child, nested, out);
680
+ collectSectioned(child, nested, out, take);
286
681
  }
287
682
  }
288
683
  function parseTableWrap(tableWrap, sectionTitle) {
289
684
  const id = attrOf(tableWrap, 'id');
290
685
  const label = textContent(findOne(tableWrap, 'label')) || undefined;
291
- const caption = textContent(findOne(tableWrap, 'caption')) || undefined;
686
+ const caption = renderCaption(findOne(tableWrap, 'caption'));
292
687
  const footnotes = textContent(findOne(tableWrap, 'table-wrap-foot')) || undefined;
293
688
  // Some deposits offer both renderings inside <alternatives>; prefer the markup.
294
689
  const table = findOne(tableWrap, 'table') ?? findOne(findOne(tableWrap, 'alternatives'), 'table');
@@ -416,6 +811,57 @@ function extractTableRows(table) {
416
811
  headerRowCount++;
417
812
  return { headerRowCount, rows: parsed.map((row) => row.cells) };
418
813
  }
814
+ // ─── Assets ─────────────────────────────────────────────────────────────────
815
+ /**
816
+ * Extract every `<fig>` and `<supplementary-material>` under `root` in document
817
+ * order — pass the `<article>` node.
818
+ *
819
+ * Whole-article, for the reason the table walk is: 17% of figures and 23% of
820
+ * supplementary material sit outside `<body>` entirely — `<floats-group>`
821
+ * deposits, `<back>/<sec>` and `<app-group>/<app>` placements, and figures
822
+ * hanging off the abstract in `<front>`. Each asset names the innermost
823
+ * enclosing `<sec>` wherever that sits, an untitled one inheriting its parent's
824
+ * title, so the reader keeps a positional cue; an asset inside no `<sec>` at all
825
+ * carries no section name. (#130)
826
+ */
827
+ export function extractPmcAssets(root) {
828
+ if (!root)
829
+ return [];
830
+ const assets = [];
831
+ collectSectioned(root, undefined, assets, (child, tag, sectionTitle) => {
832
+ const assetType = ASSET_TAG_TYPES[tag];
833
+ return assetType ? parseAsset(child, assetType, sectionTitle) : undefined;
834
+ });
835
+ return assets;
836
+ }
837
+ /** Tags lifted into `assets[]`, mapped to the type they are reported as. */
838
+ const ASSET_TAG_TYPES = {
839
+ fig: 'figure',
840
+ 'supplementary-material': 'supplementary-material',
841
+ };
842
+ function parseAsset(node, assetType, sectionTitle) {
843
+ const id = attrOf(node, 'id');
844
+ // Both `fig` and `supplementary-material` carry the pointer on a child
845
+ // element: a `<graphic>` for images, a `<media>` for everything else.
846
+ const pointer = assetPointer(node);
847
+ const href = pointer ? attrOf(pointer, 'xlink:href') : undefined;
848
+ // `label?, caption?` are in the JATS content model of `<media>` and
849
+ // `<graphic>` as well as of the asset element, and a common deposit style
850
+ // hangs them there instead — 19 of 84 supplementary items in a 68-record draw
851
+ // carry their caption on the `<media>` and nothing on the element itself.
852
+ // Reading direct children alone returns a pointer with no text at all. The
853
+ // asset's own label and caption win where it deposits them. (#130)
854
+ const label = assetLabel(node);
855
+ const caption = renderCaption(findOne(node, 'caption')) ?? renderCaption(findOne(pointer, 'caption'));
856
+ return {
857
+ assetType,
858
+ ...(id && { id }),
859
+ ...(label && { label }),
860
+ ...(caption && { caption }),
861
+ ...(sectionTitle && { sectionTitle }),
862
+ ...(href && { href }),
863
+ };
864
+ }
419
865
  // ─── References ─────────────────────────────────────────────────────────────
420
866
  /**
421
867
  * `pub-id-type` → human label, so the trailing identifiers in a rendered
@@ -671,6 +1117,7 @@ export function parsePmcArticle(articleNode) {
671
1117
  const sections = extractBodySections(body);
672
1118
  const references = extractReferences(articleNode);
673
1119
  const tables = extractPmcTables(articleNode);
1120
+ const assets = extractPmcAssets(articleNode);
674
1121
  const normalizedPmcId = !pmcId ? '' : pmcId.startsWith('PMC') ? pmcId : `PMC${pmcId}`;
675
1122
  const articleType = attrOf(articleNode, 'article-type');
676
1123
  return {
@@ -687,6 +1134,7 @@ export function parsePmcArticle(articleNode) {
687
1134
  sections,
688
1135
  ...(references.length > 0 && { references }),
689
1136
  ...(tables.length > 0 && { tables }),
1137
+ ...(assets.length > 0 && { assets }),
690
1138
  ...(articleType && { articleType }),
691
1139
  pmcUrl: `https://www.ncbi.nlm.nih.gov/pmc/articles/${normalizedPmcId}/`,
692
1140
  ...(pmid && { pubmedUrl: `https://pubmed.ncbi.nlm.nih.gov/${pmid}/` }),