@cyanheads/pubmed-mcp-server 2.10.8 → 2.10.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +1 -1
- package/CLAUDE.md +1 -1
- package/README.md +6 -4
- package/dist/mcp-server/tools/definitions/_text.d.ts +23 -1
- package/dist/mcp-server/tools/definitions/_text.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/_text.js +25 -1
- package/dist/mcp-server/tools/definitions/_text.js.map +1 -1
- package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.d.ts +35 -0
- package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.js +657 -44
- package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.js.map +1 -1
- package/dist/mcp-server/tools/definitions/find-related.tool.d.ts +8 -0
- package/dist/mcp-server/tools/definitions/find-related.tool.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/find-related.tool.js +131 -23
- package/dist/mcp-server/tools/definitions/find-related.tool.js.map +1 -1
- package/dist/mcp-server/tools/definitions/lookup-mesh.tool.d.ts +6 -0
- package/dist/mcp-server/tools/definitions/lookup-mesh.tool.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/lookup-mesh.tool.js +14 -3
- package/dist/mcp-server/tools/definitions/lookup-mesh.tool.js.map +1 -1
- package/dist/mcp-server/tools/definitions/search-articles.tool.d.ts +6 -0
- package/dist/mcp-server/tools/definitions/search-articles.tool.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/search-articles.tool.js +15 -3
- package/dist/mcp-server/tools/definitions/search-articles.tool.js.map +1 -1
- package/dist/mcp-server/tools/definitions/spell-check.tool.d.ts +6 -0
- package/dist/mcp-server/tools/definitions/spell-check.tool.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/spell-check.tool.js +15 -3
- package/dist/mcp-server/tools/definitions/spell-check.tool.js.map +1 -1
- package/dist/services/error-contracts.d.ts +22 -2
- package/dist/services/error-contracts.d.ts.map +1 -1
- package/dist/services/error-contracts.js +24 -2
- package/dist/services/error-contracts.js.map +1 -1
- package/dist/services/europe-pmc/europe-pmc-service.d.ts +3 -3
- package/dist/services/europe-pmc/europe-pmc-service.d.ts.map +1 -1
- package/dist/services/europe-pmc/europe-pmc-service.js +8 -16
- package/dist/services/europe-pmc/europe-pmc-service.js.map +1 -1
- package/dist/services/ncbi/parsing/ordered-xml-parser-options.d.ts +44 -0
- package/dist/services/ncbi/parsing/ordered-xml-parser-options.d.ts.map +1 -0
- package/dist/services/ncbi/parsing/ordered-xml-parser-options.js +41 -0
- package/dist/services/ncbi/parsing/ordered-xml-parser-options.js.map +1 -0
- package/dist/services/ncbi/parsing/pmc-article-parser.d.ts +70 -10
- package/dist/services/ncbi/parsing/pmc-article-parser.d.ts.map +1 -1
- package/dist/services/ncbi/parsing/pmc-article-parser.js +795 -79
- package/dist/services/ncbi/parsing/pmc-article-parser.js.map +1 -1
- package/dist/services/ncbi/parsing/pmc-xml-helpers.d.ts +18 -0
- package/dist/services/ncbi/parsing/pmc-xml-helpers.d.ts.map +1 -1
- package/dist/services/ncbi/parsing/pmc-xml-helpers.js +39 -3
- package/dist/services/ncbi/parsing/pmc-xml-helpers.js.map +1 -1
- package/dist/services/ncbi/response-handler.d.ts +3 -9
- package/dist/services/ncbi/response-handler.d.ts.map +1 -1
- package/dist/services/ncbi/response-handler.js +6 -29
- package/dist/services/ncbi/response-handler.js.map +1 -1
- package/dist/services/ncbi/types.d.ts +82 -0
- package/dist/services/ncbi/types.d.ts.map +1 -1
- package/dist/services/openalex/api-client.d.ts +13 -4
- package/dist/services/openalex/api-client.d.ts.map +1 -1
- package/dist/services/openalex/api-client.js +19 -8
- package/dist/services/openalex/api-client.js.map +1 -1
- package/dist/services/openalex/openalex-service.d.ts +34 -17
- package/dist/services/openalex/openalex-service.d.ts.map +1 -1
- package/dist/services/openalex/openalex-service.js +119 -35
- package/dist/services/openalex/openalex-service.js.map +1 -1
- package/dist/services/openalex/types.d.ts +34 -1
- package/dist/services/openalex/types.d.ts.map +1 -1
- package/dist/services/openalex/types.js +20 -0
- package/dist/services/openalex/types.js.map +1 -1
- package/package.json +1 -1
- package/server.json +3 -3
|
@@ -7,7 +7,101 @@
|
|
|
7
7
|
* inline children and drops body sections for markup-heavy articles.
|
|
8
8
|
* @module src/services/ncbi/parsing/pmc-article-parser
|
|
9
9
|
*/
|
|
10
|
-
import { attrOf, childrenOf, findAll, findOne, isTextNode, rawTextContent, tagNameOf, textContent, textOf, } from './pmc-xml-helpers.js';
|
|
10
|
+
import { attrOf, childrenOf, findAll, findAllDescendants, findOne, isTextNode, rawTextContent, tagNameOf, textContent, textContentExcluding, textOf, } from './pmc-xml-helpers.js';
|
|
11
|
+
/**
|
|
12
|
+
* Block content extracted into a field of its own and therefore left out of the
|
|
13
|
+
* prose a section contributes. Each one nested inside a `<p>` would otherwise be
|
|
14
|
+
* flattened into the surrounding sentence — a `<table-wrap>` concatenating
|
|
15
|
+
* adjacent cell values into numbers that never existed, a `<fig>` gluing its
|
|
16
|
+
* label and caption onto the sentence terminator before it. (#111, #130)
|
|
17
|
+
*/
|
|
18
|
+
const LIFTED_BLOCK_TAGS = new Set([
|
|
19
|
+
'table-wrap',
|
|
20
|
+
'fig',
|
|
21
|
+
'supplementary-material',
|
|
22
|
+
]);
|
|
23
|
+
/**
|
|
24
|
+
* JATS block-level elements, which interrupt prose rather than reading inside
|
|
25
|
+
* it. Membership decides *placement*, not representation: every one of these
|
|
26
|
+
* flushes the prose run in progress and contributes at block position, whether
|
|
27
|
+
* it renders (a `<list>`), lifts to its own field leaving a marker (a `<fig>`),
|
|
28
|
+
* or contributes nothing (a `<table-wrap>`, already in `tables[]`). Anything not
|
|
29
|
+
* named here is inline markup — `<italic>`, `<xref>`, `<sup>`,
|
|
30
|
+
* `<inline-formula>` — and reads inside the sentence as it should.
|
|
31
|
+
*
|
|
32
|
+
* The set is the JATS block-display class rather than the tags observed in any
|
|
33
|
+
* one draw: an element the walk does not enumerate degrades to flattened text at
|
|
34
|
+
* block position, and getting it there is what stops the fusion. (#130)
|
|
35
|
+
*
|
|
36
|
+
* Membership governs only an element nested inside another. Every caller that
|
|
37
|
+
* walks a container renders its children one at a time, so a direct `<sec>` or
|
|
38
|
+
* `<body>` child already contributes at block position whether or not it is
|
|
39
|
+
* named here; the live question per tag is what a paragraph-internal occurrence
|
|
40
|
+
* should do. `address`, `related-article` and `related-object` are in the JATS
|
|
41
|
+
* model both as display blocks and inline within a `<p>`, and they stay: a
|
|
42
|
+
* wrong split costs a paragraph break with every character still present and in
|
|
43
|
+
* order, while a wrong fusion fabricates adjacency the source never had, which
|
|
44
|
+
* is the defect this walk exists to prevent — when a tag reads both ways,
|
|
45
|
+
* splitting is the recoverable error.
|
|
46
|
+
*
|
|
47
|
+
* `<alternatives>` is the one element deliberately absent. It is a container
|
|
48
|
+
* for equivalent renderings of a single object and never a block in its own
|
|
49
|
+
* right, so its placement is whatever holds it: naming it here broke the
|
|
50
|
+
* standard `<inline-formula><alternatives><tex-math/><mml:math/></alternatives>`
|
|
51
|
+
* deposit out of the sentence it belonged to and split that sentence in two.
|
|
52
|
+
* `<disp-formula>`, `<table-wrap>` and `<fig>` resolve their own
|
|
53
|
+
* `<alternatives>` children, so nothing depends on it flushing the run. (#130)
|
|
54
|
+
*/
|
|
55
|
+
const BLOCK_TAGS = new Set([
|
|
56
|
+
...LIFTED_BLOCK_TAGS,
|
|
57
|
+
'address',
|
|
58
|
+
'array',
|
|
59
|
+
'boxed-text',
|
|
60
|
+
'chem-struct-wrap',
|
|
61
|
+
'code',
|
|
62
|
+
'def-list',
|
|
63
|
+
'disp-formula',
|
|
64
|
+
'disp-formula-group',
|
|
65
|
+
'disp-quote',
|
|
66
|
+
'fig-group',
|
|
67
|
+
'graphic',
|
|
68
|
+
'list',
|
|
69
|
+
'media',
|
|
70
|
+
'preformat',
|
|
71
|
+
'ref-list',
|
|
72
|
+
'related-article',
|
|
73
|
+
'related-object',
|
|
74
|
+
'speech',
|
|
75
|
+
'statement',
|
|
76
|
+
'table-wrap-group',
|
|
77
|
+
'verse-group',
|
|
78
|
+
]);
|
|
79
|
+
/** True when a lifted block sits anywhere in this subtree. */
|
|
80
|
+
function containsLiftedBlock(node) {
|
|
81
|
+
return childrenOf(node).some((child) => LIFTED_BLOCK_TAGS.has(tagNameOf(child) ?? '') || containsLiftedBlock(child));
|
|
82
|
+
}
|
|
83
|
+
/**
|
|
84
|
+
* True when a `<sec>` carried block content that was extracted into its own
|
|
85
|
+
* field — a `<table-wrap>` or `<fig>` sitting beside its paragraphs, or nested
|
|
86
|
+
* inside one.
|
|
87
|
+
*
|
|
88
|
+
* This is what separates a section emptied by the lift from one that never had
|
|
89
|
+
* readable prose. The first is a real heading a reader needs in order to place
|
|
90
|
+
* the table or figure that names it; the second is a structural wrapper — a
|
|
91
|
+
* `<sec>` around a `<ref-list>`, say — which would arrive as a stray empty
|
|
92
|
+
* "References" entry if every empty section survived. Only this section's own
|
|
93
|
+
* children are considered: a `<table-wrap>` deeper down belongs to the nested
|
|
94
|
+
* `<sec>` that holds it, and that section survives on its own account.
|
|
95
|
+
* (#111, #116, #130)
|
|
96
|
+
*/
|
|
97
|
+
function hasLiftedBlockContent(sec) {
|
|
98
|
+
return childrenOf(sec).some((child) => {
|
|
99
|
+
const tag = tagNameOf(child) ?? '';
|
|
100
|
+
if (LIFTED_BLOCK_TAGS.has(tag))
|
|
101
|
+
return true;
|
|
102
|
+
return tag === 'p' && containsLiftedBlock(child);
|
|
103
|
+
});
|
|
104
|
+
}
|
|
11
105
|
// ─── Article IDs ────────────────────────────────────────────────────────────
|
|
12
106
|
function extractArticleId(articleMeta, pubIdType) {
|
|
13
107
|
if (!articleMeta)
|
|
@@ -108,10 +202,28 @@ function extractPubDate(articleMeta) {
|
|
|
108
202
|
};
|
|
109
203
|
}
|
|
110
204
|
// ─── Abstract & Keywords ────────────────────────────────────────────────────
|
|
205
|
+
/**
|
|
206
|
+
* Read the article's own abstract.
|
|
207
|
+
*
|
|
208
|
+
* JATS permits several `<abstract>` elements under `<article-meta>`,
|
|
209
|
+
* distinguished by `@abstract-type`, and publishers routinely deposit a
|
|
210
|
+
* graphical abstract, author highlights or an executive summary alongside the
|
|
211
|
+
* real one — 21 of 68 records in a validation draw carry more than one, and in
|
|
212
|
+
* 4 of those the first is not the untyped element. The untyped one is the
|
|
213
|
+
* article's own abstract, so it is preferred and a typed one used only when the
|
|
214
|
+
* record deposits nothing else, mirroring the ladder {@link extractPubDate}
|
|
215
|
+
* applies to `<pub-date>`. (#134)
|
|
216
|
+
*
|
|
217
|
+
* Content is read through the same flow walk the body uses, so a `<fig>` in an
|
|
218
|
+
* abstract leaves its marker and its caption reaches `assets[]` alone —
|
|
219
|
+
* {@link extractPmcAssets} already walks `<front>` — instead of concatenating
|
|
220
|
+
* into the prose, and a `<list>` renders as list lines rather than a token run.
|
|
221
|
+
*/
|
|
111
222
|
function extractAbstract(articleMeta) {
|
|
112
223
|
if (!articleMeta)
|
|
113
224
|
return;
|
|
114
|
-
const
|
|
225
|
+
const abstracts = findAll(articleMeta, 'abstract');
|
|
226
|
+
const abstractNode = abstracts.find((node) => !attrOf(node, 'abstract-type')) ?? abstracts[0];
|
|
115
227
|
if (!abstractNode)
|
|
116
228
|
return;
|
|
117
229
|
const sections = findAll(abstractNode, 'sec');
|
|
@@ -119,10 +231,7 @@ function extractAbstract(articleMeta) {
|
|
|
119
231
|
const parts = [];
|
|
120
232
|
for (const sec of sections) {
|
|
121
233
|
const title = textContent(findOne(sec, 'title'));
|
|
122
|
-
const text =
|
|
123
|
-
.map((p) => textContent(p))
|
|
124
|
-
.filter(Boolean)
|
|
125
|
-
.join(' ');
|
|
234
|
+
const text = abstractProse(sec);
|
|
126
235
|
if (title && text)
|
|
127
236
|
parts.push(`${title}: ${text}`);
|
|
128
237
|
else if (text)
|
|
@@ -130,14 +239,20 @@ function extractAbstract(articleMeta) {
|
|
|
130
239
|
}
|
|
131
240
|
return parts.join('\n\n').trim() || undefined;
|
|
132
241
|
}
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
242
|
+
return abstractProse(abstractNode) || undefined;
|
|
243
|
+
}
|
|
244
|
+
/**
|
|
245
|
+
* The prose of an `<abstract>` or one of its `<sec>`s: every child but the
|
|
246
|
+
* heading the caller reports separately, space-joined. The single space is the
|
|
247
|
+
* spacing the paragraph-only read this replaces already produced, so a record
|
|
248
|
+
* depositing one untyped abstract of plain `<p>`s comes back byte-identical.
|
|
249
|
+
*/
|
|
250
|
+
function abstractProse(node) {
|
|
251
|
+
const prose = childrenOf(node).filter((child) => {
|
|
252
|
+
const tag = tagNameOf(child);
|
|
253
|
+
return tag !== 'title' && tag !== 'label';
|
|
254
|
+
});
|
|
255
|
+
return flowBlocks(prose, STATEMENT_BLOCK_TAGS).join(' ');
|
|
141
256
|
}
|
|
142
257
|
function extractKeywords(articleMeta) {
|
|
143
258
|
if (!articleMeta)
|
|
@@ -152,51 +267,354 @@ function extractKeywords(articleMeta) {
|
|
|
152
267
|
}
|
|
153
268
|
return keywords;
|
|
154
269
|
}
|
|
270
|
+
// ─── Block Rendering ────────────────────────────────────────────────────────
|
|
271
|
+
/** Indent one nesting level of a `<list>`/`<def-list>` nested in a `<list-item>`. */
|
|
272
|
+
const NESTED_LIST_INDENT = ' ';
|
|
273
|
+
/** Close the prose run in progress, discarding it when it holds only whitespace. */
|
|
274
|
+
function flushRun(flow) {
|
|
275
|
+
const text = flow.run.replace(/\s+/g, ' ').trim();
|
|
276
|
+
if (text)
|
|
277
|
+
flow.out.push(text);
|
|
278
|
+
flow.run = '';
|
|
279
|
+
}
|
|
280
|
+
/**
|
|
281
|
+
* Walk a mixed-content child list, accumulating inline text into a prose run and
|
|
282
|
+
* emitting every block-level element at its own document position instead.
|
|
283
|
+
*
|
|
284
|
+
* The recursion through non-block elements is what makes the split reliable: a
|
|
285
|
+
* `<fig>` nested two elements deep inside a `<p>` still flushes the sentence
|
|
286
|
+
* before it rather than being flattened into the middle of one. Inline markup —
|
|
287
|
+
* `<italic>`, `<xref>`, `<sup>` — is transparent here and reads in place, which
|
|
288
|
+
* is the behavior {@link textContent} already gave paragraphs. (#130)
|
|
289
|
+
*
|
|
290
|
+
* `blockTags` names what interrupts the run: {@link BLOCK_TAGS} for article
|
|
291
|
+
* prose, {@link STATEMENT_BLOCK_TAGS} for a container whose `<title>` and `<p>`
|
|
292
|
+
* children are separate statements rather than one continuous sentence.
|
|
293
|
+
*/
|
|
294
|
+
function walkFlow(nodes, flow, blockTags) {
|
|
295
|
+
for (const node of nodes) {
|
|
296
|
+
if (isTextNode(node)) {
|
|
297
|
+
flow.run += textOf(node);
|
|
298
|
+
continue;
|
|
299
|
+
}
|
|
300
|
+
const tag = tagNameOf(node) ?? '';
|
|
301
|
+
if (blockTags.has(tag)) {
|
|
302
|
+
flushRun(flow);
|
|
303
|
+
const rendered = renderBlock(node);
|
|
304
|
+
if (rendered)
|
|
305
|
+
flow.out.push(rendered);
|
|
306
|
+
continue;
|
|
307
|
+
}
|
|
308
|
+
walkFlow(childrenOf(node), flow, blockTags);
|
|
309
|
+
}
|
|
310
|
+
}
|
|
311
|
+
/** Ordered text blocks contributed by one `<p>` — or any other flow container. */
|
|
312
|
+
function flowBlocks(nodes, blockTags = BLOCK_TAGS) {
|
|
313
|
+
const flow = { out: [], run: '' };
|
|
314
|
+
walkFlow(nodes, flow, blockTags);
|
|
315
|
+
flushRun(flow);
|
|
316
|
+
return flow.out;
|
|
317
|
+
}
|
|
318
|
+
/**
|
|
319
|
+
* Block tags for a container whose `<title>` and `<p>` children each carry a
|
|
320
|
+
* statement of their own — a `<caption>`, an `<abstract>`, an abstract `<sec>`.
|
|
321
|
+
* Their children carry no punctuation between them, so the concatenating read
|
|
322
|
+
* ran a caption's title straight into its first sentence
|
|
323
|
+
* (`…observational constraints6.Cloud susceptibilities…`, 60 of 342 captions in
|
|
324
|
+
* a 68-record draw) and would run two sibling paragraphs together. Only a
|
|
325
|
+
* `<title>` or `<p>` boundary separates: inline markup between two text runs
|
|
326
|
+
* stays transparent, so `Expression of <italic>NF1</italic> across 12 tissues.`
|
|
327
|
+
* still reads as one sentence. (#111, #130, #134)
|
|
328
|
+
*/
|
|
329
|
+
const STATEMENT_BLOCK_TAGS = new Set([...BLOCK_TAGS, 'title', 'p']);
|
|
330
|
+
/** A `<caption>`'s title and paragraphs, each rendered as section prose is, space-joined. */
|
|
331
|
+
function renderCaption(caption) {
|
|
332
|
+
if (!caption)
|
|
333
|
+
return;
|
|
334
|
+
return flowBlocks(childrenOf(caption), STATEMENT_BLOCK_TAGS).join(' ') || undefined;
|
|
335
|
+
}
|
|
336
|
+
/**
|
|
337
|
+
* Render one block element as the text it contributes to the enclosing section.
|
|
338
|
+
*
|
|
339
|
+
* The default arm is the point of the dispatch: an element this parser does not
|
|
340
|
+
* enumerate degrades to its flattened text at block position rather than to
|
|
341
|
+
* silence, and never lands inside a neighbouring sentence. `<table-wrap>`,
|
|
342
|
+
* `<fig>` and `<supplementary-material>` are carried by `tables[]` / `assets[]`,
|
|
343
|
+
* so the first contributes nothing and the other two leave a marker where they
|
|
344
|
+
* sat; `<ref-list>` is carried by `references[]` and likewise contributes
|
|
345
|
+
* nothing, which is what keeps a `<sec>` that only wraps one from surviving as
|
|
346
|
+
* an empty "References" heading. (#116, #130)
|
|
347
|
+
*/
|
|
348
|
+
function renderBlock(node) {
|
|
349
|
+
switch (tagNameOf(node)) {
|
|
350
|
+
case 'table-wrap':
|
|
351
|
+
case 'ref-list':
|
|
352
|
+
return '';
|
|
353
|
+
case 'fig':
|
|
354
|
+
return assetMarker(node, 'Figure');
|
|
355
|
+
case 'supplementary-material':
|
|
356
|
+
return assetMarker(node, 'Supplementary');
|
|
357
|
+
case 'list':
|
|
358
|
+
return renderList(node, 0);
|
|
359
|
+
case 'def-list':
|
|
360
|
+
return renderDefList(node, 0);
|
|
361
|
+
case 'disp-quote':
|
|
362
|
+
return renderDispQuote(node);
|
|
363
|
+
case 'boxed-text':
|
|
364
|
+
return renderBoxedText(node);
|
|
365
|
+
case 'preformat':
|
|
366
|
+
return renderPreformat(node);
|
|
367
|
+
case 'disp-formula':
|
|
368
|
+
return renderDispFormula(node);
|
|
369
|
+
default:
|
|
370
|
+
return flowBlocks(childrenOf(node)).join('\n\n');
|
|
371
|
+
}
|
|
372
|
+
}
|
|
373
|
+
/**
|
|
374
|
+
* The `<graphic>`/`<media>` pointer an asset hangs its file on, and the element
|
|
375
|
+
* a deposit may hang the label and caption on instead of on the asset itself.
|
|
376
|
+
*/
|
|
377
|
+
function assetPointer(node) {
|
|
378
|
+
return findOne(node, 'graphic') ?? findOne(node, 'media');
|
|
379
|
+
}
|
|
380
|
+
/**
|
|
381
|
+
* An asset's display label: its own `<label>`, else one the deposit hung on the
|
|
382
|
+
* pointer. Shared by {@link assetMarker} and {@link parseAsset} so the marker
|
|
383
|
+
* left in the section text and the `label` reported in `assets[]` cannot
|
|
384
|
+
* disagree — the tool layer removes the marker by rebuilding it from that label,
|
|
385
|
+
* so a label resolved one way here and another way there strands the marker in
|
|
386
|
+
* the prose under `includeAssets: false`. (#130)
|
|
387
|
+
*/
|
|
388
|
+
function assetLabel(node) {
|
|
389
|
+
return (textContent(findOne(node, 'label')) ||
|
|
390
|
+
textContent(findOne(assetPointer(node), 'label')) ||
|
|
391
|
+
undefined);
|
|
392
|
+
}
|
|
393
|
+
/**
|
|
394
|
+
* The marker a lifted asset leaves at the position it occupied. Prose refers to
|
|
395
|
+
* figures and supplements by label far more often than to tables, so removing
|
|
396
|
+
* the anchor while leaving every `<xref>` pointing at it would cost more than
|
|
397
|
+
* the ~18 characters the marker spends. (#130)
|
|
398
|
+
*/
|
|
399
|
+
function assetMarker(node, kind) {
|
|
400
|
+
const label = assetLabel(node);
|
|
401
|
+
return label ? `[${kind}: ${label}]` : `[${kind}]`;
|
|
402
|
+
}
|
|
403
|
+
/**
|
|
404
|
+
* Render a `<list>`: `@list-type` picks the item marker (`order` numbers,
|
|
405
|
+
* `simple` leaves the item bare, everything else bullets), the optional
|
|
406
|
+
* `<title>` takes a line of its own above the items, and a `<list>` or
|
|
407
|
+
* `<def-list>` nested inside a `<list-item>` — both in the JATS content model —
|
|
408
|
+
* indents one level per depth. (#130)
|
|
409
|
+
*/
|
|
410
|
+
function renderList(list, depth) {
|
|
411
|
+
const pad = NESTED_LIST_INDENT.repeat(depth);
|
|
412
|
+
const listType = attrOf(list, 'list-type');
|
|
413
|
+
const lines = [];
|
|
414
|
+
const title = textContent(findOne(list, 'title'));
|
|
415
|
+
if (title)
|
|
416
|
+
lines.push(`${pad}${title}`);
|
|
417
|
+
let ordinal = 0;
|
|
418
|
+
for (const item of findAll(list, 'list-item')) {
|
|
419
|
+
ordinal += 1;
|
|
420
|
+
const marker = listType === 'simple' ? '' : listType === 'order' ? `${ordinal}. ` : '- ';
|
|
421
|
+
const parts = [];
|
|
422
|
+
const nested = [];
|
|
423
|
+
for (const child of childrenOf(item)) {
|
|
424
|
+
const tag = tagNameOf(child);
|
|
425
|
+
if (tag === 'label')
|
|
426
|
+
continue;
|
|
427
|
+
if (tag === 'list')
|
|
428
|
+
nested.push(renderList(child, depth + 1));
|
|
429
|
+
else if (tag === 'def-list')
|
|
430
|
+
nested.push(renderDefList(child, depth + 1));
|
|
431
|
+
else
|
|
432
|
+
parts.push(...flowBlocks([child]));
|
|
433
|
+
}
|
|
434
|
+
const text = parts.join(' ');
|
|
435
|
+
if (text)
|
|
436
|
+
lines.push(`${pad}${marker}${text}`);
|
|
437
|
+
for (const block of nested)
|
|
438
|
+
if (block)
|
|
439
|
+
lines.push(block);
|
|
440
|
+
}
|
|
441
|
+
return lines.join('\n');
|
|
442
|
+
}
|
|
443
|
+
/** Render a `<def-list>`: its title on a line of its own, then `- term — definition` per item. */
|
|
444
|
+
function renderDefList(defList, depth) {
|
|
445
|
+
const pad = NESTED_LIST_INDENT.repeat(depth);
|
|
446
|
+
const lines = [];
|
|
447
|
+
const title = textContent(findOne(defList, 'title'));
|
|
448
|
+
if (title)
|
|
449
|
+
lines.push(`${pad}${title}`);
|
|
450
|
+
for (const item of findAll(defList, 'def-item')) {
|
|
451
|
+
const term = textContent(findOne(item, 'term'));
|
|
452
|
+
const definition = textContent(findOne(item, 'def'));
|
|
453
|
+
const entry = [term, definition].filter(Boolean).join(' — ');
|
|
454
|
+
if (entry)
|
|
455
|
+
lines.push(`${pad}- ${entry}`);
|
|
456
|
+
}
|
|
457
|
+
return lines.join('\n');
|
|
458
|
+
}
|
|
459
|
+
/** Render a `<disp-quote>`: every line quoted, its `<attrib>` as a trailing attribution line. */
|
|
460
|
+
function renderDispQuote(quote) {
|
|
461
|
+
const lines = [];
|
|
462
|
+
for (const child of childrenOf(quote)) {
|
|
463
|
+
if (tagNameOf(child) === 'attrib')
|
|
464
|
+
continue;
|
|
465
|
+
for (const block of flowBlocks([child])) {
|
|
466
|
+
for (const line of block.split('\n'))
|
|
467
|
+
lines.push(`> ${line}`);
|
|
468
|
+
}
|
|
469
|
+
}
|
|
470
|
+
const attrib = textContent(findOne(quote, 'attrib'));
|
|
471
|
+
if (attrib)
|
|
472
|
+
lines.push(`> — ${attrib}`);
|
|
473
|
+
return lines.join('\n');
|
|
474
|
+
}
|
|
475
|
+
/**
|
|
476
|
+
* Render a `<boxed-text>` by flattening it. Almost every one is a section
|
|
477
|
+
* container rather than a captioned box — 13 of the 14 in a 68-record draw carry
|
|
478
|
+
* `<sec>` children and nothing else — so rendering it as a caption plus
|
|
479
|
+
* paragraphs would drop the nested headings entirely. (#130)
|
|
480
|
+
*/
|
|
481
|
+
function renderBoxedText(boxedText) {
|
|
482
|
+
const blocks = [];
|
|
483
|
+
for (const child of childrenOf(boxedText)) {
|
|
484
|
+
if (tagNameOf(child) === 'sec')
|
|
485
|
+
blocks.push(...flattenSec(child));
|
|
486
|
+
else
|
|
487
|
+
blocks.push(...flowBlocks([child]));
|
|
488
|
+
}
|
|
489
|
+
return blocks.join('\n\n');
|
|
490
|
+
}
|
|
491
|
+
/**
|
|
492
|
+
* A `<sec>` subtree as flat text blocks: each heading on its own line above its
|
|
493
|
+
* prose, descendants following in document order. Used where a section has no
|
|
494
|
+
* node of its own to live in — inside a `<boxed-text>` — mirroring how the tool
|
|
495
|
+
* layer flattens sections past the depth its schema carries.
|
|
496
|
+
*/
|
|
497
|
+
function flattenSec(sec) {
|
|
498
|
+
const title = textContent(findOne(sec, 'title'));
|
|
499
|
+
const label = textContent(findOne(sec, 'label'));
|
|
500
|
+
const heading = title ? (label ? `${label} ${title}` : title) : '';
|
|
501
|
+
const blocks = [];
|
|
502
|
+
const nested = [];
|
|
503
|
+
for (const child of childrenOf(sec)) {
|
|
504
|
+
const tag = tagNameOf(child);
|
|
505
|
+
if (tag === 'title' || tag === 'label')
|
|
506
|
+
continue;
|
|
507
|
+
if (tag === 'sec')
|
|
508
|
+
nested.push(...flattenSec(child));
|
|
509
|
+
else
|
|
510
|
+
blocks.push(...flowBlocks([child]));
|
|
511
|
+
}
|
|
512
|
+
const head = [heading, blocks.join('\n\n')].filter(Boolean).join('\n');
|
|
513
|
+
return [...(head ? [head] : []), ...nested];
|
|
514
|
+
}
|
|
515
|
+
/**
|
|
516
|
+
* Render a `<preformat>` as a fenced block, read raw. It carries
|
|
517
|
+
* `xml:space="preserve"` and, in legacy deposits, the whole article as OCR text
|
|
518
|
+
* whose meaning lives in its line breaks and column spacing — {@link textContent}
|
|
519
|
+
* collapses both. Only the surrounding whitespace is trimmed, which is the XML
|
|
520
|
+
* indentation the element was serialized with rather than content. (#130)
|
|
521
|
+
*/
|
|
522
|
+
function renderPreformat(preformat) {
|
|
523
|
+
const raw = rawTextContent(preformat).trim();
|
|
524
|
+
return raw ? `\`\`\`\n${raw}\n\`\`\`` : '';
|
|
525
|
+
}
|
|
526
|
+
/** Children of a `<disp-formula>` that are not its fallback text. */
|
|
527
|
+
const DISP_FORMULA_NON_BODY = new Set(['label', 'graphic', 'media']);
|
|
528
|
+
/**
|
|
529
|
+
* Render a `<disp-formula>`: its label, then a `<tex-math>` where the deposit
|
|
530
|
+
* carries one (directly or under `<alternatives>`) and the flattened content
|
|
531
|
+
* otherwise — usually `<mml:math>`, which is 69 of the 78 formulae in a
|
|
532
|
+
* 68-record draw against 3 for `<tex-math>`. A graphic-only deposit has no
|
|
533
|
+
* fallback text and contributes nothing at all rather than a bare label on an
|
|
534
|
+
* otherwise empty line. (#130)
|
|
535
|
+
*/
|
|
536
|
+
function renderDispFormula(formula) {
|
|
537
|
+
const texMath = findOne(formula, 'tex-math') ?? findOne(findOne(formula, 'alternatives'), 'tex-math');
|
|
538
|
+
const body = textContent(texMath) || textContentExcluding(formula, DISP_FORMULA_NON_BODY);
|
|
539
|
+
if (!body)
|
|
540
|
+
return '';
|
|
541
|
+
const label = textContent(findOne(formula, 'label'));
|
|
542
|
+
return [label, body].filter(Boolean).join(' ');
|
|
543
|
+
}
|
|
155
544
|
// ─── Body Sections ──────────────────────────────────────────────────────────
|
|
156
545
|
/**
|
|
157
546
|
* Extract body sections from a `<body>` node, walking children in document order.
|
|
158
|
-
* Consecutive bare `<p>` siblings
|
|
159
|
-
*
|
|
160
|
-
*
|
|
547
|
+
* Consecutive bare `<p>` siblings — and any block element sitting directly under
|
|
548
|
+
* `<body>` — are collected into an untitled section so articles with mixed
|
|
549
|
+
* structure (direct paragraphs + trailing `<sec>`, common in
|
|
550
|
+
* manuscript-submitted PMC deposits) preserve their main text.
|
|
551
|
+
*
|
|
552
|
+
* A `<body>` whose whole content is one block and no `<sec>` therefore yields a
|
|
553
|
+
* section carrying that block's text, rather than the empty list the tool layer
|
|
554
|
+
* reads as an article with no body at all. Legacy
|
|
555
|
+
* `<preformat preformat-type="pmc-ocr-text">` deposits, which put the entire
|
|
556
|
+
* article in one such element, are the case that matters. (#130)
|
|
161
557
|
*/
|
|
162
558
|
export function extractBodySections(body) {
|
|
163
559
|
if (!body)
|
|
164
560
|
return [];
|
|
165
561
|
const sections = [];
|
|
166
|
-
let
|
|
562
|
+
let pendingBlocks = [];
|
|
167
563
|
const flushPending = () => {
|
|
168
|
-
if (
|
|
169
|
-
sections.push({ text:
|
|
170
|
-
|
|
564
|
+
if (pendingBlocks.length > 0) {
|
|
565
|
+
sections.push({ text: pendingBlocks.join('\n\n') });
|
|
566
|
+
pendingBlocks = [];
|
|
171
567
|
}
|
|
172
568
|
};
|
|
173
569
|
for (const child of childrenOf(body)) {
|
|
174
|
-
|
|
175
|
-
if (tag === 'p') {
|
|
176
|
-
const text = textContent(child);
|
|
177
|
-
if (text)
|
|
178
|
-
pendingParagraphs.push(text);
|
|
179
|
-
}
|
|
180
|
-
else if (tag === 'sec') {
|
|
570
|
+
if (tagNameOf(child) === 'sec') {
|
|
181
571
|
flushPending();
|
|
182
572
|
const section = extractSection(child);
|
|
183
573
|
if (section)
|
|
184
574
|
sections.push(section);
|
|
575
|
+
continue;
|
|
185
576
|
}
|
|
577
|
+
pendingBlocks.push(...flowBlocks([child]));
|
|
186
578
|
}
|
|
187
579
|
flushPending();
|
|
188
580
|
return sections;
|
|
189
581
|
}
|
|
582
|
+
/**
|
|
583
|
+
* Read one `<sec>`, walking every child in document order. The first `<title>`
|
|
584
|
+
* and `<label>` are the section's own metadata, a `<sec>` is a subsection, and
|
|
585
|
+
* everything else — `<p>` and block elements alike — contributes text at the
|
|
586
|
+
* position it occupies. Reading `<p>` and `<sec>` alone is what dropped a
|
|
587
|
+
* section's lists, figures, formulae and boxed text outright. (#130)
|
|
588
|
+
*/
|
|
190
589
|
function extractSection(sec) {
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
const
|
|
194
|
-
const
|
|
195
|
-
const
|
|
196
|
-
|
|
197
|
-
|
|
590
|
+
let title;
|
|
591
|
+
let label;
|
|
592
|
+
const textParts = [];
|
|
593
|
+
const subsections = [];
|
|
594
|
+
for (const child of childrenOf(sec)) {
|
|
595
|
+
const tag = tagNameOf(child);
|
|
596
|
+
if (tag === 'title') {
|
|
597
|
+
title ??= textContent(child) || undefined;
|
|
598
|
+
continue;
|
|
599
|
+
}
|
|
600
|
+
if (tag === 'label') {
|
|
601
|
+
label ??= textContent(child) || undefined;
|
|
602
|
+
continue;
|
|
603
|
+
}
|
|
604
|
+
if (tag === 'sec') {
|
|
605
|
+
const subsection = extractSection(child);
|
|
606
|
+
if (subsection)
|
|
607
|
+
subsections.push(subsection);
|
|
608
|
+
continue;
|
|
609
|
+
}
|
|
610
|
+
textParts.push(...flowBlocks([child]));
|
|
611
|
+
}
|
|
198
612
|
const text = textParts.join('\n\n');
|
|
199
|
-
|
|
613
|
+
// A section left empty *because* its block content was lifted into `tables[]`
|
|
614
|
+
// survives as a heading-only entry: the heading is what places the table for a
|
|
615
|
+
// reader, and dropping it also erased the section from a `sections` filter's
|
|
616
|
+
// reach. A section that never carried prose still drops. (#111)
|
|
617
|
+
if (!text && subsections.length === 0 && !hasLiftedBlockContent(sec))
|
|
200
618
|
return null;
|
|
201
619
|
return {
|
|
202
620
|
...(title && { title }),
|
|
@@ -205,6 +623,239 @@ function extractSection(sec) {
|
|
|
205
623
|
...(subsections.length > 0 && { subsections }),
|
|
206
624
|
};
|
|
207
625
|
}
|
|
626
|
+
// ─── Tables ─────────────────────────────────────────────────────────────────
|
|
627
|
+
/**
|
|
628
|
+
* Extract every `<table-wrap>` under `root` in document order — pass the
|
|
629
|
+
* `<article>` node.
|
|
630
|
+
*
|
|
631
|
+
* The walk covers the whole article rather than `<body>` alone: about a quarter
|
|
632
|
+
* of real tables sit in `<floats-group>`, `<back>/<sec>`, `<table-wrap-group>`
|
|
633
|
+
* or `<app-group>/<app>`, so a body-scoped extractor drops them. Each table
|
|
634
|
+
* names the innermost enclosing `<sec>` wherever that sits — a back-matter or
|
|
635
|
+
* appendix section counts, and naming it is the reader's only positional cue
|
|
636
|
+
* there. Only a table inside no `<sec>` at all, such as a `<floats-group>`
|
|
637
|
+
* deposit, carries no section name.
|
|
638
|
+
*
|
|
639
|
+
* Only XHTML `<tr>`/`<td>`/`<th>` bodies are read. **Decision, not an
|
|
640
|
+
* oversight:** across a 283-table survey of open-access records, 276 were XHTML,
|
|
641
|
+
* 7 were graphic-only deposits and 0 used the CALS `<tgroup>` model. Graphic-only
|
|
642
|
+
* and CALS bodies therefore take the unextractable path — returned with their
|
|
643
|
+
* label and caption and a reason — rather than adding a second table model no
|
|
644
|
+
* observed record needs. Revisit only if CALS shows up in the wild. (#111)
|
|
645
|
+
*/
|
|
646
|
+
export function extractPmcTables(root) {
|
|
647
|
+
if (!root)
|
|
648
|
+
return [];
|
|
649
|
+
const tables = [];
|
|
650
|
+
collectSectioned(root, undefined, tables, (child, tag, sectionTitle) => tag === 'table-wrap' ? parseTableWrap(child, sectionTitle) : undefined);
|
|
651
|
+
return tables;
|
|
652
|
+
}
|
|
653
|
+
/**
|
|
654
|
+
* Walk a subtree in document order, collecting whatever `take` recognizes and
|
|
655
|
+
* carrying the innermost enclosing `<sec>` title down to it. An untitled `<sec>`
|
|
656
|
+
* keeps its parent's title rather than dropping the reader's only positional
|
|
657
|
+
* cue, and an element inside no `<sec>` at all gets none. A recognized element
|
|
658
|
+
* is not descended into — it parses its own subtree.
|
|
659
|
+
*
|
|
660
|
+
* Shared by the table and asset walks so the section-title rule has one
|
|
661
|
+
* definition and cannot drift between them.
|
|
662
|
+
*/
|
|
663
|
+
function collectSectioned(node, sectionTitle, out, take) {
|
|
664
|
+
for (const child of childrenOf(node)) {
|
|
665
|
+
const tag = tagNameOf(child);
|
|
666
|
+
if (!tag)
|
|
667
|
+
continue;
|
|
668
|
+
const collected = take(child, tag, sectionTitle);
|
|
669
|
+
if (collected) {
|
|
670
|
+
out.push(collected);
|
|
671
|
+
continue;
|
|
672
|
+
}
|
|
673
|
+
const nested = tag === 'sec' ? textContent(findOne(child, 'title')) || sectionTitle : sectionTitle;
|
|
674
|
+
collectSectioned(child, nested, out, take);
|
|
675
|
+
}
|
|
676
|
+
}
|
|
677
|
+
function parseTableWrap(tableWrap, sectionTitle) {
|
|
678
|
+
const id = attrOf(tableWrap, 'id');
|
|
679
|
+
const label = textContent(findOne(tableWrap, 'label')) || undefined;
|
|
680
|
+
const caption = renderCaption(findOne(tableWrap, 'caption'));
|
|
681
|
+
const footnotes = textContent(findOne(tableWrap, 'table-wrap-foot')) || undefined;
|
|
682
|
+
// Some deposits offer both renderings inside <alternatives>; prefer the markup.
|
|
683
|
+
const table = findOne(tableWrap, 'table') ?? findOne(findOne(tableWrap, 'alternatives'), 'table');
|
|
684
|
+
const { rows, headerRowCount } = table
|
|
685
|
+
? extractTableRows(table)
|
|
686
|
+
: { rows: [], headerRowCount: 0 };
|
|
687
|
+
return {
|
|
688
|
+
...(id && { id }),
|
|
689
|
+
...(label && { label }),
|
|
690
|
+
...(caption && { caption }),
|
|
691
|
+
...(sectionTitle && { sectionTitle }),
|
|
692
|
+
headerRowCount,
|
|
693
|
+
rows,
|
|
694
|
+
...(footnotes && { footnotes }),
|
|
695
|
+
...(rows.length === 0 && {
|
|
696
|
+
unextractableReason: classifyUnextractable(tableWrap, table),
|
|
697
|
+
}),
|
|
698
|
+
};
|
|
699
|
+
}
|
|
700
|
+
function classifyUnextractable(tableWrap, table) {
|
|
701
|
+
if (findOne(tableWrap, 'tgroup') || findOne(table, 'tgroup'))
|
|
702
|
+
return 'cals-tgroup';
|
|
703
|
+
if (findOne(tableWrap, 'graphic') || findOne(findOne(tableWrap, 'alternatives'), 'graphic')) {
|
|
704
|
+
return 'graphic-only';
|
|
705
|
+
}
|
|
706
|
+
return 'no-rows';
|
|
707
|
+
}
|
|
708
|
+
/**
|
|
709
|
+
* Widest grid a single table row may occupy, and the ceiling every declared
|
|
710
|
+
* span is clamped to. Real deposits are far narrower — the widest row observed
|
|
711
|
+
* across a live sample was 15 columns — so this bounds a malformed or hostile
|
|
712
|
+
* span rather than limiting any genuine table: `colspan="99999999"` costs a
|
|
713
|
+
* bounded row instead of an unbounded allocation, and a `rowspan` that large
|
|
714
|
+
* carries a bounded number of rows down. How many rows a table has is set by its
|
|
715
|
+
* source `<tr>` elements and needs no cap of its own. (#111)
|
|
716
|
+
*/
|
|
717
|
+
export const MAX_TABLE_COLUMNS = 512;
|
|
718
|
+
/** A `colspan`/`rowspan` value, clamped to a sane grid. Anything unparseable is 1. */
|
|
719
|
+
function spanOf(cell, name) {
|
|
720
|
+
const declared = Number.parseInt(attrOf(cell, name) ?? '', 10);
|
|
721
|
+
if (!Number.isFinite(declared) || declared < 1)
|
|
722
|
+
return 1;
|
|
723
|
+
return Math.min(declared, MAX_TABLE_COLUMNS);
|
|
724
|
+
}
|
|
725
|
+
/**
|
|
726
|
+
* Read an XHTML `<table>` into rows of cell text, expanding `colspan` and
|
|
727
|
+
* `rowspan` so every entry in a row is one grid column.
|
|
728
|
+
*
|
|
729
|
+
* A cell covering N columns occupies N entries and one covering M rows occupies
|
|
730
|
+
* its column in the M rows below it, repeating its text across the cells it
|
|
731
|
+
* genuinely covers. Repeating is the honest flattening: the value belongs to
|
|
732
|
+
* each of those positions, and a reader scanning a column finds it there. The
|
|
733
|
+
* alternative — keeping the source cell count — leaves the column a value
|
|
734
|
+
* belongs to unrecoverable downstream, so a renderer padding rows from the left
|
|
735
|
+
* puts values under the wrong headers. (#111)
|
|
736
|
+
*
|
|
737
|
+
* Header rows are those in `<thead>` plus any leading row made entirely of
|
|
738
|
+
* `<th>` in a table that declares no `<thead>`.
|
|
739
|
+
*/
|
|
740
|
+
function extractTableRows(table) {
|
|
741
|
+
const parsed = [];
|
|
742
|
+
/** Live `rowspan` obligations, indexed by grid column. */
|
|
743
|
+
const carried = [];
|
|
744
|
+
/** Consume the carried cells sitting at and after `col`, contiguously. */
|
|
745
|
+
const drainCarried = (row, col) => {
|
|
746
|
+
let at = col;
|
|
747
|
+
while (at < MAX_TABLE_COLUMNS) {
|
|
748
|
+
const carry = carried[at];
|
|
749
|
+
if (!carry)
|
|
750
|
+
return at;
|
|
751
|
+
row[at] = carry.value;
|
|
752
|
+
carry.remaining -= 1;
|
|
753
|
+
if (carry.remaining <= 0)
|
|
754
|
+
carried[at] = undefined;
|
|
755
|
+
at += 1;
|
|
756
|
+
}
|
|
757
|
+
return at;
|
|
758
|
+
};
|
|
759
|
+
const pushRow = (tr, inHead) => {
|
|
760
|
+
const row = [];
|
|
761
|
+
let col = 0;
|
|
762
|
+
let sourceCells = 0;
|
|
763
|
+
let allHeaderCells = true;
|
|
764
|
+
for (const cell of childrenOf(tr)) {
|
|
765
|
+
const tag = tagNameOf(cell);
|
|
766
|
+
if (tag !== 'td' && tag !== 'th')
|
|
767
|
+
continue;
|
|
768
|
+
if (tag === 'td')
|
|
769
|
+
allHeaderCells = false;
|
|
770
|
+
sourceCells += 1;
|
|
771
|
+
col = drainCarried(row, col);
|
|
772
|
+
const value = textContent(cell);
|
|
773
|
+
const rowspan = spanOf(cell, 'rowspan');
|
|
774
|
+
const width = Math.min(spanOf(cell, 'colspan'), MAX_TABLE_COLUMNS - col);
|
|
775
|
+
for (let i = 0; i < width; i++) {
|
|
776
|
+
row[col] = value;
|
|
777
|
+
if (rowspan > 1)
|
|
778
|
+
carried[col] = { remaining: rowspan - 1, value };
|
|
779
|
+
col += 1;
|
|
780
|
+
}
|
|
781
|
+
}
|
|
782
|
+
// Carried cells past the last source cell still hold their columns. Walk out
|
|
783
|
+
// to the rightmost live obligation so they land where they belong; the gaps
|
|
784
|
+
// in between are columns this row genuinely left empty.
|
|
785
|
+
const rightmost = carried.reduce((last, carry, i) => (carry ? i : last), -1);
|
|
786
|
+
while (col <= rightmost)
|
|
787
|
+
col = carried[col] ? drainCarried(row, col) : col + 1;
|
|
788
|
+
if (sourceCells === 0)
|
|
789
|
+
return;
|
|
790
|
+
for (let i = 0; i < row.length; i++)
|
|
791
|
+
row[i] ??= '';
|
|
792
|
+
parsed.push({ cells: row, header: inHead || allHeaderCells });
|
|
793
|
+
};
|
|
794
|
+
for (const child of childrenOf(table)) {
|
|
795
|
+
const tag = tagNameOf(child);
|
|
796
|
+
if (tag === 'tr')
|
|
797
|
+
pushRow(child, false);
|
|
798
|
+
else if (tag === 'thead' || tag === 'tbody' || tag === 'tfoot') {
|
|
799
|
+
for (const tr of findAll(child, 'tr'))
|
|
800
|
+
pushRow(tr, tag === 'thead');
|
|
801
|
+
}
|
|
802
|
+
}
|
|
803
|
+
let headerRowCount = 0;
|
|
804
|
+
while (parsed[headerRowCount]?.header)
|
|
805
|
+
headerRowCount++;
|
|
806
|
+
return { headerRowCount, rows: parsed.map((row) => row.cells) };
|
|
807
|
+
}
|
|
808
|
+
// ─── Assets ─────────────────────────────────────────────────────────────────
|
|
809
|
+
/**
|
|
810
|
+
* Extract every `<fig>` and `<supplementary-material>` under `root` in document
|
|
811
|
+
* order — pass the `<article>` node.
|
|
812
|
+
*
|
|
813
|
+
* Whole-article, for the reason the table walk is: 17% of figures and 23% of
|
|
814
|
+
* supplementary material sit outside `<body>` entirely — `<floats-group>`
|
|
815
|
+
* deposits, `<back>/<sec>` and `<app-group>/<app>` placements, and figures
|
|
816
|
+
* hanging off the abstract in `<front>`. Each asset names the innermost
|
|
817
|
+
* enclosing `<sec>` wherever that sits, an untitled one inheriting its parent's
|
|
818
|
+
* title, so the reader keeps a positional cue; an asset inside no `<sec>` at all
|
|
819
|
+
* carries no section name. (#130)
|
|
820
|
+
*/
|
|
821
|
+
export function extractPmcAssets(root) {
|
|
822
|
+
if (!root)
|
|
823
|
+
return [];
|
|
824
|
+
const assets = [];
|
|
825
|
+
collectSectioned(root, undefined, assets, (child, tag, sectionTitle) => {
|
|
826
|
+
const assetType = ASSET_TAG_TYPES[tag];
|
|
827
|
+
return assetType ? parseAsset(child, assetType, sectionTitle) : undefined;
|
|
828
|
+
});
|
|
829
|
+
return assets;
|
|
830
|
+
}
|
|
831
|
+
/** Tags lifted into `assets[]`, mapped to the type they are reported as. */
|
|
832
|
+
const ASSET_TAG_TYPES = {
|
|
833
|
+
fig: 'figure',
|
|
834
|
+
'supplementary-material': 'supplementary-material',
|
|
835
|
+
};
|
|
836
|
+
function parseAsset(node, assetType, sectionTitle) {
|
|
837
|
+
const id = attrOf(node, 'id');
|
|
838
|
+
// Both `fig` and `supplementary-material` carry the pointer on a child
|
|
839
|
+
// element: a `<graphic>` for images, a `<media>` for everything else.
|
|
840
|
+
const pointer = assetPointer(node);
|
|
841
|
+
const href = pointer ? attrOf(pointer, 'xlink:href') : undefined;
|
|
842
|
+
// `label?, caption?` are in the JATS content model of `<media>` and
|
|
843
|
+
// `<graphic>` as well as of the asset element, and a common deposit style
|
|
844
|
+
// hangs them there instead — 19 of 84 supplementary items in a 68-record draw
|
|
845
|
+
// carry their caption on the `<media>` and nothing on the element itself.
|
|
846
|
+
// Reading direct children alone returns a pointer with no text at all. The
|
|
847
|
+
// asset's own label and caption win where it deposits them. (#130)
|
|
848
|
+
const label = assetLabel(node);
|
|
849
|
+
const caption = renderCaption(findOne(node, 'caption')) ?? renderCaption(findOne(pointer, 'caption'));
|
|
850
|
+
return {
|
|
851
|
+
assetType,
|
|
852
|
+
...(id && { id }),
|
|
853
|
+
...(label && { label }),
|
|
854
|
+
...(caption && { caption }),
|
|
855
|
+
...(sectionTitle && { sectionTitle }),
|
|
856
|
+
...(href && { href }),
|
|
857
|
+
};
|
|
858
|
+
}
|
|
208
859
|
// ─── References ─────────────────────────────────────────────────────────────
|
|
209
860
|
/**
|
|
210
861
|
* `pub-id-type` → human label, so the trailing identifiers in a rendered
|
|
@@ -263,6 +914,49 @@ function renderElementCitation(node) {
|
|
|
263
914
|
}
|
|
264
915
|
return parts.join(' ');
|
|
265
916
|
}
|
|
917
|
+
/**
|
|
918
|
+
* Author-name wrappers a `<mixed-citation>` can carry as a direct child. Their
|
|
919
|
+
* parts routinely sit adjacent with zero characters between them, so the
|
|
920
|
+
* sibling-element spacing rule has to reach inside them rather than stopping at
|
|
921
|
+
* the wrapper. Scoped to these three deliberately: recursing into arbitrary
|
|
922
|
+
* elements would change how inline markup inside `<article-title>` and friends
|
|
923
|
+
* renders. (#124)
|
|
924
|
+
*/
|
|
925
|
+
const NAME_WRAPPER_TAGS = new Set(['name', 'string-name', 'person-group']);
|
|
926
|
+
/**
|
|
927
|
+
* Serialize a `<name>` / `<string-name>` / `<person-group>` subtree, applying
|
|
928
|
+
* the same rule `renderMixedCitation` applies to its own children: a single
|
|
929
|
+
* space between two adjacent elements that carry nothing between them, and
|
|
930
|
+
* source text emitted verbatim. `<person-group>` separates its names with real
|
|
931
|
+
* `, ` text nodes and `<string-name>` usually separates surname from given
|
|
932
|
+
* names with a newline, so a fixed separator would double punctuation the
|
|
933
|
+
* source already has — spacing only the zero-gap transitions leaves those
|
|
934
|
+
* records byte-identical. Returns raw text; the caller collapses whitespace
|
|
935
|
+
* once over the finished citation.
|
|
936
|
+
*/
|
|
937
|
+
function renderNameWrapper(node) {
|
|
938
|
+
let rendered = '';
|
|
939
|
+
let prevWasElement = false;
|
|
940
|
+
for (const child of childrenOf(node)) {
|
|
941
|
+
if (isTextNode(child)) {
|
|
942
|
+
const raw = textOf(child);
|
|
943
|
+
if (raw) {
|
|
944
|
+
rendered += raw;
|
|
945
|
+
prevWasElement = false;
|
|
946
|
+
}
|
|
947
|
+
continue;
|
|
948
|
+
}
|
|
949
|
+
const tag = tagNameOf(child) ?? '';
|
|
950
|
+
const part = NAME_WRAPPER_TAGS.has(tag) ? renderNameWrapper(child) : rawTextContent(child);
|
|
951
|
+
if (!part)
|
|
952
|
+
continue;
|
|
953
|
+
if (prevWasElement)
|
|
954
|
+
rendered += ' ';
|
|
955
|
+
rendered += part;
|
|
956
|
+
prevWasElement = true;
|
|
957
|
+
}
|
|
958
|
+
return rendered;
|
|
959
|
+
}
|
|
266
960
|
/**
|
|
267
961
|
* True when the citation text emitted so far already ends with a literal prefix
|
|
268
962
|
* naming this `pub-id-type` (`doi:`, `PMID `, `pmcid.`), so prepending the label
|
|
@@ -276,13 +970,14 @@ function hasLiteralIdPrefix(rendered, pubIdType) {
|
|
|
276
970
|
* Render a `<mixed-citation>` as a readable string.
|
|
277
971
|
*
|
|
278
972
|
* Mixed citations carry their punctuation in the text nodes between elements,
|
|
279
|
-
* so a flat `textContent()` reads correctly almost everywhere.
|
|
280
|
-
* defeat it,
|
|
281
|
-
* `<pub-id>`s fuse into one unreadable token (#115),
|
|
282
|
-
* straight into the volume that follows it (#123)
|
|
283
|
-
*
|
|
284
|
-
*
|
|
285
|
-
*
|
|
973
|
+
* so a flat `textContent()` reads correctly almost everywhere. Three adjacencies
|
|
974
|
+
* defeat it, all with zero characters between the elements: consecutive typed
|
|
975
|
+
* `<pub-id>`s fuse into one unreadable token (#115), an inline title runs
|
|
976
|
+
* straight into the volume that follows it (#123), and a `<surname>` glues onto
|
|
977
|
+
* the `<given-names>` beside it inside an author-name wrapper (#124). Walk the
|
|
978
|
+
* direct children so those cases can be repaired without touching `textContent`,
|
|
979
|
+
* which abstracts, titles, and body paragraphs share and where zero-gap
|
|
980
|
+
* adjacency is often intentional.
|
|
286
981
|
*
|
|
287
982
|
* Typed `<pub-id>`s are labeled (unless literal prefix text already names the
|
|
288
983
|
* type), and a single space separates two adjacent elements that carry nothing
|
|
@@ -305,9 +1000,10 @@ function renderMixedCitation(node) {
|
|
|
305
1000
|
}
|
|
306
1001
|
continue;
|
|
307
1002
|
}
|
|
1003
|
+
const tag = tagNameOf(child) ?? '';
|
|
308
1004
|
let part;
|
|
309
1005
|
let labeled = false;
|
|
310
|
-
if (
|
|
1006
|
+
if (tag === 'pub-id') {
|
|
311
1007
|
const value = textContent(child);
|
|
312
1008
|
if (!value)
|
|
313
1009
|
continue;
|
|
@@ -321,6 +1017,9 @@ function renderMixedCitation(node) {
|
|
|
321
1017
|
part = value;
|
|
322
1018
|
}
|
|
323
1019
|
}
|
|
1020
|
+
else if (NAME_WRAPPER_TAGS.has(tag)) {
|
|
1021
|
+
part = renderNameWrapper(child);
|
|
1022
|
+
}
|
|
324
1023
|
else {
|
|
325
1024
|
part = rawTextContent(child);
|
|
326
1025
|
}
|
|
@@ -332,43 +1031,57 @@ function renderMixedCitation(node) {
|
|
|
332
1031
|
return rendered.replace(/\s+/g, ' ').trim();
|
|
333
1032
|
}
|
|
334
1033
|
/**
|
|
335
|
-
* Extract references from
|
|
336
|
-
*
|
|
337
|
-
*
|
|
338
|
-
*
|
|
339
|
-
* `
|
|
1034
|
+
* Extract references from anywhere under `root` — pass the `<article>` node to
|
|
1035
|
+
* cover a whole document, or a `<back>` node to scope the search to it.
|
|
1036
|
+
*
|
|
1037
|
+
* `<ref-list>` placement is not uniform: roughly 60% of Europe PMC deposits nest
|
|
1038
|
+
* it under `body/sec/sec` and the rest put it directly under `<back>`, so a
|
|
1039
|
+
* search scoped to a direct `<back>` child misses the majority. Every
|
|
1040
|
+
* `<ref-list>` descendant is collected in document order instead, and a `<ref>`
|
|
1041
|
+
* id already seen is skipped so a document exposing the same list under both
|
|
1042
|
+
* containers still yields each reference once. (#116)
|
|
1043
|
+
*
|
|
1044
|
+
* Prefers `<mixed-citation>` over `<element-citation>`, descending into
|
|
1045
|
+
* `<citation-alternatives>` when a ref carries both forms there rather than as
|
|
1046
|
+
* direct children of `<ref>`. Both forms are rendered child-by-child so adjacent
|
|
1047
|
+
* elements stay separable (see `renderMixedCitation` and
|
|
1048
|
+
* `renderElementCitation`).
|
|
340
1049
|
*/
|
|
341
|
-
export function extractReferences(
|
|
342
|
-
if (!
|
|
343
|
-
return [];
|
|
344
|
-
const refList = findOne(back, 'ref-list');
|
|
345
|
-
if (!refList)
|
|
1050
|
+
export function extractReferences(root) {
|
|
1051
|
+
if (!root)
|
|
346
1052
|
return [];
|
|
347
1053
|
const results = [];
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
citation =
|
|
1054
|
+
const seenIds = new Set();
|
|
1055
|
+
for (const refList of findAllDescendants(root, 'ref-list')) {
|
|
1056
|
+
for (const ref of findAll(refList, 'ref')) {
|
|
1057
|
+
const id = attrOf(ref, 'id');
|
|
1058
|
+
if (id && seenIds.has(id))
|
|
1059
|
+
continue;
|
|
1060
|
+
// JATS wraps the two citation forms in <citation-alternatives> (the NLM
|
|
1061
|
+
// construct carrying both a structured <element-citation> and a readable
|
|
1062
|
+
// <mixed-citation>). findOne matches direct children only, so resolve that
|
|
1063
|
+
// container first; refs with a direct citation node fall through to `ref`.
|
|
1064
|
+
const container = findOne(ref, 'citation-alternatives') ?? ref;
|
|
1065
|
+
const mixedCitation = findOne(container, 'mixed-citation');
|
|
1066
|
+
const elementCitation = findOne(container, 'element-citation');
|
|
1067
|
+
let citation = '';
|
|
1068
|
+
if (mixedCitation) {
|
|
1069
|
+
citation = renderMixedCitation(mixedCitation);
|
|
1070
|
+
}
|
|
1071
|
+
else if (elementCitation) {
|
|
1072
|
+
citation = renderElementCitation(elementCitation);
|
|
1073
|
+
}
|
|
1074
|
+
if (!citation)
|
|
1075
|
+
continue;
|
|
1076
|
+
if (id)
|
|
1077
|
+
seenIds.add(id);
|
|
1078
|
+
const label = textContent(findOne(ref, 'label')) || undefined;
|
|
1079
|
+
results.push({
|
|
1080
|
+
...(id && { id }),
|
|
1081
|
+
...(label && { label }),
|
|
1082
|
+
citation,
|
|
1083
|
+
});
|
|
362
1084
|
}
|
|
363
|
-
if (!citation)
|
|
364
|
-
continue;
|
|
365
|
-
const id = attrOf(ref, 'id');
|
|
366
|
-
const label = textContent(findOne(ref, 'label')) || undefined;
|
|
367
|
-
results.push({
|
|
368
|
-
...(id && { id }),
|
|
369
|
-
...(label && { label }),
|
|
370
|
-
citation,
|
|
371
|
-
});
|
|
372
1085
|
}
|
|
373
1086
|
return results;
|
|
374
1087
|
}
|
|
@@ -384,7 +1097,6 @@ export function parsePmcArticle(articleNode) {
|
|
|
384
1097
|
const articleMeta = findOne(front, 'article-meta');
|
|
385
1098
|
const journalMeta = findOne(front, 'journal-meta');
|
|
386
1099
|
const body = findOne(articleNode, 'body');
|
|
387
|
-
const back = findOne(articleNode, 'back');
|
|
388
1100
|
const pmcId = extractArticleId(articleMeta, 'pmcid') ?? extractArticleId(articleMeta, 'pmc-uid') ?? '';
|
|
389
1101
|
const pmid = extractArticleId(articleMeta, 'pmid');
|
|
390
1102
|
const doi = extractArticleId(articleMeta, 'doi');
|
|
@@ -397,7 +1109,9 @@ export function parsePmcArticle(articleNode) {
|
|
|
397
1109
|
const abstract = extractAbstract(articleMeta);
|
|
398
1110
|
const keywords = extractKeywords(articleMeta);
|
|
399
1111
|
const sections = extractBodySections(body);
|
|
400
|
-
const references = extractReferences(
|
|
1112
|
+
const references = extractReferences(articleNode);
|
|
1113
|
+
const tables = extractPmcTables(articleNode);
|
|
1114
|
+
const assets = extractPmcAssets(articleNode);
|
|
401
1115
|
const normalizedPmcId = !pmcId ? '' : pmcId.startsWith('PMC') ? pmcId : `PMC${pmcId}`;
|
|
402
1116
|
const articleType = attrOf(articleNode, 'article-type');
|
|
403
1117
|
return {
|
|
@@ -413,6 +1127,8 @@ export function parsePmcArticle(articleNode) {
|
|
|
413
1127
|
...(keywords.length > 0 && { keywords }),
|
|
414
1128
|
sections,
|
|
415
1129
|
...(references.length > 0 && { references }),
|
|
1130
|
+
...(tables.length > 0 && { tables }),
|
|
1131
|
+
...(assets.length > 0 && { assets }),
|
|
416
1132
|
...(articleType && { articleType }),
|
|
417
1133
|
pmcUrl: `https://www.ncbi.nlm.nih.gov/pmc/articles/${normalizedPmcId}/`,
|
|
418
1134
|
...(pmid && { pubmedUrl: `https://pubmed.ncbi.nlm.nih.gov/${pmid}/` }),
|