@cyanheads/pubmed-mcp-server 2.10.9 → 2.10.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. package/AGENTS.md +1 -1
  2. package/CLAUDE.md +1 -1
  3. package/README.md +13 -5
  4. package/dist/mcp-server/tools/definitions/_schemas.d.ts +16 -0
  5. package/dist/mcp-server/tools/definitions/_schemas.d.ts.map +1 -1
  6. package/dist/mcp-server/tools/definitions/_schemas.js +20 -0
  7. package/dist/mcp-server/tools/definitions/_schemas.js.map +1 -1
  8. package/dist/mcp-server/tools/definitions/convert-ids.tool.d.ts +6 -0
  9. package/dist/mcp-server/tools/definitions/convert-ids.tool.d.ts.map +1 -1
  10. package/dist/mcp-server/tools/definitions/convert-ids.tool.js +28 -3
  11. package/dist/mcp-server/tools/definitions/convert-ids.tool.js.map +1 -1
  12. package/dist/mcp-server/tools/definitions/fetch-articles.tool.d.ts +27 -0
  13. package/dist/mcp-server/tools/definitions/fetch-articles.tool.d.ts.map +1 -1
  14. package/dist/mcp-server/tools/definitions/fetch-articles.tool.js +166 -24
  15. package/dist/mcp-server/tools/definitions/fetch-articles.tool.js.map +1 -1
  16. package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.d.ts +18 -0
  17. package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.d.ts.map +1 -1
  18. package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.js +307 -68
  19. package/dist/mcp-server/tools/definitions/fetch-fulltext.tool.js.map +1 -1
  20. package/dist/mcp-server/tools/definitions/find-related.tool.d.ts +12 -0
  21. package/dist/mcp-server/tools/definitions/find-related.tool.d.ts.map +1 -1
  22. package/dist/mcp-server/tools/definitions/find-related.tool.js +171 -27
  23. package/dist/mcp-server/tools/definitions/find-related.tool.js.map +1 -1
  24. package/dist/mcp-server/tools/definitions/format-citations.tool.d.ts.map +1 -1
  25. package/dist/mcp-server/tools/definitions/format-citations.tool.js +11 -13
  26. package/dist/mcp-server/tools/definitions/format-citations.tool.js.map +1 -1
  27. package/dist/mcp-server/tools/definitions/lookup-citation.tool.d.ts.map +1 -1
  28. package/dist/mcp-server/tools/definitions/lookup-citation.tool.js +42 -8
  29. package/dist/mcp-server/tools/definitions/lookup-citation.tool.js.map +1 -1
  30. package/dist/mcp-server/tools/definitions/lookup-mesh.tool.d.ts +6 -0
  31. package/dist/mcp-server/tools/definitions/lookup-mesh.tool.d.ts.map +1 -1
  32. package/dist/mcp-server/tools/definitions/lookup-mesh.tool.js +14 -3
  33. package/dist/mcp-server/tools/definitions/lookup-mesh.tool.js.map +1 -1
  34. package/dist/mcp-server/tools/definitions/search-articles.tool.d.ts +10 -0
  35. package/dist/mcp-server/tools/definitions/search-articles.tool.d.ts.map +1 -1
  36. package/dist/mcp-server/tools/definitions/search-articles.tool.js +52 -5
  37. package/dist/mcp-server/tools/definitions/search-articles.tool.js.map +1 -1
  38. package/dist/mcp-server/tools/definitions/spell-check.tool.d.ts +6 -0
  39. package/dist/mcp-server/tools/definitions/spell-check.tool.d.ts.map +1 -1
  40. package/dist/mcp-server/tools/definitions/spell-check.tool.js +15 -3
  41. package/dist/mcp-server/tools/definitions/spell-check.tool.js.map +1 -1
  42. package/dist/services/error-contracts.d.ts +39 -2
  43. package/dist/services/error-contracts.d.ts.map +1 -1
  44. package/dist/services/error-contracts.js +43 -2
  45. package/dist/services/error-contracts.js.map +1 -1
  46. package/dist/services/ncbi/formatting/citation-formatter.d.ts +0 -29
  47. package/dist/services/ncbi/formatting/citation-formatter.d.ts.map +1 -1
  48. package/dist/services/ncbi/formatting/citation-formatter.js +381 -30
  49. package/dist/services/ncbi/formatting/citation-formatter.js.map +1 -1
  50. package/dist/services/ncbi/ncbi-service.d.ts +3 -2
  51. package/dist/services/ncbi/ncbi-service.d.ts.map +1 -1
  52. package/dist/services/ncbi/ncbi-service.js +22 -10
  53. package/dist/services/ncbi/ncbi-service.js.map +1 -1
  54. package/dist/services/ncbi/parsing/article-parser.d.ts +45 -2
  55. package/dist/services/ncbi/parsing/article-parser.d.ts.map +1 -1
  56. package/dist/services/ncbi/parsing/article-parser.js +208 -1
  57. package/dist/services/ncbi/parsing/article-parser.js.map +1 -1
  58. package/dist/services/ncbi/parsing/esummary-parser.d.ts.map +1 -1
  59. package/dist/services/ncbi/parsing/esummary-parser.js +32 -1
  60. package/dist/services/ncbi/parsing/esummary-parser.js.map +1 -1
  61. package/dist/services/ncbi/parsing/pmc-article-parser.d.ts +24 -4
  62. package/dist/services/ncbi/parsing/pmc-article-parser.d.ts.map +1 -1
  63. package/dist/services/ncbi/parsing/pmc-article-parser.js +505 -57
  64. package/dist/services/ncbi/parsing/pmc-article-parser.js.map +1 -1
  65. package/dist/services/ncbi/response-handler.d.ts.map +1 -1
  66. package/dist/services/ncbi/response-handler.js +42 -1
  67. package/dist/services/ncbi/response-handler.js.map +1 -1
  68. package/dist/services/ncbi/types.d.ts +215 -2
  69. package/dist/services/ncbi/types.d.ts.map +1 -1
  70. package/dist/services/openalex/api-client.d.ts +13 -4
  71. package/dist/services/openalex/api-client.d.ts.map +1 -1
  72. package/dist/services/openalex/api-client.js +19 -8
  73. package/dist/services/openalex/api-client.js.map +1 -1
  74. package/dist/services/openalex/openalex-service.d.ts +34 -17
  75. package/dist/services/openalex/openalex-service.d.ts.map +1 -1
  76. package/dist/services/openalex/openalex-service.js +119 -35
  77. package/dist/services/openalex/openalex-service.js.map +1 -1
  78. package/dist/services/openalex/types.d.ts +34 -1
  79. package/dist/services/openalex/types.d.ts.map +1 -1
  80. package/dist/services/openalex/types.js +20 -0
  81. package/dist/services/openalex/types.js.map +1 -1
  82. package/package.json +1 -1
  83. package/server.json +3 -3
@@ -30,7 +30,7 @@ import { ensureArray } from '../../../services/ncbi/parsing/xml-helpers.js';
30
30
  import { getUnpaywallService, } from '../../../services/unpaywall/unpaywall-service.js';
31
31
  import { fitWholeItems } from './_budget.js';
32
32
  import { conceptMeta, EDAM_DATA_RETRIEVAL, SCHEMA_SCHOLARLY_ARTICLE } from './_concepts.js';
33
- import { pmidStringSchema } from './_schemas.js';
33
+ import { doiStringSchema, pmcidStringSchema, pmidStringSchema } from './_schemas.js';
34
34
  import { escapeMarkdownInline, escapeMarkdownTableCell, sliceCodeUnits } from './_text.js';
35
35
  function normalizePmcId(id) {
36
36
  return id.replace(/^PMC/i, '');
@@ -152,18 +152,99 @@ function applyTableFilters(article, filters) {
152
152
  return withTables(article, article.tables.filter((t) => t.sectionTitle !== undefined && surviving.has(t.sectionTitle)));
153
153
  }
154
154
  /**
155
- * Apply the requested section/reference/table filters, then clamp the section
156
- * tree to the depth the output schema carries. All of it runs here so every path
155
+ * Replace an article's asset list, dropping the field entirely when nothing is
156
+ * left — the same rule {@link withTables} follows, and for the same reason: an
157
+ * empty array claims the article deposits no figures, which a filtered-to-nothing
158
+ * list does not mean.
159
+ */
160
+ function withAssets(article, assets) {
161
+ const { assets: _replaced, ...rest } = article;
162
+ return (assets.length > 0 ? { ...rest, assets } : rest);
163
+ }
164
+ /**
165
+ * The positional marker the parser leaves in section text where an asset was
166
+ * lifted out — `[Figure: Fig. 1]`, `[Supplementary: Table S3]`, or the bare
167
+ * `[Figure]` / `[Supplementary]` when the deposit carries no label.
168
+ *
169
+ * Mirrors `assetMarker` in `pmc-article-parser.ts`, which is the only producer.
170
+ * The two must stay byte-identical: this is what {@link stripAssetMarkers}
171
+ * removes, and a marker built any other way would leave the real one in place.
172
+ * Derived per asset rather than matched as a pattern, so prose that happens to
173
+ * carry a bracketed word is never touched. (#130)
174
+ */
175
+ function assetMarkerText(asset) {
176
+ const kind = asset.assetType === 'figure' ? 'Figure' : 'Supplementary';
177
+ return asset.label ? `[${kind}: ${asset.label}]` : `[${kind}]`;
178
+ }
179
+ /**
180
+ * Remove every marker in `markers` from one text field and close the gap it
181
+ * leaves: trailing whitespace on a line the marker ended, and the blank line a
182
+ * marker that stood alone as its own block leaves behind.
183
+ */
184
+ function stripAssetMarkers(text, markers) {
185
+ if (!text)
186
+ return text;
187
+ let out = text;
188
+ for (const marker of markers)
189
+ out = out.split(marker).join('');
190
+ if (out === text)
191
+ return text;
192
+ return out
193
+ .split('\n')
194
+ .map((line) => line.replace(/[ \t]+$/, ''))
195
+ .join('\n')
196
+ .replace(/\n{3,}/g, '\n\n')
197
+ .trim();
198
+ }
199
+ /** Strip asset markers from a section and every subsection beneath it. */
200
+ function sectionWithoutMarkers(section, markers) {
201
+ const text = stripAssetMarkers(section.text, markers);
202
+ const subsections = section.subsections?.map((sub) => sectionWithoutMarkers(sub, markers));
203
+ if (text === section.text && subsections === undefined)
204
+ return section;
205
+ return { ...section, text, ...(subsections && { subsections }) };
206
+ }
207
+ /**
208
+ * Narrow the asset list to what the request asked for, mirroring {@link
209
+ * applyTableFilters}: `includeAssets: false` is the wholesale off switch, and an
210
+ * active `sections` filter keeps an asset whose named section survived while
211
+ * dropping one that names no section at all — a `<floats-group>` deposit — since
212
+ * the caller asked for named headings and it belongs to none.
213
+ *
214
+ * Where it goes beyond the table counterpart: turning assets off also removes the
215
+ * positional markers the parser left in the section text. The marker is the
216
+ * lift's anchor, and without the array to resolve it against it points at
217
+ * nothing. Prose-shaped blocks — lists, quotes, formulae — are section text
218
+ * rather than assets and this switch never touches them. (#130)
219
+ */
220
+ function applyAssetFilters(article, filters) {
221
+ const assets = article.assets;
222
+ if (!assets?.length)
223
+ return article;
224
+ if (!filters.includeAssets) {
225
+ const markers = assets.map(assetMarkerText);
226
+ return withAssets({ ...article, sections: article.sections.map((s) => sectionWithoutMarkers(s, markers)) }, []);
227
+ }
228
+ if (!filters.sections?.length)
229
+ return article;
230
+ const surviving = matchedSectionTitles(article.sections, filters.sections.map(lowerCase));
231
+ return withAssets(article, assets.filter((a) => a.sectionTitle !== undefined && surviving.has(a.sectionTitle)));
232
+ }
233
+ /**
234
+ * Apply the requested section/reference/table/asset filters, then clamp the
235
+ * section tree to the depth the output schema carries. All of it runs here so every path
157
236
  * producing a `pmc` article — PMC EFetch and the Europe PMC stage — shares one
158
237
  * shape, and the budget helpers downstream count the text that will actually
159
238
  * survive validation. (#112)
160
239
  *
161
- * Order is load-bearing. Tables are matched against the section tree *after*
162
- * `filterSections` and the `maxSections` slice, so they narrow with exactly what
163
- * the response returns, and *before* `clampSectionDepth`, which folds sections
164
- * past {@link MAX_SECTION_DEPTH} into a parent's text — their titles vanish from
165
- * the output while a table still names them, so a title set built after the
166
- * clamp would drop tables that should have survived. (#111)
240
+ * Order is load-bearing. Tables and assets are matched against the section tree
241
+ * *after* `filterSections` and the `maxSections` slice, so they narrow with
242
+ * exactly what the response returns, and *before* `clampSectionDepth`, which
243
+ * folds sections past {@link MAX_SECTION_DEPTH} into a parent's text — their
244
+ * titles vanish from the output while a table or figure still names them, so a
245
+ * title set built after the clamp would drop entries that should have survived.
246
+ * Stripping the asset markers also has to precede the clamp, or a marker in a
247
+ * folded-away section survives in the text it was folded into. (#111, #130)
167
248
  */
168
249
  function applyPmcFilters(article, filters) {
169
250
  let out = article;
@@ -178,6 +259,7 @@ function applyPmcFilters(article, filters) {
178
259
  out = rest;
179
260
  }
180
261
  out = applyTableFilters(out, filters);
262
+ out = applyAssetFilters(out, filters);
181
263
  return { ...out, sections: clampSectionDepth(out.sections) };
182
264
  }
183
265
  /**
@@ -287,6 +369,10 @@ const JournalSchema = z
287
369
  volume: z.string().optional().describe('Volume number'),
288
370
  issue: z.string().optional().describe('Issue number'),
289
371
  pages: z.string().optional().describe('Page range'),
372
+ elocationId: z
373
+ .string()
374
+ .optional()
375
+ .describe('Electronic article locator from JATS `<elocation-id>` — the publisher-assigned article number (e.g. "e20542"). Journals that assign article numbers deposit no `<fpage>`, so this is the only locator on roughly half of PMC records. Never a substitute for `pages`; JATS carries no type attribute, so there is no counterpart to the `elocationIdType` that `pubmed_fetch_articles` reports.'),
290
376
  })
291
377
  .describe('Journal information');
292
378
  const ReferenceSchema = z
@@ -338,6 +424,44 @@ const TableSchema = z
338
424
  .describe('Why `rows` is empty — set only then. graphic-only: the table was deposited as an image with no underlying markup. cals-tgroup: the table uses the CALS `<tgroup>` model, which this server does not extract (0 of 283 tables in an open-access survey used it). no-rows: the markup carried no rows. The label and caption are still returned, so a table that could not be read is visible rather than silently missing.'),
339
425
  })
340
426
  .describe('One table from the article, with its cells, caption, and owning section');
427
+ /**
428
+ * One `<fig>` or `<supplementary-material>`, hung off the article beside
429
+ * {@link TableSchema} and for the same reason: a sixth of them sit in
430
+ * `<floats-group>`, `<back>` or an appendix with no `<sec>` to attach to, so an
431
+ * asset inside a section names it in {@link sectionTitle} rather than being
432
+ * placed by position.
433
+ *
434
+ * Scalar leaves only. `articles[]` → the article union → `assets[]` → a leaf is
435
+ * six of the eight hops the `format-parity` sentinel walker allows; a nested
436
+ * object here (a `files[]` list, say) spends the remaining two and leaves nothing
437
+ * for a later field. Re-run `bun run lint:mcp` after any change to this shape.
438
+ *
439
+ * There is no media-type field and no per-asset unextractable reason: `mimetype`
440
+ * appears on no observed deposit, and an uncaptioned supplement is still fully
441
+ * described by its `id` and `href`, unlike a table that promises cells and
442
+ * carries none. (#130)
443
+ */
444
+ const AssetSchema = z
445
+ .object({
446
+ assetType: z
447
+ .enum(['figure', 'supplementary-material'])
448
+ .describe('Which captioned element this came from — `figure` for a `<fig>`, `supplementary-material` for a `<supplementary-material>` deposit'),
449
+ label: z.string().optional().describe('Display label as printed, e.g. `Fig. 1`'),
450
+ caption: z.string().optional().describe('Caption text, with the label excluded'),
451
+ id: z
452
+ .string()
453
+ .optional()
454
+ .describe('JATS `id` attribute — the target body-text cross-references point at'),
455
+ sectionTitle: z
456
+ .string()
457
+ .optional()
458
+ .describe('Title of the innermost section enclosing the asset, wherever that section sits — body, `<back>` matter, or an appendix all count. Absent for an asset inside no section at all, such as a `<floats-group>` deposit.'),
459
+ href: z
460
+ .string()
461
+ .optional()
462
+ .describe('The `<graphic>`/`<media>` `@xlink:href` exactly as deposited — a pointer into the PMC deposit (`MOL2-20-1253-g001.jpg`), not a fetchable URL. No absolute form of it resolves; read the rendered article at `pmcUrl` instead. Absent when the deposit names no file.'),
463
+ })
464
+ .describe('One figure or supplementary-material item, with its caption, pointer, and section');
341
465
  const PublicationDateSchema = z
342
466
  .object({
343
467
  year: z.string().optional().describe('Publication year'),
@@ -377,6 +501,10 @@ const PmcArticleSchema = z
377
501
  .array(TableSchema)
378
502
  .optional()
379
503
  .describe('Every `<table-wrap>` the article carries, in document order — from the body and from `<floats-group>`, `<back>` and appendices alike. Absent when the article deposits none, when `includeTables` is false, or when a `sections` filter left none standing.'),
504
+ assets: z
505
+ .array(AssetSchema)
506
+ .optional()
507
+ .describe('Every `<fig>` and `<supplementary-material>` the article carries, in document order — from the body and from `<floats-group>`, `<back>` and appendices alike. Each one lifted from the body leaves a `[Figure: <label>]` or `[Supplementary: <label>]` marker at its position in the section text, so reading order survives the lift. Absent when the article deposits none, when `includeAssets` is false, or when a `sections` filter left none standing.'),
380
508
  references: z.array(ReferenceSchema).optional().describe('Reference list'),
381
509
  epmcId: z
382
510
  .string()
@@ -441,12 +569,13 @@ const UnavailableReasonSchema = z
441
569
  'no-epmc-fulltext',
442
570
  'no-body',
443
571
  'no-doi',
572
+ 'doi-lookup-failed',
444
573
  'no-oa',
445
574
  'fetch-failed',
446
575
  'parse-failed',
447
576
  'service-error',
448
577
  ])
449
- .describe('Why no full text was returned — the most specific signal any tier that answered reported. not-found: upstream returned no record for this ID. no-pmc-fallback-disabled: every tier was skipped (`triedTiers` is all `not-attempted`) — typically because EPMC (`EUROPEPMC_ENABLED`) and Unpaywall (`UNPAYWALL_EMAIL`) are not configured. no-epmc-fulltext: EPMC indexed the record but publishes no fullTextXML. no-body: the record was retrieved but carries front matter and abstract only, with no body sections — use `pubmed_fetch_articles` for the metadata. no-doi: no DOI to query Unpaywall. no-oa: Unpaywall has no OA copy. fetch-failed: download failed. parse-failed: extraction empty. service-error: upstream server failure (threw, timed out, or returned malformed data). A reason never means the chain ran to completion — read `unqueriedTiers` for that.');
578
+ .describe('Why no full text was returned — the most specific signal any tier that answered reported. not-found: upstream returned no record for this ID. no-pmc-fallback-disabled: every tier was skipped (`triedTiers` is all `not-attempted`) — typically because EPMC (`EUROPEPMC_ENABLED`) and Unpaywall (`UNPAYWALL_EMAIL`) are not configured. no-epmc-fulltext: EPMC indexed the record but publishes no fullTextXML. no-body: the record was retrieved but carries front matter and abstract only, with no body sections — use `pubmed_fetch_articles` for the metadata. no-doi: the DOI lookup ran and this record has none, so Unpaywall could not be queried. doi-lookup-failed: the DOI lookup itself errored, so whether a DOI exists is unknown and Unpaywall was never reached — retry the request; unlike no-doi this is a transient failure, not a settled answer. no-oa: Unpaywall has no OA copy. fetch-failed: download failed. parse-failed: extraction empty. service-error: upstream server failure (threw, timed out, or returned malformed data). A reason never means the chain ran to completion — read `unqueriedTiers` for that.');
450
579
  const UnqueriedTierSchema = z
451
580
  .enum(['europepmc', 'unpaywall'])
452
581
  .describe('A fallback tier this deployment has not configured');
@@ -457,12 +586,13 @@ const TierOutcomeSchema = z
457
586
  'no-fulltext',
458
587
  'no-body',
459
588
  'no-doi',
589
+ 'doi-lookup-failed',
460
590
  'no-oa',
461
591
  'fetch-failed',
462
592
  'parse-failed',
463
593
  'service-error',
464
594
  ])
465
- .describe('Per-tier outcome. not-attempted: tier was skipped. miss: tier returned no record. no-fulltext: EPMC indexed the record but publishes no fullTextXML. no-body: the tier returned a record with front matter and abstract but no body sections, so the chain continued. no-doi: no DOI to query Unpaywall. no-oa: Unpaywall reports no open-access copy. fetch-failed: OA copy download failed. parse-failed: extraction produced empty content. service-error: tier service threw.');
595
+ .describe('Per-tier outcome. not-attempted: tier was skipped. miss: tier returned no record. no-fulltext: EPMC indexed the record but publishes no fullTextXML. no-body: the tier returned a record with front matter and abstract but no body sections, so the chain continued. no-doi: the DOI lookup ran and this record has none, so Unpaywall could not be queried. doi-lookup-failed: the DOI lookup itself errored, so whether a DOI exists is unknown and Unpaywall was never reached — retry the request. no-oa: Unpaywall reports no open-access copy. fetch-failed: OA copy download failed. parse-failed: extraction produced empty content. service-error: tier service threw.');
466
596
  const TriedTierSchema = z
467
597
  .object({
468
598
  tier: z.enum(['pmc', 'europepmc', 'unpaywall']).describe('Which tier in the resolution chain'),
@@ -525,6 +655,14 @@ const TruncatedArticleSchema = z
525
655
  .array(z.string())
526
656
  .optional()
527
657
  .describe("The dropped tables by name, in document order — each table's label, else its `id`, else `table <n>` for its position in the article. Names the tables a bare count only hints at, the way `deferred.ids` names deferred articles. Every table from the first that did not fit onward is here: admission stops at that table rather than skipping ahead to a smaller one, so these are contiguous. Absent when none were dropped."),
658
+ omittedAssets: z
659
+ .number()
660
+ .optional()
661
+ .describe('Figures and supplementary items this article dropped whole because the budget left no room once sections and tables were served. An asset is never returned with a truncated caption, so it is either returned complete or counted here. Absent when none were dropped.'),
662
+ omittedAssetNames: z
663
+ .array(z.string())
664
+ .optional()
665
+ .describe("The dropped assets by name, in document order — each asset's label, else its `id`, else `asset <n>` for its position in the article. Contiguous for the same reason `omittedTableNames` is: admission stops at the first asset that did not fit rather than skipping ahead to a smaller one. Absent when none were dropped."),
528
666
  })
529
667
  .describe('Character accounting for one article the budget shortened');
530
668
  const TruncationSchema = z
@@ -550,6 +688,10 @@ const TruncationSchema = z
550
688
  .number()
551
689
  .optional()
552
690
  .describe('Tables dropped whole across every budgeted article, because the budget left no room once body sections were served. Absent when none were dropped. Re-request the affected articles with a higher `maxCharacters`, or with `sections` narrowed, to receive them.'),
691
+ omittedAssets: z
692
+ .number()
693
+ .optional()
694
+ .describe('Figures and supplementary items dropped whole across every budgeted article, because the budget left no room once body sections and tables were served. Absent when none were dropped. Re-request the affected articles with a higher `maxCharacters`, or with `sections` narrowed, to receive them.'),
553
695
  articles: z
554
696
  .array(TruncatedArticleSchema)
555
697
  .describe('Per-article accounting, covering only the articles the budget shortened'),
@@ -609,34 +751,37 @@ function tableCharacters(table) {
609
751
  return totalLength([table.label, table.caption, table.footnotes, ...table.rows.flat()]);
610
752
  }
611
753
  /**
612
- * Admit tables in document order until the allowance is spent, then drop the
613
- * rest whole and name them.
754
+ * Characters an asset costs the budget: the text it carries — label, caption,
755
+ * and the `href` pointer. Positional metadata is excluded, exactly as {@link
756
+ * tableCharacters} excludes `sectionTitle` and `id`: it places the asset rather
757
+ * than being content the caller asked for. An asset is admitted or dropped
758
+ * whole — a caption cut in half is a caption that says something else — so there
759
+ * is no partial measure to take. (#130)
760
+ */
761
+ function assetCharacters(asset) {
762
+ return totalLength([asset.label, asset.caption, asset.href]);
763
+ }
764
+ /**
765
+ * Admit tables, then assets, in document order until the allowance is spent,
766
+ * then drop the rest whole and name them. The split is {@link fitWholeItems}'s
767
+ * prefix cut, the same one `maxResponseCharacters` applies to whole articles.
614
768
  *
615
- * Admission stops at the first table that does not fit rather than skipping past
769
+ * Admission stops at the first entry that does not fit rather than skipping past
616
770
  * it to a smaller one further down: the returned set stays a document-order
617
771
  * prefix, so a caller reading it knows where the response stopped instead of
618
- * receiving a late table with nothing saying the earlier ones exist. A table
619
- * that does not fit is never cut either — half a grid reads as a complete one
772
+ * receiving a late entry with nothing saying the earlier ones exist. An entry
773
+ * that does not fit is never cut either — half a grid reads as a complete table
620
774
  * carrying values that were never deposited, the defect the table extraction
621
- * exists to fix. Both mirror how `maxResponseCharacters` defers a whole article
622
- * rather than half-populating it. (#111)
775
+ * exists to fix, and half a caption says something the deposit does not.
776
+ * (#111, #130)
623
777
  */
624
- function fitTables(tables, allowance) {
625
- const kept = [];
626
- let spent = 0;
627
- for (const [index, table] of tables.entries()) {
628
- const size = tableCharacters(table);
629
- if (spent + size > allowance) {
630
- return {
631
- kept,
632
- omittedNames: tables.slice(index).map((t, i) => tableDisplayName(t, index + i)),
633
- spent,
634
- };
635
- }
636
- kept.push(table);
637
- spent += size;
638
- }
639
- return { kept, omittedNames: [], spent };
778
+ function fitWholeNamed(items, allowance, measure, name) {
779
+ const fit = fitWholeItems(items, allowance, measure);
780
+ return {
781
+ kept: fit.kept,
782
+ omittedNames: fit.deferred.map((item, i) => name(item, fit.kept.length + i)),
783
+ spent: fit.keptCharacters,
784
+ };
640
785
  }
641
786
  /**
642
787
  * Rebuild a section subtree from `fitted`, consuming one entry per node in the
@@ -728,11 +873,11 @@ function allotSectionBudgets(sizes, budget) {
728
873
  * or cut — the budget spends on body text and table content, keeping every
729
874
  * article citable.
730
875
  *
731
- * Body sections are served first and tables spend what `maxCharacters` leaves,
732
- * in document order: a table that does not fit is dropped whole and counted,
733
- * never truncated into a partial grid. With no `maxCharacters` — a bare
734
- * `maxCharactersPerSection` request — nothing bounds the tables and every one is
735
- * kept. (#111)
876
+ * Body sections are served first, then tables, then assets, each spending what
877
+ * `maxCharacters` has left, in document order: an item that does not fit is
878
+ * dropped whole and counted, never truncated into a partial grid or a caption cut
879
+ * short. With no `maxCharacters` — a bare `maxCharactersPerSection` request —
880
+ * nothing bounds either list and every entry is kept. (#111, #130)
736
881
  *
737
882
  * Returns the article untouched (same object identity) when no budget was
738
883
  * requested or nothing exceeded it. A section left with zero characters is
@@ -742,12 +887,15 @@ function allotSectionBudgets(sizes, budget) {
742
887
  */
743
888
  function applyPmcBudget(article, budget) {
744
889
  const tables = article.tables ?? [];
745
- if (!budgetRequested(budget) || (article.sections.length === 0 && tables.length === 0)) {
890
+ const assets = article.assets ?? [];
891
+ if (!budgetRequested(budget) ||
892
+ (article.sections.length === 0 && tables.length === 0 && assets.length === 0)) {
746
893
  return { article, omittedSections: 0 };
747
894
  }
748
895
  const sizes = article.sections.map(sectionCharacters);
749
896
  const tablesOriginal = tables.reduce((sum, table) => sum + tableCharacters(table), 0);
750
- const originalCharacters = sizes.reduce((sum, size) => sum + size, 0) + tablesOriginal;
897
+ const assetsOriginal = assets.reduce((sum, asset) => sum + assetCharacters(asset), 0);
898
+ const originalCharacters = sizes.reduce((sum, size) => sum + size, 0) + tablesOriginal + assetsOriginal;
751
899
  const allowances = allotSectionBudgets(sizes, budget);
752
900
  const kept = [];
753
901
  const sectionReports = [];
@@ -770,19 +918,26 @@ function applyPmcBudget(article, budget) {
770
918
  }
771
919
  kept.push(withFittedTexts(section, fitted, { i: 0 }));
772
920
  });
773
- // Sections are served first; the tables spend whatever `maxCharacters` has
774
- // left. A bare per-section budget sets no total, so nothing bounds them.
775
- const tableAllowance = budget.maxCharacters === undefined
921
+ // Sections are served first, then tables, then assets — each spending whatever
922
+ // `maxCharacters` has left. A bare per-section budget sets no total, so nothing
923
+ // bounds either list.
924
+ const remainingAllowance = () => budget.maxCharacters === undefined
776
925
  ? Number.POSITIVE_INFINITY
777
926
  : Math.max(budget.maxCharacters - returnedCharacters, 0);
778
- const fittedTables = fitTables(tables, tableAllowance);
927
+ const fittedTables = fitWholeNamed(tables, remainingAllowance(), tableCharacters, tableDisplayName);
779
928
  returnedCharacters += fittedTables.spent;
780
929
  const omittedTables = fittedTables.omittedNames.length;
781
- if (returnedCharacters === originalCharacters && omittedSections === 0 && omittedTables === 0) {
930
+ const fittedAssets = fitWholeNamed(assets, remainingAllowance(), assetCharacters, assetDisplayName);
931
+ returnedCharacters += fittedAssets.spent;
932
+ const omittedAssets = fittedAssets.omittedNames.length;
933
+ if (returnedCharacters === originalCharacters &&
934
+ omittedSections === 0 &&
935
+ omittedTables === 0 &&
936
+ omittedAssets === 0) {
782
937
  return { article, omittedSections: 0 };
783
938
  }
784
939
  return {
785
- article: withTables({ ...article, sections: kept }, fittedTables.kept),
940
+ article: withAssets(withTables({ ...article, sections: kept }, fittedTables.kept), fittedAssets.kept),
786
941
  omittedSections,
787
942
  truncation: {
788
943
  originalCharacters,
@@ -792,6 +947,10 @@ function applyPmcBudget(article, budget) {
792
947
  omittedTables,
793
948
  omittedTableNames: fittedTables.omittedNames,
794
949
  }),
950
+ ...(omittedAssets > 0 && {
951
+ omittedAssets,
952
+ omittedAssetNames: fittedAssets.omittedNames,
953
+ }),
795
954
  },
796
955
  };
797
956
  }
@@ -821,6 +980,10 @@ const UNEXTRACTABLE_TABLE_EXPLANATIONS = {
821
980
  function tableDisplayName(table, index) {
822
981
  return table.label ?? table.id ?? `table ${index + 1}`;
823
982
  }
983
+ /** How an asset is named in a notice: its label, else its id, else its position. */
984
+ function assetDisplayName(asset, index) {
985
+ return asset.label ?? asset.id ?? `asset ${index + 1}`;
986
+ }
824
987
  /**
825
988
  * Collect the returned tables that carry no cell values. Read off the articles
826
989
  * the response actually ships — after every filter and both budgets — so a table
@@ -884,13 +1047,19 @@ function buildTruncationNotice(truncation) {
884
1047
  const omittedTables = truncation.omittedTables
885
1048
  ? ` ${truncation.omittedTables} table(s) were dropped whole rather than cut mid-row: ${droppedNames.join(', ')}.`
886
1049
  : '';
1050
+ // Assets are admitted last and only whole — a caption cut in half says
1051
+ // something the deposit does not — so name them on the same terms.
1052
+ const droppedAssetNames = truncation.articles.flatMap((a) => a.omittedAssetNames ?? []);
1053
+ const omittedAssets = truncation.omittedAssets
1054
+ ? ` ${truncation.omittedAssets} figure/supplementary item(s) were dropped whole rather than returned with a shortened caption: ${droppedAssetNames.join(', ')}.`
1055
+ : '';
887
1056
  // Name only the budgets the request actually set — pointing at `maxCharacters`
888
1057
  // when the caller only capped per-section sends them to a knob that is unset.
889
1058
  const knobs = [
890
1059
  truncation.maxCharacters !== undefined ? '`maxCharacters`' : undefined,
891
1060
  truncation.maxCharactersPerSection !== undefined ? '`maxCharactersPerSection`' : undefined,
892
1061
  ].filter((k) => k !== undefined);
893
- return `Full text was shortened to fit the requested character budget: ${truncation.returnedCharacters} of ${truncation.originalCharacters} body characters returned across ${subject} in ${truncation.mode} mode.${omitted}${omittedTables} See \`truncation\` for per-article and per-section counts, and raise ${knobs.join(' or ')} or narrow \`sections\` to retrieve more.`;
1062
+ return `Full text was shortened to fit the requested character budget: ${truncation.returnedCharacters} of ${truncation.originalCharacters} body characters returned across ${subject} in ${truncation.mode} mode.${omitted}${omittedTables}${omittedAssets} See \`truncation\` for per-article and per-section counts, and raise ${knobs.join(' or ')} or narrow \`sections\` to retrieve more.`;
894
1063
  }
895
1064
  /**
896
1065
  * Compose the recovery notice for articles the whole-response budget withheld.
@@ -969,9 +1138,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
969
1138
  input: z
970
1139
  .object({
971
1140
  pmcids: z
972
- .array(z
973
- .string()
974
- .regex(/^(?:PMC)?\d+$/i, 'PMC ID must be digits, optionally prefixed with "PMC" (e.g. "PMC9575052" or "9575052")'))
1141
+ .array(pmcidStringSchema)
975
1142
  .min(1)
976
1143
  .max(10)
977
1144
  .optional()
@@ -983,11 +1150,11 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
983
1150
  .optional()
984
1151
  .describe('PubMed IDs. Provide exactly one of `pmcids`, `pmids`, or `dois`. Articles in PMC are returned as structured JATS; articles not in PMC fall through to Europe PMC (when EPMC has a `fullTextXML`), then to Unpaywall when `UNPAYWALL_EMAIL` is set and a DOI is available.'),
985
1152
  dois: z
986
- .array(z.string().min(3))
1153
+ .array(doiStringSchema)
987
1154
  .min(1)
988
1155
  .max(10)
989
1156
  .optional()
990
- .describe('DOIs to resolve (e.g. ["10.21203/rs.3.rs-9010375/v1"]). Provide exactly one of `pmcids`, `pmids`, or `dois`. Resolved to a PMCID via the PMC ID Converter and returned as structured JATS when the article is in PMC; DOIs with no PMC counterpart (preprints, EPMC-only OA) fall through to Europe PMC, then Unpaywall, when those layers are enabled.'),
1157
+ .describe('DOIs to resolve (e.g. ["10.21203/rs.3.rs-9010375/v1"]), one per element. Provide exactly one of `pmcids`, `pmids`, or `dois`. Resolved to a PMCID via the PMC ID Converter and returned as structured JATS when the article is in PMC; DOIs with no PMC counterpart (preprints, EPMC-only OA) fall through to Europe PMC, then Unpaywall, when those layers are enabled.'),
991
1158
  includeReferences: z
992
1159
  .boolean()
993
1160
  .default(false)
@@ -996,6 +1163,10 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
996
1163
  .boolean()
997
1164
  .default(true)
998
1165
  .describe("Include the article's tables — cells, captions, labels and footnotes. On by default because a dropped table takes its numbers with it. Table-dense articles pay for it: rendered tables typically add 12–17% to an article record and can more than double it. Set false to omit them, or cap the cost with `maxCharacters`, which drops tables it cannot fit whole. Applies to `source=pmc` results only."),
1166
+ includeAssets: z
1167
+ .boolean()
1168
+ .default(true)
1169
+ .describe("Include the article's figures and supplementary material — `assets[]`, each with its label, caption, enclosing section and deposit pointer. On by default because it is cheaper than tables: a median asset-bearing article grows about 10%, and the body prose already refers to these by label. Set false to omit them, which also removes the `[Figure: …]` / `[Supplementary: …]` markers from the section text, since without the array they point at nothing. Prose-shaped blocks — lists, definition lists, block quotes, boxed text, preformatted blocks, displayed formulae — are section text rather than assets and this switch never affects them. Applies to `source=pmc` results only."),
999
1170
  maxSections: z
1000
1171
  .number()
1001
1172
  .int()
@@ -1006,14 +1177,14 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1006
1177
  sections: z
1007
1178
  .array(z.string())
1008
1179
  .optional()
1009
- .describe('Filter to specific sections by title (e.g. ["Introduction", "Methods", "Results", "Discussion"]). A term matches a section or subsection title at any nesting depth, case-insensitively, as a substring — "resul" matches "Results". A section whose own title matches is returned whole; one kept only because a nested subsection matched keeps its heading as a breadcrumb, with its own text cleared and only the matching branch beneath it. Tables narrow with the filter: one whose section did not survive, or that names no section, is dropped. Applies to `source=pmc` results only.'),
1180
+ .describe('Filter to specific sections by title (e.g. ["Introduction", "Methods", "Results", "Discussion"]). A term matches a section or subsection title at any nesting depth, case-insensitively, as a substring — "resul" matches "Results". A section whose own title matches is returned whole; one kept only because a nested subsection matched keeps its heading as a breadcrumb, with its own text cleared and only the matching branch beneath it. Tables and assets narrow with the filter: one whose section did not survive, or that names no section, is dropped. Applies to `source=pmc` results only.'),
1010
1181
  maxCharacters: z
1011
1182
  .number()
1012
1183
  .int()
1013
1184
  .min(1)
1014
1185
  .max(1_000_000)
1015
1186
  .optional()
1016
- .describe('Per-article budget for body text, in characters. Counts `source=pmc` section and subsection text plus table label, caption, cell and footnote text, or the `source=unpaywall` `content` body; titles, abstracts, identifiers, and references are never counted or shortened. The counted unit is that text alone — the Markdown grid `content[]` renders around the cells (pipes, padding, the divider row, headings) is scaffolding this budget does not measure, so a table renders longer than it costs here. Sections are served first and tables spend what is left, in document order — admission stops at the first table that does not fit, and every table from there on is dropped whole rather than cut mid-row, counted in `truncation.omittedTables` and named in `truncation.articles[].omittedTableNames`. Applied after `sections`, `maxSections`, `includeReferences`, and `includeTables`, so semantic filtering is unaffected. This knob alone bounds only bodies: the response-wide ceiling it implies is this value times the number of articles returned, plus every uncounted field. Use `maxResponseCharacters` for a true whole-response ceiling. Omit for the full body.'),
1187
+ .describe('Per-article budget for body text, in characters. Counts `source=pmc` section and subsection text — which carries the inline blocks the parser renders in place, such as lists, definition lists, block quotes, boxed text, preformatted blocks and displayed formulae — plus table label, caption, cell and footnote text and asset label, caption and `href` text; or the `source=unpaywall` `content` body. Titles, abstracts, identifiers, and references are never counted or shortened. The counted unit is that text alone — the Markdown grid `content[]` renders around the cells (pipes, padding, the divider row, headings) is scaffolding this budget does not measure, so a table renders longer than it costs here. Sections are served first, then tables, then assets, each spending what is left, in document order — admission stops at the first entry that does not fit, and every entry from there on is dropped whole rather than cut mid-row or returned with a shortened caption, counted in `truncation.omittedTables` / `truncation.omittedAssets` and named in `truncation.articles[].omittedTableNames` / `omittedAssetNames`. Applied after `sections`, `maxSections`, `includeReferences`, `includeTables`, and `includeAssets`, so semantic filtering is unaffected. This knob alone bounds only bodies: the response-wide ceiling it implies is this value times the number of articles returned, plus every uncounted field. Use `maxResponseCharacters` for a true whole-response ceiling. Omit for the full body.'),
1017
1188
  maxCharactersPerSection: z
1018
1189
  .number()
1019
1190
  .int()
@@ -1405,6 +1576,17 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1405
1576
  // ── Stage 3: Unpaywall fallback ─────────────────────────────────────────
1406
1577
  const unpaywall = getUnpaywallService();
1407
1578
  const fallbackArticles = [];
1579
+ // Detail of a DOI-backfill lookup that threw, per branch. A candidate still
1580
+ // DOI-less after one of these is an unknown, not a settled absence: it
1581
+ // reports `doi-lookup-failed` rather than `no-doi`. (#119)
1582
+ let pmcidDoiLookupFailure;
1583
+ let pmidDoiLookupFailure;
1584
+ /** The Unpaywall tier entry for a candidate that reached the stage with no DOI. */
1585
+ const doilessTierEntry = (failure) => ({
1586
+ tier: 'unpaywall',
1587
+ outcome: failure ? 'doi-lookup-failed' : 'no-doi',
1588
+ ...(failure && { detail: failure }),
1589
+ });
1408
1590
  // `pmcids` input reaches Unpaywall on the DOI the chain already holds: the
1409
1591
  // EPMC stage searches by PMCID and its hit carries one, captured on non-hit
1410
1592
  // outcomes too. PMCIDs EPMC never resolved fall back to the PMC ID
@@ -1447,8 +1629,9 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1447
1629
  });
1448
1630
  }
1449
1631
  catch (error) {
1632
+ pmcidDoiLookupFailure = error instanceof Error ? error.message : String(error);
1450
1633
  ctx.log.warning('Failed to resolve PMCID → DOI for the Unpaywall fallback', {
1451
- error: error instanceof Error ? error.message : String(error),
1634
+ error: pmcidDoiLookupFailure,
1452
1635
  pmcidCount: needDoi.length,
1453
1636
  });
1454
1637
  }
@@ -1462,12 +1645,15 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1462
1645
  pmcId,
1463
1646
  result: candidate.doi
1464
1647
  ? await resolveUnpaywall({ pmcId, doi: candidate.doi, budget }, unpaywall, ctx)
1465
- : { unavailable: { reason: 'no-doi' } },
1648
+ : undefined,
1466
1649
  };
1467
1650
  }));
1468
1651
  for (const { pmcId, result } of outcomes) {
1469
1652
  const inputId = pmcidToInputId.get(pmcId) ?? pmcId;
1470
- if ('article' in result) {
1653
+ if (result === undefined) {
1654
+ chainByInput.get(inputId)?.push(doilessTierEntry(pmcidDoiLookupFailure));
1655
+ }
1656
+ else if ('article' in result) {
1471
1657
  fallbackArticles.push(result.article);
1472
1658
  inputIdByArticle.set(result.article, inputId);
1473
1659
  if (result.truncation)
@@ -1501,19 +1687,21 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1501
1687
  });
1502
1688
  }
1503
1689
  catch (error) {
1690
+ pmidDoiLookupFailure = error instanceof Error ? error.message : String(error);
1504
1691
  ctx.log.warning('Failed to batch-fetch DOIs from PubMed for Unpaywall fallback', {
1505
- error: error instanceof Error ? error.message : String(error),
1692
+ error: pmidDoiLookupFailure,
1506
1693
  pmidCount: needDoi.length,
1507
1694
  });
1508
1695
  }
1509
1696
  }
1510
1697
  if (!unpaywall) {
1511
- // `fetchPubmedDois` has already run, so the DOI state is settled here.
1512
- // A candidate with no DOI could not have reached Unpaywall configured
1513
- // or not — that is `no-doi`, a real answer, not an incomplete search.
1698
+ // `fetchPubmedDois` has already run, so a candidate with no DOI could
1699
+ // not have reached Unpaywall configured or not — that is `no-doi`, a
1700
+ // real answer, not an incomplete search. Unless the lookup itself
1701
+ // threw, in which case the DOI state is unknown rather than absent.
1514
1702
  for (const c of pmidFallbackCandidates) {
1515
1703
  if (!c.doi) {
1516
- chainByInput.get(c.pmid)?.push({ tier: 'unpaywall', outcome: 'no-doi' });
1704
+ chainByInput.get(c.pmid)?.push(doilessTierEntry(pmidDoiLookupFailure));
1517
1705
  continue;
1518
1706
  }
1519
1707
  chainByInput.get(c.pmid)?.push({
@@ -1529,10 +1717,13 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1529
1717
  candidate,
1530
1718
  result: candidate.doi
1531
1719
  ? await resolveUnpaywall({ pmid: candidate.pmid, doi: candidate.doi, budget }, unpaywall, ctx)
1532
- : { unavailable: { reason: 'no-doi' } },
1720
+ : undefined,
1533
1721
  })));
1534
1722
  for (const { candidate, result } of outcomes) {
1535
- if ('article' in result) {
1723
+ if (result === undefined) {
1724
+ chainByInput.get(candidate.pmid)?.push(doilessTierEntry(pmidDoiLookupFailure));
1725
+ }
1726
+ else if ('article' in result) {
1536
1727
  fallbackArticles.push(result.article);
1537
1728
  inputIdByArticle.set(result.article, candidate.pmid);
1538
1729
  if (result.truncation)
@@ -1651,6 +1842,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1651
1842
  // stage, so a deferred article's dropped tables leave the roll-up together
1652
1843
  // with its entry when the splice above removes it. (#111)
1653
1844
  const omittedTables = truncatedArticles.reduce((n, a) => n + (a.omittedTables ?? 0), 0);
1845
+ const omittedAssets = truncatedArticles.reduce((n, a) => n + (a.omittedAssets ?? 0), 0);
1654
1846
  // Rolled up only when the budget actually removed characters, so an
1655
1847
  // under-budget request returns exactly what it did before the budget
1656
1848
  // controls existed. (#81)
@@ -1665,6 +1857,7 @@ export const fetchFulltextTool = tool('pubmed_fetch_fulltext', {
1665
1857
  returnedCharacters: truncatedArticles.reduce((n, a) => n + a.returnedCharacters, 0),
1666
1858
  omittedSections,
1667
1859
  ...(omittedTables > 0 && { omittedTables }),
1860
+ ...(omittedAssets > 0 && { omittedAssets }),
1668
1861
  articles: truncatedArticles,
1669
1862
  }
1670
1863
  : undefined;
@@ -1836,8 +2029,11 @@ async function runEpmcStage(epmc, args) {
1836
2029
  pmidOutcomes.set(run.c.pmid, run.outcome);
1837
2030
  if (run.article)
1838
2031
  collectHit(run.c.pmid, { ...run, article: run.article });
2032
+ // A DOI the EPMC hit carried is evidence the next stage needs, whatever the
2033
+ // fetch outcome was — merge it in like the pmcid branch below, so Unpaywall
2034
+ // gets it without a redundant PubMed metadata round-trip. (#119)
1839
2035
  else
1840
- remainingPmid.push(run.c);
2036
+ remainingPmid.push(run.doi && !run.c.doi ? { ...run.c, doi: run.doi } : run.c);
1841
2037
  }
1842
2038
  for (const run of pmcidResults) {
1843
2039
  pmcidOutcomes.set(run.c.normalized, run.outcome);
@@ -2144,6 +2340,7 @@ function unpaywallReasonToTierOutcome(reason) {
2144
2340
  switch (reason) {
2145
2341
  case 'no-body':
2146
2342
  case 'no-doi':
2343
+ case 'doi-lookup-failed':
2147
2344
  case 'no-oa':
2148
2345
  case 'fetch-failed':
2149
2346
  case 'parse-failed':
@@ -2195,6 +2392,11 @@ function reasonFromChain(chain) {
2195
2392
  return 'no-body';
2196
2393
  case 'unpaywall:no-doi':
2197
2394
  return 'no-doi';
2395
+ // Its own case, never folded into the `unpaywall:no-doi` → `not-found`
2396
+ // collapse above: that collapse says record absence is the specific signal,
2397
+ // and a lookup that never answered has established no absence at all. (#119)
2398
+ case 'unpaywall:doi-lookup-failed':
2399
+ return 'doi-lookup-failed';
2198
2400
  case 'unpaywall:no-oa':
2199
2401
  return 'no-oa';
2200
2402
  case 'unpaywall:fetch-failed':
@@ -2238,7 +2440,8 @@ function formatUnqueriedTiers(tiers, chain) {
2238
2440
  */
2239
2441
  function formatTruncation(t, lines) {
2240
2442
  const tablesOmitted = t.omittedTables === undefined ? '' : `; ${t.omittedTables} table(s) omitted whole`;
2241
- lines.push(`\n**Truncated (${t.mode} mode):** ${t.returnedCharacters} of ${t.originalCharacters} body characters returned across ${t.articles.length} article(s); ${t.omittedSections} section(s) omitted${tablesOmitted}`);
2443
+ const assetsOmitted = t.omittedAssets === undefined ? '' : `; ${t.omittedAssets} asset(s) omitted whole`;
2444
+ lines.push(`\n**Truncated (${t.mode} mode):** ${t.returnedCharacters} of ${t.originalCharacters} body characters returned across ${t.articles.length} article(s); ${t.omittedSections} section(s) omitted${tablesOmitted}${assetsOmitted}`);
2242
2445
  const budgets = [
2243
2446
  t.maxCharacters === undefined ? undefined : `maxCharacters ${t.maxCharacters}`,
2244
2447
  t.maxCharactersPerSection === undefined
@@ -2251,7 +2454,10 @@ function formatTruncation(t, lines) {
2251
2454
  const tablesDropped = a.omittedTables === undefined
2252
2455
  ? ''
2253
2456
  : `, ${a.omittedTables} table(s) dropped whole: ${(a.omittedTableNames ?? []).join(', ')}`;
2254
- lines.push(`- ${a.id} (${a.source}): ${a.returnedCharacters} of ${a.originalCharacters} characters${tablesDropped}`);
2457
+ const assetsDropped = a.omittedAssets === undefined
2458
+ ? ''
2459
+ : `, ${a.omittedAssets} asset(s) dropped whole: ${(a.omittedAssetNames ?? []).join(', ')}`;
2460
+ lines.push(`- ${a.id} (${a.source}): ${a.returnedCharacters} of ${a.originalCharacters} characters${tablesDropped}${assetsDropped}`);
2255
2461
  for (const s of a.sections ?? []) {
2256
2462
  lines.push(` - ${s.title ?? 'untitled section'} — ${s.returnedCharacters} of ${s.originalCharacters} characters (truncated: ${s.truncated})`);
2257
2463
  }
@@ -2287,6 +2493,8 @@ function formatPmcArticle(a, lines, truncation) {
2287
2493
  parts.push(`**${a.journal.volume}**${a.journal.issue ? `(${a.journal.issue})` : ''}`);
2288
2494
  if (a.journal.pages)
2289
2495
  parts.push(a.journal.pages);
2496
+ if (a.journal.elocationId)
2497
+ parts.push(a.journal.elocationId);
2290
2498
  if (a.journal.issn)
2291
2499
  parts.push(`ISSN ${a.journal.issn}`);
2292
2500
  if (parts.length)
@@ -2322,6 +2530,8 @@ function formatPmcArticle(a, lines, truncation) {
2322
2530
  formatSection(sec, lines, 4);
2323
2531
  if (a.tables?.length)
2324
2532
  formatTables(a.tables, lines);
2533
+ if (a.assets?.length)
2534
+ formatAssets(a.assets, lines);
2325
2535
  if (a.references?.length) {
2326
2536
  lines.push(`\n#### References (${a.references.length})`);
2327
2537
  for (const ref of a.references) {
@@ -2361,6 +2571,35 @@ function formatTables(tables, lines) {
2361
2571
  lines.push(`\nFootnotes: ${escapeMarkdownInline(table.footnotes)}`);
2362
2572
  }
2363
2573
  }
2574
+ /**
2575
+ * Render every figure and supplementary item, so a `content[]` reader gets the
2576
+ * same caption, pointer and placement `structuredContent` carries rather than a
2577
+ * count saying they exist. Each field of `AssetSchema` appears here — that is
2578
+ * what `format-parity` verifies. (#130)
2579
+ *
2580
+ * Label, caption and `href` go through {@link escapeMarkdownInline}: the parser
2581
+ * stores the raw upstream text, and a caption carrying `[`, `*` or a tag-shaped
2582
+ * `<` would otherwise form a link, emphasis or raw HTML at the render boundary.
2583
+ *
2584
+ * The note about `href` is worth its line: the value is a filename inside the PMC
2585
+ * deposit, and an agent that reads it as a URL will spend a request on a 404 for
2586
+ * every figure in the article.
2587
+ */
2588
+ function formatAssets(assets, lines) {
2589
+ lines.push(`\n#### Assets (${assets.length})`);
2590
+ lines.push(`\n> \`file\` is the pointer exactly as deposited — a name inside the PMC deposit, not a fetchable URL. Open the article at the PMC link above to view it.`);
2591
+ for (const asset of assets) {
2592
+ const heading = [asset.label, asset.caption].filter(Boolean).join(' — ');
2593
+ lines.push(`\n##### ${escapeMarkdownInline(heading || 'Asset')}`);
2594
+ const meta = [
2595
+ asset.assetType,
2596
+ asset.sectionTitle ? `Section: ${escapeMarkdownInline(asset.sectionTitle)}` : undefined,
2597
+ asset.id ? `id: ${escapeMarkdownInline(asset.id)}` : undefined,
2598
+ asset.href ? `file: ${escapeMarkdownInline(asset.href)}` : undefined,
2599
+ ].filter((part) => part !== undefined);
2600
+ lines.push(`*${meta.join(' · ')}*`);
2601
+ }
2602
+ }
2364
2603
  /**
2365
2604
  * The meta-line clause describing a table's header rows.
2366
2605
  *