@enhansome/core 1.8.1 → 1.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,5 +1,6 @@
1
1
  import { RepoInfoDetails } from './github.js';
2
2
  import { Logger } from './logger.js';
3
+ import type { Heading, Link, List, Parent, Root } from 'mdast';
3
4
  export interface JsonOutput {
4
5
  items: JsonSection[];
5
6
  metadata: JsonMetadata;
@@ -58,3 +59,8 @@ export declare function processMarkdownContent(originalContent: string, token: s
58
59
  finalContent: string;
59
60
  jsonData: JsonOutput;
60
61
  }>;
62
+ export declare function normalizeGitHubUrls(tree: Root): void;
63
+ export declare function findFirstGitHubLink(node: Parent): Link | undefined;
64
+ export declare function countLinkedItems(listNode: List): number;
65
+ export declare function isStructuralHeading(node: Heading): boolean;
66
+ export declare function findTitleSlotIndex(tree: Root): number;
package/dist/markdown.js CHANGED
@@ -85,6 +85,7 @@ export async function processMarkdownContent(originalContent, token, replacement
85
85
  const contentAfterReplacements = applyTextReplacements(originalContent, replacements.filter(rule => rule.type !== 'branding'), log);
86
86
  const processor = unified().use(remarkParse).use(remarkGfm);
87
87
  const tree = processor.parse(contentAfterReplacements);
88
+ normalizeGitHubUrls(tree);
88
89
  const githubUrls = collectGitHubLinks(tree);
89
90
  const repoInfoMap = await fetchTargetData(githubUrls, repos);
90
91
  // The title derives from the *source* repository (originalRepository), never
@@ -176,6 +177,112 @@ function collectGitHubLinks(tree) {
176
177
  });
177
178
  return urls;
178
179
  }
180
+ // Input normalization, same family as fixRelativeLinks: make GitHub repos a
181
+ // source expresses WITHOUT markdown links visible as real link nodes, so
182
+ // every downstream consumer — repo fetch, entry tests, gates, badges — sees
183
+ // the shape a markdown link would have produced. Two families:
184
+ //
185
+ // (1) Bare scheme-less `github.com/owner/repo` text — GFM autolinks only
186
+ // scheme-full URLs, so these stay plain text. The link's label is the
187
+ // URL text; only owner/repo are consumed (a deeper path like `/tree/main`
188
+ // stays in the trailing text — the identity reads the first two segments
189
+ // either way).
190
+ // (2) Inline `<a href="…">…</a>` anchors — remark emits the open and close
191
+ // tags as separate html nodes around the label's inline nodes, so the
192
+ // pair is rewrapped as one link. Anchors inside BLOCK html (`<details>`
193
+ // summaries, centered banners) keep their raw form: that html is the
194
+ // structure the walk already reads (detailsSummaryTitle), and rewriting
195
+ // it is a different decision than link visibility.
196
+ //
197
+ // Text inside code spans/blocks never linkifies (code has no text nodes) and
198
+ // text inside an existing link label is skipped — nested links are not a
199
+ // thing. Non-GitHub and org-only anchors stay raw html.
200
+ const BARE_GITHUB_URL = /(?:^|(?<=[\s(>\[]))(?:www\.)?github\.com\/[A-Za-z0-9_.-]+\/[A-Za-z0-9_.-]+/g;
201
+ const ANCHOR_OPEN = /^<a\s[^>]*href=(["'])([^"']*)\1[^>]*>$/;
202
+ const ANCHOR_CLOSE = /^<\/a\s*>$/;
203
+ export function normalizeGitHubUrls(tree) {
204
+ const walk = (node, inLink) => {
205
+ if (inLink) {
206
+ return;
207
+ }
208
+ const parent = node;
209
+ if (!Array.isArray(parent.children)) {
210
+ return;
211
+ }
212
+ for (const child of parent.children) {
213
+ walk(child, child.type === 'link');
214
+ }
215
+ parent.children = normalizeInlineChildren(parent.children);
216
+ };
217
+ walk(tree, false);
218
+ }
219
+ // One parent's inline run: bare-URL text nodes split into text/link/text,
220
+ // anchor open+close pairs rewrapped as a link. The anchor's inner nodes are
221
+ // taken verbatim — they are the label, and splitting them would nest links.
222
+ function normalizeInlineChildren(children) {
223
+ const out = [];
224
+ for (let i = 0; i < children.length; i++) {
225
+ const child = children[i];
226
+ if (child.type === 'html') {
227
+ const anchor = ANCHOR_OPEN.exec(child.value.trim());
228
+ const url = anchor && parseGitHubUrl(anchor[2]) ? anchor[2] : null;
229
+ if (url) {
230
+ const closeIndex = children.findIndex((candidate, j) => j > i &&
231
+ candidate.type === 'html' &&
232
+ ANCHOR_CLOSE.test(candidate.value.trim()));
233
+ if (closeIndex !== -1) {
234
+ out.push({
235
+ type: 'link',
236
+ url,
237
+ children: children.slice(i + 1, closeIndex),
238
+ });
239
+ i = closeIndex;
240
+ continue;
241
+ }
242
+ }
243
+ out.push(child);
244
+ continue;
245
+ }
246
+ if (child.type === 'text') {
247
+ out.push(...linkifyBareUrls(child));
248
+ continue;
249
+ }
250
+ out.push(child);
251
+ }
252
+ return out;
253
+ }
254
+ function linkifyBareUrls(node) {
255
+ const matches = [...node.value.matchAll(BARE_GITHUB_URL)];
256
+ if (matches.length === 0) {
257
+ return [node];
258
+ }
259
+ const out = [];
260
+ let cursor = 0;
261
+ for (const match of matches) {
262
+ const label = match[0].replace(/\.+$/, '');
263
+ const url = `https://${label.replace(/^www\./, '')}`;
264
+ if (!parseGitHubUrl(url)) {
265
+ continue;
266
+ }
267
+ const start = match.index;
268
+ if (start > cursor) {
269
+ out.push({
270
+ type: 'text',
271
+ value: node.value.slice(cursor, start),
272
+ });
273
+ }
274
+ out.push({
275
+ type: 'link',
276
+ url,
277
+ children: [{ type: 'text', value: label }],
278
+ });
279
+ cursor = start + label.length;
280
+ }
281
+ if (cursor < node.value.length) {
282
+ out.push({ type: 'text', value: node.value.slice(cursor) });
283
+ }
284
+ return out;
285
+ }
179
286
  function createBadgeText(info) {
180
287
  if (info.archived) {
181
288
  return ' ⚠️ Archived';
@@ -192,27 +299,41 @@ function createBadgeText(info) {
192
299
  }
193
300
  return ` ${parts.join(' | ')}`;
194
301
  }
195
- function findFirstGitHubLink(node) {
196
- let linkUrl;
197
- visit(node, 'link', (linkNode) => {
198
- if (!linkUrl && parseGitHubUrl(linkNode.url)) {
199
- linkUrl = linkNode.url;
302
+ // The first link in a subtree that points at a GitHub repo. Returns the link
303
+ // node itself so callers can take both its URL (identity) and its text
304
+ // (title fallbacks).
305
+ export function findFirstGitHubLink(node) {
306
+ let linkNode;
307
+ visit(node, 'link', (candidate) => {
308
+ if (!linkNode && parseGitHubUrl(candidate.url)) {
309
+ linkNode = candidate;
200
310
  }
201
311
  });
202
- return linkUrl;
312
+ return linkNode;
203
313
  }
204
314
  // The GitHub link that represents a list item: the FIRST GitHub link in the
205
- // item's OWN paragraph only. A nested-descendant link is deliberately ignored
206
- // — it belongs to a child, not to this item. An item is a GitHub node iff its
207
- // own paragraph links to a GitHub repo; an item with no own GitHub link but
315
+ // item's OWN paragraphs, in document order. Paper-list entries carry their
316
+ // identity link ([[Code]](github)) in a paragraph after the title one, so
317
+ // every own paragraph is scanned — the title still comes from the first
318
+ // paragraph alone. A nested-descendant link is deliberately ignored — it
319
+ // belongs to a child, not to this item. An item is a GitHub node iff its own
320
+ // paragraphs link to a GitHub repo; an item with no own GitHub link but
208
321
  // nested GitHub children is a group, not a node borrowing a child's identity.
209
322
  //
210
- // A GitHub link that is secondary within the paragraph (e.g.
323
+ // A GitHub link that is secondary within a paragraph (e.g.
211
324
  // `[name](marketplace) … [On GitHub](github)`) is still the item's own link, so
212
325
  // `findFirstGitHubLink` over the paragraph finds it correctly.
213
326
  function findOwnGitHubLink(itemNode) {
214
- const paragraph = itemNode.children.find((child) => child.type === 'paragraph');
215
- return paragraph ? findFirstGitHubLink(paragraph) : undefined;
327
+ for (const child of itemNode.children) {
328
+ if (child.type !== 'paragraph') {
329
+ continue;
330
+ }
331
+ const link = findFirstGitHubLink(child);
332
+ if (link) {
333
+ return link;
334
+ }
335
+ }
336
+ return undefined;
216
337
  }
217
338
  function fixRelativeLinks(tree, relativeLinkPrefix) {
218
339
  if (!relativeLinkPrefix) {
@@ -274,15 +395,89 @@ function compareByRepoInfo(by, a, b) {
274
395
  const timeB = b.pushed_at ? new Date(b.pushed_at).getTime() : 0;
275
396
  return timeB - timeA;
276
397
  }
277
- function processListRecursively(listNode, repoInfoMap, sortOptions, isNested = false) {
278
- // The section-sparsity gate counts items whose SUBTREE contains a GitHub
279
- // link, not items with an own-paragraph link only. A category with no own
280
- // link but nested GitHub children still counts (it becomes a group);
281
- // switching to `findOwnGitHubLink` would silently drop purely categorical
282
- // sections, so it deliberately diverges from the identity resolver below.
283
- const itemsWithGitHubLinks = listNode.children.filter(item => !!findFirstGitHubLink(item));
284
- if (!isNested && itemsWithGitHubLinks.length < sortOptions.minLinks) {
285
- return [];
398
+ // The minLinks noise gate counts items whose SUBTREE contains a GitHub link,
399
+ // not items with an own-paragraph link only. A category with no own link but
400
+ // nested GitHub children still counts (it becomes a group); switching to
401
+ // `findOwnGitHubLink` would silently drop purely categorical sections, so it
402
+ // deliberately diverges from the identity resolver.
403
+ export function countLinkedItems(listNode) {
404
+ return listNode.children.filter(item => !!findFirstGitHubLink(item)).length;
405
+ }
406
+ // The table counterpart of countLinkedItems for the section gate: rows whose
407
+ // subtree contains a GitHub link. No header-row skip — a card grid's first row
408
+ // sits above the delimiter and is content, while pure-label header rows carry
409
+ // no links and count zero either way.
410
+ function countLinkedRows(tableNode) {
411
+ return tableNode.children.filter(row => row.type === 'tableRow' && !!findFirstGitHubLink(row)).length;
412
+ }
413
+ // One entry's title and description, split from its own inlines: leading text
414
+ // up to and including the first link is the title (prefix tags like
415
+ // "[UPDATED]" survive), the prose trailing it the description. The link is
416
+ // found through emphasis wrappers — a bolded card link
417
+ // (`**[Name](repo)** - description`) splits like a plain one. With no link
418
+ // the whole text is the title — the description must never echo the title
419
+ // back.
420
+ function splitEntryText(inlines) {
421
+ const linkIndex = inlines.findIndex(child => {
422
+ let hasLink = false;
423
+ visit(child, 'link', () => {
424
+ hasLink = true;
425
+ });
426
+ return hasLink;
427
+ });
428
+ if (linkIndex === -1) {
429
+ return { title: getInlineText(inlines), description: '' };
430
+ }
431
+ return {
432
+ title: getInlineText(inlines.slice(0, linkIndex + 1)),
433
+ description: stripLeadingNoise(getInlineText(inlines.slice(linkIndex + 1))),
434
+ };
435
+ }
436
+ // The emission decision every entry source shares — list items, table rows,
437
+ // and the paragraph/blockquote sources to come: an own GitHub link that
438
+ // resolved makes an item; a dead own link drops the entry and lifts its
439
+ // children to the nearest live parent; no own link with children makes a
440
+ // group — never a `repo_info`, that's the identity-borrowing bug; no own link
441
+ // and no children is a non-GitHub leaf, kept in markdown, dropped from JSON.
442
+ // TODO(future): preserve non-GitHub leaves in a separate shape.
443
+ function emitEntryNodes(githubUrl, repoInfo, text, childrenJson) {
444
+ if (githubUrl && repoInfo) {
445
+ return [
446
+ {
447
+ node_type: 'item',
448
+ title: text.title,
449
+ description: text.description || null,
450
+ children: childrenJson,
451
+ repo_info: toRepoInfo(repoInfo),
452
+ },
453
+ ];
454
+ }
455
+ if (githubUrl) {
456
+ return childrenJson;
457
+ }
458
+ if (childrenJson.length > 0) {
459
+ return [
460
+ {
461
+ node_type: 'group',
462
+ title: text.title,
463
+ description: text.description || null,
464
+ children: childrenJson,
465
+ },
466
+ ];
467
+ }
468
+ return [];
469
+ }
470
+ function processListRecursively(listNode, repoInfoMap, sortOptions, isNested = false,
471
+ // The caller's section-scope gate decision (sectionGatePasses). Absent for
472
+ // nested lists (emitted under their parent regardless) and for top-level
473
+ // lists with no open container (preamble), which gate per list.
474
+ sectionGateOpen) {
475
+ if (!isNested) {
476
+ const gateOpen = sectionGateOpen ??
477
+ countLinkedItems(listNode) >= sortOptions.minLinks;
478
+ if (!gateOpen) {
479
+ return [];
480
+ }
286
481
  }
287
482
  // Zip each item with its emitted JSON nodes and repo info so one sort orders
288
483
  // both the rendered AST and the emitted JSON. `emitted` is empty for
@@ -290,59 +485,31 @@ function processListRecursively(listNode, repoInfoMap, sortOptions, isNested = f
290
485
  // AST, dropped from JSON.
291
486
  const entries = [];
292
487
  for (const itemNode of listNode.children) {
293
- const githubUrl = findOwnGitHubLink(itemNode);
488
+ const ownLink = findOwnGitHubLink(itemNode);
489
+ const githubUrl = ownLink?.url;
294
490
  const repoInfo = githubUrl ? (repoInfoMap.get(githubUrl) ?? null) : null;
295
- const nestedLists = itemNode.children.filter((child) => child.type === 'list');
296
- const childrenJson = nestedLists.flatMap(nestedList => processListRecursively(nestedList, repoInfoMap, sortOptions, true));
297
- let title = '';
298
- let description = '';
299
- const paragraph = itemNode.children.find(p => p.type === 'paragraph');
300
- if (paragraph) {
301
- const linkIndex = paragraph.children.findIndex((c) => c.type === 'link');
302
- if (linkIndex !== -1) {
303
- // Title = leading text + the first link's text, so prefix tags like
304
- // "[UPDATED]" survive; description is the prose trailing the link.
305
- title = getInlineText(paragraph.children.slice(0, linkIndex + 1));
306
- description = stripLeadingNoise(getInlineText(paragraph.children.slice(linkIndex + 1)));
307
- }
308
- else {
309
- // No link: the whole paragraph is the title, so leave description empty
310
- // (avoid echoing the title back).
311
- title = getInlineText(paragraph.children);
312
- description = '';
313
- }
314
- }
315
- // No-own-link items with children become groups — NEVER give them a
316
- // `repo_info`, that's the identity-borrowing bug. No-own-link, no-child
317
- // items are non-GitHub leaves: kept in markdown, dropped from JSON.
318
- // TODO(future): preserve non-GitHub leaves in a separate shape.
319
- let emitted = [];
320
- if (githubUrl && repoInfo) {
321
- emitted = [
322
- {
323
- node_type: 'item',
324
- title,
325
- description: description || null,
326
- children: childrenJson,
327
- repo_info: toRepoInfo(repoInfo),
328
- },
329
- ];
330
- }
331
- else if (githubUrl) {
332
- // Dead target: the item itself is not emitted; its children lift to this
333
- // list's level — the nearest live parent.
334
- emitted = childrenJson;
335
- }
336
- else if (childrenJson.length > 0) {
337
- emitted = [
338
- {
339
- node_type: 'group',
340
- title,
341
- description: description || null,
342
- children: childrenJson,
343
- },
344
- ];
491
+ // Nested content is the item's children: deeper lists as before, plus
492
+ // tables — the AnimeResearch shape wraps a <details><summary> block and
493
+ // its table inside one list item. Both emit under the parent item
494
+ // regardless of the gate: the top-level call already gated the section.
495
+ const nestedContent = itemNode.children.filter((child) => child.type === 'list' || child.type === 'table');
496
+ const childrenJson = nestedContent.flatMap(child => child.type === 'list'
497
+ ? processListRecursively(child, repoInfoMap, sortOptions, true)
498
+ : processTableRows(child, repoInfoMap, true));
499
+ // Title/description split on the FIRST paragraph only — a paper-list
500
+ // entry's identity link may live in a later paragraph (findOwnGitHubLink)
501
+ // while its title text stays the leading one.
502
+ const paragraph = itemNode.children.find((child) => child.type === 'paragraph');
503
+ const entryText = splitEntryText(paragraph?.children ?? []);
504
+ // A details-wrapped item carries its visible text in the summary, not in
505
+ // a paragraph — that text is the group's title.
506
+ if (!entryText.title) {
507
+ entryText.title = listItemSummaryTitle(itemNode);
345
508
  }
509
+ // The shared title fallbacks (an inline-code link label carries no text
510
+ // nodes, so the split alone can leave an empty title).
511
+ entryText.title = entryTitle(entryText.title, ownLink, repoInfo);
512
+ const emitted = emitEntryNodes(githubUrl, repoInfo, entryText, childrenJson);
346
513
  entries.push({ emitted, node: itemNode, repoInfo });
347
514
  }
348
515
  if (sortOptions.by) {
@@ -352,6 +519,285 @@ function processListRecursively(listNode, repoInfoMap, sortOptions, isNested = f
352
519
  listNode.children = entries.map(entry => entry.node);
353
520
  return entries.flatMap(entry => entry.emitted);
354
521
  }
522
+ // Title with the fallbacks every entry source shares: the split's own text
523
+ // when it names something; else the own link's label when meaningful; else —
524
+ // for a degenerate base (rank/year cell, URL label, tag word) owner/name, for
525
+ // an absent base (an image-only link) the repo name. A degenerate base with
526
+ // no live repo link (a group) keeps its text: nothing better exists.
527
+ function entryTitle(base, ownLink, repoInfo) {
528
+ if (base && !isDegenerateTitle(base)) {
529
+ return base;
530
+ }
531
+ const linkText = ownLink ? getInlineText(ownLink.children) : '';
532
+ if (isMeaningfulLinkText(linkText)) {
533
+ return linkText;
534
+ }
535
+ if (base === '') {
536
+ return repoInfo?.repo ?? '';
537
+ }
538
+ return repoInfo ? `${repoInfo.owner}/${repoInfo.repo}` : base;
539
+ }
540
+ // Table rows are entries under the nearest open container. The common shape —
541
+ // scala's `[name](repo) | description`, spec tables with the link in a later
542
+ // column — holds ONE repo per row: the title is the first cell's text (the
543
+ // own link's text, then the repo name, when the first cell is empty), the
544
+ // description the remaining cells' text (badge images carry no text nodes, so
545
+ // they never pollute it). Card grids pin several repos per row, each cell its
546
+ // own card, so those emit one entry per link-bearing cell. A linked row above
547
+ // the delimiter is content like any other (grid tables have no label header;
548
+ // pure-label header rows carry no links and emit nothing). Rows stay in source
549
+ // order — unlike the unranked lists the product sorts, a table's row order is
550
+ // part of its meaning.
551
+ function processTableRows(tableNode, repoInfoMap, gateOpen) {
552
+ if (!gateOpen) {
553
+ return [];
554
+ }
555
+ const items = [];
556
+ for (const row of tableNode.children) {
557
+ if (row.type !== 'tableRow') {
558
+ continue;
559
+ }
560
+ const cellLinks = row.children.map(cell => findFirstGitHubLink(cell));
561
+ const distinctUrls = new Set(cellLinks.flatMap(link => (link ? [link.url] : [])));
562
+ if (distinctUrls.size === 0) {
563
+ continue;
564
+ }
565
+ if (distinctUrls.size === 1) {
566
+ const ownLink = cellLinks.find((link) => !!link);
567
+ if (!ownLink) {
568
+ continue;
569
+ }
570
+ const repoInfo = repoInfoMap.get(ownLink.url) ?? null;
571
+ const firstCellText = splitEntryText(row.children[0]?.children ?? []);
572
+ // A degenerate first cell (a rank/year column) moves the title source to
573
+ // the link's own cell: that cell no longer feeds the description — its
574
+ // label became the title — and the degenerate cell drops as noise.
575
+ const linkCellIndex = isDegenerateTitle(firstCellText.title)
576
+ ? cellLinks.findIndex(link => link === ownLink)
577
+ : -1;
578
+ const titleText = linkCellIndex === -1
579
+ ? firstCellText
580
+ : splitEntryText(row.children[linkCellIndex].children);
581
+ const description = [
582
+ titleText.description,
583
+ ...row.children
584
+ .slice(1)
585
+ .filter((_cell, index) => index + 1 !== linkCellIndex)
586
+ .map(cell => getNodeText(cell)),
587
+ ]
588
+ .join(' ')
589
+ .trim();
590
+ items.push(...emitEntryNodes(ownLink.url, repoInfo, {
591
+ title: entryTitle(titleText.title, ownLink, repoInfo),
592
+ description,
593
+ }, []));
594
+ continue;
595
+ }
596
+ const emittedUrls = new Set();
597
+ for (const [cellIndex, ownLink] of cellLinks.entries()) {
598
+ if (!ownLink || emittedUrls.has(ownLink.url)) {
599
+ continue;
600
+ }
601
+ emittedUrls.add(ownLink.url);
602
+ const repoInfo = repoInfoMap.get(ownLink.url) ?? null;
603
+ const cellText = splitEntryText(row.children[cellIndex].children);
604
+ items.push(...emitEntryNodes(ownLink.url, repoInfo, {
605
+ title: entryTitle(cellText.title, ownLink, repoInfo),
606
+ description: cellText.description,
607
+ }, []));
608
+ }
609
+ }
610
+ return items;
611
+ }
612
+ // A top-level paragraph can BE an entry, not only feed container prose. Two
613
+ // corpus families qualify (progress/empty-tree-parses.md, step 4): a GitHub
614
+ // link LEADING the paragraph behind a short entry label, or the paragraph
615
+ // ending in a tag cluster whose GitHub link carries the identity — the
616
+ // paper-list shape whose first link points at the paper and a [Code]/[Github]
617
+ // tag at the end, name/author lines ending in that tag, and dated lines
618
+ // ("… [Github] 4 Feb 2023"). Prose that mentions a repo ("Please see
619
+ // CONTRIBUTING", "See also [repo]", intro text ending in a bare repo URL) is
620
+ // neither and stays description. Calibrated on the 2,293-README corpus plus
621
+ // the fixture fleet, not intuition: an entry label is a TAG (empty, CJK/emoji,
622
+ // bracketed, or colon-terminated) — never bare English prose — and the
623
+ // identity link must carry text, so the ubiquitous image-only awesome badge
624
+ // is not an entry.
625
+ const ENTRY_LABEL_MAX = 15;
626
+ const ENTRY_TRAILING_MAX = 3;
627
+ const TAG_LINK_TEXT = /^[\[\]()*\s:_-]*(?:source\s+code|code|github|repo|source|src|project|paper|page|web|site|home|official|notebook|demo|data|docs|implementation|arxiv)\b[\[\]()*\s:_-]*$/i;
628
+ const URL_LINK_TEXT = /^https?:\/\/\S+$/i;
629
+ // A title that names nothing — the degenerate families every entry source can
630
+ // emit: a pure number/punctuation run (a table's rank or year column, "15."
631
+ // / "2023" / "2025-05" / "-"), a URL (a link whose label is the URL itself,
632
+ // scheme or scheme-less `github.com/…`), or a bare tag word ("GitHub",
633
+ // "Source code" — best-of's generated lines). CJK labels are letters
634
+ // (`\p{L}`) and stay titles.
635
+ const URL_TITLE = /^(?:[a-z][a-z0-9+.-]*:\/\/|www\.)\S+$|^github\.com\/\S+$/i;
636
+ function isDegenerateTitle(title) {
637
+ const trimmed = title.trim();
638
+ return (!!trimmed &&
639
+ (URL_TITLE.test(trimmed) ||
640
+ TAG_LINK_TEXT.test(trimmed) ||
641
+ !/[\p{L}]/u.test(trimmed)));
642
+ }
643
+ // A link label that can serve as a title. The degenerate shapes are excluded
644
+ // twice over — a URL label stays a URL, and a tag label names the link's role
645
+ // ("[Github](repo)"), not the repo.
646
+ function isMeaningfulLinkText(text) {
647
+ return text !== '' && !isDegenerateTitle(text);
648
+ }
649
+ // Dated entry lines end their tag cluster with a publication date ("4 Feb
650
+ // 2023", "19 Apr 2022", "2023") — a real corpus family (dated paper/tutorial
651
+ // lists), not a sentence continuation.
652
+ const DATE_TRAILING = /^(\d{1,2}[ -])?([a-zà-ÿ]{3,12}[ -])?\d{2,4}$/i;
653
+ // Boilerplate navigation line present in most mirror READMEs; casing and
654
+ // the optional "the" vary ("⬆ back to top", "⬆️ Back to Top",
655
+ // "Back to the Top").
656
+ const BACK_TO_TOP = /back to (?:the )?top/i;
657
+ // Tag-shaped leading label: empty, non-ASCII (CJK labels like 项目地址:,
658
+ // emoji markers), bracketed ([6] Diffusion:), or colon-terminated (Demo:).
659
+ // English prose ("Please see", "See also", "Inspired by the") is none of
660
+ // these — those are cross-reference sentences, not entry labels.
661
+ function isEntryLabel(label) {
662
+ return (label === '' ||
663
+ /[^\x00-\x7f]/.test(label) ||
664
+ label.startsWith('[') ||
665
+ /[::]$/.test(label));
666
+ }
667
+ // Flattened view of a paragraph's (or heading's) inline content for the entry
668
+ // test: the first link, the free text before it, and the free text after the
669
+ // link cluster (emphasis recursed into; link labels, images, and inline html
670
+ // carry no positional weight — a bare [Code] tag's label is not trailing
671
+ // prose).
672
+ function scanInlines(paragraph) {
673
+ let firstLink;
674
+ let label = '';
675
+ let trailing = '';
676
+ const walk = (node) => {
677
+ if (node.type === 'link') {
678
+ if (firstLink) {
679
+ trailing = '';
680
+ }
681
+ firstLink = firstLink ?? node;
682
+ return;
683
+ }
684
+ if (node.type === 'text') {
685
+ if (firstLink) {
686
+ trailing += node.value;
687
+ }
688
+ else {
689
+ label += node.value;
690
+ }
691
+ return;
692
+ }
693
+ for (const child of node.children ?? []) {
694
+ walk(child);
695
+ }
696
+ };
697
+ for (const child of paragraph.children) {
698
+ walk(child);
699
+ }
700
+ return { firstLink, label: label.trim(), trailing: trailing.trim() };
701
+ }
702
+ // True when nothing of substance follows the paragraph's last link: pure
703
+ // punctuation, or a date (see DATE_TRAILING).
704
+ function tagClusterEndsParagraph(trailing) {
705
+ if (trailing.length <= ENTRY_TRAILING_MAX) {
706
+ return true;
707
+ }
708
+ return DATE_TRAILING.test(trailing.replace(/^[\[\]()*\s:;,_-]+/, ''));
709
+ }
710
+ // The GitHub link that makes a top-level paragraph — or a heading inside a
711
+ // blockquote — an entry, if it is one (see the family comment above). Null
712
+ // for plain prose.
713
+ function paragraphEntryLink(paragraph) {
714
+ const githubLink = findFirstGitHubLink(paragraph);
715
+ if (!githubLink) {
716
+ return undefined;
717
+ }
718
+ const labelText = getInlineText(githubLink.children);
719
+ if (!labelText) {
720
+ return undefined;
721
+ }
722
+ const scan = scanInlines(paragraph);
723
+ if (scan.firstLink === githubLink &&
724
+ scan.label.length <= ENTRY_LABEL_MAX &&
725
+ isEntryLabel(scan.label)) {
726
+ return githubLink;
727
+ }
728
+ if (tagClusterEndsParagraph(scan.trailing) &&
729
+ (TAG_LINK_TEXT.test(labelText) ||
730
+ (URL_LINK_TEXT.test(labelText) && scan.firstLink !== githubLink))) {
731
+ return githubLink;
732
+ }
733
+ return undefined;
734
+ }
735
+ function blockquoteEntries(blockquote) {
736
+ const faces = [];
737
+ let sawParagraph = false;
738
+ for (const child of blockquote.children) {
739
+ if (child.type === 'paragraph') {
740
+ if (sawParagraph) {
741
+ continue;
742
+ }
743
+ sawParagraph = true;
744
+ const link = paragraphEntryLink(child);
745
+ if (link) {
746
+ faces.push({ inlines: child.children, link });
747
+ }
748
+ }
749
+ else if (child.type === 'heading') {
750
+ const link = paragraphEntryLink(child);
751
+ if (link) {
752
+ faces.push({ inlines: child.children, link });
753
+ }
754
+ }
755
+ }
756
+ return faces;
757
+ }
758
+ // The shared entry emission for the non-list sources (paragraph entries,
759
+ // blockquote cards): resolve, split, title-fallback — the same decisions the
760
+ // list and table paths make through the same helpers.
761
+ function entryNodesFor(ownLink, inlines, repoInfoMap) {
762
+ const repoInfo = repoInfoMap.get(ownLink.url) ?? null;
763
+ const entryText = splitEntryText(inlines);
764
+ return emitEntryNodes(ownLink.url, repoInfo, {
765
+ title: entryTitle(entryText.title, ownLink, repoInfo),
766
+ description: entryText.description,
767
+ }, []);
768
+ }
769
+ // The <details><summary>…</summary> collapsible-section idiom: the summary
770
+ // text delimits structure like a heading would (java's generated README,
771
+ // paper-list tables behind "1.1 <topic>" summaries). kbd chips inside the
772
+ // summary ("5 projects") are metadata, not the title. A details block with
773
+ // no summary, or an empty one, delimits nothing.
774
+ const DETAILS_SUMMARY = /<details[^>]*>[\s\S]*?<summary[^>]*>([\s\S]*?)<\/summary>/i;
775
+ const DETAILS_CLOSE = /^\s*<\/details>/i;
776
+ function detailsSummaryTitle(htmlValue) {
777
+ const match = DETAILS_SUMMARY.exec(htmlValue);
778
+ if (!match) {
779
+ return null;
780
+ }
781
+ const title = match[1]
782
+ .replace(/<kbd>[\s\S]*?<\/kbd>/gi, '')
783
+ .replace(/<[^>]+>/g, ' ')
784
+ .replace(/\s+/g, ' ')
785
+ .trim();
786
+ return title || null;
787
+ }
788
+ // The summary text of a <details><summary> block inside a list item, when the
789
+ // item has no paragraph text of its own.
790
+ function listItemSummaryTitle(itemNode) {
791
+ for (const child of itemNode.children) {
792
+ if (child.type === 'html') {
793
+ const title = detailsSummaryTitle(child.value);
794
+ if (title) {
795
+ return title;
796
+ }
797
+ }
798
+ }
799
+ return '';
800
+ }
355
801
  const INVALID_TITLE_PATTERNS = [
356
802
  /^contributing/i,
357
803
  /^license$/i,
@@ -404,7 +850,7 @@ const TOC_TITLE_PATTERNS = [/^contents$/i, /^table of contents$/i];
404
850
  // A heading that delimits content structure. Text-less headings (a bare `#`,
405
851
  // an image-only heading) are spacers in real docs; TOC headings are structure
406
852
  // mirrors. Neither participates in the section tree.
407
- function isStructuralHeading(node) {
853
+ export function isStructuralHeading(node) {
408
854
  const title = getNodeText(node);
409
855
  return (!!title && !TOC_TITLE_PATTERNS.some(pattern => pattern.test(title.trim())));
410
856
  }
@@ -425,6 +871,17 @@ function findTitleHeadingIndex(tree) {
425
871
  node.depth === 1 &&
426
872
  isValidTitle(getNodeText(node)));
427
873
  }
874
+ // The heading branding owns even when it isn't a *valid* title: a generic
875
+ // first H1 ("# Guides", "# Contents") is still the de-facto title slot —
876
+ // applyBrandingToTree replaces it — so the section tree must not treat it
877
+ // as a section wrapping the whole document.
878
+ export function findTitleSlotIndex(tree) {
879
+ const titleHeadingIndex = findTitleHeadingIndex(tree);
880
+ if (titleHeadingIndex !== -1) {
881
+ return titleHeadingIndex;
882
+ }
883
+ return tree.children.findIndex((node) => node.type === 'heading' && node.depth === 1);
884
+ }
428
885
  /**
429
886
  * Makes the rendered H1 match metadata.title, replacing whichever heading
430
887
  * occupies the title slot (a valid title H1, else the first generic H1, else a
@@ -461,13 +918,7 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepository) {
461
918
  let documentTitle = titleHeadingIndex === -1
462
919
  ? ''
463
920
  : getNodeText(tree.children[titleHeadingIndex]);
464
- // The heading branding owns even when it isn't a *valid* title: a generic
465
- // first H1 ("# Guides", "# Contents") is still the de-facto title slot —
466
- // applyBrandingToTree replaces it — so the section tree must not treat it
467
- // as a section wrapping the whole document.
468
- const titleSlotIndex = titleHeadingIndex !== -1
469
- ? titleHeadingIndex
470
- : tree.children.findIndex((node) => node.type === 'heading' && node.depth === 1);
921
+ const titleSlotIndex = findTitleSlotIndex(tree);
471
922
  // Derive a subject from the *source* repository name when no valid H1 is
472
923
  // present. Using the source — not the enhanced/mirror repo — keeps the org
473
924
  // name out of the title.
@@ -478,8 +929,34 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepository) {
478
929
  }
479
930
  }
480
931
  const sectionDepth = findSectionDepth(tree, titleSlotIndex);
481
- const sections = [];
932
+ const gateForSection = sectionGatePasses(tree, titleSlotIndex, sectionDepth, repoInfoMap, sortOptions);
933
+ // Sections finalize out of document order when a details-section closes
934
+ // inside its parent section's span, so each carries the document index it
935
+ // opened at for the document-order sort at the end.
936
+ const sectionRecords = [];
482
937
  const stack = [];
938
+ // Entry emission shared by the paragraph and blockquote-card faces: the
939
+ // implicit section opens for a containerless entry (the stack-bottom
940
+ // invariant), the minLinks gate is the section aggregate, and a gated or
941
+ // dead entry emits nothing and stays out of the description (mirrors dead
942
+ // list items).
943
+ const emitStandaloneEntry = (ownLink, inlines, atIndex) => {
944
+ if (stack.length === 0) {
945
+ openImplicitSection(stack, atIndex);
946
+ }
947
+ if (gateForSection(stack[0])) {
948
+ stack[stack.length - 1].children.push(...entryNodesFor(ownLink, inlines, repoInfoMap));
949
+ }
950
+ };
951
+ // A standalone entry restating the repo of an open item container (crypto's
952
+ // generated cards repeat the repo URL in a line under the link-heading) is
953
+ // that item's description, not a second item for the same repo. The
954
+ // repoInfoMap resolves both URLs through one per-repo memo, so identity
955
+ // comparison holds even for aliased spellings.
956
+ const rementionsOpenItem = (link) => {
957
+ const repoInfo = repoInfoMap.get(link.url);
958
+ return !!repoInfo && stack.some(container => container.repoInfo === repoInfo);
959
+ };
483
960
  for (let i = 0; i < tree.children.length; i++) {
484
961
  const node = tree.children[i];
485
962
  if (node.type === 'heading') {
@@ -489,32 +966,117 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepository) {
489
966
  if (i === titleSlotIndex || !isStructuralHeading(node)) {
490
967
  continue;
491
968
  }
492
- closeContainers(stack, node.depth, sections);
493
- openContainer(stack, node, sectionDepth, repoInfoMap);
969
+ const headingEntry = entryHeadingInfo(node, repoInfoMap);
970
+ // The promotion flavor of the entry test: a textless identity link is
971
+ // a title-line badge, not an entry (see entryHeadingInfo).
972
+ const promotedEntry = headingEntry && getInlineText(headingEntry.link.children)
973
+ ? headingEntry
974
+ : null;
975
+ closeContainers(stack, node.depth, sectionRecords, !!promotedEntry);
976
+ openContainer(stack, node, i, sectionDepth, promotedEntry, headingEntry);
977
+ }
978
+ else if (node.type === 'paragraph') {
979
+ const text = getNodeText(node);
980
+ // Boilerplate "back to top" lines are neither entries nor description.
981
+ const ownLink = BACK_TO_TOP.test(text) ? undefined : paragraphEntryLink(node);
982
+ // An entry paragraph behaves like a one-item list; a failed gate, a
983
+ // dead target, or a re-mention of the enclosing item leaves plain
984
+ // description.
985
+ if (ownLink && !rementionsOpenItem(ownLink)) {
986
+ emitStandaloneEntry(ownLink, node.children, i);
987
+ }
988
+ else {
989
+ const container = stack[stack.length - 1];
990
+ // Avoid adding boilerplate "back to top" links to descriptions.
991
+ if (container && text && !BACK_TO_TOP.test(text)) {
992
+ container.description = container.description
993
+ ? `${container.description}\n${text}`
994
+ : text;
995
+ }
996
+ }
494
997
  }
495
- else if (node.type === 'paragraph' || node.type === 'blockquote') {
998
+ else if (node.type === 'blockquote') {
496
999
  const text = getNodeText(node);
497
- const container = stack[stack.length - 1];
498
- // Avoid adding boilerplate "back to top" links to descriptions.
499
- if (container && text && !text.includes('back to top')) {
500
- container.description = container.description
501
- ? `${container.description}\n${text}`
502
- : text;
1000
+ // A card blockquote is the blockquote face of entries — one per
1001
+ // qualifying paragraph/heading child; any other quote is container
1002
+ // prose.
1003
+ const faces = BACK_TO_TOP.test(text)
1004
+ ? []
1005
+ : blockquoteEntries(node).filter(face => !rementionsOpenItem(face.link));
1006
+ if (faces.length > 0) {
1007
+ for (const face of faces) {
1008
+ emitStandaloneEntry(face.link, face.inlines, i);
1009
+ }
1010
+ }
1011
+ else {
1012
+ const container = stack[stack.length - 1];
1013
+ // Avoid adding boilerplate "back to top" links to descriptions.
1014
+ if (container && text && !BACK_TO_TOP.test(text)) {
1015
+ container.description = container.description
1016
+ ? `${container.description}\n${text}`
1017
+ : text;
1018
+ }
1019
+ }
1020
+ }
1021
+ else if (node.type === 'html') {
1022
+ // A details-summary block opens a section like a heading would; the
1023
+ // close tag (or the next summary) ends it — never the enclosing
1024
+ // section, which keeps collecting after the collapsible block.
1025
+ const summaryTitle = detailsSummaryTitle(node.value);
1026
+ if (summaryTitle) {
1027
+ closeInnermostDetails(stack, sectionRecords);
1028
+ // Container depths never decrease going up the stack. When the open
1029
+ // containers sit deeper than sectionDepth (a mid-document H1 defines
1030
+ // sectionDepth while the content sections run deeper), the
1031
+ // details-section joins at the current depth — pushing at the
1032
+ // shallower sectionDepth would invert the stack and strand the gate
1033
+ // (which reads the stack bottom) on a tiny outer section.
1034
+ const joinDepth = stack.length === 0
1035
+ ? sectionDepth
1036
+ : Math.max(sectionDepth, stack[stack.length - 1].headingDepth);
1037
+ stack.push({
1038
+ children: [],
1039
+ description: '',
1040
+ headingDepth: joinDepth,
1041
+ headingIndex: i,
1042
+ kind: 'section',
1043
+ openedByDetails: true,
1044
+ title: summaryTitle,
1045
+ });
1046
+ }
1047
+ else if (DETAILS_CLOSE.test(node.value)) {
1048
+ closeInnermostDetails(stack, sectionRecords);
503
1049
  }
504
1050
  }
505
1051
  else if (node.type === 'list') {
1052
+ // A list with no open container still means content: synthesize the
1053
+ // implicit section for it (see openImplicitSection). Afterwards the
1054
+ // stack bottom is always a section — implicit or real — which is the
1055
+ // invariant the gate below and closeContainers rely on.
1056
+ if (stack.length === 0) {
1057
+ openImplicitSection(stack, i);
1058
+ }
506
1059
  // Every list inside the open container contributes items — a section is
507
- // not closed by its first list. With no open container (preamble), the
508
- // list is not part of any JSON section, but its AST is still sorted so
509
- // the rendered markdown matches.
510
- const items = processListRecursively(node, repoInfoMap, sortOptions);
511
- const container = stack[stack.length - 1];
512
- if (container) {
513
- container.children.push(...items);
1060
+ // not closed by its first list — and the minLinks gate is decided per
1061
+ // section, against the whole section subtree.
1062
+ const items = processListRecursively(node, repoInfoMap, sortOptions, false, gateForSection(stack[0]));
1063
+ stack[stack.length - 1].children.push(...items);
1064
+ }
1065
+ else if (node.type === 'table') {
1066
+ // Tables hold entries the same way lists do — the implicit section opens
1067
+ // for one with no open container (the stack-bottom invariant), and the
1068
+ // minLinks gate is the same section aggregate.
1069
+ if (stack.length === 0) {
1070
+ openImplicitSection(stack, i);
514
1071
  }
1072
+ const items = processTableRows(node, repoInfoMap, gateForSection(stack[0]));
1073
+ stack[stack.length - 1].children.push(...items);
515
1074
  }
516
1075
  }
517
- closeContainers(stack, 0, sections);
1076
+ closeContainers(stack, 0, sectionRecords);
1077
+ const sections = sectionRecords
1078
+ .sort((a, b) => a.headingIndex - b.headingIndex)
1079
+ .map(record => record.section);
518
1080
  return { sections, title: documentTitle, titleHeadingIndex };
519
1081
  }
520
1082
  // The heading depth that opens top-level sections: the shallowest structural
@@ -534,87 +1096,308 @@ function findSectionDepth(tree, titleSlotIndex) {
534
1096
  });
535
1097
  return depth;
536
1098
  }
537
- // A heading whose only link is a live GitHub link represents a resource, not a
538
- // container — the link-heading pattern (`#### [Repo](github…)`). Badge images
539
- // wrapped in links or multiple links disqualify (more than one link means the
540
- // heading is not "the" resource), as does a dead target.
541
- function soleLiveHeadingLink(heading, repoInfoMap) {
542
- const links = heading.children.filter((child) => child.type === 'link');
543
- if (links.length !== 1) {
1099
+ function entryHeadingInfo(heading, repoInfoMap) {
1100
+ const link = findFirstGitHubLink(heading);
1101
+ if (!link) {
544
1102
  return null;
545
1103
  }
546
- return repoInfoMap.get(links[0].url) ?? null;
1104
+ const repoInfo = repoInfoMap.get(link.url);
1105
+ return repoInfo ? { link, repoInfo } : null;
547
1106
  }
548
- function openContainer(stack, heading, sectionDepth, repoInfoMap) {
1107
+ function openContainer(stack, heading, headingIndex, sectionDepth, promotedEntry, headingEntry) {
549
1108
  const title = getNodeText(heading);
550
1109
  // Sections sit at the section level — and any heading met with an empty
551
1110
  // stack is promoted: a deeper heading before the first section (orphan
552
- // subheading) still owns its subtree, and a link-heading at section level
553
- // becomes a section rather than a top-level item, which the contract has no
554
- // place for.
1111
+ // subheading) still owns its subtree. A promoted ENTRY heading (see
1112
+ // entryHeadingInfo) is an item, not a section: the JSON contract has no
1113
+ // top-level items, so a synthesized section wraps the whole run of them.
555
1114
  if (stack.length === 0 || heading.depth === sectionDepth) {
1115
+ if (promotedEntry) {
1116
+ if (stack.length === 0) {
1117
+ openSynthesizedSection(stack, headingIndex, sectionDepth);
1118
+ }
1119
+ stack.push({
1120
+ children: [],
1121
+ description: '',
1122
+ headingDepth: heading.depth,
1123
+ headingIndex,
1124
+ kind: 'item',
1125
+ repoInfo: promotedEntry.repoInfo,
1126
+ title,
1127
+ });
1128
+ return;
1129
+ }
556
1130
  stack.push({
557
1131
  children: [],
558
1132
  description: '',
559
1133
  headingDepth: heading.depth,
1134
+ headingIndex,
560
1135
  kind: 'section',
561
1136
  title,
562
1137
  });
563
1138
  return;
564
1139
  }
565
- const repoInfo = soleLiveHeadingLink(heading, repoInfoMap);
1140
+ // A deeper entry heading opens an item container the same way a promoted
1141
+ // one does — children and prose below it collect as its content.
566
1142
  stack.push({
567
1143
  children: [],
568
1144
  description: '',
569
1145
  headingDepth: heading.depth,
570
- kind: repoInfo ? 'item' : 'group',
571
- repoInfo: repoInfo ?? undefined,
1146
+ headingIndex,
1147
+ kind: headingEntry ? 'item' : 'group',
1148
+ repoInfo: headingEntry?.repoInfo,
572
1149
  title,
573
1150
  });
574
1151
  }
575
- // Finalize every container a heading of `depth` closes (same-or-shallower),
576
- // bottom-up so each finalized node lands in its parent. Pruning falls out of
577
- // the finalize rule: a section/group whose children array is empty (no items
578
- // anywhere beneath — lists only return item-bearing nodes, and empty children
579
- // were never appended) is dropped; an item always survives, it IS the content.
580
- // The stack bottom is always a section (openContainer's promotion guarantees
581
- // it), so a finalized group/item always has a parent to land in.
582
- function closeContainers(stack, depth, sections) {
583
- while (stack.length > 0 &&
584
- stack[stack.length - 1].headingDepth >= depth) {
585
- const container = stack.pop();
586
- if (container.children.length === 0 && container.kind !== 'item') {
587
- continue;
1152
+ // A document can hold registry content with no heading above it — headingless
1153
+ // docs, TOC-only docs, or preamble entries (list, table) before the first
1154
+ // section heading. Rather than dropping them, the first one synthesizes the
1155
+ // section it implicitly belongs to ("Overview"). It behaves exactly like a
1156
+ // real section:
1157
+ // closed by the first structural heading (or document end), gated with its
1158
+ // whole subtree, and pruned when empty.
1159
+ function openImplicitSection(stack, atIndex) {
1160
+ stack.push({
1161
+ children: [],
1162
+ description: '',
1163
+ // Infinity so ANY structural heading closes it in closeContainers and
1164
+ // ends its span in the section gate — the implicit section never reaches
1165
+ // past the preamble.
1166
+ headingDepth: Infinity,
1167
+ // One before the opening list so the gate's scan (from headingIndex + 1)
1168
+ // counts that list itself.
1169
+ headingIndex: atIndex - 1,
1170
+ kind: 'section',
1171
+ title: 'Overview',
1172
+ });
1173
+ }
1174
+ // The section synthesized around a run of promoted entry headings
1175
+ // (openContainer's entry branch). It behaves like the implicit section — same
1176
+ // "Overview" title, gated by its subtree — but sits at sectionDepth, so the
1177
+ // first plain heading at that level ends the run, and closeContainers'
1178
+ // stopAtSynthesized keeps the entry headings' own pops from closing it:
1179
+ // consecutive entry headings are siblings INSIDE it.
1180
+ function openSynthesizedSection(stack, firstEntryIndex, depth) {
1181
+ stack.push({
1182
+ children: [],
1183
+ description: '',
1184
+ headingDepth: depth,
1185
+ // One before the first entry heading so the gate's scan (from
1186
+ // headingIndex + 1) counts that heading itself.
1187
+ headingIndex: firstEntryIndex - 1,
1188
+ kind: 'section',
1189
+ openedBySynthesis: true,
1190
+ title: 'Overview',
1191
+ });
1192
+ }
1193
+ // The minLinks gate scoped to a SECTION: every entry source in the section
1194
+ // counts together (list items, table rows, entry paragraphs, entry headings,
1195
+ // blockquote faces) — best-of-style documents put each entry in its own
1196
+ // single-item list, which the per-list gate dropped one by one. The section's
1197
+ // subtree runs from its heading to the first structural heading at or above
1198
+ // its depth (the same heading that would close it in closeContainers).
1199
+ //
1200
+ // Heading-per-entry documents (FBI-tools' `### name` + link paragraph, FLOSS's
1201
+ // `## game` + link cluster) put exactly ONE entry in each section, so every
1202
+ // section fails the gate alone and the whole document drops. They are not
1203
+ // sparse sections but flat entry lists delimited by headings: a section at
1204
+ // sectionDepth that fails alone is re-gated against its RUN — the maximal
1205
+ // sequence of adjacent section spans each holding at most one entry. The
1206
+ // noise floor survives: a single one-entry section between multi-entry
1207
+ // sections is a run of one and stays dropped.
1208
+ function sectionGatePasses(tree, titleSlotIndex, sectionDepth, repoInfoMap, sortOptions) {
1209
+ const cache = new Map();
1210
+ // Linked entries from scanStart to the first structural heading that would
1211
+ // close a container of closeDepth. An entry heading counts as an entry
1212
+ // wherever it sits inside the span — it opens an item, not a container —
1213
+ // except at the scanned section's own depth, where the walk closes that
1214
+ // section when the entry run starts (a same-depth entry heading is content
1215
+ // only for the synthesized wrapper, hence sameDepthEntriesAreContent).
1216
+ const scanCount = (scanStart, closeDepth, sameDepthEntriesAreContent) => {
1217
+ let linkedEntries = 0;
1218
+ for (let j = scanStart; j < tree.children.length; j++) {
1219
+ const node = tree.children[j];
1220
+ if (node.type === 'heading') {
1221
+ if (j === titleSlotIndex || !isStructuralHeading(node)) {
1222
+ continue;
1223
+ }
1224
+ if (entryHeadingInfo(node, repoInfoMap)) {
1225
+ if (sameDepthEntriesAreContent || node.depth > closeDepth) {
1226
+ linkedEntries += 1;
1227
+ }
1228
+ else {
1229
+ break;
1230
+ }
1231
+ continue;
1232
+ }
1233
+ if (node.depth <= closeDepth) {
1234
+ break;
1235
+ }
1236
+ continue;
1237
+ }
1238
+ if (node.type === 'list') {
1239
+ linkedEntries += countLinkedItems(node);
1240
+ }
1241
+ else if (node.type === 'table') {
1242
+ linkedEntries += countLinkedRows(node);
1243
+ }
1244
+ else if (node.type === 'paragraph' && paragraphEntryLink(node)) {
1245
+ linkedEntries += 1;
1246
+ }
1247
+ else if (node.type === 'blockquote') {
1248
+ linkedEntries += blockquoteEntries(node).length;
1249
+ }
588
1250
  }
589
- const description = container.description || null;
590
- if (container.kind === 'section') {
591
- sections.push({
592
- description,
593
- items: container.children,
594
- title: container.title,
595
- });
596
- continue;
1251
+ return linkedEntries;
1252
+ };
1253
+ // A boundary's entry count for the run walk: the heading itself when it is
1254
+ // an entry heading, plus its span's entries (the span closes at the first
1255
+ // heading at or above the boundary's OWN depth, exactly where the walk
1256
+ // closes that section).
1257
+ const spanCountCache = new Map();
1258
+ const spanCount = (boundaryIndex) => {
1259
+ const cached = spanCountCache.get(boundaryIndex);
1260
+ if (cached !== undefined) {
1261
+ return cached;
597
1262
  }
598
- const parent = stack[stack.length - 1];
599
- // Non-section containers always have an open parent (stack invariant), and
600
- // kind === 'item' exactly when repoInfo is set.
601
- if (container.repoInfo) {
602
- parent.children.push({
603
- children: container.children,
604
- description,
605
- node_type: 'item',
606
- repo_info: toRepoInfo(container.repoInfo),
607
- title: container.title,
1263
+ const heading = tree.children[boundaryIndex];
1264
+ const count = (entryHeadingInfo(heading, repoInfoMap) ? 1 : 0) +
1265
+ scanCount(boundaryIndex + 1, heading.depth, false);
1266
+ spanCountCache.set(boundaryIndex, count);
1267
+ return count;
1268
+ };
1269
+ // The section boundaries in document order, by the walk's own promotion
1270
+ // rule (openContainer): a heading opens a section when it sits at
1271
+ // sectionDepth or arrives with nothing open — FLOSS-style documents have a
1272
+ // lone late H1 that pins sectionDepth at 1 while every `## game` heading
1273
+ // alternately closes its predecessor (stack empties) and is promoted. The
1274
+ // depth simulation mirrors closeContainers/openContainer exactly.
1275
+ let boundaries;
1276
+ const sectionBoundaries = () => {
1277
+ if (!boundaries) {
1278
+ const indices = [];
1279
+ const openDepths = [];
1280
+ tree.children.forEach((node, i) => {
1281
+ if (node.type !== 'heading' || i === titleSlotIndex) {
1282
+ return;
1283
+ }
1284
+ if (!isStructuralHeading(node)) {
1285
+ return;
1286
+ }
1287
+ while (openDepths.length > 0 &&
1288
+ openDepths[openDepths.length - 1] >= node.depth) {
1289
+ openDepths.pop();
1290
+ }
1291
+ if (openDepths.length === 0 || node.depth === sectionDepth) {
1292
+ indices.push(i);
1293
+ }
1294
+ openDepths.push(node.depth);
608
1295
  });
1296
+ boundaries = indices;
609
1297
  }
610
- else {
611
- parent.children.push({
612
- children: container.children,
613
- description,
614
- node_type: 'group',
615
- title: container.title,
616
- });
1298
+ return boundaries;
1299
+ };
1300
+ const runTotal = (section) => {
1301
+ const list = sectionBoundaries();
1302
+ const k = list.indexOf(section.headingIndex);
1303
+ if (k === -1) {
1304
+ return spanCount(section.headingIndex);
1305
+ }
1306
+ let total = spanCount(section.headingIndex);
1307
+ for (const direction of [-1, 1]) {
1308
+ for (let m = k + direction; m >= 0 && m < list.length; m += direction) {
1309
+ const count = spanCount(list[m]);
1310
+ if (count > 1) {
1311
+ break;
1312
+ }
1313
+ total += count;
1314
+ }
1315
+ }
1316
+ return total;
1317
+ };
1318
+ return section => {
1319
+ const cached = cache.get(section.headingIndex);
1320
+ if (cached !== undefined) {
1321
+ return cached;
1322
+ }
1323
+ const own = scanCount(section.headingIndex + 1, section.headingDepth, !!section.openedBySynthesis);
1324
+ const passes = own >= sortOptions.minLinks ||
1325
+ (!section.openedBySynthesis &&
1326
+ sectionBoundaries().includes(section.headingIndex) &&
1327
+ runTotal(section) >= sortOptions.minLinks);
1328
+ cache.set(section.headingIndex, passes);
1329
+ return passes;
1330
+ };
1331
+ }
1332
+ // Prune-or-emit one popped container: a section/group whose children array is
1333
+ // empty (no items anywhere beneath — the entry sources only return
1334
+ // item-bearing nodes) is dropped; an item always survives, it IS the content.
1335
+ // Non-section containers always have an open parent (stack invariant), and
1336
+ // kind === 'item' exactly when repoInfo is set.
1337
+ function finalizeContainer(container, stack, sectionRecords) {
1338
+ if (container.children.length === 0 && container.kind !== 'item') {
1339
+ return;
1340
+ }
1341
+ const description = container.description || null;
1342
+ if (container.kind === 'section') {
1343
+ sectionRecords.push({
1344
+ headingIndex: container.headingIndex,
1345
+ section: { description, items: container.children, title: container.title },
1346
+ });
1347
+ return;
1348
+ }
1349
+ const parent = stack[stack.length - 1];
1350
+ if (container.repoInfo) {
1351
+ parent.children.push({
1352
+ children: container.children,
1353
+ description,
1354
+ node_type: 'item',
1355
+ repo_info: toRepoInfo(container.repoInfo),
1356
+ title: container.title,
1357
+ });
1358
+ }
1359
+ else {
1360
+ parent.children.push({
1361
+ children: container.children,
1362
+ description,
1363
+ node_type: 'group',
1364
+ title: container.title,
1365
+ });
1366
+ }
1367
+ }
1368
+ // Finalize every container a heading of `depth` closes (same-or-shallower),
1369
+ // bottom-up so each finalized node lands in its parent. The stack bottom is
1370
+ // always a section (openContainer's promotion guarantees it), so a finalized
1371
+ // group/item always has a parent to land in. stopAtSynthesized: the pop an
1372
+ // entry heading triggers must stop at the synthesized section wrapping its
1373
+ // run — the next entry heading of the run lands back inside it.
1374
+ function closeContainers(stack, depth, sectionRecords, stopAtSynthesized = false) {
1375
+ while (stack.length > 0 &&
1376
+ stack[stack.length - 1].headingDepth >= depth) {
1377
+ if (stopAtSynthesized && stack[stack.length - 1].openedBySynthesis) {
1378
+ break;
617
1379
  }
1380
+ finalizeContainer(stack.pop(), stack, sectionRecords);
1381
+ }
1382
+ }
1383
+ // Ends the innermost open details-section and everything opened inside it,
1384
+ // leaving the enclosing containers untouched — unlike a heading close, a
1385
+ // details boundary never ends its parent section, so content after the
1386
+ // collapsible block keeps collecting under it. A stray </details> (no
1387
+ // details-section open) is a no-op.
1388
+ function closeInnermostDetails(stack, sectionRecords) {
1389
+ let detailsIndex = -1;
1390
+ for (let s = stack.length - 1; s >= 0; s--) {
1391
+ if (stack[s].openedByDetails) {
1392
+ detailsIndex = s;
1393
+ break;
1394
+ }
1395
+ }
1396
+ if (detailsIndex === -1) {
1397
+ return;
1398
+ }
1399
+ while (stack.length > detailsIndex) {
1400
+ finalizeContainer(stack.pop(), stack, sectionRecords);
618
1401
  }
619
1402
  }
620
1403
  function serializeAst(tree, originalContent) {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@enhansome/core",
3
- "version": "1.8.1",
3
+ "version": "1.10.0",
4
4
  "description": "Library core for enhansome — enhance markdown with GitHub star counts.",
5
5
  "repository": {
6
6
  "type": "git",