@enhansome/core 1.8.1 → 1.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/markdown.d.ts +6 -0
- package/dist/markdown.js +934 -151
- package/package.json +1 -1
package/dist/markdown.d.ts
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { RepoInfoDetails } from './github.js';
|
|
2
2
|
import { Logger } from './logger.js';
|
|
3
|
+
import type { Heading, Link, List, Parent, Root } from 'mdast';
|
|
3
4
|
export interface JsonOutput {
|
|
4
5
|
items: JsonSection[];
|
|
5
6
|
metadata: JsonMetadata;
|
|
@@ -58,3 +59,8 @@ export declare function processMarkdownContent(originalContent: string, token: s
|
|
|
58
59
|
finalContent: string;
|
|
59
60
|
jsonData: JsonOutput;
|
|
60
61
|
}>;
|
|
62
|
+
export declare function normalizeGitHubUrls(tree: Root): void;
|
|
63
|
+
export declare function findFirstGitHubLink(node: Parent): Link | undefined;
|
|
64
|
+
export declare function countLinkedItems(listNode: List): number;
|
|
65
|
+
export declare function isStructuralHeading(node: Heading): boolean;
|
|
66
|
+
export declare function findTitleSlotIndex(tree: Root): number;
|
package/dist/markdown.js
CHANGED
|
@@ -85,6 +85,7 @@ export async function processMarkdownContent(originalContent, token, replacement
|
|
|
85
85
|
const contentAfterReplacements = applyTextReplacements(originalContent, replacements.filter(rule => rule.type !== 'branding'), log);
|
|
86
86
|
const processor = unified().use(remarkParse).use(remarkGfm);
|
|
87
87
|
const tree = processor.parse(contentAfterReplacements);
|
|
88
|
+
normalizeGitHubUrls(tree);
|
|
88
89
|
const githubUrls = collectGitHubLinks(tree);
|
|
89
90
|
const repoInfoMap = await fetchTargetData(githubUrls, repos);
|
|
90
91
|
// The title derives from the *source* repository (originalRepository), never
|
|
@@ -176,6 +177,112 @@ function collectGitHubLinks(tree) {
|
|
|
176
177
|
});
|
|
177
178
|
return urls;
|
|
178
179
|
}
|
|
180
|
+
// Input normalization, same family as fixRelativeLinks: make GitHub repos a
|
|
181
|
+
// source expresses WITHOUT markdown links visible as real link nodes, so
|
|
182
|
+
// every downstream consumer — repo fetch, entry tests, gates, badges — sees
|
|
183
|
+
// the shape a markdown link would have produced. Two families:
|
|
184
|
+
//
|
|
185
|
+
// (1) Bare scheme-less `github.com/owner/repo` text — GFM autolinks only
|
|
186
|
+
// scheme-full URLs, so these stay plain text. The link's label is the
|
|
187
|
+
// URL text; only owner/repo are consumed (a deeper path like `/tree/main`
|
|
188
|
+
// stays in the trailing text — the identity reads the first two segments
|
|
189
|
+
// either way).
|
|
190
|
+
// (2) Inline `<a href="…">…</a>` anchors — remark emits the open and close
|
|
191
|
+
// tags as separate html nodes around the label's inline nodes, so the
|
|
192
|
+
// pair is rewrapped as one link. Anchors inside BLOCK html (`<details>`
|
|
193
|
+
// summaries, centered banners) keep their raw form: that html is the
|
|
194
|
+
// structure the walk already reads (detailsSummaryTitle), and rewriting
|
|
195
|
+
// it is a different decision than link visibility.
|
|
196
|
+
//
|
|
197
|
+
// Text inside code spans/blocks never linkifies (code has no text nodes) and
|
|
198
|
+
// text inside an existing link label is skipped — nested links are not a
|
|
199
|
+
// thing. Non-GitHub and org-only anchors stay raw html.
|
|
200
|
+
const BARE_GITHUB_URL = /(?:^|(?<=[\s(>\[]))(?:www\.)?github\.com\/[A-Za-z0-9_.-]+\/[A-Za-z0-9_.-]+/g;
|
|
201
|
+
const ANCHOR_OPEN = /^<a\s[^>]*href=(["'])([^"']*)\1[^>]*>$/;
|
|
202
|
+
const ANCHOR_CLOSE = /^<\/a\s*>$/;
|
|
203
|
+
export function normalizeGitHubUrls(tree) {
|
|
204
|
+
const walk = (node, inLink) => {
|
|
205
|
+
if (inLink) {
|
|
206
|
+
return;
|
|
207
|
+
}
|
|
208
|
+
const parent = node;
|
|
209
|
+
if (!Array.isArray(parent.children)) {
|
|
210
|
+
return;
|
|
211
|
+
}
|
|
212
|
+
for (const child of parent.children) {
|
|
213
|
+
walk(child, child.type === 'link');
|
|
214
|
+
}
|
|
215
|
+
parent.children = normalizeInlineChildren(parent.children);
|
|
216
|
+
};
|
|
217
|
+
walk(tree, false);
|
|
218
|
+
}
|
|
219
|
+
// One parent's inline run: bare-URL text nodes split into text/link/text,
|
|
220
|
+
// anchor open+close pairs rewrapped as a link. The anchor's inner nodes are
|
|
221
|
+
// taken verbatim — they are the label, and splitting them would nest links.
|
|
222
|
+
function normalizeInlineChildren(children) {
|
|
223
|
+
const out = [];
|
|
224
|
+
for (let i = 0; i < children.length; i++) {
|
|
225
|
+
const child = children[i];
|
|
226
|
+
if (child.type === 'html') {
|
|
227
|
+
const anchor = ANCHOR_OPEN.exec(child.value.trim());
|
|
228
|
+
const url = anchor && parseGitHubUrl(anchor[2]) ? anchor[2] : null;
|
|
229
|
+
if (url) {
|
|
230
|
+
const closeIndex = children.findIndex((candidate, j) => j > i &&
|
|
231
|
+
candidate.type === 'html' &&
|
|
232
|
+
ANCHOR_CLOSE.test(candidate.value.trim()));
|
|
233
|
+
if (closeIndex !== -1) {
|
|
234
|
+
out.push({
|
|
235
|
+
type: 'link',
|
|
236
|
+
url,
|
|
237
|
+
children: children.slice(i + 1, closeIndex),
|
|
238
|
+
});
|
|
239
|
+
i = closeIndex;
|
|
240
|
+
continue;
|
|
241
|
+
}
|
|
242
|
+
}
|
|
243
|
+
out.push(child);
|
|
244
|
+
continue;
|
|
245
|
+
}
|
|
246
|
+
if (child.type === 'text') {
|
|
247
|
+
out.push(...linkifyBareUrls(child));
|
|
248
|
+
continue;
|
|
249
|
+
}
|
|
250
|
+
out.push(child);
|
|
251
|
+
}
|
|
252
|
+
return out;
|
|
253
|
+
}
|
|
254
|
+
function linkifyBareUrls(node) {
|
|
255
|
+
const matches = [...node.value.matchAll(BARE_GITHUB_URL)];
|
|
256
|
+
if (matches.length === 0) {
|
|
257
|
+
return [node];
|
|
258
|
+
}
|
|
259
|
+
const out = [];
|
|
260
|
+
let cursor = 0;
|
|
261
|
+
for (const match of matches) {
|
|
262
|
+
const label = match[0].replace(/\.+$/, '');
|
|
263
|
+
const url = `https://${label.replace(/^www\./, '')}`;
|
|
264
|
+
if (!parseGitHubUrl(url)) {
|
|
265
|
+
continue;
|
|
266
|
+
}
|
|
267
|
+
const start = match.index;
|
|
268
|
+
if (start > cursor) {
|
|
269
|
+
out.push({
|
|
270
|
+
type: 'text',
|
|
271
|
+
value: node.value.slice(cursor, start),
|
|
272
|
+
});
|
|
273
|
+
}
|
|
274
|
+
out.push({
|
|
275
|
+
type: 'link',
|
|
276
|
+
url,
|
|
277
|
+
children: [{ type: 'text', value: label }],
|
|
278
|
+
});
|
|
279
|
+
cursor = start + label.length;
|
|
280
|
+
}
|
|
281
|
+
if (cursor < node.value.length) {
|
|
282
|
+
out.push({ type: 'text', value: node.value.slice(cursor) });
|
|
283
|
+
}
|
|
284
|
+
return out;
|
|
285
|
+
}
|
|
179
286
|
function createBadgeText(info) {
|
|
180
287
|
if (info.archived) {
|
|
181
288
|
return ' ⚠️ Archived';
|
|
@@ -192,27 +299,41 @@ function createBadgeText(info) {
|
|
|
192
299
|
}
|
|
193
300
|
return ` ${parts.join(' | ')}`;
|
|
194
301
|
}
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
302
|
+
// The first link in a subtree that points at a GitHub repo. Returns the link
|
|
303
|
+
// node itself so callers can take both its URL (identity) and its text
|
|
304
|
+
// (title fallbacks).
|
|
305
|
+
export function findFirstGitHubLink(node) {
|
|
306
|
+
let linkNode;
|
|
307
|
+
visit(node, 'link', (candidate) => {
|
|
308
|
+
if (!linkNode && parseGitHubUrl(candidate.url)) {
|
|
309
|
+
linkNode = candidate;
|
|
200
310
|
}
|
|
201
311
|
});
|
|
202
|
-
return
|
|
312
|
+
return linkNode;
|
|
203
313
|
}
|
|
204
314
|
// The GitHub link that represents a list item: the FIRST GitHub link in the
|
|
205
|
-
// item's OWN
|
|
206
|
-
//
|
|
207
|
-
// own paragraph
|
|
315
|
+
// item's OWN paragraphs, in document order. Paper-list entries carry their
|
|
316
|
+
// identity link ([[Code]](github)) in a paragraph after the title one, so
|
|
317
|
+
// every own paragraph is scanned — the title still comes from the first
|
|
318
|
+
// paragraph alone. A nested-descendant link is deliberately ignored — it
|
|
319
|
+
// belongs to a child, not to this item. An item is a GitHub node iff its own
|
|
320
|
+
// paragraphs link to a GitHub repo; an item with no own GitHub link but
|
|
208
321
|
// nested GitHub children is a group, not a node borrowing a child's identity.
|
|
209
322
|
//
|
|
210
|
-
// A GitHub link that is secondary within
|
|
323
|
+
// A GitHub link that is secondary within a paragraph (e.g.
|
|
211
324
|
// `[name](marketplace) … [On GitHub](github)`) is still the item's own link, so
|
|
212
325
|
// `findFirstGitHubLink` over the paragraph finds it correctly.
|
|
213
326
|
function findOwnGitHubLink(itemNode) {
|
|
214
|
-
const
|
|
215
|
-
|
|
327
|
+
for (const child of itemNode.children) {
|
|
328
|
+
if (child.type !== 'paragraph') {
|
|
329
|
+
continue;
|
|
330
|
+
}
|
|
331
|
+
const link = findFirstGitHubLink(child);
|
|
332
|
+
if (link) {
|
|
333
|
+
return link;
|
|
334
|
+
}
|
|
335
|
+
}
|
|
336
|
+
return undefined;
|
|
216
337
|
}
|
|
217
338
|
function fixRelativeLinks(tree, relativeLinkPrefix) {
|
|
218
339
|
if (!relativeLinkPrefix) {
|
|
@@ -274,15 +395,89 @@ function compareByRepoInfo(by, a, b) {
|
|
|
274
395
|
const timeB = b.pushed_at ? new Date(b.pushed_at).getTime() : 0;
|
|
275
396
|
return timeB - timeA;
|
|
276
397
|
}
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
398
|
+
// The minLinks noise gate counts items whose SUBTREE contains a GitHub link,
|
|
399
|
+
// not items with an own-paragraph link only. A category with no own link but
|
|
400
|
+
// nested GitHub children still counts (it becomes a group); switching to
|
|
401
|
+
// `findOwnGitHubLink` would silently drop purely categorical sections, so it
|
|
402
|
+
// deliberately diverges from the identity resolver.
|
|
403
|
+
export function countLinkedItems(listNode) {
|
|
404
|
+
return listNode.children.filter(item => !!findFirstGitHubLink(item)).length;
|
|
405
|
+
}
|
|
406
|
+
// The table counterpart of countLinkedItems for the section gate: rows whose
|
|
407
|
+
// subtree contains a GitHub link. No header-row skip — a card grid's first row
|
|
408
|
+
// sits above the delimiter and is content, while pure-label header rows carry
|
|
409
|
+
// no links and count zero either way.
|
|
410
|
+
function countLinkedRows(tableNode) {
|
|
411
|
+
return tableNode.children.filter(row => row.type === 'tableRow' && !!findFirstGitHubLink(row)).length;
|
|
412
|
+
}
|
|
413
|
+
// One entry's title and description, split from its own inlines: leading text
|
|
414
|
+
// up to and including the first link is the title (prefix tags like
|
|
415
|
+
// "[UPDATED]" survive), the prose trailing it the description. The link is
|
|
416
|
+
// found through emphasis wrappers — a bolded card link
|
|
417
|
+
// (`**[Name](repo)** - description`) splits like a plain one. With no link
|
|
418
|
+
// the whole text is the title — the description must never echo the title
|
|
419
|
+
// back.
|
|
420
|
+
function splitEntryText(inlines) {
|
|
421
|
+
const linkIndex = inlines.findIndex(child => {
|
|
422
|
+
let hasLink = false;
|
|
423
|
+
visit(child, 'link', () => {
|
|
424
|
+
hasLink = true;
|
|
425
|
+
});
|
|
426
|
+
return hasLink;
|
|
427
|
+
});
|
|
428
|
+
if (linkIndex === -1) {
|
|
429
|
+
return { title: getInlineText(inlines), description: '' };
|
|
430
|
+
}
|
|
431
|
+
return {
|
|
432
|
+
title: getInlineText(inlines.slice(0, linkIndex + 1)),
|
|
433
|
+
description: stripLeadingNoise(getInlineText(inlines.slice(linkIndex + 1))),
|
|
434
|
+
};
|
|
435
|
+
}
|
|
436
|
+
// The emission decision every entry source shares — list items, table rows,
|
|
437
|
+
// and the paragraph/blockquote sources to come: an own GitHub link that
|
|
438
|
+
// resolved makes an item; a dead own link drops the entry and lifts its
|
|
439
|
+
// children to the nearest live parent; no own link with children makes a
|
|
440
|
+
// group — never a `repo_info`, that's the identity-borrowing bug; no own link
|
|
441
|
+
// and no children is a non-GitHub leaf, kept in markdown, dropped from JSON.
|
|
442
|
+
// TODO(future): preserve non-GitHub leaves in a separate shape.
|
|
443
|
+
function emitEntryNodes(githubUrl, repoInfo, text, childrenJson) {
|
|
444
|
+
if (githubUrl && repoInfo) {
|
|
445
|
+
return [
|
|
446
|
+
{
|
|
447
|
+
node_type: 'item',
|
|
448
|
+
title: text.title,
|
|
449
|
+
description: text.description || null,
|
|
450
|
+
children: childrenJson,
|
|
451
|
+
repo_info: toRepoInfo(repoInfo),
|
|
452
|
+
},
|
|
453
|
+
];
|
|
454
|
+
}
|
|
455
|
+
if (githubUrl) {
|
|
456
|
+
return childrenJson;
|
|
457
|
+
}
|
|
458
|
+
if (childrenJson.length > 0) {
|
|
459
|
+
return [
|
|
460
|
+
{
|
|
461
|
+
node_type: 'group',
|
|
462
|
+
title: text.title,
|
|
463
|
+
description: text.description || null,
|
|
464
|
+
children: childrenJson,
|
|
465
|
+
},
|
|
466
|
+
];
|
|
467
|
+
}
|
|
468
|
+
return [];
|
|
469
|
+
}
|
|
470
|
+
function processListRecursively(listNode, repoInfoMap, sortOptions, isNested = false,
|
|
471
|
+
// The caller's section-scope gate decision (sectionGatePasses). Absent for
|
|
472
|
+
// nested lists (emitted under their parent regardless) and for top-level
|
|
473
|
+
// lists with no open container (preamble), which gate per list.
|
|
474
|
+
sectionGateOpen) {
|
|
475
|
+
if (!isNested) {
|
|
476
|
+
const gateOpen = sectionGateOpen ??
|
|
477
|
+
countLinkedItems(listNode) >= sortOptions.minLinks;
|
|
478
|
+
if (!gateOpen) {
|
|
479
|
+
return [];
|
|
480
|
+
}
|
|
286
481
|
}
|
|
287
482
|
// Zip each item with its emitted JSON nodes and repo info so one sort orders
|
|
288
483
|
// both the rendered AST and the emitted JSON. `emitted` is empty for
|
|
@@ -290,59 +485,31 @@ function processListRecursively(listNode, repoInfoMap, sortOptions, isNested = f
|
|
|
290
485
|
// AST, dropped from JSON.
|
|
291
486
|
const entries = [];
|
|
292
487
|
for (const itemNode of listNode.children) {
|
|
293
|
-
const
|
|
488
|
+
const ownLink = findOwnGitHubLink(itemNode);
|
|
489
|
+
const githubUrl = ownLink?.url;
|
|
294
490
|
const repoInfo = githubUrl ? (repoInfoMap.get(githubUrl) ?? null) : null;
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
const
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
description = '';
|
|
313
|
-
}
|
|
314
|
-
}
|
|
315
|
-
// No-own-link items with children become groups — NEVER give them a
|
|
316
|
-
// `repo_info`, that's the identity-borrowing bug. No-own-link, no-child
|
|
317
|
-
// items are non-GitHub leaves: kept in markdown, dropped from JSON.
|
|
318
|
-
// TODO(future): preserve non-GitHub leaves in a separate shape.
|
|
319
|
-
let emitted = [];
|
|
320
|
-
if (githubUrl && repoInfo) {
|
|
321
|
-
emitted = [
|
|
322
|
-
{
|
|
323
|
-
node_type: 'item',
|
|
324
|
-
title,
|
|
325
|
-
description: description || null,
|
|
326
|
-
children: childrenJson,
|
|
327
|
-
repo_info: toRepoInfo(repoInfo),
|
|
328
|
-
},
|
|
329
|
-
];
|
|
330
|
-
}
|
|
331
|
-
else if (githubUrl) {
|
|
332
|
-
// Dead target: the item itself is not emitted; its children lift to this
|
|
333
|
-
// list's level — the nearest live parent.
|
|
334
|
-
emitted = childrenJson;
|
|
335
|
-
}
|
|
336
|
-
else if (childrenJson.length > 0) {
|
|
337
|
-
emitted = [
|
|
338
|
-
{
|
|
339
|
-
node_type: 'group',
|
|
340
|
-
title,
|
|
341
|
-
description: description || null,
|
|
342
|
-
children: childrenJson,
|
|
343
|
-
},
|
|
344
|
-
];
|
|
491
|
+
// Nested content is the item's children: deeper lists as before, plus
|
|
492
|
+
// tables — the AnimeResearch shape wraps a <details><summary> block and
|
|
493
|
+
// its table inside one list item. Both emit under the parent item
|
|
494
|
+
// regardless of the gate: the top-level call already gated the section.
|
|
495
|
+
const nestedContent = itemNode.children.filter((child) => child.type === 'list' || child.type === 'table');
|
|
496
|
+
const childrenJson = nestedContent.flatMap(child => child.type === 'list'
|
|
497
|
+
? processListRecursively(child, repoInfoMap, sortOptions, true)
|
|
498
|
+
: processTableRows(child, repoInfoMap, true));
|
|
499
|
+
// Title/description split on the FIRST paragraph only — a paper-list
|
|
500
|
+
// entry's identity link may live in a later paragraph (findOwnGitHubLink)
|
|
501
|
+
// while its title text stays the leading one.
|
|
502
|
+
const paragraph = itemNode.children.find((child) => child.type === 'paragraph');
|
|
503
|
+
const entryText = splitEntryText(paragraph?.children ?? []);
|
|
504
|
+
// A details-wrapped item carries its visible text in the summary, not in
|
|
505
|
+
// a paragraph — that text is the group's title.
|
|
506
|
+
if (!entryText.title) {
|
|
507
|
+
entryText.title = listItemSummaryTitle(itemNode);
|
|
345
508
|
}
|
|
509
|
+
// The shared title fallbacks (an inline-code link label carries no text
|
|
510
|
+
// nodes, so the split alone can leave an empty title).
|
|
511
|
+
entryText.title = entryTitle(entryText.title, ownLink, repoInfo);
|
|
512
|
+
const emitted = emitEntryNodes(githubUrl, repoInfo, entryText, childrenJson);
|
|
346
513
|
entries.push({ emitted, node: itemNode, repoInfo });
|
|
347
514
|
}
|
|
348
515
|
if (sortOptions.by) {
|
|
@@ -352,6 +519,285 @@ function processListRecursively(listNode, repoInfoMap, sortOptions, isNested = f
|
|
|
352
519
|
listNode.children = entries.map(entry => entry.node);
|
|
353
520
|
return entries.flatMap(entry => entry.emitted);
|
|
354
521
|
}
|
|
522
|
+
// Title with the fallbacks every entry source shares: the split's own text
|
|
523
|
+
// when it names something; else the own link's label when meaningful; else —
|
|
524
|
+
// for a degenerate base (rank/year cell, URL label, tag word) owner/name, for
|
|
525
|
+
// an absent base (an image-only link) the repo name. A degenerate base with
|
|
526
|
+
// no live repo link (a group) keeps its text: nothing better exists.
|
|
527
|
+
function entryTitle(base, ownLink, repoInfo) {
|
|
528
|
+
if (base && !isDegenerateTitle(base)) {
|
|
529
|
+
return base;
|
|
530
|
+
}
|
|
531
|
+
const linkText = ownLink ? getInlineText(ownLink.children) : '';
|
|
532
|
+
if (isMeaningfulLinkText(linkText)) {
|
|
533
|
+
return linkText;
|
|
534
|
+
}
|
|
535
|
+
if (base === '') {
|
|
536
|
+
return repoInfo?.repo ?? '';
|
|
537
|
+
}
|
|
538
|
+
return repoInfo ? `${repoInfo.owner}/${repoInfo.repo}` : base;
|
|
539
|
+
}
|
|
540
|
+
// Table rows are entries under the nearest open container. The common shape —
|
|
541
|
+
// scala's `[name](repo) | description`, spec tables with the link in a later
|
|
542
|
+
// column — holds ONE repo per row: the title is the first cell's text (the
|
|
543
|
+
// own link's text, then the repo name, when the first cell is empty), the
|
|
544
|
+
// description the remaining cells' text (badge images carry no text nodes, so
|
|
545
|
+
// they never pollute it). Card grids pin several repos per row, each cell its
|
|
546
|
+
// own card, so those emit one entry per link-bearing cell. A linked row above
|
|
547
|
+
// the delimiter is content like any other (grid tables have no label header;
|
|
548
|
+
// pure-label header rows carry no links and emit nothing). Rows stay in source
|
|
549
|
+
// order — unlike the unranked lists the product sorts, a table's row order is
|
|
550
|
+
// part of its meaning.
|
|
551
|
+
function processTableRows(tableNode, repoInfoMap, gateOpen) {
|
|
552
|
+
if (!gateOpen) {
|
|
553
|
+
return [];
|
|
554
|
+
}
|
|
555
|
+
const items = [];
|
|
556
|
+
for (const row of tableNode.children) {
|
|
557
|
+
if (row.type !== 'tableRow') {
|
|
558
|
+
continue;
|
|
559
|
+
}
|
|
560
|
+
const cellLinks = row.children.map(cell => findFirstGitHubLink(cell));
|
|
561
|
+
const distinctUrls = new Set(cellLinks.flatMap(link => (link ? [link.url] : [])));
|
|
562
|
+
if (distinctUrls.size === 0) {
|
|
563
|
+
continue;
|
|
564
|
+
}
|
|
565
|
+
if (distinctUrls.size === 1) {
|
|
566
|
+
const ownLink = cellLinks.find((link) => !!link);
|
|
567
|
+
if (!ownLink) {
|
|
568
|
+
continue;
|
|
569
|
+
}
|
|
570
|
+
const repoInfo = repoInfoMap.get(ownLink.url) ?? null;
|
|
571
|
+
const firstCellText = splitEntryText(row.children[0]?.children ?? []);
|
|
572
|
+
// A degenerate first cell (a rank/year column) moves the title source to
|
|
573
|
+
// the link's own cell: that cell no longer feeds the description — its
|
|
574
|
+
// label became the title — and the degenerate cell drops as noise.
|
|
575
|
+
const linkCellIndex = isDegenerateTitle(firstCellText.title)
|
|
576
|
+
? cellLinks.findIndex(link => link === ownLink)
|
|
577
|
+
: -1;
|
|
578
|
+
const titleText = linkCellIndex === -1
|
|
579
|
+
? firstCellText
|
|
580
|
+
: splitEntryText(row.children[linkCellIndex].children);
|
|
581
|
+
const description = [
|
|
582
|
+
titleText.description,
|
|
583
|
+
...row.children
|
|
584
|
+
.slice(1)
|
|
585
|
+
.filter((_cell, index) => index + 1 !== linkCellIndex)
|
|
586
|
+
.map(cell => getNodeText(cell)),
|
|
587
|
+
]
|
|
588
|
+
.join(' ')
|
|
589
|
+
.trim();
|
|
590
|
+
items.push(...emitEntryNodes(ownLink.url, repoInfo, {
|
|
591
|
+
title: entryTitle(titleText.title, ownLink, repoInfo),
|
|
592
|
+
description,
|
|
593
|
+
}, []));
|
|
594
|
+
continue;
|
|
595
|
+
}
|
|
596
|
+
const emittedUrls = new Set();
|
|
597
|
+
for (const [cellIndex, ownLink] of cellLinks.entries()) {
|
|
598
|
+
if (!ownLink || emittedUrls.has(ownLink.url)) {
|
|
599
|
+
continue;
|
|
600
|
+
}
|
|
601
|
+
emittedUrls.add(ownLink.url);
|
|
602
|
+
const repoInfo = repoInfoMap.get(ownLink.url) ?? null;
|
|
603
|
+
const cellText = splitEntryText(row.children[cellIndex].children);
|
|
604
|
+
items.push(...emitEntryNodes(ownLink.url, repoInfo, {
|
|
605
|
+
title: entryTitle(cellText.title, ownLink, repoInfo),
|
|
606
|
+
description: cellText.description,
|
|
607
|
+
}, []));
|
|
608
|
+
}
|
|
609
|
+
}
|
|
610
|
+
return items;
|
|
611
|
+
}
|
|
612
|
+
// A top-level paragraph can BE an entry, not only feed container prose. Two
|
|
613
|
+
// corpus families qualify (progress/empty-tree-parses.md, step 4): a GitHub
|
|
614
|
+
// link LEADING the paragraph behind a short entry label, or the paragraph
|
|
615
|
+
// ending in a tag cluster whose GitHub link carries the identity — the
|
|
616
|
+
// paper-list shape whose first link points at the paper and a [Code]/[Github]
|
|
617
|
+
// tag at the end, name/author lines ending in that tag, and dated lines
|
|
618
|
+
// ("… [Github] 4 Feb 2023"). Prose that mentions a repo ("Please see
|
|
619
|
+
// CONTRIBUTING", "See also [repo]", intro text ending in a bare repo URL) is
|
|
620
|
+
// neither and stays description. Calibrated on the 2,293-README corpus plus
|
|
621
|
+
// the fixture fleet, not intuition: an entry label is a TAG (empty, CJK/emoji,
|
|
622
|
+
// bracketed, or colon-terminated) — never bare English prose — and the
|
|
623
|
+
// identity link must carry text, so the ubiquitous image-only awesome badge
|
|
624
|
+
// is not an entry.
|
|
625
|
+
const ENTRY_LABEL_MAX = 15;
|
|
626
|
+
const ENTRY_TRAILING_MAX = 3;
|
|
627
|
+
const TAG_LINK_TEXT = /^[\[\]()*\s:_-]*(?:source\s+code|code|github|repo|source|src|project|paper|page|web|site|home|official|notebook|demo|data|docs|implementation|arxiv)\b[\[\]()*\s:_-]*$/i;
|
|
628
|
+
const URL_LINK_TEXT = /^https?:\/\/\S+$/i;
|
|
629
|
+
// A title that names nothing — the degenerate families every entry source can
|
|
630
|
+
// emit: a pure number/punctuation run (a table's rank or year column, "15."
|
|
631
|
+
// / "2023" / "2025-05" / "-"), a URL (a link whose label is the URL itself,
|
|
632
|
+
// scheme or scheme-less `github.com/…`), or a bare tag word ("GitHub",
|
|
633
|
+
// "Source code" — best-of's generated lines). CJK labels are letters
|
|
634
|
+
// (`\p{L}`) and stay titles.
|
|
635
|
+
const URL_TITLE = /^(?:[a-z][a-z0-9+.-]*:\/\/|www\.)\S+$|^github\.com\/\S+$/i;
|
|
636
|
+
function isDegenerateTitle(title) {
|
|
637
|
+
const trimmed = title.trim();
|
|
638
|
+
return (!!trimmed &&
|
|
639
|
+
(URL_TITLE.test(trimmed) ||
|
|
640
|
+
TAG_LINK_TEXT.test(trimmed) ||
|
|
641
|
+
!/[\p{L}]/u.test(trimmed)));
|
|
642
|
+
}
|
|
643
|
+
// A link label that can serve as a title. The degenerate shapes are excluded
|
|
644
|
+
// twice over — a URL label stays a URL, and a tag label names the link's role
|
|
645
|
+
// ("[Github](repo)"), not the repo.
|
|
646
|
+
function isMeaningfulLinkText(text) {
|
|
647
|
+
return text !== '' && !isDegenerateTitle(text);
|
|
648
|
+
}
|
|
649
|
+
// Dated entry lines end their tag cluster with a publication date ("4 Feb
|
|
650
|
+
// 2023", "19 Apr 2022", "2023") — a real corpus family (dated paper/tutorial
|
|
651
|
+
// lists), not a sentence continuation.
|
|
652
|
+
const DATE_TRAILING = /^(\d{1,2}[ -])?([a-zà-ÿ]{3,12}[ -])?\d{2,4}$/i;
|
|
653
|
+
// Boilerplate navigation line present in most mirror READMEs; casing and
|
|
654
|
+
// the optional "the" vary ("⬆ back to top", "⬆️ Back to Top",
|
|
655
|
+
// "Back to the Top").
|
|
656
|
+
const BACK_TO_TOP = /back to (?:the )?top/i;
|
|
657
|
+
// Tag-shaped leading label: empty, non-ASCII (CJK labels like 项目地址:,
|
|
658
|
+
// emoji markers), bracketed ([6] Diffusion:), or colon-terminated (Demo:).
|
|
659
|
+
// English prose ("Please see", "See also", "Inspired by the") is none of
|
|
660
|
+
// these — those are cross-reference sentences, not entry labels.
|
|
661
|
+
function isEntryLabel(label) {
|
|
662
|
+
return (label === '' ||
|
|
663
|
+
/[^\x00-\x7f]/.test(label) ||
|
|
664
|
+
label.startsWith('[') ||
|
|
665
|
+
/[::]$/.test(label));
|
|
666
|
+
}
|
|
667
|
+
// Flattened view of a paragraph's (or heading's) inline content for the entry
|
|
668
|
+
// test: the first link, the free text before it, and the free text after the
|
|
669
|
+
// link cluster (emphasis recursed into; link labels, images, and inline html
|
|
670
|
+
// carry no positional weight — a bare [Code] tag's label is not trailing
|
|
671
|
+
// prose).
|
|
672
|
+
function scanInlines(paragraph) {
|
|
673
|
+
let firstLink;
|
|
674
|
+
let label = '';
|
|
675
|
+
let trailing = '';
|
|
676
|
+
const walk = (node) => {
|
|
677
|
+
if (node.type === 'link') {
|
|
678
|
+
if (firstLink) {
|
|
679
|
+
trailing = '';
|
|
680
|
+
}
|
|
681
|
+
firstLink = firstLink ?? node;
|
|
682
|
+
return;
|
|
683
|
+
}
|
|
684
|
+
if (node.type === 'text') {
|
|
685
|
+
if (firstLink) {
|
|
686
|
+
trailing += node.value;
|
|
687
|
+
}
|
|
688
|
+
else {
|
|
689
|
+
label += node.value;
|
|
690
|
+
}
|
|
691
|
+
return;
|
|
692
|
+
}
|
|
693
|
+
for (const child of node.children ?? []) {
|
|
694
|
+
walk(child);
|
|
695
|
+
}
|
|
696
|
+
};
|
|
697
|
+
for (const child of paragraph.children) {
|
|
698
|
+
walk(child);
|
|
699
|
+
}
|
|
700
|
+
return { firstLink, label: label.trim(), trailing: trailing.trim() };
|
|
701
|
+
}
|
|
702
|
+
// True when nothing of substance follows the paragraph's last link: pure
|
|
703
|
+
// punctuation, or a date (see DATE_TRAILING).
|
|
704
|
+
function tagClusterEndsParagraph(trailing) {
|
|
705
|
+
if (trailing.length <= ENTRY_TRAILING_MAX) {
|
|
706
|
+
return true;
|
|
707
|
+
}
|
|
708
|
+
return DATE_TRAILING.test(trailing.replace(/^[\[\]()*\s:;,_-]+/, ''));
|
|
709
|
+
}
|
|
710
|
+
// The GitHub link that makes a top-level paragraph — or a heading inside a
|
|
711
|
+
// blockquote — an entry, if it is one (see the family comment above). Null
|
|
712
|
+
// for plain prose.
|
|
713
|
+
function paragraphEntryLink(paragraph) {
|
|
714
|
+
const githubLink = findFirstGitHubLink(paragraph);
|
|
715
|
+
if (!githubLink) {
|
|
716
|
+
return undefined;
|
|
717
|
+
}
|
|
718
|
+
const labelText = getInlineText(githubLink.children);
|
|
719
|
+
if (!labelText) {
|
|
720
|
+
return undefined;
|
|
721
|
+
}
|
|
722
|
+
const scan = scanInlines(paragraph);
|
|
723
|
+
if (scan.firstLink === githubLink &&
|
|
724
|
+
scan.label.length <= ENTRY_LABEL_MAX &&
|
|
725
|
+
isEntryLabel(scan.label)) {
|
|
726
|
+
return githubLink;
|
|
727
|
+
}
|
|
728
|
+
if (tagClusterEndsParagraph(scan.trailing) &&
|
|
729
|
+
(TAG_LINK_TEXT.test(labelText) ||
|
|
730
|
+
(URL_LINK_TEXT.test(labelText) && scan.firstLink !== githubLink))) {
|
|
731
|
+
return githubLink;
|
|
732
|
+
}
|
|
733
|
+
return undefined;
|
|
734
|
+
}
|
|
735
|
+
function blockquoteEntries(blockquote) {
|
|
736
|
+
const faces = [];
|
|
737
|
+
let sawParagraph = false;
|
|
738
|
+
for (const child of blockquote.children) {
|
|
739
|
+
if (child.type === 'paragraph') {
|
|
740
|
+
if (sawParagraph) {
|
|
741
|
+
continue;
|
|
742
|
+
}
|
|
743
|
+
sawParagraph = true;
|
|
744
|
+
const link = paragraphEntryLink(child);
|
|
745
|
+
if (link) {
|
|
746
|
+
faces.push({ inlines: child.children, link });
|
|
747
|
+
}
|
|
748
|
+
}
|
|
749
|
+
else if (child.type === 'heading') {
|
|
750
|
+
const link = paragraphEntryLink(child);
|
|
751
|
+
if (link) {
|
|
752
|
+
faces.push({ inlines: child.children, link });
|
|
753
|
+
}
|
|
754
|
+
}
|
|
755
|
+
}
|
|
756
|
+
return faces;
|
|
757
|
+
}
|
|
758
|
+
// The shared entry emission for the non-list sources (paragraph entries,
|
|
759
|
+
// blockquote cards): resolve, split, title-fallback — the same decisions the
|
|
760
|
+
// list and table paths make through the same helpers.
|
|
761
|
+
function entryNodesFor(ownLink, inlines, repoInfoMap) {
|
|
762
|
+
const repoInfo = repoInfoMap.get(ownLink.url) ?? null;
|
|
763
|
+
const entryText = splitEntryText(inlines);
|
|
764
|
+
return emitEntryNodes(ownLink.url, repoInfo, {
|
|
765
|
+
title: entryTitle(entryText.title, ownLink, repoInfo),
|
|
766
|
+
description: entryText.description,
|
|
767
|
+
}, []);
|
|
768
|
+
}
|
|
769
|
+
// The <details><summary>…</summary> collapsible-section idiom: the summary
|
|
770
|
+
// text delimits structure like a heading would (java's generated README,
|
|
771
|
+
// paper-list tables behind "1.1 <topic>" summaries). kbd chips inside the
|
|
772
|
+
// summary ("5 projects") are metadata, not the title. A details block with
|
|
773
|
+
// no summary, or an empty one, delimits nothing.
|
|
774
|
+
const DETAILS_SUMMARY = /<details[^>]*>[\s\S]*?<summary[^>]*>([\s\S]*?)<\/summary>/i;
|
|
775
|
+
const DETAILS_CLOSE = /^\s*<\/details>/i;
|
|
776
|
+
function detailsSummaryTitle(htmlValue) {
|
|
777
|
+
const match = DETAILS_SUMMARY.exec(htmlValue);
|
|
778
|
+
if (!match) {
|
|
779
|
+
return null;
|
|
780
|
+
}
|
|
781
|
+
const title = match[1]
|
|
782
|
+
.replace(/<kbd>[\s\S]*?<\/kbd>/gi, '')
|
|
783
|
+
.replace(/<[^>]+>/g, ' ')
|
|
784
|
+
.replace(/\s+/g, ' ')
|
|
785
|
+
.trim();
|
|
786
|
+
return title || null;
|
|
787
|
+
}
|
|
788
|
+
// The summary text of a <details><summary> block inside a list item, when the
|
|
789
|
+
// item has no paragraph text of its own.
|
|
790
|
+
function listItemSummaryTitle(itemNode) {
|
|
791
|
+
for (const child of itemNode.children) {
|
|
792
|
+
if (child.type === 'html') {
|
|
793
|
+
const title = detailsSummaryTitle(child.value);
|
|
794
|
+
if (title) {
|
|
795
|
+
return title;
|
|
796
|
+
}
|
|
797
|
+
}
|
|
798
|
+
}
|
|
799
|
+
return '';
|
|
800
|
+
}
|
|
355
801
|
const INVALID_TITLE_PATTERNS = [
|
|
356
802
|
/^contributing/i,
|
|
357
803
|
/^license$/i,
|
|
@@ -404,7 +850,7 @@ const TOC_TITLE_PATTERNS = [/^contents$/i, /^table of contents$/i];
|
|
|
404
850
|
// A heading that delimits content structure. Text-less headings (a bare `#`,
|
|
405
851
|
// an image-only heading) are spacers in real docs; TOC headings are structure
|
|
406
852
|
// mirrors. Neither participates in the section tree.
|
|
407
|
-
function isStructuralHeading(node) {
|
|
853
|
+
export function isStructuralHeading(node) {
|
|
408
854
|
const title = getNodeText(node);
|
|
409
855
|
return (!!title && !TOC_TITLE_PATTERNS.some(pattern => pattern.test(title.trim())));
|
|
410
856
|
}
|
|
@@ -425,6 +871,17 @@ function findTitleHeadingIndex(tree) {
|
|
|
425
871
|
node.depth === 1 &&
|
|
426
872
|
isValidTitle(getNodeText(node)));
|
|
427
873
|
}
|
|
874
|
+
// The heading branding owns even when it isn't a *valid* title: a generic
|
|
875
|
+
// first H1 ("# Guides", "# Contents") is still the de-facto title slot —
|
|
876
|
+
// applyBrandingToTree replaces it — so the section tree must not treat it
|
|
877
|
+
// as a section wrapping the whole document.
|
|
878
|
+
export function findTitleSlotIndex(tree) {
|
|
879
|
+
const titleHeadingIndex = findTitleHeadingIndex(tree);
|
|
880
|
+
if (titleHeadingIndex !== -1) {
|
|
881
|
+
return titleHeadingIndex;
|
|
882
|
+
}
|
|
883
|
+
return tree.children.findIndex((node) => node.type === 'heading' && node.depth === 1);
|
|
884
|
+
}
|
|
428
885
|
/**
|
|
429
886
|
* Makes the rendered H1 match metadata.title, replacing whichever heading
|
|
430
887
|
* occupies the title slot (a valid title H1, else the first generic H1, else a
|
|
@@ -461,13 +918,7 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepository) {
|
|
|
461
918
|
let documentTitle = titleHeadingIndex === -1
|
|
462
919
|
? ''
|
|
463
920
|
: getNodeText(tree.children[titleHeadingIndex]);
|
|
464
|
-
|
|
465
|
-
// first H1 ("# Guides", "# Contents") is still the de-facto title slot —
|
|
466
|
-
// applyBrandingToTree replaces it — so the section tree must not treat it
|
|
467
|
-
// as a section wrapping the whole document.
|
|
468
|
-
const titleSlotIndex = titleHeadingIndex !== -1
|
|
469
|
-
? titleHeadingIndex
|
|
470
|
-
: tree.children.findIndex((node) => node.type === 'heading' && node.depth === 1);
|
|
921
|
+
const titleSlotIndex = findTitleSlotIndex(tree);
|
|
471
922
|
// Derive a subject from the *source* repository name when no valid H1 is
|
|
472
923
|
// present. Using the source — not the enhanced/mirror repo — keeps the org
|
|
473
924
|
// name out of the title.
|
|
@@ -478,8 +929,34 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepository) {
|
|
|
478
929
|
}
|
|
479
930
|
}
|
|
480
931
|
const sectionDepth = findSectionDepth(tree, titleSlotIndex);
|
|
481
|
-
const
|
|
932
|
+
const gateForSection = sectionGatePasses(tree, titleSlotIndex, sectionDepth, repoInfoMap, sortOptions);
|
|
933
|
+
// Sections finalize out of document order when a details-section closes
|
|
934
|
+
// inside its parent section's span, so each carries the document index it
|
|
935
|
+
// opened at for the document-order sort at the end.
|
|
936
|
+
const sectionRecords = [];
|
|
482
937
|
const stack = [];
|
|
938
|
+
// Entry emission shared by the paragraph and blockquote-card faces: the
|
|
939
|
+
// implicit section opens for a containerless entry (the stack-bottom
|
|
940
|
+
// invariant), the minLinks gate is the section aggregate, and a gated or
|
|
941
|
+
// dead entry emits nothing and stays out of the description (mirrors dead
|
|
942
|
+
// list items).
|
|
943
|
+
const emitStandaloneEntry = (ownLink, inlines, atIndex) => {
|
|
944
|
+
if (stack.length === 0) {
|
|
945
|
+
openImplicitSection(stack, atIndex);
|
|
946
|
+
}
|
|
947
|
+
if (gateForSection(stack[0])) {
|
|
948
|
+
stack[stack.length - 1].children.push(...entryNodesFor(ownLink, inlines, repoInfoMap));
|
|
949
|
+
}
|
|
950
|
+
};
|
|
951
|
+
// A standalone entry restating the repo of an open item container (crypto's
|
|
952
|
+
// generated cards repeat the repo URL in a line under the link-heading) is
|
|
953
|
+
// that item's description, not a second item for the same repo. The
|
|
954
|
+
// repoInfoMap resolves both URLs through one per-repo memo, so identity
|
|
955
|
+
// comparison holds even for aliased spellings.
|
|
956
|
+
const rementionsOpenItem = (link) => {
|
|
957
|
+
const repoInfo = repoInfoMap.get(link.url);
|
|
958
|
+
return !!repoInfo && stack.some(container => container.repoInfo === repoInfo);
|
|
959
|
+
};
|
|
483
960
|
for (let i = 0; i < tree.children.length; i++) {
|
|
484
961
|
const node = tree.children[i];
|
|
485
962
|
if (node.type === 'heading') {
|
|
@@ -489,32 +966,117 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepository) {
|
|
|
489
966
|
if (i === titleSlotIndex || !isStructuralHeading(node)) {
|
|
490
967
|
continue;
|
|
491
968
|
}
|
|
492
|
-
|
|
493
|
-
|
|
969
|
+
const headingEntry = entryHeadingInfo(node, repoInfoMap);
|
|
970
|
+
// The promotion flavor of the entry test: a textless identity link is
|
|
971
|
+
// a title-line badge, not an entry (see entryHeadingInfo).
|
|
972
|
+
const promotedEntry = headingEntry && getInlineText(headingEntry.link.children)
|
|
973
|
+
? headingEntry
|
|
974
|
+
: null;
|
|
975
|
+
closeContainers(stack, node.depth, sectionRecords, !!promotedEntry);
|
|
976
|
+
openContainer(stack, node, i, sectionDepth, promotedEntry, headingEntry);
|
|
977
|
+
}
|
|
978
|
+
else if (node.type === 'paragraph') {
|
|
979
|
+
const text = getNodeText(node);
|
|
980
|
+
// Boilerplate "back to top" lines are neither entries nor description.
|
|
981
|
+
const ownLink = BACK_TO_TOP.test(text) ? undefined : paragraphEntryLink(node);
|
|
982
|
+
// An entry paragraph behaves like a one-item list; a failed gate, a
|
|
983
|
+
// dead target, or a re-mention of the enclosing item leaves plain
|
|
984
|
+
// description.
|
|
985
|
+
if (ownLink && !rementionsOpenItem(ownLink)) {
|
|
986
|
+
emitStandaloneEntry(ownLink, node.children, i);
|
|
987
|
+
}
|
|
988
|
+
else {
|
|
989
|
+
const container = stack[stack.length - 1];
|
|
990
|
+
// Avoid adding boilerplate "back to top" links to descriptions.
|
|
991
|
+
if (container && text && !BACK_TO_TOP.test(text)) {
|
|
992
|
+
container.description = container.description
|
|
993
|
+
? `${container.description}\n${text}`
|
|
994
|
+
: text;
|
|
995
|
+
}
|
|
996
|
+
}
|
|
494
997
|
}
|
|
495
|
-
else if (node.type === '
|
|
998
|
+
else if (node.type === 'blockquote') {
|
|
496
999
|
const text = getNodeText(node);
|
|
497
|
-
|
|
498
|
-
//
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
1000
|
+
// A card blockquote is the blockquote face of entries — one per
|
|
1001
|
+
// qualifying paragraph/heading child; any other quote is container
|
|
1002
|
+
// prose.
|
|
1003
|
+
const faces = BACK_TO_TOP.test(text)
|
|
1004
|
+
? []
|
|
1005
|
+
: blockquoteEntries(node).filter(face => !rementionsOpenItem(face.link));
|
|
1006
|
+
if (faces.length > 0) {
|
|
1007
|
+
for (const face of faces) {
|
|
1008
|
+
emitStandaloneEntry(face.link, face.inlines, i);
|
|
1009
|
+
}
|
|
1010
|
+
}
|
|
1011
|
+
else {
|
|
1012
|
+
const container = stack[stack.length - 1];
|
|
1013
|
+
// Avoid adding boilerplate "back to top" links to descriptions.
|
|
1014
|
+
if (container && text && !BACK_TO_TOP.test(text)) {
|
|
1015
|
+
container.description = container.description
|
|
1016
|
+
? `${container.description}\n${text}`
|
|
1017
|
+
: text;
|
|
1018
|
+
}
|
|
1019
|
+
}
|
|
1020
|
+
}
|
|
1021
|
+
else if (node.type === 'html') {
|
|
1022
|
+
// A details-summary block opens a section like a heading would; the
|
|
1023
|
+
// close tag (or the next summary) ends it — never the enclosing
|
|
1024
|
+
// section, which keeps collecting after the collapsible block.
|
|
1025
|
+
const summaryTitle = detailsSummaryTitle(node.value);
|
|
1026
|
+
if (summaryTitle) {
|
|
1027
|
+
closeInnermostDetails(stack, sectionRecords);
|
|
1028
|
+
// Container depths never decrease going up the stack. When the open
|
|
1029
|
+
// containers sit deeper than sectionDepth (a mid-document H1 defines
|
|
1030
|
+
// sectionDepth while the content sections run deeper), the
|
|
1031
|
+
// details-section joins at the current depth — pushing at the
|
|
1032
|
+
// shallower sectionDepth would invert the stack and strand the gate
|
|
1033
|
+
// (which reads the stack bottom) on a tiny outer section.
|
|
1034
|
+
const joinDepth = stack.length === 0
|
|
1035
|
+
? sectionDepth
|
|
1036
|
+
: Math.max(sectionDepth, stack[stack.length - 1].headingDepth);
|
|
1037
|
+
stack.push({
|
|
1038
|
+
children: [],
|
|
1039
|
+
description: '',
|
|
1040
|
+
headingDepth: joinDepth,
|
|
1041
|
+
headingIndex: i,
|
|
1042
|
+
kind: 'section',
|
|
1043
|
+
openedByDetails: true,
|
|
1044
|
+
title: summaryTitle,
|
|
1045
|
+
});
|
|
1046
|
+
}
|
|
1047
|
+
else if (DETAILS_CLOSE.test(node.value)) {
|
|
1048
|
+
closeInnermostDetails(stack, sectionRecords);
|
|
503
1049
|
}
|
|
504
1050
|
}
|
|
505
1051
|
else if (node.type === 'list') {
|
|
1052
|
+
// A list with no open container still means content: synthesize the
|
|
1053
|
+
// implicit section for it (see openImplicitSection). Afterwards the
|
|
1054
|
+
// stack bottom is always a section — implicit or real — which is the
|
|
1055
|
+
// invariant the gate below and closeContainers rely on.
|
|
1056
|
+
if (stack.length === 0) {
|
|
1057
|
+
openImplicitSection(stack, i);
|
|
1058
|
+
}
|
|
506
1059
|
// Every list inside the open container contributes items — a section is
|
|
507
|
-
// not closed by its first list
|
|
508
|
-
//
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
1060
|
+
// not closed by its first list — and the minLinks gate is decided per
|
|
1061
|
+
// section, against the whole section subtree.
|
|
1062
|
+
const items = processListRecursively(node, repoInfoMap, sortOptions, false, gateForSection(stack[0]));
|
|
1063
|
+
stack[stack.length - 1].children.push(...items);
|
|
1064
|
+
}
|
|
1065
|
+
else if (node.type === 'table') {
|
|
1066
|
+
// Tables hold entries the same way lists do — the implicit section opens
|
|
1067
|
+
// for one with no open container (the stack-bottom invariant), and the
|
|
1068
|
+
// minLinks gate is the same section aggregate.
|
|
1069
|
+
if (stack.length === 0) {
|
|
1070
|
+
openImplicitSection(stack, i);
|
|
514
1071
|
}
|
|
1072
|
+
const items = processTableRows(node, repoInfoMap, gateForSection(stack[0]));
|
|
1073
|
+
stack[stack.length - 1].children.push(...items);
|
|
515
1074
|
}
|
|
516
1075
|
}
|
|
517
|
-
closeContainers(stack, 0,
|
|
1076
|
+
closeContainers(stack, 0, sectionRecords);
|
|
1077
|
+
const sections = sectionRecords
|
|
1078
|
+
.sort((a, b) => a.headingIndex - b.headingIndex)
|
|
1079
|
+
.map(record => record.section);
|
|
518
1080
|
return { sections, title: documentTitle, titleHeadingIndex };
|
|
519
1081
|
}
|
|
520
1082
|
// The heading depth that opens top-level sections: the shallowest structural
|
|
@@ -534,87 +1096,308 @@ function findSectionDepth(tree, titleSlotIndex) {
|
|
|
534
1096
|
});
|
|
535
1097
|
return depth;
|
|
536
1098
|
}
|
|
537
|
-
|
|
538
|
-
|
|
539
|
-
|
|
540
|
-
// heading is not "the" resource), as does a dead target.
|
|
541
|
-
function soleLiveHeadingLink(heading, repoInfoMap) {
|
|
542
|
-
const links = heading.children.filter((child) => child.type === 'link');
|
|
543
|
-
if (links.length !== 1) {
|
|
1099
|
+
function entryHeadingInfo(heading, repoInfoMap) {
|
|
1100
|
+
const link = findFirstGitHubLink(heading);
|
|
1101
|
+
if (!link) {
|
|
544
1102
|
return null;
|
|
545
1103
|
}
|
|
546
|
-
|
|
1104
|
+
const repoInfo = repoInfoMap.get(link.url);
|
|
1105
|
+
return repoInfo ? { link, repoInfo } : null;
|
|
547
1106
|
}
|
|
548
|
-
function openContainer(stack, heading, sectionDepth,
|
|
1107
|
+
function openContainer(stack, heading, headingIndex, sectionDepth, promotedEntry, headingEntry) {
|
|
549
1108
|
const title = getNodeText(heading);
|
|
550
1109
|
// Sections sit at the section level — and any heading met with an empty
|
|
551
1110
|
// stack is promoted: a deeper heading before the first section (orphan
|
|
552
|
-
// subheading) still owns its subtree
|
|
553
|
-
//
|
|
554
|
-
//
|
|
1111
|
+
// subheading) still owns its subtree. A promoted ENTRY heading (see
|
|
1112
|
+
// entryHeadingInfo) is an item, not a section: the JSON contract has no
|
|
1113
|
+
// top-level items, so a synthesized section wraps the whole run of them.
|
|
555
1114
|
if (stack.length === 0 || heading.depth === sectionDepth) {
|
|
1115
|
+
if (promotedEntry) {
|
|
1116
|
+
if (stack.length === 0) {
|
|
1117
|
+
openSynthesizedSection(stack, headingIndex, sectionDepth);
|
|
1118
|
+
}
|
|
1119
|
+
stack.push({
|
|
1120
|
+
children: [],
|
|
1121
|
+
description: '',
|
|
1122
|
+
headingDepth: heading.depth,
|
|
1123
|
+
headingIndex,
|
|
1124
|
+
kind: 'item',
|
|
1125
|
+
repoInfo: promotedEntry.repoInfo,
|
|
1126
|
+
title,
|
|
1127
|
+
});
|
|
1128
|
+
return;
|
|
1129
|
+
}
|
|
556
1130
|
stack.push({
|
|
557
1131
|
children: [],
|
|
558
1132
|
description: '',
|
|
559
1133
|
headingDepth: heading.depth,
|
|
1134
|
+
headingIndex,
|
|
560
1135
|
kind: 'section',
|
|
561
1136
|
title,
|
|
562
1137
|
});
|
|
563
1138
|
return;
|
|
564
1139
|
}
|
|
565
|
-
|
|
1140
|
+
// A deeper entry heading opens an item container the same way a promoted
|
|
1141
|
+
// one does — children and prose below it collect as its content.
|
|
566
1142
|
stack.push({
|
|
567
1143
|
children: [],
|
|
568
1144
|
description: '',
|
|
569
1145
|
headingDepth: heading.depth,
|
|
570
|
-
|
|
571
|
-
|
|
1146
|
+
headingIndex,
|
|
1147
|
+
kind: headingEntry ? 'item' : 'group',
|
|
1148
|
+
repoInfo: headingEntry?.repoInfo,
|
|
572
1149
|
title,
|
|
573
1150
|
});
|
|
574
1151
|
}
|
|
575
|
-
//
|
|
576
|
-
//
|
|
577
|
-
//
|
|
578
|
-
//
|
|
579
|
-
//
|
|
580
|
-
//
|
|
581
|
-
//
|
|
582
|
-
function
|
|
583
|
-
|
|
584
|
-
|
|
585
|
-
|
|
586
|
-
|
|
587
|
-
|
|
1152
|
+
// A document can hold registry content with no heading above it — headingless
|
|
1153
|
+
// docs, TOC-only docs, or preamble entries (list, table) before the first
|
|
1154
|
+
// section heading. Rather than dropping them, the first one synthesizes the
|
|
1155
|
+
// section it implicitly belongs to ("Overview"). It behaves exactly like a
|
|
1156
|
+
// real section:
|
|
1157
|
+
// closed by the first structural heading (or document end), gated with its
|
|
1158
|
+
// whole subtree, and pruned when empty.
|
|
1159
|
+
function openImplicitSection(stack, atIndex) {
|
|
1160
|
+
stack.push({
|
|
1161
|
+
children: [],
|
|
1162
|
+
description: '',
|
|
1163
|
+
// Infinity so ANY structural heading closes it in closeContainers and
|
|
1164
|
+
// ends its span in the section gate — the implicit section never reaches
|
|
1165
|
+
// past the preamble.
|
|
1166
|
+
headingDepth: Infinity,
|
|
1167
|
+
// One before the opening list so the gate's scan (from headingIndex + 1)
|
|
1168
|
+
// counts that list itself.
|
|
1169
|
+
headingIndex: atIndex - 1,
|
|
1170
|
+
kind: 'section',
|
|
1171
|
+
title: 'Overview',
|
|
1172
|
+
});
|
|
1173
|
+
}
|
|
1174
|
+
// The section synthesized around a run of promoted entry headings
|
|
1175
|
+
// (openContainer's entry branch). It behaves like the implicit section — same
|
|
1176
|
+
// "Overview" title, gated by its subtree — but sits at sectionDepth, so the
|
|
1177
|
+
// first plain heading at that level ends the run, and closeContainers'
|
|
1178
|
+
// stopAtSynthesized keeps the entry headings' own pops from closing it:
|
|
1179
|
+
// consecutive entry headings are siblings INSIDE it.
|
|
1180
|
+
function openSynthesizedSection(stack, firstEntryIndex, depth) {
|
|
1181
|
+
stack.push({
|
|
1182
|
+
children: [],
|
|
1183
|
+
description: '',
|
|
1184
|
+
headingDepth: depth,
|
|
1185
|
+
// One before the first entry heading so the gate's scan (from
|
|
1186
|
+
// headingIndex + 1) counts that heading itself.
|
|
1187
|
+
headingIndex: firstEntryIndex - 1,
|
|
1188
|
+
kind: 'section',
|
|
1189
|
+
openedBySynthesis: true,
|
|
1190
|
+
title: 'Overview',
|
|
1191
|
+
});
|
|
1192
|
+
}
|
|
1193
|
+
// The minLinks gate scoped to a SECTION: every entry source in the section
|
|
1194
|
+
// counts together (list items, table rows, entry paragraphs, entry headings,
|
|
1195
|
+
// blockquote faces) — best-of-style documents put each entry in its own
|
|
1196
|
+
// single-item list, which the per-list gate dropped one by one. The section's
|
|
1197
|
+
// subtree runs from its heading to the first structural heading at or above
|
|
1198
|
+
// its depth (the same heading that would close it in closeContainers).
|
|
1199
|
+
//
|
|
1200
|
+
// Heading-per-entry documents (FBI-tools' `### name` + link paragraph, FLOSS's
|
|
1201
|
+
// `## game` + link cluster) put exactly ONE entry in each section, so every
|
|
1202
|
+
// section fails the gate alone and the whole document drops. They are not
|
|
1203
|
+
// sparse sections but flat entry lists delimited by headings: a section at
|
|
1204
|
+
// sectionDepth that fails alone is re-gated against its RUN — the maximal
|
|
1205
|
+
// sequence of adjacent section spans each holding at most one entry. The
|
|
1206
|
+
// noise floor survives: a single one-entry section between multi-entry
|
|
1207
|
+
// sections is a run of one and stays dropped.
|
|
1208
|
+
function sectionGatePasses(tree, titleSlotIndex, sectionDepth, repoInfoMap, sortOptions) {
|
|
1209
|
+
const cache = new Map();
|
|
1210
|
+
// Linked entries from scanStart to the first structural heading that would
|
|
1211
|
+
// close a container of closeDepth. An entry heading counts as an entry
|
|
1212
|
+
// wherever it sits inside the span — it opens an item, not a container —
|
|
1213
|
+
// except at the scanned section's own depth, where the walk closes that
|
|
1214
|
+
// section when the entry run starts (a same-depth entry heading is content
|
|
1215
|
+
// only for the synthesized wrapper, hence sameDepthEntriesAreContent).
|
|
1216
|
+
const scanCount = (scanStart, closeDepth, sameDepthEntriesAreContent) => {
|
|
1217
|
+
let linkedEntries = 0;
|
|
1218
|
+
for (let j = scanStart; j < tree.children.length; j++) {
|
|
1219
|
+
const node = tree.children[j];
|
|
1220
|
+
if (node.type === 'heading') {
|
|
1221
|
+
if (j === titleSlotIndex || !isStructuralHeading(node)) {
|
|
1222
|
+
continue;
|
|
1223
|
+
}
|
|
1224
|
+
if (entryHeadingInfo(node, repoInfoMap)) {
|
|
1225
|
+
if (sameDepthEntriesAreContent || node.depth > closeDepth) {
|
|
1226
|
+
linkedEntries += 1;
|
|
1227
|
+
}
|
|
1228
|
+
else {
|
|
1229
|
+
break;
|
|
1230
|
+
}
|
|
1231
|
+
continue;
|
|
1232
|
+
}
|
|
1233
|
+
if (node.depth <= closeDepth) {
|
|
1234
|
+
break;
|
|
1235
|
+
}
|
|
1236
|
+
continue;
|
|
1237
|
+
}
|
|
1238
|
+
if (node.type === 'list') {
|
|
1239
|
+
linkedEntries += countLinkedItems(node);
|
|
1240
|
+
}
|
|
1241
|
+
else if (node.type === 'table') {
|
|
1242
|
+
linkedEntries += countLinkedRows(node);
|
|
1243
|
+
}
|
|
1244
|
+
else if (node.type === 'paragraph' && paragraphEntryLink(node)) {
|
|
1245
|
+
linkedEntries += 1;
|
|
1246
|
+
}
|
|
1247
|
+
else if (node.type === 'blockquote') {
|
|
1248
|
+
linkedEntries += blockquoteEntries(node).length;
|
|
1249
|
+
}
|
|
588
1250
|
}
|
|
589
|
-
|
|
590
|
-
|
|
591
|
-
|
|
592
|
-
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
|
|
1251
|
+
return linkedEntries;
|
|
1252
|
+
};
|
|
1253
|
+
// A boundary's entry count for the run walk: the heading itself when it is
|
|
1254
|
+
// an entry heading, plus its span's entries (the span closes at the first
|
|
1255
|
+
// heading at or above the boundary's OWN depth, exactly where the walk
|
|
1256
|
+
// closes that section).
|
|
1257
|
+
const spanCountCache = new Map();
|
|
1258
|
+
const spanCount = (boundaryIndex) => {
|
|
1259
|
+
const cached = spanCountCache.get(boundaryIndex);
|
|
1260
|
+
if (cached !== undefined) {
|
|
1261
|
+
return cached;
|
|
597
1262
|
}
|
|
598
|
-
const
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
|
|
603
|
-
|
|
604
|
-
|
|
605
|
-
|
|
606
|
-
|
|
607
|
-
|
|
1263
|
+
const heading = tree.children[boundaryIndex];
|
|
1264
|
+
const count = (entryHeadingInfo(heading, repoInfoMap) ? 1 : 0) +
|
|
1265
|
+
scanCount(boundaryIndex + 1, heading.depth, false);
|
|
1266
|
+
spanCountCache.set(boundaryIndex, count);
|
|
1267
|
+
return count;
|
|
1268
|
+
};
|
|
1269
|
+
// The section boundaries in document order, by the walk's own promotion
|
|
1270
|
+
// rule (openContainer): a heading opens a section when it sits at
|
|
1271
|
+
// sectionDepth or arrives with nothing open — FLOSS-style documents have a
|
|
1272
|
+
// lone late H1 that pins sectionDepth at 1 while every `## game` heading
|
|
1273
|
+
// alternately closes its predecessor (stack empties) and is promoted. The
|
|
1274
|
+
// depth simulation mirrors closeContainers/openContainer exactly.
|
|
1275
|
+
let boundaries;
|
|
1276
|
+
const sectionBoundaries = () => {
|
|
1277
|
+
if (!boundaries) {
|
|
1278
|
+
const indices = [];
|
|
1279
|
+
const openDepths = [];
|
|
1280
|
+
tree.children.forEach((node, i) => {
|
|
1281
|
+
if (node.type !== 'heading' || i === titleSlotIndex) {
|
|
1282
|
+
return;
|
|
1283
|
+
}
|
|
1284
|
+
if (!isStructuralHeading(node)) {
|
|
1285
|
+
return;
|
|
1286
|
+
}
|
|
1287
|
+
while (openDepths.length > 0 &&
|
|
1288
|
+
openDepths[openDepths.length - 1] >= node.depth) {
|
|
1289
|
+
openDepths.pop();
|
|
1290
|
+
}
|
|
1291
|
+
if (openDepths.length === 0 || node.depth === sectionDepth) {
|
|
1292
|
+
indices.push(i);
|
|
1293
|
+
}
|
|
1294
|
+
openDepths.push(node.depth);
|
|
608
1295
|
});
|
|
1296
|
+
boundaries = indices;
|
|
609
1297
|
}
|
|
610
|
-
|
|
611
|
-
|
|
612
|
-
|
|
613
|
-
|
|
614
|
-
|
|
615
|
-
|
|
616
|
-
|
|
1298
|
+
return boundaries;
|
|
1299
|
+
};
|
|
1300
|
+
const runTotal = (section) => {
|
|
1301
|
+
const list = sectionBoundaries();
|
|
1302
|
+
const k = list.indexOf(section.headingIndex);
|
|
1303
|
+
if (k === -1) {
|
|
1304
|
+
return spanCount(section.headingIndex);
|
|
1305
|
+
}
|
|
1306
|
+
let total = spanCount(section.headingIndex);
|
|
1307
|
+
for (const direction of [-1, 1]) {
|
|
1308
|
+
for (let m = k + direction; m >= 0 && m < list.length; m += direction) {
|
|
1309
|
+
const count = spanCount(list[m]);
|
|
1310
|
+
if (count > 1) {
|
|
1311
|
+
break;
|
|
1312
|
+
}
|
|
1313
|
+
total += count;
|
|
1314
|
+
}
|
|
1315
|
+
}
|
|
1316
|
+
return total;
|
|
1317
|
+
};
|
|
1318
|
+
return section => {
|
|
1319
|
+
const cached = cache.get(section.headingIndex);
|
|
1320
|
+
if (cached !== undefined) {
|
|
1321
|
+
return cached;
|
|
1322
|
+
}
|
|
1323
|
+
const own = scanCount(section.headingIndex + 1, section.headingDepth, !!section.openedBySynthesis);
|
|
1324
|
+
const passes = own >= sortOptions.minLinks ||
|
|
1325
|
+
(!section.openedBySynthesis &&
|
|
1326
|
+
sectionBoundaries().includes(section.headingIndex) &&
|
|
1327
|
+
runTotal(section) >= sortOptions.minLinks);
|
|
1328
|
+
cache.set(section.headingIndex, passes);
|
|
1329
|
+
return passes;
|
|
1330
|
+
};
|
|
1331
|
+
}
|
|
1332
|
+
// Prune-or-emit one popped container: a section/group whose children array is
|
|
1333
|
+
// empty (no items anywhere beneath — the entry sources only return
|
|
1334
|
+
// item-bearing nodes) is dropped; an item always survives, it IS the content.
|
|
1335
|
+
// Non-section containers always have an open parent (stack invariant), and
|
|
1336
|
+
// kind === 'item' exactly when repoInfo is set.
|
|
1337
|
+
function finalizeContainer(container, stack, sectionRecords) {
|
|
1338
|
+
if (container.children.length === 0 && container.kind !== 'item') {
|
|
1339
|
+
return;
|
|
1340
|
+
}
|
|
1341
|
+
const description = container.description || null;
|
|
1342
|
+
if (container.kind === 'section') {
|
|
1343
|
+
sectionRecords.push({
|
|
1344
|
+
headingIndex: container.headingIndex,
|
|
1345
|
+
section: { description, items: container.children, title: container.title },
|
|
1346
|
+
});
|
|
1347
|
+
return;
|
|
1348
|
+
}
|
|
1349
|
+
const parent = stack[stack.length - 1];
|
|
1350
|
+
if (container.repoInfo) {
|
|
1351
|
+
parent.children.push({
|
|
1352
|
+
children: container.children,
|
|
1353
|
+
description,
|
|
1354
|
+
node_type: 'item',
|
|
1355
|
+
repo_info: toRepoInfo(container.repoInfo),
|
|
1356
|
+
title: container.title,
|
|
1357
|
+
});
|
|
1358
|
+
}
|
|
1359
|
+
else {
|
|
1360
|
+
parent.children.push({
|
|
1361
|
+
children: container.children,
|
|
1362
|
+
description,
|
|
1363
|
+
node_type: 'group',
|
|
1364
|
+
title: container.title,
|
|
1365
|
+
});
|
|
1366
|
+
}
|
|
1367
|
+
}
|
|
1368
|
+
// Finalize every container a heading of `depth` closes (same-or-shallower),
|
|
1369
|
+
// bottom-up so each finalized node lands in its parent. The stack bottom is
|
|
1370
|
+
// always a section (openContainer's promotion guarantees it), so a finalized
|
|
1371
|
+
// group/item always has a parent to land in. stopAtSynthesized: the pop an
|
|
1372
|
+
// entry heading triggers must stop at the synthesized section wrapping its
|
|
1373
|
+
// run — the next entry heading of the run lands back inside it.
|
|
1374
|
+
function closeContainers(stack, depth, sectionRecords, stopAtSynthesized = false) {
|
|
1375
|
+
while (stack.length > 0 &&
|
|
1376
|
+
stack[stack.length - 1].headingDepth >= depth) {
|
|
1377
|
+
if (stopAtSynthesized && stack[stack.length - 1].openedBySynthesis) {
|
|
1378
|
+
break;
|
|
617
1379
|
}
|
|
1380
|
+
finalizeContainer(stack.pop(), stack, sectionRecords);
|
|
1381
|
+
}
|
|
1382
|
+
}
|
|
1383
|
+
// Ends the innermost open details-section and everything opened inside it,
|
|
1384
|
+
// leaving the enclosing containers untouched — unlike a heading close, a
|
|
1385
|
+
// details boundary never ends its parent section, so content after the
|
|
1386
|
+
// collapsible block keeps collecting under it. A stray </details> (no
|
|
1387
|
+
// details-section open) is a no-op.
|
|
1388
|
+
function closeInnermostDetails(stack, sectionRecords) {
|
|
1389
|
+
let detailsIndex = -1;
|
|
1390
|
+
for (let s = stack.length - 1; s >= 0; s--) {
|
|
1391
|
+
if (stack[s].openedByDetails) {
|
|
1392
|
+
detailsIndex = s;
|
|
1393
|
+
break;
|
|
1394
|
+
}
|
|
1395
|
+
}
|
|
1396
|
+
if (detailsIndex === -1) {
|
|
1397
|
+
return;
|
|
1398
|
+
}
|
|
1399
|
+
while (stack.length > detailsIndex) {
|
|
1400
|
+
finalizeContainer(stack.pop(), stack, sectionRecords);
|
|
618
1401
|
}
|
|
619
1402
|
}
|
|
620
1403
|
function serializeAst(tree, originalContent) {
|