@enhansome/core 1.8.0 → 1.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/github.d.ts CHANGED
@@ -76,6 +76,12 @@ export interface MakeOctokitOptions {
76
76
  */
77
77
  export declare function makeOctokit(token: string, { log, maxRetries, maxWaitSeconds, throttle, }?: MakeOctokitOptions): GithubClient;
78
78
  export declare function getRepoInfo(octokit: GithubClient, owner: string, repo: string): Promise<RepoInfoDetails>;
79
+ /**
80
+ * The repo's numeric GitHub id, for the JSON metadata — consumers key stable
81
+ * node ids on it. Returns null (logged) instead of failing the run: the id is
82
+ * metadata, not a gate; consumers fall back when it is missing.
83
+ */
84
+ export declare function getRepoId(octokit: GithubClient, owner: string, repo: string): Promise<null | number>;
79
85
  export declare function getReadme(octokit: GithubClient, owner: string, repo: string, format?: 'html' | 'raw'): Promise<string>;
80
86
  /** Root file/directory names — the compile-manifest gate reads them to tell a
81
87
  * directory of resources from a repo that IS the deliverable. A `path: ''`
package/dist/github.js CHANGED
@@ -82,6 +82,20 @@ export async function getRepoInfo(octokit, owner, repo) {
82
82
  topics: data.topics ?? [],
83
83
  };
84
84
  }
85
+ /**
86
+ * The repo's numeric GitHub id, for the JSON metadata — consumers key stable
87
+ * node ids on it. Returns null (logged) instead of failing the run: the id is
88
+ * metadata, not a gate; consumers fall back when it is missing.
89
+ */
90
+ export async function getRepoId(octokit, owner, repo) {
91
+ try {
92
+ return (await getRepoInfo(octokit, owner, repo)).id;
93
+ }
94
+ catch (error) {
95
+ octokit.log.error(`Failed to fetch repo id for ${owner}/${repo}: ${formatRequestError(error)}`);
96
+ return null;
97
+ }
98
+ }
85
99
  export async function getReadme(octokit, owner, repo, format = 'raw') {
86
100
  octokit.log.debug(`Fetching ${format} README for ${owner}/${repo}`);
87
101
  const response = await octokit.rest.repos.getReadme({
package/dist/index.d.ts CHANGED
@@ -1,4 +1,4 @@
1
- export { formatRequestError, getLatestCommitSha, getReadme, getRepoInfo, getRootEntryNames, makeOctokit, parseGitHubUrl, parseOwnerRepo, } from './github.js';
1
+ export { formatRequestError, getLatestCommitSha, getReadme, getRepoId, getRepoInfo, getRootEntryNames, makeOctokit, parseGitHubUrl, parseOwnerRepo, } from './github.js';
2
2
  export type { GithubClient, MakeOctokitOptions, RepoIdentifier, RepoInfoDetails, ThrottleOptions, } from './github.js';
3
3
  export type { Logger } from './logger.js';
4
4
  export { consoleLog, silentLog } from './logger.js';
package/dist/index.js CHANGED
@@ -1,4 +1,4 @@
1
- export { formatRequestError, getLatestCommitSha, getReadme, getRepoInfo, getRootEntryNames, makeOctokit, parseGitHubUrl, parseOwnerRepo, } from './github.js';
1
+ export { formatRequestError, getLatestCommitSha, getReadme, getRepoId, getRepoInfo, getRootEntryNames, makeOctokit, parseGitHubUrl, parseOwnerRepo, } from './github.js';
2
2
  export { consoleLog, silentLog } from './logger.js';
3
3
  export { toRepoInfo } from './markdown.js';
4
4
  export { enhance } from './orchestrator.js';
@@ -1,5 +1,6 @@
1
1
  import { RepoInfoDetails } from './github.js';
2
2
  import { Logger } from './logger.js';
3
+ import type { Heading, Link, List, Parent, Root } from 'mdast';
3
4
  export interface JsonOutput {
4
5
  items: JsonSection[];
5
6
  metadata: JsonMetadata;
@@ -45,6 +46,7 @@ export interface JsonMetadata {
45
46
  enhanced_repository_description: null | string;
46
47
  last_updated: string;
47
48
  original_repository: string;
49
+ original_repository_id: null | number;
48
50
  original_repository_sha: null | string;
49
51
  title: string;
50
52
  }
@@ -53,7 +55,12 @@ export interface JsonSection {
53
55
  items: JsonNode[];
54
56
  title: string;
55
57
  }
56
- export declare function processMarkdownContent(originalContent: string, token: string, replacements: ReplacementRule[] | undefined, sortOptions: SortOptions | undefined, originalRepository: string, relativeLinkPrefix?: string, enhancedRepository?: string, enhancedRepositoryDescription?: string, originalRepositorySha?: string, now?: Date, log?: Logger): Promise<{
58
+ export declare function processMarkdownContent(originalContent: string, token: string, replacements: ReplacementRule[] | undefined, sortOptions: SortOptions | undefined, originalRepository: string, relativeLinkPrefix?: string, enhancedRepository?: string, enhancedRepositoryDescription?: string, originalRepositorySha?: string, originalRepositoryId?: number, now?: Date, log?: Logger): Promise<{
57
59
  finalContent: string;
58
60
  jsonData: JsonOutput;
59
61
  }>;
62
+ export declare function normalizeGitHubUrls(tree: Root): void;
63
+ export declare function findFirstGitHubLink(node: Parent): Link | undefined;
64
+ export declare function countLinkedItems(listNode: List): number;
65
+ export declare function isStructuralHeading(node: Heading): boolean;
66
+ export declare function findTitleSlotIndex(tree: Root): number;
package/dist/markdown.js CHANGED
@@ -79,12 +79,13 @@ async function fetchTargetData(urls, repos) {
79
79
  log.info(`Target fetch: ${repoInfoMap.size}/${urls.size} repo-info ok in ${Date.now() - fetchStart}ms (concurrency ${FETCH_CONCURRENCY}).`);
80
80
  return repoInfoMap;
81
81
  }
82
- export async function processMarkdownContent(originalContent, token, replacements = [], sortOptions = { by: '', minLinks: 2 }, originalRepository, relativeLinkPrefix = '', enhancedRepository, enhancedRepositoryDescription, originalRepositorySha, now = new Date(), log = consoleLog) {
82
+ export async function processMarkdownContent(originalContent, token, replacements = [], sortOptions = { by: '', minLinks: 2 }, originalRepository, relativeLinkPrefix = '', enhancedRepository, enhancedRepositoryDescription, originalRepositorySha, originalRepositoryId, now = new Date(), log = consoleLog) {
83
83
  const repos = createRepoInfoLookup(token, log);
84
84
  const brandingEnabled = replacements.some(rule => rule.type === 'branding');
85
85
  const contentAfterReplacements = applyTextReplacements(originalContent, replacements.filter(rule => rule.type !== 'branding'), log);
86
86
  const processor = unified().use(remarkParse).use(remarkGfm);
87
87
  const tree = processor.parse(contentAfterReplacements);
88
+ normalizeGitHubUrls(tree);
88
89
  const githubUrls = collectGitHubLinks(tree);
89
90
  const repoInfoMap = await fetchTargetData(githubUrls, repos);
90
91
  // The title derives from the *source* repository (originalRepository), never
@@ -99,6 +100,7 @@ export async function processMarkdownContent(originalContent, token, replacement
99
100
  const metadata = {
100
101
  last_updated: now.toISOString(),
101
102
  original_repository: originalRepository.trim(),
103
+ original_repository_id: originalRepositoryId ?? null,
102
104
  original_repository_sha: (originalRepositorySha?.trim() ?? '') || null,
103
105
  enhanced_repository: (enhancedRepository?.trim() ?? '') || null,
104
106
  enhanced_repository_description: (enhancedRepositoryDescription?.trim() ?? '') || null,
@@ -175,6 +177,112 @@ function collectGitHubLinks(tree) {
175
177
  });
176
178
  return urls;
177
179
  }
180
+ // Input normalization, same family as fixRelativeLinks: make GitHub repos a
181
+ // source expresses WITHOUT markdown links visible as real link nodes, so
182
+ // every downstream consumer — repo fetch, entry tests, gates, badges — sees
183
+ // the shape a markdown link would have produced. Two families:
184
+ //
185
+ // (1) Bare scheme-less `github.com/owner/repo` text — GFM autolinks only
186
+ // scheme-full URLs, so these stay plain text. The link's label is the
187
+ // URL text; only owner/repo are consumed (a deeper path like `/tree/main`
188
+ // stays in the trailing text — the identity reads the first two segments
189
+ // either way).
190
+ // (2) Inline `<a href="…">…</a>` anchors — remark emits the open and close
191
+ // tags as separate html nodes around the label's inline nodes, so the
192
+ // pair is rewrapped as one link. Anchors inside BLOCK html (`<details>`
193
+ // summaries, centered banners) keep their raw form: that html is the
194
+ // structure the walk already reads (detailsSummaryTitle), and rewriting
195
+ // it is a different decision than link visibility.
196
+ //
197
+ // Text inside code spans/blocks never linkifies (code has no text nodes) and
198
+ // text inside an existing link label is skipped — nested links are not a
199
+ // thing. Non-GitHub and org-only anchors stay raw html.
200
+ const BARE_GITHUB_URL = /(?:^|(?<=[\s(>\[]))(?:www\.)?github\.com\/[A-Za-z0-9_.-]+\/[A-Za-z0-9_.-]+/g;
201
+ const ANCHOR_OPEN = /^<a\s[^>]*href=(["'])([^"']*)\1[^>]*>$/;
202
+ const ANCHOR_CLOSE = /^<\/a\s*>$/;
203
+ export function normalizeGitHubUrls(tree) {
204
+ const walk = (node, inLink) => {
205
+ if (inLink) {
206
+ return;
207
+ }
208
+ const parent = node;
209
+ if (!Array.isArray(parent.children)) {
210
+ return;
211
+ }
212
+ for (const child of parent.children) {
213
+ walk(child, child.type === 'link');
214
+ }
215
+ parent.children = normalizeInlineChildren(parent.children);
216
+ };
217
+ walk(tree, false);
218
+ }
219
+ // One parent's inline run: bare-URL text nodes split into text/link/text,
220
+ // anchor open+close pairs rewrapped as a link. The anchor's inner nodes are
221
+ // taken verbatim — they are the label, and splitting them would nest links.
222
+ function normalizeInlineChildren(children) {
223
+ const out = [];
224
+ for (let i = 0; i < children.length; i++) {
225
+ const child = children[i];
226
+ if (child.type === 'html') {
227
+ const anchor = ANCHOR_OPEN.exec(child.value.trim());
228
+ const url = anchor && parseGitHubUrl(anchor[2]) ? anchor[2] : null;
229
+ if (url) {
230
+ const closeIndex = children.findIndex((candidate, j) => j > i &&
231
+ candidate.type === 'html' &&
232
+ ANCHOR_CLOSE.test(candidate.value.trim()));
233
+ if (closeIndex !== -1) {
234
+ out.push({
235
+ type: 'link',
236
+ url,
237
+ children: children.slice(i + 1, closeIndex),
238
+ });
239
+ i = closeIndex;
240
+ continue;
241
+ }
242
+ }
243
+ out.push(child);
244
+ continue;
245
+ }
246
+ if (child.type === 'text') {
247
+ out.push(...linkifyBareUrls(child));
248
+ continue;
249
+ }
250
+ out.push(child);
251
+ }
252
+ return out;
253
+ }
254
+ function linkifyBareUrls(node) {
255
+ const matches = [...node.value.matchAll(BARE_GITHUB_URL)];
256
+ if (matches.length === 0) {
257
+ return [node];
258
+ }
259
+ const out = [];
260
+ let cursor = 0;
261
+ for (const match of matches) {
262
+ const label = match[0].replace(/\.+$/, '');
263
+ const url = `https://${label.replace(/^www\./, '')}`;
264
+ if (!parseGitHubUrl(url)) {
265
+ continue;
266
+ }
267
+ const start = match.index;
268
+ if (start > cursor) {
269
+ out.push({
270
+ type: 'text',
271
+ value: node.value.slice(cursor, start),
272
+ });
273
+ }
274
+ out.push({
275
+ type: 'link',
276
+ url,
277
+ children: [{ type: 'text', value: label }],
278
+ });
279
+ cursor = start + label.length;
280
+ }
281
+ if (cursor < node.value.length) {
282
+ out.push({ type: 'text', value: node.value.slice(cursor) });
283
+ }
284
+ return out;
285
+ }
178
286
  function createBadgeText(info) {
179
287
  if (info.archived) {
180
288
  return ' ⚠️ Archived';
@@ -191,27 +299,41 @@ function createBadgeText(info) {
191
299
  }
192
300
  return ` ${parts.join(' | ')}`;
193
301
  }
194
- function findFirstGitHubLink(node) {
195
- let linkUrl;
196
- visit(node, 'link', (linkNode) => {
197
- if (!linkUrl && parseGitHubUrl(linkNode.url)) {
198
- linkUrl = linkNode.url;
302
+ // The first link in a subtree that points at a GitHub repo. Returns the link
303
+ // node itself so callers can take both its URL (identity) and its text
304
+ // (title fallbacks).
305
+ export function findFirstGitHubLink(node) {
306
+ let linkNode;
307
+ visit(node, 'link', (candidate) => {
308
+ if (!linkNode && parseGitHubUrl(candidate.url)) {
309
+ linkNode = candidate;
199
310
  }
200
311
  });
201
- return linkUrl;
312
+ return linkNode;
202
313
  }
203
314
  // The GitHub link that represents a list item: the FIRST GitHub link in the
204
- // item's OWN paragraph only. A nested-descendant link is deliberately ignored
205
- // — it belongs to a child, not to this item. An item is a GitHub node iff its
206
- // own paragraph links to a GitHub repo; an item with no own GitHub link but
315
+ // item's OWN paragraphs, in document order. Paper-list entries carry their
316
+ // identity link ([[Code]](github)) in a paragraph after the title one, so
317
+ // every own paragraph is scanned — the title still comes from the first
318
+ // paragraph alone. A nested-descendant link is deliberately ignored — it
319
+ // belongs to a child, not to this item. An item is a GitHub node iff its own
320
+ // paragraphs link to a GitHub repo; an item with no own GitHub link but
207
321
  // nested GitHub children is a group, not a node borrowing a child's identity.
208
322
  //
209
- // A GitHub link that is secondary within the paragraph (e.g.
323
+ // A GitHub link that is secondary within a paragraph (e.g.
210
324
  // `[name](marketplace) … [On GitHub](github)`) is still the item's own link, so
211
325
  // `findFirstGitHubLink` over the paragraph finds it correctly.
212
326
  function findOwnGitHubLink(itemNode) {
213
- const paragraph = itemNode.children.find((child) => child.type === 'paragraph');
214
- return paragraph ? findFirstGitHubLink(paragraph) : undefined;
327
+ for (const child of itemNode.children) {
328
+ if (child.type !== 'paragraph') {
329
+ continue;
330
+ }
331
+ const link = findFirstGitHubLink(child);
332
+ if (link) {
333
+ return link;
334
+ }
335
+ }
336
+ return undefined;
215
337
  }
216
338
  function fixRelativeLinks(tree, relativeLinkPrefix) {
217
339
  if (!relativeLinkPrefix) {
@@ -273,15 +395,89 @@ function compareByRepoInfo(by, a, b) {
273
395
  const timeB = b.pushed_at ? new Date(b.pushed_at).getTime() : 0;
274
396
  return timeB - timeA;
275
397
  }
276
- function processListRecursively(listNode, repoInfoMap, sortOptions, isNested = false) {
277
- // The section-sparsity gate counts items whose SUBTREE contains a GitHub
278
- // link, not items with an own-paragraph link only. A category with no own
279
- // link but nested GitHub children still counts (it becomes a group);
280
- // switching to `findOwnGitHubLink` would silently drop purely categorical
281
- // sections, so it deliberately diverges from the identity resolver below.
282
- const itemsWithGitHubLinks = listNode.children.filter(item => !!findFirstGitHubLink(item));
283
- if (!isNested && itemsWithGitHubLinks.length < sortOptions.minLinks) {
284
- return [];
398
+ // The minLinks noise gate counts items whose SUBTREE contains a GitHub link,
399
+ // not items with an own-paragraph link only. A category with no own link but
400
+ // nested GitHub children still counts (it becomes a group); switching to
401
+ // `findOwnGitHubLink` would silently drop purely categorical sections, so it
402
+ // deliberately diverges from the identity resolver.
403
+ export function countLinkedItems(listNode) {
404
+ return listNode.children.filter(item => !!findFirstGitHubLink(item)).length;
405
+ }
406
+ // The table counterpart of countLinkedItems for the section gate: rows whose
407
+ // subtree contains a GitHub link. No header-row skip — a card grid's first row
408
+ // sits above the delimiter and is content, while pure-label header rows carry
409
+ // no links and count zero either way.
410
+ function countLinkedRows(tableNode) {
411
+ return tableNode.children.filter(row => row.type === 'tableRow' && !!findFirstGitHubLink(row)).length;
412
+ }
413
+ // One entry's title and description, split from its own inlines: leading text
414
+ // up to and including the first link is the title (prefix tags like
415
+ // "[UPDATED]" survive), the prose trailing it the description. The link is
416
+ // found through emphasis wrappers — a bolded card link
417
+ // (`**[Name](repo)** - description`) splits like a plain one. With no link
418
+ // the whole text is the title — the description must never echo the title
419
+ // back.
420
+ function splitEntryText(inlines) {
421
+ const linkIndex = inlines.findIndex(child => {
422
+ let hasLink = false;
423
+ visit(child, 'link', () => {
424
+ hasLink = true;
425
+ });
426
+ return hasLink;
427
+ });
428
+ if (linkIndex === -1) {
429
+ return { title: getInlineText(inlines), description: '' };
430
+ }
431
+ return {
432
+ title: getInlineText(inlines.slice(0, linkIndex + 1)),
433
+ description: stripLeadingNoise(getInlineText(inlines.slice(linkIndex + 1))),
434
+ };
435
+ }
436
+ // The emission decision every entry source shares — list items, table rows,
437
+ // and the paragraph/blockquote sources to come: an own GitHub link that
438
+ // resolved makes an item; a dead own link drops the entry and lifts its
439
+ // children to the nearest live parent; no own link with children makes a
440
+ // group — never a `repo_info`, that's the identity-borrowing bug; no own link
441
+ // and no children is a non-GitHub leaf, kept in markdown, dropped from JSON.
442
+ // TODO(future): preserve non-GitHub leaves in a separate shape.
443
+ function emitEntryNodes(githubUrl, repoInfo, text, childrenJson) {
444
+ if (githubUrl && repoInfo) {
445
+ return [
446
+ {
447
+ node_type: 'item',
448
+ title: text.title,
449
+ description: text.description || null,
450
+ children: childrenJson,
451
+ repo_info: toRepoInfo(repoInfo),
452
+ },
453
+ ];
454
+ }
455
+ if (githubUrl) {
456
+ return childrenJson;
457
+ }
458
+ if (childrenJson.length > 0) {
459
+ return [
460
+ {
461
+ node_type: 'group',
462
+ title: text.title,
463
+ description: text.description || null,
464
+ children: childrenJson,
465
+ },
466
+ ];
467
+ }
468
+ return [];
469
+ }
470
+ function processListRecursively(listNode, repoInfoMap, sortOptions, isNested = false,
471
+ // The caller's section-scope gate decision (sectionGatePasses). Absent for
472
+ // nested lists (emitted under their parent regardless) and for top-level
473
+ // lists with no open container (preamble), which gate per list.
474
+ sectionGateOpen) {
475
+ if (!isNested) {
476
+ const gateOpen = sectionGateOpen ??
477
+ countLinkedItems(listNode) >= sortOptions.minLinks;
478
+ if (!gateOpen) {
479
+ return [];
480
+ }
285
481
  }
286
482
  // Zip each item with its emitted JSON nodes and repo info so one sort orders
287
483
  // both the rendered AST and the emitted JSON. `emitted` is empty for
@@ -289,59 +485,31 @@ function processListRecursively(listNode, repoInfoMap, sortOptions, isNested = f
289
485
  // AST, dropped from JSON.
290
486
  const entries = [];
291
487
  for (const itemNode of listNode.children) {
292
- const githubUrl = findOwnGitHubLink(itemNode);
488
+ const ownLink = findOwnGitHubLink(itemNode);
489
+ const githubUrl = ownLink?.url;
293
490
  const repoInfo = githubUrl ? (repoInfoMap.get(githubUrl) ?? null) : null;
294
- const nestedLists = itemNode.children.filter((child) => child.type === 'list');
295
- const childrenJson = nestedLists.flatMap(nestedList => processListRecursively(nestedList, repoInfoMap, sortOptions, true));
296
- let title = '';
297
- let description = '';
298
- const paragraph = itemNode.children.find(p => p.type === 'paragraph');
299
- if (paragraph) {
300
- const linkIndex = paragraph.children.findIndex((c) => c.type === 'link');
301
- if (linkIndex !== -1) {
302
- // Title = leading text + the first link's text, so prefix tags like
303
- // "[UPDATED]" survive; description is the prose trailing the link.
304
- title = getInlineText(paragraph.children.slice(0, linkIndex + 1));
305
- description = stripLeadingNoise(getInlineText(paragraph.children.slice(linkIndex + 1)));
306
- }
307
- else {
308
- // No link: the whole paragraph is the title, so leave description empty
309
- // (avoid echoing the title back).
310
- title = getInlineText(paragraph.children);
311
- description = '';
312
- }
313
- }
314
- // No-own-link items with children become groups — NEVER give them a
315
- // `repo_info`, that's the identity-borrowing bug. No-own-link, no-child
316
- // items are non-GitHub leaves: kept in markdown, dropped from JSON.
317
- // TODO(future): preserve non-GitHub leaves in a separate shape.
318
- let emitted = [];
319
- if (githubUrl && repoInfo) {
320
- emitted = [
321
- {
322
- node_type: 'item',
323
- title,
324
- description: description || null,
325
- children: childrenJson,
326
- repo_info: toRepoInfo(repoInfo),
327
- },
328
- ];
329
- }
330
- else if (githubUrl) {
331
- // Dead target: the item itself is not emitted; its children lift to this
332
- // list's level — the nearest live parent.
333
- emitted = childrenJson;
334
- }
335
- else if (childrenJson.length > 0) {
336
- emitted = [
337
- {
338
- node_type: 'group',
339
- title,
340
- description: description || null,
341
- children: childrenJson,
342
- },
343
- ];
491
+ // Nested content is the item's children: deeper lists as before, plus
492
+ // tables — the AnimeResearch shape wraps a <details><summary> block and
493
+ // its table inside one list item. Both emit under the parent item
494
+ // regardless of the gate: the top-level call already gated the section.
495
+ const nestedContent = itemNode.children.filter((child) => child.type === 'list' || child.type === 'table');
496
+ const childrenJson = nestedContent.flatMap(child => child.type === 'list'
497
+ ? processListRecursively(child, repoInfoMap, sortOptions, true)
498
+ : processTableRows(child, repoInfoMap, true));
499
+ // Title/description split on the FIRST paragraph only — a paper-list
500
+ // entry's identity link may live in a later paragraph (findOwnGitHubLink)
501
+ // while its title text stays the leading one.
502
+ const paragraph = itemNode.children.find((child) => child.type === 'paragraph');
503
+ const entryText = splitEntryText(paragraph?.children ?? []);
504
+ // A details-wrapped item carries its visible text in the summary, not in
505
+ // a paragraph — that text is the group's title.
506
+ if (!entryText.title) {
507
+ entryText.title = listItemSummaryTitle(itemNode);
344
508
  }
509
+ // The shared title fallbacks (an inline-code link label carries no text
510
+ // nodes, so the split alone can leave an empty title).
511
+ entryText.title = entryTitle(entryText.title, ownLink, repoInfo);
512
+ const emitted = emitEntryNodes(githubUrl, repoInfo, entryText, childrenJson);
345
513
  entries.push({ emitted, node: itemNode, repoInfo });
346
514
  }
347
515
  if (sortOptions.by) {
@@ -351,6 +519,244 @@ function processListRecursively(listNode, repoInfoMap, sortOptions, isNested = f
351
519
  listNode.children = entries.map(entry => entry.node);
352
520
  return entries.flatMap(entry => entry.emitted);
353
521
  }
522
+ // Title with the fallbacks every entry source shares when the split yields no
523
+ // text (e.g. an image-only card link): the own link's text, then the repo
524
+ // name.
525
+ function entryTitle(base, ownLink, repoInfo) {
526
+ return (base ||
527
+ (ownLink ? getInlineText(ownLink.children) : '') ||
528
+ repoInfo?.repo ||
529
+ '');
530
+ }
531
+ // Table rows are entries under the nearest open container. The common shape —
532
+ // scala's `[name](repo) | description`, spec tables with the link in a later
533
+ // column — holds ONE repo per row: the title is the first cell's text (the
534
+ // own link's text, then the repo name, when the first cell is empty), the
535
+ // description the remaining cells' text (badge images carry no text nodes, so
536
+ // they never pollute it). Card grids pin several repos per row, each cell its
537
+ // own card, so those emit one entry per link-bearing cell. A linked row above
538
+ // the delimiter is content like any other (grid tables have no label header;
539
+ // pure-label header rows carry no links and emit nothing). Rows stay in source
540
+ // order — unlike the unranked lists the product sorts, a table's row order is
541
+ // part of its meaning.
542
+ function processTableRows(tableNode, repoInfoMap, gateOpen) {
543
+ if (!gateOpen) {
544
+ return [];
545
+ }
546
+ const items = [];
547
+ for (const row of tableNode.children) {
548
+ if (row.type !== 'tableRow') {
549
+ continue;
550
+ }
551
+ const cellLinks = row.children.map(cell => findFirstGitHubLink(cell));
552
+ const distinctUrls = new Set(cellLinks.flatMap(link => (link ? [link.url] : [])));
553
+ if (distinctUrls.size === 0) {
554
+ continue;
555
+ }
556
+ if (distinctUrls.size === 1) {
557
+ const ownLink = cellLinks.find((link) => !!link);
558
+ if (!ownLink) {
559
+ continue;
560
+ }
561
+ const repoInfo = repoInfoMap.get(ownLink.url) ?? null;
562
+ const firstCellText = splitEntryText(row.children[0]?.children ?? []);
563
+ const restCellsText = row.children
564
+ .slice(1)
565
+ .map(cell => getNodeText(cell))
566
+ .filter(Boolean);
567
+ items.push(...emitEntryNodes(ownLink.url, repoInfo, {
568
+ title: entryTitle(firstCellText.title, ownLink, repoInfo),
569
+ description: [firstCellText.description, ...restCellsText]
570
+ .join(' ')
571
+ .trim(),
572
+ }, []));
573
+ continue;
574
+ }
575
+ const emittedUrls = new Set();
576
+ for (const [cellIndex, ownLink] of cellLinks.entries()) {
577
+ if (!ownLink || emittedUrls.has(ownLink.url)) {
578
+ continue;
579
+ }
580
+ emittedUrls.add(ownLink.url);
581
+ const repoInfo = repoInfoMap.get(ownLink.url) ?? null;
582
+ const cellText = splitEntryText(row.children[cellIndex].children);
583
+ items.push(...emitEntryNodes(ownLink.url, repoInfo, {
584
+ title: entryTitle(cellText.title, ownLink, repoInfo),
585
+ description: cellText.description,
586
+ }, []));
587
+ }
588
+ }
589
+ return items;
590
+ }
591
+ // A top-level paragraph can BE an entry, not only feed container prose. Two
592
+ // corpus families qualify (progress/empty-tree-parses.md, step 4): a GitHub
593
+ // link LEADING the paragraph behind a short entry label, or the paragraph
594
+ // ending in a tag cluster whose GitHub link carries the identity — the
595
+ // paper-list shape whose first link points at the paper and a [Code]/[Github]
596
+ // tag at the end, name/author lines ending in that tag, and dated lines
597
+ // ("… [Github] 4 Feb 2023"). Prose that mentions a repo ("Please see
598
+ // CONTRIBUTING", "See also [repo]", intro text ending in a bare repo URL) is
599
+ // neither and stays description. Calibrated on the 2,293-README corpus plus
600
+ // the fixture fleet, not intuition: an entry label is a TAG (empty, CJK/emoji,
601
+ // bracketed, or colon-terminated) — never bare English prose — and the
602
+ // identity link must carry text, so the ubiquitous image-only awesome badge
603
+ // is not an entry.
604
+ const ENTRY_LABEL_MAX = 15;
605
+ const ENTRY_TRAILING_MAX = 3;
606
+ const TAG_LINK_TEXT = /^[\[\]()*\s:_-]*(?:source\s+code|code|github|repo|source|src|project|paper|page|web|site|home|official|notebook|demo|data|docs|implementation|arxiv)\b[\[\]()*\s:_-]*$/i;
607
+ const URL_LINK_TEXT = /^https?:\/\/\S+$/i;
608
+ // Dated entry lines end their tag cluster with a publication date ("4 Feb
609
+ // 2023", "19 Apr 2022", "2023") — a real corpus family (dated paper/tutorial
610
+ // lists), not a sentence continuation.
611
+ const DATE_TRAILING = /^(\d{1,2}[ -])?([a-zà-ÿ]{3,12}[ -])?\d{2,4}$/i;
612
+ // Boilerplate navigation line present in most mirror READMEs; casing and
613
+ // the optional "the" vary ("⬆ back to top", "⬆️ Back to Top",
614
+ // "Back to the Top").
615
+ const BACK_TO_TOP = /back to (?:the )?top/i;
616
+ // Tag-shaped leading label: empty, non-ASCII (CJK labels like 项目地址:,
617
+ // emoji markers), bracketed ([6] Diffusion:), or colon-terminated (Demo:).
618
+ // English prose ("Please see", "See also", "Inspired by the") is none of
619
+ // these — those are cross-reference sentences, not entry labels.
620
+ function isEntryLabel(label) {
621
+ return (label === '' ||
622
+ /[^\x00-\x7f]/.test(label) ||
623
+ label.startsWith('[') ||
624
+ /[::]$/.test(label));
625
+ }
626
+ // Flattened view of a paragraph's (or heading's) inline content for the entry
627
+ // test: the first link, the free text before it, and the free text after the
628
+ // link cluster (emphasis recursed into; link labels, images, and inline html
629
+ // carry no positional weight — a bare [Code] tag's label is not trailing
630
+ // prose).
631
+ function scanInlines(paragraph) {
632
+ let firstLink;
633
+ let label = '';
634
+ let trailing = '';
635
+ const walk = (node) => {
636
+ if (node.type === 'link') {
637
+ if (firstLink) {
638
+ trailing = '';
639
+ }
640
+ firstLink = firstLink ?? node;
641
+ return;
642
+ }
643
+ if (node.type === 'text') {
644
+ if (firstLink) {
645
+ trailing += node.value;
646
+ }
647
+ else {
648
+ label += node.value;
649
+ }
650
+ return;
651
+ }
652
+ for (const child of node.children ?? []) {
653
+ walk(child);
654
+ }
655
+ };
656
+ for (const child of paragraph.children) {
657
+ walk(child);
658
+ }
659
+ return { firstLink, label: label.trim(), trailing: trailing.trim() };
660
+ }
661
+ // True when nothing of substance follows the paragraph's last link: pure
662
+ // punctuation, or a date (see DATE_TRAILING).
663
+ function tagClusterEndsParagraph(trailing) {
664
+ if (trailing.length <= ENTRY_TRAILING_MAX) {
665
+ return true;
666
+ }
667
+ return DATE_TRAILING.test(trailing.replace(/^[\[\]()*\s:;,_-]+/, ''));
668
+ }
669
+ // The GitHub link that makes a top-level paragraph — or a heading inside a
670
+ // blockquote — an entry, if it is one (see the family comment above). Null
671
+ // for plain prose.
672
+ function paragraphEntryLink(paragraph) {
673
+ const githubLink = findFirstGitHubLink(paragraph);
674
+ if (!githubLink) {
675
+ return undefined;
676
+ }
677
+ const labelText = getInlineText(githubLink.children);
678
+ if (!labelText) {
679
+ return undefined;
680
+ }
681
+ const scan = scanInlines(paragraph);
682
+ if (scan.firstLink === githubLink &&
683
+ scan.label.length <= ENTRY_LABEL_MAX &&
684
+ isEntryLabel(scan.label)) {
685
+ return githubLink;
686
+ }
687
+ if (tagClusterEndsParagraph(scan.trailing) &&
688
+ (TAG_LINK_TEXT.test(labelText) ||
689
+ (URL_LINK_TEXT.test(labelText) && scan.firstLink !== githubLink))) {
690
+ return githubLink;
691
+ }
692
+ return undefined;
693
+ }
694
+ function blockquoteEntries(blockquote) {
695
+ const faces = [];
696
+ let sawParagraph = false;
697
+ for (const child of blockquote.children) {
698
+ if (child.type === 'paragraph') {
699
+ if (sawParagraph) {
700
+ continue;
701
+ }
702
+ sawParagraph = true;
703
+ const link = paragraphEntryLink(child);
704
+ if (link) {
705
+ faces.push({ inlines: child.children, link });
706
+ }
707
+ }
708
+ else if (child.type === 'heading') {
709
+ const link = paragraphEntryLink(child);
710
+ if (link) {
711
+ faces.push({ inlines: child.children, link });
712
+ }
713
+ }
714
+ }
715
+ return faces;
716
+ }
717
+ // The shared entry emission for the non-list sources (paragraph entries,
718
+ // blockquote cards): resolve, split, title-fallback — the same decisions the
719
+ // list and table paths make through the same helpers.
720
+ function entryNodesFor(ownLink, inlines, repoInfoMap) {
721
+ const repoInfo = repoInfoMap.get(ownLink.url) ?? null;
722
+ const entryText = splitEntryText(inlines);
723
+ return emitEntryNodes(ownLink.url, repoInfo, {
724
+ title: entryTitle(entryText.title, ownLink, repoInfo),
725
+ description: entryText.description,
726
+ }, []);
727
+ }
728
+ // The <details><summary>…</summary> collapsible-section idiom: the summary
729
+ // text delimits structure like a heading would (java's generated README,
730
+ // paper-list tables behind "1.1 <topic>" summaries). kbd chips inside the
731
+ // summary ("5 projects") are metadata, not the title. A details block with
732
+ // no summary, or an empty one, delimits nothing.
733
+ const DETAILS_SUMMARY = /<details[^>]*>[\s\S]*?<summary[^>]*>([\s\S]*?)<\/summary>/i;
734
+ const DETAILS_CLOSE = /^\s*<\/details>/i;
735
+ function detailsSummaryTitle(htmlValue) {
736
+ const match = DETAILS_SUMMARY.exec(htmlValue);
737
+ if (!match) {
738
+ return null;
739
+ }
740
+ const title = match[1]
741
+ .replace(/<kbd>[\s\S]*?<\/kbd>/gi, '')
742
+ .replace(/<[^>]+>/g, ' ')
743
+ .replace(/\s+/g, ' ')
744
+ .trim();
745
+ return title || null;
746
+ }
747
+ // The summary text of a <details><summary> block inside a list item, when the
748
+ // item has no paragraph text of its own.
749
+ function listItemSummaryTitle(itemNode) {
750
+ for (const child of itemNode.children) {
751
+ if (child.type === 'html') {
752
+ const title = detailsSummaryTitle(child.value);
753
+ if (title) {
754
+ return title;
755
+ }
756
+ }
757
+ }
758
+ return '';
759
+ }
354
760
  const INVALID_TITLE_PATTERNS = [
355
761
  /^contributing/i,
356
762
  /^license$/i,
@@ -403,7 +809,7 @@ const TOC_TITLE_PATTERNS = [/^contents$/i, /^table of contents$/i];
403
809
  // A heading that delimits content structure. Text-less headings (a bare `#`,
404
810
  // an image-only heading) are spacers in real docs; TOC headings are structure
405
811
  // mirrors. Neither participates in the section tree.
406
- function isStructuralHeading(node) {
812
+ export function isStructuralHeading(node) {
407
813
  const title = getNodeText(node);
408
814
  return (!!title && !TOC_TITLE_PATTERNS.some(pattern => pattern.test(title.trim())));
409
815
  }
@@ -424,6 +830,17 @@ function findTitleHeadingIndex(tree) {
424
830
  node.depth === 1 &&
425
831
  isValidTitle(getNodeText(node)));
426
832
  }
833
+ // The heading branding owns even when it isn't a *valid* title: a generic
834
+ // first H1 ("# Guides", "# Contents") is still the de-facto title slot —
835
+ // applyBrandingToTree replaces it — so the section tree must not treat it
836
+ // as a section wrapping the whole document.
837
+ export function findTitleSlotIndex(tree) {
838
+ const titleHeadingIndex = findTitleHeadingIndex(tree);
839
+ if (titleHeadingIndex !== -1) {
840
+ return titleHeadingIndex;
841
+ }
842
+ return tree.children.findIndex((node) => node.type === 'heading' && node.depth === 1);
843
+ }
427
844
  /**
428
845
  * Makes the rendered H1 match metadata.title, replacing whichever heading
429
846
  * occupies the title slot (a valid title H1, else the first generic H1, else a
@@ -460,13 +877,7 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepository) {
460
877
  let documentTitle = titleHeadingIndex === -1
461
878
  ? ''
462
879
  : getNodeText(tree.children[titleHeadingIndex]);
463
- // The heading branding owns even when it isn't a *valid* title: a generic
464
- // first H1 ("# Guides", "# Contents") is still the de-facto title slot —
465
- // applyBrandingToTree replaces it — so the section tree must not treat it
466
- // as a section wrapping the whole document.
467
- const titleSlotIndex = titleHeadingIndex !== -1
468
- ? titleHeadingIndex
469
- : tree.children.findIndex((node) => node.type === 'heading' && node.depth === 1);
880
+ const titleSlotIndex = findTitleSlotIndex(tree);
470
881
  // Derive a subject from the *source* repository name when no valid H1 is
471
882
  // present. Using the source — not the enhanced/mirror repo — keeps the org
472
883
  // name out of the title.
@@ -477,8 +888,34 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepository) {
477
888
  }
478
889
  }
479
890
  const sectionDepth = findSectionDepth(tree, titleSlotIndex);
480
- const sections = [];
891
+ const gateForSection = sectionGatePasses(tree, titleSlotIndex, sectionDepth, repoInfoMap, sortOptions);
892
+ // Sections finalize out of document order when a details-section closes
893
+ // inside its parent section's span, so each carries the document index it
894
+ // opened at for the document-order sort at the end.
895
+ const sectionRecords = [];
481
896
  const stack = [];
897
+ // Entry emission shared by the paragraph and blockquote-card faces: the
898
+ // implicit section opens for a containerless entry (the stack-bottom
899
+ // invariant), the minLinks gate is the section aggregate, and a gated or
900
+ // dead entry emits nothing and stays out of the description (mirrors dead
901
+ // list items).
902
+ const emitStandaloneEntry = (ownLink, inlines, atIndex) => {
903
+ if (stack.length === 0) {
904
+ openImplicitSection(stack, atIndex);
905
+ }
906
+ if (gateForSection(stack[0])) {
907
+ stack[stack.length - 1].children.push(...entryNodesFor(ownLink, inlines, repoInfoMap));
908
+ }
909
+ };
910
+ // A standalone entry restating the repo of an open item container (crypto's
911
+ // generated cards repeat the repo URL in a line under the link-heading) is
912
+ // that item's description, not a second item for the same repo. The
913
+ // repoInfoMap resolves both URLs through one per-repo memo, so identity
914
+ // comparison holds even for aliased spellings.
915
+ const rementionsOpenItem = (link) => {
916
+ const repoInfo = repoInfoMap.get(link.url);
917
+ return !!repoInfo && stack.some(container => container.repoInfo === repoInfo);
918
+ };
482
919
  for (let i = 0; i < tree.children.length; i++) {
483
920
  const node = tree.children[i];
484
921
  if (node.type === 'heading') {
@@ -488,32 +925,117 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepository) {
488
925
  if (i === titleSlotIndex || !isStructuralHeading(node)) {
489
926
  continue;
490
927
  }
491
- closeContainers(stack, node.depth, sections);
492
- openContainer(stack, node, sectionDepth, repoInfoMap);
928
+ const headingEntry = entryHeadingInfo(node, repoInfoMap);
929
+ // The promotion flavor of the entry test: a textless identity link is
930
+ // a title-line badge, not an entry (see entryHeadingInfo).
931
+ const promotedEntry = headingEntry && getInlineText(headingEntry.link.children)
932
+ ? headingEntry
933
+ : null;
934
+ closeContainers(stack, node.depth, sectionRecords, !!promotedEntry);
935
+ openContainer(stack, node, i, sectionDepth, promotedEntry, headingEntry);
936
+ }
937
+ else if (node.type === 'paragraph') {
938
+ const text = getNodeText(node);
939
+ // Boilerplate "back to top" lines are neither entries nor description.
940
+ const ownLink = BACK_TO_TOP.test(text) ? undefined : paragraphEntryLink(node);
941
+ // An entry paragraph behaves like a one-item list; a failed gate, a
942
+ // dead target, or a re-mention of the enclosing item leaves plain
943
+ // description.
944
+ if (ownLink && !rementionsOpenItem(ownLink)) {
945
+ emitStandaloneEntry(ownLink, node.children, i);
946
+ }
947
+ else {
948
+ const container = stack[stack.length - 1];
949
+ // Avoid adding boilerplate "back to top" links to descriptions.
950
+ if (container && text && !BACK_TO_TOP.test(text)) {
951
+ container.description = container.description
952
+ ? `${container.description}\n${text}`
953
+ : text;
954
+ }
955
+ }
493
956
  }
494
- else if (node.type === 'paragraph' || node.type === 'blockquote') {
957
+ else if (node.type === 'blockquote') {
495
958
  const text = getNodeText(node);
496
- const container = stack[stack.length - 1];
497
- // Avoid adding boilerplate "back to top" links to descriptions.
498
- if (container && text && !text.includes('back to top')) {
499
- container.description = container.description
500
- ? `${container.description}\n${text}`
501
- : text;
959
+ // A card blockquote is the blockquote face of entries — one per
960
+ // qualifying paragraph/heading child; any other quote is container
961
+ // prose.
962
+ const faces = BACK_TO_TOP.test(text)
963
+ ? []
964
+ : blockquoteEntries(node).filter(face => !rementionsOpenItem(face.link));
965
+ if (faces.length > 0) {
966
+ for (const face of faces) {
967
+ emitStandaloneEntry(face.link, face.inlines, i);
968
+ }
969
+ }
970
+ else {
971
+ const container = stack[stack.length - 1];
972
+ // Avoid adding boilerplate "back to top" links to descriptions.
973
+ if (container && text && !BACK_TO_TOP.test(text)) {
974
+ container.description = container.description
975
+ ? `${container.description}\n${text}`
976
+ : text;
977
+ }
978
+ }
979
+ }
980
+ else if (node.type === 'html') {
981
+ // A details-summary block opens a section like a heading would; the
982
+ // close tag (or the next summary) ends it — never the enclosing
983
+ // section, which keeps collecting after the collapsible block.
984
+ const summaryTitle = detailsSummaryTitle(node.value);
985
+ if (summaryTitle) {
986
+ closeInnermostDetails(stack, sectionRecords);
987
+ // Container depths never decrease going up the stack. When the open
988
+ // containers sit deeper than sectionDepth (a mid-document H1 defines
989
+ // sectionDepth while the content sections run deeper), the
990
+ // details-section joins at the current depth — pushing at the
991
+ // shallower sectionDepth would invert the stack and strand the gate
992
+ // (which reads the stack bottom) on a tiny outer section.
993
+ const joinDepth = stack.length === 0
994
+ ? sectionDepth
995
+ : Math.max(sectionDepth, stack[stack.length - 1].headingDepth);
996
+ stack.push({
997
+ children: [],
998
+ description: '',
999
+ headingDepth: joinDepth,
1000
+ headingIndex: i,
1001
+ kind: 'section',
1002
+ openedByDetails: true,
1003
+ title: summaryTitle,
1004
+ });
1005
+ }
1006
+ else if (DETAILS_CLOSE.test(node.value)) {
1007
+ closeInnermostDetails(stack, sectionRecords);
502
1008
  }
503
1009
  }
504
1010
  else if (node.type === 'list') {
1011
+ // A list with no open container still means content: synthesize the
1012
+ // implicit section for it (see openImplicitSection). Afterwards the
1013
+ // stack bottom is always a section — implicit or real — which is the
1014
+ // invariant the gate below and closeContainers rely on.
1015
+ if (stack.length === 0) {
1016
+ openImplicitSection(stack, i);
1017
+ }
505
1018
  // Every list inside the open container contributes items — a section is
506
- // not closed by its first list. With no open container (preamble), the
507
- // list is not part of any JSON section, but its AST is still sorted so
508
- // the rendered markdown matches.
509
- const items = processListRecursively(node, repoInfoMap, sortOptions);
510
- const container = stack[stack.length - 1];
511
- if (container) {
512
- container.children.push(...items);
1019
+ // not closed by its first list — and the minLinks gate is decided per
1020
+ // section, against the whole section subtree.
1021
+ const items = processListRecursively(node, repoInfoMap, sortOptions, false, gateForSection(stack[0]));
1022
+ stack[stack.length - 1].children.push(...items);
1023
+ }
1024
+ else if (node.type === 'table') {
1025
+ // Tables hold entries the same way lists do — the implicit section opens
1026
+ // for one with no open container (the stack-bottom invariant), and the
1027
+ // minLinks gate is the same section aggregate.
1028
+ if (stack.length === 0) {
1029
+ openImplicitSection(stack, i);
513
1030
  }
1031
+ const items = processTableRows(node, repoInfoMap, gateForSection(stack[0]));
1032
+ stack[stack.length - 1].children.push(...items);
514
1033
  }
515
1034
  }
516
- closeContainers(stack, 0, sections);
1035
+ closeContainers(stack, 0, sectionRecords);
1036
+ const sections = sectionRecords
1037
+ .sort((a, b) => a.headingIndex - b.headingIndex)
1038
+ .map(record => record.section);
517
1039
  return { sections, title: documentTitle, titleHeadingIndex };
518
1040
  }
519
1041
  // The heading depth that opens top-level sections: the shallowest structural
@@ -533,87 +1055,308 @@ function findSectionDepth(tree, titleSlotIndex) {
533
1055
  });
534
1056
  return depth;
535
1057
  }
536
- // A heading whose only link is a live GitHub link represents a resource, not a
537
- // container — the link-heading pattern (`#### [Repo](github…)`). Badge images
538
- // wrapped in links or multiple links disqualify (more than one link means the
539
- // heading is not "the" resource), as does a dead target.
540
- function soleLiveHeadingLink(heading, repoInfoMap) {
541
- const links = heading.children.filter((child) => child.type === 'link');
542
- if (links.length !== 1) {
1058
+ function entryHeadingInfo(heading, repoInfoMap) {
1059
+ const link = findFirstGitHubLink(heading);
1060
+ if (!link) {
543
1061
  return null;
544
1062
  }
545
- return repoInfoMap.get(links[0].url) ?? null;
1063
+ const repoInfo = repoInfoMap.get(link.url);
1064
+ return repoInfo ? { link, repoInfo } : null;
546
1065
  }
547
- function openContainer(stack, heading, sectionDepth, repoInfoMap) {
1066
+ function openContainer(stack, heading, headingIndex, sectionDepth, promotedEntry, headingEntry) {
548
1067
  const title = getNodeText(heading);
549
1068
  // Sections sit at the section level — and any heading met with an empty
550
1069
  // stack is promoted: a deeper heading before the first section (orphan
551
- // subheading) still owns its subtree, and a link-heading at section level
552
- // becomes a section rather than a top-level item, which the contract has no
553
- // place for.
1070
+ // subheading) still owns its subtree. A promoted ENTRY heading (see
1071
+ // entryHeadingInfo) is an item, not a section: the JSON contract has no
1072
+ // top-level items, so a synthesized section wraps the whole run of them.
554
1073
  if (stack.length === 0 || heading.depth === sectionDepth) {
1074
+ if (promotedEntry) {
1075
+ if (stack.length === 0) {
1076
+ openSynthesizedSection(stack, headingIndex, sectionDepth);
1077
+ }
1078
+ stack.push({
1079
+ children: [],
1080
+ description: '',
1081
+ headingDepth: heading.depth,
1082
+ headingIndex,
1083
+ kind: 'item',
1084
+ repoInfo: promotedEntry.repoInfo,
1085
+ title,
1086
+ });
1087
+ return;
1088
+ }
555
1089
  stack.push({
556
1090
  children: [],
557
1091
  description: '',
558
1092
  headingDepth: heading.depth,
1093
+ headingIndex,
559
1094
  kind: 'section',
560
1095
  title,
561
1096
  });
562
1097
  return;
563
1098
  }
564
- const repoInfo = soleLiveHeadingLink(heading, repoInfoMap);
1099
+ // A deeper entry heading opens an item container the same way a promoted
1100
+ // one does — children and prose below it collect as its content.
565
1101
  stack.push({
566
1102
  children: [],
567
1103
  description: '',
568
1104
  headingDepth: heading.depth,
569
- kind: repoInfo ? 'item' : 'group',
570
- repoInfo: repoInfo ?? undefined,
1105
+ headingIndex,
1106
+ kind: headingEntry ? 'item' : 'group',
1107
+ repoInfo: headingEntry?.repoInfo,
571
1108
  title,
572
1109
  });
573
1110
  }
574
- // Finalize every container a heading of `depth` closes (same-or-shallower),
575
- // bottom-up so each finalized node lands in its parent. Pruning falls out of
576
- // the finalize rule: a section/group whose children array is empty (no items
577
- // anywhere beneath — lists only return item-bearing nodes, and empty children
578
- // were never appended) is dropped; an item always survives, it IS the content.
579
- // The stack bottom is always a section (openContainer's promotion guarantees
580
- // it), so a finalized group/item always has a parent to land in.
581
- function closeContainers(stack, depth, sections) {
582
- while (stack.length > 0 &&
583
- stack[stack.length - 1].headingDepth >= depth) {
584
- const container = stack.pop();
585
- if (container.children.length === 0 && container.kind !== 'item') {
586
- continue;
1111
+ // A document can hold registry content with no heading above it — headingless
1112
+ // docs, TOC-only docs, or preamble entries (list, table) before the first
1113
+ // section heading. Rather than dropping them, the first one synthesizes the
1114
+ // section it implicitly belongs to ("Overview"). It behaves exactly like a
1115
+ // real section:
1116
+ // closed by the first structural heading (or document end), gated with its
1117
+ // whole subtree, and pruned when empty.
1118
+ function openImplicitSection(stack, atIndex) {
1119
+ stack.push({
1120
+ children: [],
1121
+ description: '',
1122
+ // Infinity so ANY structural heading closes it in closeContainers and
1123
+ // ends its span in the section gate — the implicit section never reaches
1124
+ // past the preamble.
1125
+ headingDepth: Infinity,
1126
+ // One before the opening list so the gate's scan (from headingIndex + 1)
1127
+ // counts that list itself.
1128
+ headingIndex: atIndex - 1,
1129
+ kind: 'section',
1130
+ title: 'Overview',
1131
+ });
1132
+ }
1133
+ // The section synthesized around a run of promoted entry headings
1134
+ // (openContainer's entry branch). It behaves like the implicit section — same
1135
+ // "Overview" title, gated by its subtree — but sits at sectionDepth, so the
1136
+ // first plain heading at that level ends the run, and closeContainers'
1137
+ // stopAtSynthesized keeps the entry headings' own pops from closing it:
1138
+ // consecutive entry headings are siblings INSIDE it.
1139
+ function openSynthesizedSection(stack, firstEntryIndex, depth) {
1140
+ stack.push({
1141
+ children: [],
1142
+ description: '',
1143
+ headingDepth: depth,
1144
+ // One before the first entry heading so the gate's scan (from
1145
+ // headingIndex + 1) counts that heading itself.
1146
+ headingIndex: firstEntryIndex - 1,
1147
+ kind: 'section',
1148
+ openedBySynthesis: true,
1149
+ title: 'Overview',
1150
+ });
1151
+ }
1152
+ // The minLinks gate scoped to a SECTION: every entry source in the section
1153
+ // counts together (list items, table rows, entry paragraphs, entry headings,
1154
+ // blockquote faces) — best-of-style documents put each entry in its own
1155
+ // single-item list, which the per-list gate dropped one by one. The section's
1156
+ // subtree runs from its heading to the first structural heading at or above
1157
+ // its depth (the same heading that would close it in closeContainers).
1158
+ //
1159
+ // Heading-per-entry documents (FBI-tools' `### name` + link paragraph, FLOSS's
1160
+ // `## game` + link cluster) put exactly ONE entry in each section, so every
1161
+ // section fails the gate alone and the whole document drops. They are not
1162
+ // sparse sections but flat entry lists delimited by headings: a section at
1163
+ // sectionDepth that fails alone is re-gated against its RUN — the maximal
1164
+ // sequence of adjacent section spans each holding at most one entry. The
1165
+ // noise floor survives: a single one-entry section between multi-entry
1166
+ // sections is a run of one and stays dropped.
1167
+ function sectionGatePasses(tree, titleSlotIndex, sectionDepth, repoInfoMap, sortOptions) {
1168
+ const cache = new Map();
1169
+ // Linked entries from scanStart to the first structural heading that would
1170
+ // close a container of closeDepth. An entry heading counts as an entry
1171
+ // wherever it sits inside the span — it opens an item, not a container —
1172
+ // except at the scanned section's own depth, where the walk closes that
1173
+ // section when the entry run starts (a same-depth entry heading is content
1174
+ // only for the synthesized wrapper, hence sameDepthEntriesAreContent).
1175
+ const scanCount = (scanStart, closeDepth, sameDepthEntriesAreContent) => {
1176
+ let linkedEntries = 0;
1177
+ for (let j = scanStart; j < tree.children.length; j++) {
1178
+ const node = tree.children[j];
1179
+ if (node.type === 'heading') {
1180
+ if (j === titleSlotIndex || !isStructuralHeading(node)) {
1181
+ continue;
1182
+ }
1183
+ if (entryHeadingInfo(node, repoInfoMap)) {
1184
+ if (sameDepthEntriesAreContent || node.depth > closeDepth) {
1185
+ linkedEntries += 1;
1186
+ }
1187
+ else {
1188
+ break;
1189
+ }
1190
+ continue;
1191
+ }
1192
+ if (node.depth <= closeDepth) {
1193
+ break;
1194
+ }
1195
+ continue;
1196
+ }
1197
+ if (node.type === 'list') {
1198
+ linkedEntries += countLinkedItems(node);
1199
+ }
1200
+ else if (node.type === 'table') {
1201
+ linkedEntries += countLinkedRows(node);
1202
+ }
1203
+ else if (node.type === 'paragraph' && paragraphEntryLink(node)) {
1204
+ linkedEntries += 1;
1205
+ }
1206
+ else if (node.type === 'blockquote') {
1207
+ linkedEntries += blockquoteEntries(node).length;
1208
+ }
587
1209
  }
588
- const description = container.description || null;
589
- if (container.kind === 'section') {
590
- sections.push({
591
- description,
592
- items: container.children,
593
- title: container.title,
594
- });
595
- continue;
1210
+ return linkedEntries;
1211
+ };
1212
+ // A boundary's entry count for the run walk: the heading itself when it is
1213
+ // an entry heading, plus its span's entries (the span closes at the first
1214
+ // heading at or above the boundary's OWN depth, exactly where the walk
1215
+ // closes that section).
1216
+ const spanCountCache = new Map();
1217
+ const spanCount = (boundaryIndex) => {
1218
+ const cached = spanCountCache.get(boundaryIndex);
1219
+ if (cached !== undefined) {
1220
+ return cached;
596
1221
  }
597
- const parent = stack[stack.length - 1];
598
- // Non-section containers always have an open parent (stack invariant), and
599
- // kind === 'item' exactly when repoInfo is set.
600
- if (container.repoInfo) {
601
- parent.children.push({
602
- children: container.children,
603
- description,
604
- node_type: 'item',
605
- repo_info: toRepoInfo(container.repoInfo),
606
- title: container.title,
1222
+ const heading = tree.children[boundaryIndex];
1223
+ const count = (entryHeadingInfo(heading, repoInfoMap) ? 1 : 0) +
1224
+ scanCount(boundaryIndex + 1, heading.depth, false);
1225
+ spanCountCache.set(boundaryIndex, count);
1226
+ return count;
1227
+ };
1228
+ // The section boundaries in document order, by the walk's own promotion
1229
+ // rule (openContainer): a heading opens a section when it sits at
1230
+ // sectionDepth or arrives with nothing open — FLOSS-style documents have a
1231
+ // lone late H1 that pins sectionDepth at 1 while every `## game` heading
1232
+ // alternately closes its predecessor (stack empties) and is promoted. The
1233
+ // depth simulation mirrors closeContainers/openContainer exactly.
1234
+ let boundaries;
1235
+ const sectionBoundaries = () => {
1236
+ if (!boundaries) {
1237
+ const indices = [];
1238
+ const openDepths = [];
1239
+ tree.children.forEach((node, i) => {
1240
+ if (node.type !== 'heading' || i === titleSlotIndex) {
1241
+ return;
1242
+ }
1243
+ if (!isStructuralHeading(node)) {
1244
+ return;
1245
+ }
1246
+ while (openDepths.length > 0 &&
1247
+ openDepths[openDepths.length - 1] >= node.depth) {
1248
+ openDepths.pop();
1249
+ }
1250
+ if (openDepths.length === 0 || node.depth === sectionDepth) {
1251
+ indices.push(i);
1252
+ }
1253
+ openDepths.push(node.depth);
607
1254
  });
1255
+ boundaries = indices;
608
1256
  }
609
- else {
610
- parent.children.push({
611
- children: container.children,
612
- description,
613
- node_type: 'group',
614
- title: container.title,
615
- });
1257
+ return boundaries;
1258
+ };
1259
+ const runTotal = (section) => {
1260
+ const list = sectionBoundaries();
1261
+ const k = list.indexOf(section.headingIndex);
1262
+ if (k === -1) {
1263
+ return spanCount(section.headingIndex);
616
1264
  }
1265
+ let total = spanCount(section.headingIndex);
1266
+ for (const direction of [-1, 1]) {
1267
+ for (let m = k + direction; m >= 0 && m < list.length; m += direction) {
1268
+ const count = spanCount(list[m]);
1269
+ if (count > 1) {
1270
+ break;
1271
+ }
1272
+ total += count;
1273
+ }
1274
+ }
1275
+ return total;
1276
+ };
1277
+ return section => {
1278
+ const cached = cache.get(section.headingIndex);
1279
+ if (cached !== undefined) {
1280
+ return cached;
1281
+ }
1282
+ const own = scanCount(section.headingIndex + 1, section.headingDepth, !!section.openedBySynthesis);
1283
+ const passes = own >= sortOptions.minLinks ||
1284
+ (!section.openedBySynthesis &&
1285
+ sectionBoundaries().includes(section.headingIndex) &&
1286
+ runTotal(section) >= sortOptions.minLinks);
1287
+ cache.set(section.headingIndex, passes);
1288
+ return passes;
1289
+ };
1290
+ }
1291
+ // Prune-or-emit one popped container: a section/group whose children array is
1292
+ // empty (no items anywhere beneath — the entry sources only return
1293
+ // item-bearing nodes) is dropped; an item always survives, it IS the content.
1294
+ // Non-section containers always have an open parent (stack invariant), and
1295
+ // kind === 'item' exactly when repoInfo is set.
1296
+ function finalizeContainer(container, stack, sectionRecords) {
1297
+ if (container.children.length === 0 && container.kind !== 'item') {
1298
+ return;
1299
+ }
1300
+ const description = container.description || null;
1301
+ if (container.kind === 'section') {
1302
+ sectionRecords.push({
1303
+ headingIndex: container.headingIndex,
1304
+ section: { description, items: container.children, title: container.title },
1305
+ });
1306
+ return;
1307
+ }
1308
+ const parent = stack[stack.length - 1];
1309
+ if (container.repoInfo) {
1310
+ parent.children.push({
1311
+ children: container.children,
1312
+ description,
1313
+ node_type: 'item',
1314
+ repo_info: toRepoInfo(container.repoInfo),
1315
+ title: container.title,
1316
+ });
1317
+ }
1318
+ else {
1319
+ parent.children.push({
1320
+ children: container.children,
1321
+ description,
1322
+ node_type: 'group',
1323
+ title: container.title,
1324
+ });
1325
+ }
1326
+ }
1327
+ // Finalize every container a heading of `depth` closes (same-or-shallower),
1328
+ // bottom-up so each finalized node lands in its parent. The stack bottom is
1329
+ // always a section (openContainer's promotion guarantees it), so a finalized
1330
+ // group/item always has a parent to land in. stopAtSynthesized: the pop an
1331
+ // entry heading triggers must stop at the synthesized section wrapping its
1332
+ // run — the next entry heading of the run lands back inside it.
1333
+ function closeContainers(stack, depth, sectionRecords, stopAtSynthesized = false) {
1334
+ while (stack.length > 0 &&
1335
+ stack[stack.length - 1].headingDepth >= depth) {
1336
+ if (stopAtSynthesized && stack[stack.length - 1].openedBySynthesis) {
1337
+ break;
1338
+ }
1339
+ finalizeContainer(stack.pop(), stack, sectionRecords);
1340
+ }
1341
+ }
1342
+ // Ends the innermost open details-section and everything opened inside it,
1343
+ // leaving the enclosing containers untouched — unlike a heading close, a
1344
+ // details boundary never ends its parent section, so content after the
1345
+ // collapsible block keeps collecting under it. A stray </details> (no
1346
+ // details-section open) is a no-op.
1347
+ function closeInnermostDetails(stack, sectionRecords) {
1348
+ let detailsIndex = -1;
1349
+ for (let s = stack.length - 1; s >= 0; s--) {
1350
+ if (stack[s].openedByDetails) {
1351
+ detailsIndex = s;
1352
+ break;
1353
+ }
1354
+ }
1355
+ if (detailsIndex === -1) {
1356
+ return;
1357
+ }
1358
+ while (stack.length > detailsIndex) {
1359
+ finalizeContainer(stack.pop(), stack, sectionRecords);
617
1360
  }
618
1361
  }
619
1362
  function serializeAst(tree, originalContent) {
@@ -9,6 +9,7 @@ export interface EnhanceOptions {
9
9
  log?: Logger;
10
10
  now?: Date;
11
11
  originalRepository: string;
12
+ originalRepositoryId?: number;
12
13
  originalRepositorySha?: string;
13
14
  relativeLinkPrefix?: string;
14
15
  /** Text substitutions applied to the source before it is parsed. */
@@ -1,6 +1,6 @@
1
1
  import { processMarkdownContent, } from './markdown.js';
2
2
  export async function enhance(options) {
3
- const { content, disableBranding = false, log, now = new Date(), originalRepository, originalRepositorySha, relativeLinkPrefix = '', replacements = [], sortBy = '', enhancedRepository, enhancedRepositoryDescription, token, } = options;
3
+ const { content, disableBranding = false, log, now = new Date(), originalRepository, originalRepositoryId, originalRepositorySha, relativeLinkPrefix = '', replacements = [], sortBy = '', enhancedRepository, enhancedRepositoryDescription, token, } = options;
4
4
  // Branding is an internal rule prepended to the caller's own; build a fresh
5
5
  // array so the caller's `replacements` is never mutated.
6
6
  const branding = { type: 'branding' };
@@ -9,7 +9,7 @@ export async function enhance(options) {
9
9
  by: sortBy,
10
10
  minLinks: 2,
11
11
  };
12
- const { finalContent, jsonData } = await processMarkdownContent(content, token, rules, sortOptions, originalRepository, relativeLinkPrefix, enhancedRepository, enhancedRepositoryDescription, originalRepositorySha, now, log);
12
+ const { finalContent, jsonData } = await processMarkdownContent(content, token, rules, sortOptions, originalRepository, relativeLinkPrefix, enhancedRepository, enhancedRepositoryDescription, originalRepositorySha, originalRepositoryId, now, log);
13
13
  return {
14
14
  finalContent,
15
15
  jsonData,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@enhansome/core",
3
- "version": "1.8.0",
3
+ "version": "1.9.0",
4
4
  "description": "Library core for enhansome — enhance markdown with GitHub star counts.",
5
5
  "repository": {
6
6
  "type": "git",