@enhansome/core 1.7.1 → 1.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/github.d.ts CHANGED
@@ -8,6 +8,7 @@ export type GithubClient = InstanceType<typeof HardenedOctokit>;
8
8
  export interface RepoInfoDetails {
9
9
  archived: boolean;
10
10
  description: null | string;
11
+ id: number;
11
12
  language: null | string;
12
13
  open_issues_count: number;
13
14
  owner: string;
@@ -75,6 +76,12 @@ export interface MakeOctokitOptions {
75
76
  */
76
77
  export declare function makeOctokit(token: string, { log, maxRetries, maxWaitSeconds, throttle, }?: MakeOctokitOptions): GithubClient;
77
78
  export declare function getRepoInfo(octokit: GithubClient, owner: string, repo: string): Promise<RepoInfoDetails>;
79
+ /**
80
+ * The repo's numeric GitHub id, for the JSON metadata — consumers key stable
81
+ * node ids on it. Returns null (logged) instead of failing the run: the id is
82
+ * metadata, not a gate; consumers fall back when it is missing.
83
+ */
84
+ export declare function getRepoId(octokit: GithubClient, owner: string, repo: string): Promise<null | number>;
78
85
  export declare function getReadme(octokit: GithubClient, owner: string, repo: string, format?: 'html' | 'raw'): Promise<string>;
79
86
  /** Root file/directory names — the compile-manifest gate reads them to tell a
80
87
  * directory of resources from a repo that IS the deliverable. A `path: ''`
package/dist/github.js CHANGED
@@ -72,6 +72,7 @@ export async function getRepoInfo(octokit, owner, repo) {
72
72
  return {
73
73
  archived: data.archived,
74
74
  description: data.description ?? null,
75
+ id: data.id,
75
76
  language: data.language,
76
77
  open_issues_count: data.open_issues_count,
77
78
  owner: data.owner.login,
@@ -81,6 +82,20 @@ export async function getRepoInfo(octokit, owner, repo) {
81
82
  topics: data.topics ?? [],
82
83
  };
83
84
  }
85
+ /**
86
+ * The repo's numeric GitHub id, for the JSON metadata — consumers key stable
87
+ * node ids on it. Returns null (logged) instead of failing the run: the id is
88
+ * metadata, not a gate; consumers fall back when it is missing.
89
+ */
90
+ export async function getRepoId(octokit, owner, repo) {
91
+ try {
92
+ return (await getRepoInfo(octokit, owner, repo)).id;
93
+ }
94
+ catch (error) {
95
+ octokit.log.error(`Failed to fetch repo id for ${owner}/${repo}: ${formatRequestError(error)}`);
96
+ return null;
97
+ }
98
+ }
84
99
  export async function getReadme(octokit, owner, repo, format = 'raw') {
85
100
  octokit.log.debug(`Fetching ${format} README for ${owner}/${repo}`);
86
101
  const response = await octokit.rest.repos.getReadme({
package/dist/index.d.ts CHANGED
@@ -1,4 +1,4 @@
1
- export { formatRequestError, getLatestCommitSha, getReadme, getRepoInfo, getRootEntryNames, makeOctokit, parseGitHubUrl, parseOwnerRepo, } from './github.js';
1
+ export { formatRequestError, getLatestCommitSha, getReadme, getRepoId, getRepoInfo, getRootEntryNames, makeOctokit, parseGitHubUrl, parseOwnerRepo, } from './github.js';
2
2
  export type { GithubClient, MakeOctokitOptions, RepoIdentifier, RepoInfoDetails, ThrottleOptions, } from './github.js';
3
3
  export type { Logger } from './logger.js';
4
4
  export { consoleLog, silentLog } from './logger.js';
package/dist/index.js CHANGED
@@ -1,4 +1,4 @@
1
- export { formatRequestError, getLatestCommitSha, getReadme, getRepoInfo, getRootEntryNames, makeOctokit, parseGitHubUrl, parseOwnerRepo, } from './github.js';
1
+ export { formatRequestError, getLatestCommitSha, getReadme, getRepoId, getRepoInfo, getRootEntryNames, makeOctokit, parseGitHubUrl, parseOwnerRepo, } from './github.js';
2
2
  export { consoleLog, silentLog } from './logger.js';
3
3
  export { toRepoInfo } from './markdown.js';
4
4
  export { enhance } from './orchestrator.js';
@@ -17,6 +17,7 @@ export interface SortOptions {
17
17
  }
18
18
  export interface RepoInfo {
19
19
  archived: boolean;
20
+ id: number;
20
21
  language: null | string;
21
22
  last_commit: null | string;
22
23
  owner: string;
@@ -29,7 +30,7 @@ export interface JsonItem {
29
30
  children: JsonNode[];
30
31
  description: null | string;
31
32
  node_type: 'item';
32
- repo_info?: RepoInfo;
33
+ repo_info: RepoInfo;
33
34
  title: string;
34
35
  }
35
36
  export interface JsonGroup {
@@ -44,6 +45,7 @@ export interface JsonMetadata {
44
45
  enhanced_repository_description: null | string;
45
46
  last_updated: string;
46
47
  original_repository: string;
48
+ original_repository_id: null | number;
47
49
  original_repository_sha: null | string;
48
50
  title: string;
49
51
  }
@@ -52,7 +54,7 @@ export interface JsonSection {
52
54
  items: JsonNode[];
53
55
  title: string;
54
56
  }
55
- export declare function processMarkdownContent(originalContent: string, token: string, replacements: ReplacementRule[] | undefined, sortOptions: SortOptions | undefined, originalRepository: string, relativeLinkPrefix?: string, enhancedRepository?: string, enhancedRepositoryDescription?: string, originalRepositorySha?: string, now?: Date, log?: Logger): Promise<{
57
+ export declare function processMarkdownContent(originalContent: string, token: string, replacements: ReplacementRule[] | undefined, sortOptions: SortOptions | undefined, originalRepository: string, relativeLinkPrefix?: string, enhancedRepository?: string, enhancedRepositoryDescription?: string, originalRepositorySha?: string, originalRepositoryId?: number, now?: Date, log?: Logger): Promise<{
56
58
  finalContent: string;
57
59
  jsonData: JsonOutput;
58
60
  }>;
package/dist/markdown.js CHANGED
@@ -10,6 +10,7 @@ import { consoleLog } from './logger.js';
10
10
  export function toRepoInfo(details) {
11
11
  return {
12
12
  archived: details.archived,
13
+ id: details.id,
13
14
  language: details.language,
14
15
  last_commit: details.pushed_at,
15
16
  owner: details.owner,
@@ -78,7 +79,7 @@ async function fetchTargetData(urls, repos) {
78
79
  log.info(`Target fetch: ${repoInfoMap.size}/${urls.size} repo-info ok in ${Date.now() - fetchStart}ms (concurrency ${FETCH_CONCURRENCY}).`);
79
80
  return repoInfoMap;
80
81
  }
81
- export async function processMarkdownContent(originalContent, token, replacements = [], sortOptions = { by: '', minLinks: 2 }, originalRepository, relativeLinkPrefix = '', enhancedRepository, enhancedRepositoryDescription, originalRepositorySha, now = new Date(), log = consoleLog) {
82
+ export async function processMarkdownContent(originalContent, token, replacements = [], sortOptions = { by: '', minLinks: 2 }, originalRepository, relativeLinkPrefix = '', enhancedRepository, enhancedRepositoryDescription, originalRepositorySha, originalRepositoryId, now = new Date(), log = consoleLog) {
82
83
  const repos = createRepoInfoLookup(token, log);
83
84
  const brandingEnabled = replacements.some(rule => rule.type === 'branding');
84
85
  const contentAfterReplacements = applyTextReplacements(originalContent, replacements.filter(rule => rule.type !== 'branding'), log);
@@ -98,6 +99,7 @@ export async function processMarkdownContent(originalContent, token, replacement
98
99
  const metadata = {
99
100
  last_updated: now.toISOString(),
100
101
  original_repository: originalRepository.trim(),
102
+ original_repository_id: originalRepositoryId ?? null,
101
103
  original_repository_sha: (originalRepositorySha?.trim() ?? '') || null,
102
104
  enhanced_repository: (enhancedRepository?.trim() ?? '') || null,
103
105
  enhanced_repository_description: (enhancedRepositoryDescription?.trim() ?? '') || null,
@@ -282,10 +284,10 @@ function processListRecursively(listNode, repoInfoMap, sortOptions, isNested = f
282
284
  if (!isNested && itemsWithGitHubLinks.length < sortOptions.minLinks) {
283
285
  return [];
284
286
  }
285
- // Zip each item with its JSON node and repo info so one sort orders both the
286
- // rendered AST and the emitted JSON. `json` is null only for non-GitHub
287
- // leaves (no own link, no nested GitHub children): kept in the AST, dropped
288
- // from JSON.
287
+ // Zip each item with its emitted JSON nodes and repo info so one sort orders
288
+ // both the rendered AST and the emitted JSON. `emitted` is empty for
289
+ // non-GitHub leaves (no own link, no nested GitHub children): kept in the
290
+ // AST, dropped from JSON.
289
291
  const entries = [];
290
292
  for (const itemNode of listNode.children) {
291
293
  const githubUrl = findOwnGitHubLink(itemNode);
@@ -314,37 +316,41 @@ function processListRecursively(listNode, repoInfoMap, sortOptions, isNested = f
314
316
  // `repo_info`, that's the identity-borrowing bug. No-own-link, no-child
315
317
  // items are non-GitHub leaves: kept in markdown, dropped from JSON.
316
318
  // TODO(future): preserve non-GitHub leaves in a separate shape.
317
- let jsonData = null;
318
- if (githubUrl) {
319
- const item = {
320
- node_type: 'item',
321
- title,
322
- description: description || null,
323
- children: childrenJson,
324
- };
325
- if (repoInfo) {
326
- item.repo_info = toRepoInfo(repoInfo);
327
- }
328
- jsonData = item;
319
+ let emitted = [];
320
+ if (githubUrl && repoInfo) {
321
+ emitted = [
322
+ {
323
+ node_type: 'item',
324
+ title,
325
+ description: description || null,
326
+ children: childrenJson,
327
+ repo_info: toRepoInfo(repoInfo),
328
+ },
329
+ ];
330
+ }
331
+ else if (githubUrl) {
332
+ // Dead target: the item itself is not emitted; its children lift to this
333
+ // list's level — the nearest live parent.
334
+ emitted = childrenJson;
329
335
  }
330
336
  else if (childrenJson.length > 0) {
331
- jsonData = {
332
- node_type: 'group',
333
- title,
334
- description: description || null,
335
- children: childrenJson,
336
- };
337
+ emitted = [
338
+ {
339
+ node_type: 'group',
340
+ title,
341
+ description: description || null,
342
+ children: childrenJson,
343
+ },
344
+ ];
337
345
  }
338
- entries.push({ json: jsonData, node: itemNode, repoInfo });
346
+ entries.push({ emitted, node: itemNode, repoInfo });
339
347
  }
340
348
  if (sortOptions.by) {
341
349
  entries.sort((a, b) => compareByRepoInfo(sortOptions.by, a.repoInfo, b.repoInfo));
342
350
  }
343
351
  // Reorder the AST to match the sort so rendered markdown and JSON agree.
344
352
  listNode.children = entries.map(entry => entry.node);
345
- return entries
346
- .map(entry => entry.json)
347
- .filter((json) => json !== null);
353
+ return entries.flatMap(entry => entry.emitted);
348
354
  }
349
355
  const INVALID_TITLE_PATTERNS = [
350
356
  /^contributing/i,
@@ -389,6 +395,19 @@ function isValidTitle(title) {
389
395
  }
390
396
  return !INVALID_TITLE_PATTERNS.some(pattern => pattern.test(title.trim()));
391
397
  }
398
+ // Headings that mirror structure rather than delimit it. Unlike
399
+ // INVALID_TITLE_PATTERNS (a title-detection aid — "## Tools" is a perfectly
400
+ // good content section), a TOC heading never owns content: the worst offender
401
+ // is a `# Table of Contents` H1 that would otherwise wrap the whole document
402
+ // as its "section".
403
+ const TOC_TITLE_PATTERNS = [/^contents$/i, /^table of contents$/i];
404
+ // A heading that delimits content structure. Text-less headings (a bare `#`,
405
+ // an image-only heading) are spacers in real docs; TOC headings are structure
406
+ // mirrors. Neither participates in the section tree.
407
+ function isStructuralHeading(node) {
408
+ const title = getNodeText(node);
409
+ return (!!title && !TOC_TITLE_PATTERNS.some(pattern => pattern.test(title.trim())));
410
+ }
392
411
  /** Never duplicates "Awesome": if the title already contains it, append the suffix verbatim; otherwise prefix first. */
393
412
  function brandTitle(title) {
394
413
  const trimmed = title.trim();
@@ -442,6 +461,13 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepository) {
442
461
  let documentTitle = titleHeadingIndex === -1
443
462
  ? ''
444
463
  : getNodeText(tree.children[titleHeadingIndex]);
464
+ // The heading branding owns even when it isn't a *valid* title: a generic
465
+ // first H1 ("# Guides", "# Contents") is still the de-facto title slot —
466
+ // applyBrandingToTree replaces it — so the section tree must not treat it
467
+ // as a section wrapping the whole document.
468
+ const titleSlotIndex = titleHeadingIndex !== -1
469
+ ? titleHeadingIndex
470
+ : tree.children.findIndex((node) => node.type === 'heading' && node.depth === 1);
445
471
  // Derive a subject from the *source* repository name when no valid H1 is
446
472
  // present. Using the source — not the enhanced/mirror repo — keeps the org
447
473
  // name out of the title.
@@ -451,52 +477,146 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepository) {
451
477
  documentTitle = formatRepoNameAsTitle(repoName);
452
478
  }
453
479
  }
480
+ const sectionDepth = findSectionDepth(tree, titleSlotIndex);
454
481
  const sections = [];
455
- let currentSection = null;
456
- for (const node of tree.children) {
457
- if (node.type === 'heading' && node.depth > 1) {
458
- if (currentSection) {
459
- sections.push(currentSection);
482
+ const stack = [];
483
+ for (let i = 0; i < tree.children.length; i++) {
484
+ const node = tree.children[i];
485
+ if (node.type === 'heading') {
486
+ // The title-slot H1 belongs to branding/metadata; non-structural
487
+ // headings (see isStructuralHeading) delimit nothing. Neither
488
+ // participates in the section tree.
489
+ if (i === titleSlotIndex || !isStructuralHeading(node)) {
490
+ continue;
460
491
  }
461
- currentSection = {
462
- description: '',
463
- items: [],
464
- title: getNodeText(node),
465
- };
492
+ closeContainers(stack, node.depth, sections);
493
+ openContainer(stack, node, sectionDepth, repoInfoMap);
466
494
  }
467
- else if (currentSection) {
468
- if (node.type === 'paragraph') {
469
- const paragraphText = getNodeText(node);
470
- // Avoid adding boilerplate "back to top" links to description
471
- if (!paragraphText.includes('back to top')) {
472
- if (currentSection.description) {
473
- currentSection.description += `\n${paragraphText}`;
474
- }
475
- else {
476
- currentSection.description = paragraphText;
477
- }
478
- }
479
- }
480
- else if (node.type === 'list') {
481
- const items = processListRecursively(node, repoInfoMap, sortOptions);
482
- if (items.length > 0) {
483
- currentSection.items = items;
484
- sections.push(currentSection);
485
- }
486
- currentSection = null;
495
+ else if (node.type === 'paragraph' || node.type === 'blockquote') {
496
+ const text = getNodeText(node);
497
+ const container = stack[stack.length - 1];
498
+ // Avoid adding boilerplate "back to top" links to descriptions.
499
+ if (container && text && !text.includes('back to top')) {
500
+ container.description = container.description
501
+ ? `${container.description}\n${text}`
502
+ : text;
487
503
  }
488
504
  }
489
505
  else if (node.type === 'list') {
490
- // No active section: not part of any JSON section, but still sort its AST
491
- // so the rendered markdown matches.
492
- processListRecursively(node, repoInfoMap, sortOptions);
506
+ // Every list inside the open container contributes items — a section is
507
+ // not closed by its first list. With no open container (preamble), the
508
+ // list is not part of any JSON section, but its AST is still sorted so
509
+ // the rendered markdown matches.
510
+ const items = processListRecursively(node, repoInfoMap, sortOptions);
511
+ const container = stack[stack.length - 1];
512
+ if (container) {
513
+ container.children.push(...items);
514
+ }
493
515
  }
494
516
  }
495
- if (currentSection) {
496
- sections.push(currentSection);
497
- }
517
+ closeContainers(stack, 0, sections);
498
518
  return { sections, title: documentTitle, titleHeadingIndex };
499
519
  }
520
+ // The heading depth that opens top-level sections: the shallowest structural
521
+ // heading in the document other than the title slot. H1s count — ~20% of
522
+ // mirror READMEs use `# Section` after the title H1, and hardcoding H2 would
523
+ // drop all their items. Non-structural headings are skipped here too (same
524
+ // rule as the walk). Infinity when there is no such heading (no sections).
525
+ function findSectionDepth(tree, titleSlotIndex) {
526
+ let depth = Infinity;
527
+ tree.children.forEach((node, i) => {
528
+ if (node.type === 'heading' &&
529
+ i !== titleSlotIndex &&
530
+ isStructuralHeading(node) &&
531
+ node.depth < depth) {
532
+ depth = node.depth;
533
+ }
534
+ });
535
+ return depth;
536
+ }
537
+ // A heading whose only link is a live GitHub link represents a resource, not a
538
+ // container — the link-heading pattern (`#### [Repo](github…)`). Badge images
539
+ // wrapped in links or multiple links disqualify (more than one link means the
540
+ // heading is not "the" resource), as does a dead target.
541
+ function soleLiveHeadingLink(heading, repoInfoMap) {
542
+ const links = heading.children.filter((child) => child.type === 'link');
543
+ if (links.length !== 1) {
544
+ return null;
545
+ }
546
+ return repoInfoMap.get(links[0].url) ?? null;
547
+ }
548
+ function openContainer(stack, heading, sectionDepth, repoInfoMap) {
549
+ const title = getNodeText(heading);
550
+ // Sections sit at the section level — and any heading met with an empty
551
+ // stack is promoted: a deeper heading before the first section (orphan
552
+ // subheading) still owns its subtree, and a link-heading at section level
553
+ // becomes a section rather than a top-level item, which the contract has no
554
+ // place for.
555
+ if (stack.length === 0 || heading.depth === sectionDepth) {
556
+ stack.push({
557
+ children: [],
558
+ description: '',
559
+ headingDepth: heading.depth,
560
+ kind: 'section',
561
+ title,
562
+ });
563
+ return;
564
+ }
565
+ const repoInfo = soleLiveHeadingLink(heading, repoInfoMap);
566
+ stack.push({
567
+ children: [],
568
+ description: '',
569
+ headingDepth: heading.depth,
570
+ kind: repoInfo ? 'item' : 'group',
571
+ repoInfo: repoInfo ?? undefined,
572
+ title,
573
+ });
574
+ }
575
+ // Finalize every container a heading of `depth` closes (same-or-shallower),
576
+ // bottom-up so each finalized node lands in its parent. Pruning falls out of
577
+ // the finalize rule: a section/group whose children array is empty (no items
578
+ // anywhere beneath — lists only return item-bearing nodes, and empty children
579
+ // were never appended) is dropped; an item always survives, it IS the content.
580
+ // The stack bottom is always a section (openContainer's promotion guarantees
581
+ // it), so a finalized group/item always has a parent to land in.
582
+ function closeContainers(stack, depth, sections) {
583
+ while (stack.length > 0 &&
584
+ stack[stack.length - 1].headingDepth >= depth) {
585
+ const container = stack.pop();
586
+ if (container.children.length === 0 && container.kind !== 'item') {
587
+ continue;
588
+ }
589
+ const description = container.description || null;
590
+ if (container.kind === 'section') {
591
+ sections.push({
592
+ description,
593
+ items: container.children,
594
+ title: container.title,
595
+ });
596
+ continue;
597
+ }
598
+ const parent = stack[stack.length - 1];
599
+ // Non-section containers always have an open parent (stack invariant), and
600
+ // kind === 'item' exactly when repoInfo is set.
601
+ if (container.repoInfo) {
602
+ parent.children.push({
603
+ children: container.children,
604
+ description,
605
+ node_type: 'item',
606
+ repo_info: toRepoInfo(container.repoInfo),
607
+ title: container.title,
608
+ });
609
+ }
610
+ else {
611
+ parent.children.push({
612
+ children: container.children,
613
+ description,
614
+ node_type: 'group',
615
+ title: container.title,
616
+ });
617
+ }
618
+ }
619
+ }
500
620
  function serializeAst(tree, originalContent) {
501
621
  let finalContent = unified()
502
622
  .use(remarkStringify)
@@ -9,6 +9,7 @@ export interface EnhanceOptions {
9
9
  log?: Logger;
10
10
  now?: Date;
11
11
  originalRepository: string;
12
+ originalRepositoryId?: number;
12
13
  originalRepositorySha?: string;
13
14
  relativeLinkPrefix?: string;
14
15
  /** Text substitutions applied to the source before it is parsed. */
@@ -1,6 +1,6 @@
1
1
  import { processMarkdownContent, } from './markdown.js';
2
2
  export async function enhance(options) {
3
- const { content, disableBranding = false, log, now = new Date(), originalRepository, originalRepositorySha, relativeLinkPrefix = '', replacements = [], sortBy = '', enhancedRepository, enhancedRepositoryDescription, token, } = options;
3
+ const { content, disableBranding = false, log, now = new Date(), originalRepository, originalRepositoryId, originalRepositorySha, relativeLinkPrefix = '', replacements = [], sortBy = '', enhancedRepository, enhancedRepositoryDescription, token, } = options;
4
4
  // Branding is an internal rule prepended to the caller's own; build a fresh
5
5
  // array so the caller's `replacements` is never mutated.
6
6
  const branding = { type: 'branding' };
@@ -9,7 +9,7 @@ export async function enhance(options) {
9
9
  by: sortBy,
10
10
  minLinks: 2,
11
11
  };
12
- const { finalContent, jsonData } = await processMarkdownContent(content, token, rules, sortOptions, originalRepository, relativeLinkPrefix, enhancedRepository, enhancedRepositoryDescription, originalRepositorySha, now, log);
12
+ const { finalContent, jsonData } = await processMarkdownContent(content, token, rules, sortOptions, originalRepository, relativeLinkPrefix, enhancedRepository, enhancedRepositoryDescription, originalRepositorySha, originalRepositoryId, now, log);
13
13
  return {
14
14
  finalContent,
15
15
  jsonData,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@enhansome/core",
3
- "version": "1.7.1",
3
+ "version": "1.8.1",
4
4
  "description": "Library core for enhansome — enhance markdown with GitHub star counts.",
5
5
  "repository": {
6
6
  "type": "git",