@enhansome/core 1.10.2 → 1.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/first-seen.js +1 -1
- package/dist/github.d.ts +9 -0
- package/dist/github.js +36 -2
- package/dist/index.d.ts +2 -2
- package/dist/index.js +2 -2
- package/dist/internal-links.d.ts +21 -0
- package/dist/internal-links.js +112 -0
- package/dist/markdown.d.ts +10 -3
- package/dist/markdown.js +367 -45
- package/dist/orchestrator.d.ts +2 -1
- package/dist/orchestrator.js +2 -2
- package/package.json +1 -1
package/dist/first-seen.js
CHANGED
|
@@ -9,7 +9,7 @@ export function firstSeenFor(previous, now) {
|
|
|
9
9
|
}
|
|
10
10
|
function collect(nodes, index) {
|
|
11
11
|
for (const node of nodes) {
|
|
12
|
-
if (node.node_type === 'item' && node.first_seen) {
|
|
12
|
+
if (node.node_type === 'item' && node.repo_info && node.first_seen) {
|
|
13
13
|
const existing = index.get(node.repo_info.id);
|
|
14
14
|
if (existing === undefined || node.first_seen < existing) {
|
|
15
15
|
index.set(node.repo_info.id, node.first_seen);
|
package/dist/github.d.ts
CHANGED
|
@@ -8,8 +8,10 @@ export type GithubClient = InstanceType<typeof HardenedOctokit>;
|
|
|
8
8
|
export interface RepoInfoDetails {
|
|
9
9
|
archived: boolean;
|
|
10
10
|
description: null | string;
|
|
11
|
+
homepage: null | string;
|
|
11
12
|
id: number;
|
|
12
13
|
language: null | string;
|
|
14
|
+
license: null | string;
|
|
13
15
|
open_issues_count: number;
|
|
14
16
|
owner: string;
|
|
15
17
|
pushed_at: null | string;
|
|
@@ -84,6 +86,13 @@ export declare function getRepoInfo(octokit: GithubClient, owner: string, repo:
|
|
|
84
86
|
*/
|
|
85
87
|
export declare function getRepoInfoOrNull(octokit: GithubClient, owner: string, repo: string): Promise<null | RepoInfoDetails>;
|
|
86
88
|
export declare function getReadme(octokit: GithubClient, owner: string, repo: string, format?: 'html' | 'raw'): Promise<string>;
|
|
89
|
+
/**
|
|
90
|
+
* One in-repo markdown file, for following a README's links into the source's
|
|
91
|
+
* content files. Returns null (logged) instead of failing the run: a followed
|
|
92
|
+
* file is best-effort the way repo-info failures are — only the README fetch
|
|
93
|
+
* itself can fail the run.
|
|
94
|
+
*/
|
|
95
|
+
export declare function getRepoFileOrNull(octokit: GithubClient, owner: string, repo: string, path: string): Promise<null | string>;
|
|
87
96
|
/** Root file/directory names — the compile-manifest gate reads them to tell a
|
|
88
97
|
* directory of resources from a repo that IS the deliverable. A `path: ''`
|
|
89
98
|
* listing is always an array; the single-entry branch defends against a file
|
package/dist/github.js
CHANGED
|
@@ -72,8 +72,10 @@ export async function getRepoInfo(octokit, owner, repo) {
|
|
|
72
72
|
return {
|
|
73
73
|
archived: data.archived,
|
|
74
74
|
description: data.description ?? null,
|
|
75
|
+
homepage: data.homepage || null,
|
|
75
76
|
id: data.id,
|
|
76
77
|
language: data.language,
|
|
78
|
+
license: data.license?.spdx_id ?? null,
|
|
77
79
|
open_issues_count: data.open_issues_count,
|
|
78
80
|
owner: data.owner.login,
|
|
79
81
|
pushed_at: data.pushed_at,
|
|
@@ -108,6 +110,28 @@ export async function getReadme(octokit, owner, repo, format = 'raw') {
|
|
|
108
110
|
// still describe the JSON shape - cast to string.
|
|
109
111
|
return response.data;
|
|
110
112
|
}
|
|
113
|
+
/**
|
|
114
|
+
* One in-repo markdown file, for following a README's links into the source's
|
|
115
|
+
* content files. Returns null (logged) instead of failing the run: a followed
|
|
116
|
+
* file is best-effort the way repo-info failures are — only the README fetch
|
|
117
|
+
* itself can fail the run.
|
|
118
|
+
*/
|
|
119
|
+
export async function getRepoFileOrNull(octokit, owner, repo, path) {
|
|
120
|
+
try {
|
|
121
|
+
octokit.log.debug(`Fetching ${owner}/${repo}/${path}`);
|
|
122
|
+
const { data } = await octokit.rest.repos.getContent({
|
|
123
|
+
mediaType: { format: 'raw' },
|
|
124
|
+
owner,
|
|
125
|
+
path,
|
|
126
|
+
repo,
|
|
127
|
+
});
|
|
128
|
+
return data;
|
|
129
|
+
}
|
|
130
|
+
catch (error) {
|
|
131
|
+
octokit.log.warn(`Failed to fetch ${owner}/${repo}/${path}: ${formatRequestError(error)}`);
|
|
132
|
+
return null;
|
|
133
|
+
}
|
|
134
|
+
}
|
|
111
135
|
/** Root file/directory names — the compile-manifest gate reads them to tell a
|
|
112
136
|
* directory of resources from a repo that IS the deliverable. A `path: ''`
|
|
113
137
|
* listing is always an array; the single-entry branch defends against a file
|
|
@@ -151,7 +175,17 @@ export function parseOwnerRepo(value) {
|
|
|
151
175
|
if (parts.length !== 2) {
|
|
152
176
|
return null;
|
|
153
177
|
}
|
|
154
|
-
return {
|
|
178
|
+
return {
|
|
179
|
+
owner: stripZeroWidth(parts[0]),
|
|
180
|
+
repo: stripZeroWidth(parts[1]).replace(/\.git$/, ''),
|
|
181
|
+
};
|
|
182
|
+
}
|
|
183
|
+
// Zero-width characters glued to a repo name (awesome-computer-vision links
|
|
184
|
+
// neuraltalk with an invisible BOM) are not identity; URL.pathname keeps them
|
|
185
|
+
// percent-encoded, so both spellings are stripped before the name is used.
|
|
186
|
+
const ZERO_WIDTH = /(?:%EF%BB%BF|%E2%80%8[B-F]|%E2%81%A0)|[\u200B-\u200F\u2060\uFEFF]/gi;
|
|
187
|
+
function stripZeroWidth(segment) {
|
|
188
|
+
return segment.replace(ZERO_WIDTH, '');
|
|
155
189
|
}
|
|
156
190
|
export function parseGitHubUrl(url) {
|
|
157
191
|
try {
|
|
@@ -163,7 +197,7 @@ export function parseGitHubUrl(url) {
|
|
|
163
197
|
.split('/')
|
|
164
198
|
.filter(part => part.length > 0);
|
|
165
199
|
if (pathParts.length >= 2) {
|
|
166
|
-
const owner = pathParts[0], repo = pathParts[1].replace(/\.git$/, '');
|
|
200
|
+
const owner = stripZeroWidth(pathParts[0]), repo = stripZeroWidth(pathParts[1]).replace(/\.git$/, '');
|
|
167
201
|
return { owner, repo };
|
|
168
202
|
}
|
|
169
203
|
return null;
|
package/dist/index.d.ts
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
|
-
export { formatRequestError, getLatestCommitSha, getReadme, getRepoInfo, getRepoInfoOrNull, getRootEntryNames, makeOctokit, parseGitHubUrl, parseOwnerRepo, } from './github.js';
|
|
1
|
+
export { formatRequestError, getLatestCommitSha, getReadme, getRepoFileOrNull, getRepoInfo, getRepoInfoOrNull, getRootEntryNames, makeOctokit, parseGitHubUrl, parseOwnerRepo, } from './github.js';
|
|
2
2
|
export type { GithubClient, MakeOctokitOptions, RepoIdentifier, RepoInfoDetails, ThrottleOptions, } from './github.js';
|
|
3
3
|
export type { Logger } from './logger.js';
|
|
4
4
|
export { consoleLog, silentLog } from './logger.js';
|
|
5
|
-
export { toRepoInfo } from './markdown.js';
|
|
5
|
+
export { countItems, hollowsPrevious, toRepoInfo } from './markdown.js';
|
|
6
6
|
export type { JsonGroup, JsonItem, JsonMetadata, JsonNode, JsonOutput, JsonSection, ReplacementRule, RepoInfo, } from './markdown.js';
|
|
7
7
|
export { enhance } from './orchestrator.js';
|
|
8
8
|
export type { EnhanceOptions, EnhanceResult } from './orchestrator.js';
|
package/dist/index.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
export { formatRequestError, getLatestCommitSha, getReadme, getRepoInfo, getRepoInfoOrNull, getRootEntryNames, makeOctokit, parseGitHubUrl, parseOwnerRepo, } from './github.js';
|
|
1
|
+
export { formatRequestError, getLatestCommitSha, getReadme, getRepoFileOrNull, getRepoInfo, getRepoInfoOrNull, getRootEntryNames, makeOctokit, parseGitHubUrl, parseOwnerRepo, } from './github.js';
|
|
2
2
|
export { consoleLog, silentLog } from './logger.js';
|
|
3
|
-
export { toRepoInfo } from './markdown.js';
|
|
3
|
+
export { countItems, hollowsPrevious, toRepoInfo } from './markdown.js';
|
|
4
4
|
export { enhance } from './orchestrator.js';
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
/** Bounds for one follow run: files fetched, and bytes per file. */
|
|
2
|
+
export declare const MAX_FOLLOWED_FILES = 50;
|
|
3
|
+
export declare const MAX_FOLLOWED_FILE_BYTES = 1000000;
|
|
4
|
+
/** True for a resolved repo path that must never be followed or inlined. */
|
|
5
|
+
export declare function isSkippedPath(path: string): boolean;
|
|
6
|
+
/**
|
|
7
|
+
* Resolves one link target against the referring document's directory.
|
|
8
|
+
* Accepts repo-relative (`docs/x.md`), referring-file-relative
|
|
9
|
+
* (`../guides/y.md`) and repo-root-absolute (`/x.md`) forms. Returns null for
|
|
10
|
+
* anything that is not an in-repo path: external URLs, anchors, non-markdown
|
|
11
|
+
* files, or `..` escaping the repo root.
|
|
12
|
+
*/
|
|
13
|
+
export declare function resolveRepoPath(target: string, baseDir: string): null | string;
|
|
14
|
+
/**
|
|
15
|
+
* Extracts the in-repo markdown path from a same-repo absolute GitHub URL
|
|
16
|
+
* (`https://github.com/owner/repo/blob/main/docs/x.md`), or null for links to
|
|
17
|
+
* any other host/repo.
|
|
18
|
+
*/
|
|
19
|
+
export declare function parseSameRepoBlobPath(url: string, owner: string, repo: string): null | string;
|
|
20
|
+
/** True when a link's visible text is just a filename/path, not a category name. */
|
|
21
|
+
export declare function isFilenameLabel(label: string, path: string): boolean;
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
// Path rules for following a README's links into other markdown files of the
|
|
2
|
+
// SAME repo — the "content moved out of the README" shape (android-root moved
|
|
3
|
+
// 600+ entries into docs/apps-and-modules/*.md and left a landing page).
|
|
4
|
+
// Pure string/path logic only; fetching and AST splicing live in markdown.ts.
|
|
5
|
+
/** Bounds for one follow run: files fetched, and bytes per file. */
|
|
6
|
+
export const MAX_FOLLOWED_FILES = 50;
|
|
7
|
+
export const MAX_FOLLOWED_FILE_BYTES = 1_000_000;
|
|
8
|
+
const MARKDOWN_EXT = /\.(md|markdown)$/i;
|
|
9
|
+
// Repo-convention meta documents. Only honored at the repo root and under
|
|
10
|
+
// .github/ — elsewhere a same-named file is content: android-root's
|
|
11
|
+
// docs/apps-and-modules/security.md is the Security category, and a basename
|
|
12
|
+
// rule would silently drop it.
|
|
13
|
+
const META_BASENAMES = new Set([
|
|
14
|
+
'acknowledgements',
|
|
15
|
+
'acknowledgments',
|
|
16
|
+
'authors',
|
|
17
|
+
'changes',
|
|
18
|
+
'changelog',
|
|
19
|
+
'citation',
|
|
20
|
+
'code of conduct',
|
|
21
|
+
'code_of_conduct',
|
|
22
|
+
'code-of-conduct',
|
|
23
|
+
'conduct',
|
|
24
|
+
'contributing',
|
|
25
|
+
'contributors',
|
|
26
|
+
'funding',
|
|
27
|
+
'history',
|
|
28
|
+
'issue_template',
|
|
29
|
+
'license',
|
|
30
|
+
'licence',
|
|
31
|
+
'notice',
|
|
32
|
+
'pull_request_template',
|
|
33
|
+
'security',
|
|
34
|
+
'support',
|
|
35
|
+
'todo',
|
|
36
|
+
]);
|
|
37
|
+
// The root README and its translations/self-links are alternate views of the
|
|
38
|
+
// document being enhanced, never additional content — inlining one would
|
|
39
|
+
// duplicate the whole list. Directory READMEs (pages/RISH.md aside,
|
|
40
|
+
// CLI/README.md shapes) are content and stay followable.
|
|
41
|
+
const ROOT_README = /^readme[^/]*\.(md|markdown)$/i;
|
|
42
|
+
function baseName(path) {
|
|
43
|
+
const tail = path.slice(path.lastIndexOf('/') + 1);
|
|
44
|
+
const dot = tail.lastIndexOf('.');
|
|
45
|
+
return (dot > 0 ? tail.slice(0, dot) : tail).toLowerCase();
|
|
46
|
+
}
|
|
47
|
+
/** True for a resolved repo path that must never be followed or inlined. */
|
|
48
|
+
export function isSkippedPath(path) {
|
|
49
|
+
if (!MARKDOWN_EXT.test(path)) {
|
|
50
|
+
return true;
|
|
51
|
+
}
|
|
52
|
+
if (!path.includes('/')) {
|
|
53
|
+
return ROOT_README.test(path) || META_BASENAMES.has(baseName(path));
|
|
54
|
+
}
|
|
55
|
+
return path.startsWith('.github/') && META_BASENAMES.has(baseName(path));
|
|
56
|
+
}
|
|
57
|
+
/**
|
|
58
|
+
* Resolves one link target against the referring document's directory.
|
|
59
|
+
* Accepts repo-relative (`docs/x.md`), referring-file-relative
|
|
60
|
+
* (`../guides/y.md`) and repo-root-absolute (`/x.md`) forms. Returns null for
|
|
61
|
+
* anything that is not an in-repo path: external URLs, anchors, non-markdown
|
|
62
|
+
* files, or `..` escaping the repo root.
|
|
63
|
+
*/
|
|
64
|
+
export function resolveRepoPath(target, baseDir) {
|
|
65
|
+
const stripped = target.split('#')[0].split('?')[0].trim();
|
|
66
|
+
if (!stripped || stripped.includes('://') || stripped.startsWith('mailto:')) {
|
|
67
|
+
return null;
|
|
68
|
+
}
|
|
69
|
+
const segments = (stripped.startsWith('/') ? [] : baseDir.split('/')).concat(stripped.split('/'));
|
|
70
|
+
const stack = [];
|
|
71
|
+
for (const segment of segments) {
|
|
72
|
+
if (segment === '' || segment === '.') {
|
|
73
|
+
continue;
|
|
74
|
+
}
|
|
75
|
+
if (segment === '..') {
|
|
76
|
+
if (stack.length === 0) {
|
|
77
|
+
return null;
|
|
78
|
+
}
|
|
79
|
+
stack.pop();
|
|
80
|
+
continue;
|
|
81
|
+
}
|
|
82
|
+
stack.push(segment);
|
|
83
|
+
}
|
|
84
|
+
return stack.length > 0 ? stack.join('/') : null;
|
|
85
|
+
}
|
|
86
|
+
/**
|
|
87
|
+
* Extracts the in-repo markdown path from a same-repo absolute GitHub URL
|
|
88
|
+
* (`https://github.com/owner/repo/blob/main/docs/x.md`), or null for links to
|
|
89
|
+
* any other host/repo.
|
|
90
|
+
*/
|
|
91
|
+
export function parseSameRepoBlobPath(url, owner, repo) {
|
|
92
|
+
const match = /^https?:\/\/(?:www\.)?github\.com\/([A-Za-z0-9_.-]+)\/([A-Za-z0-9_.-]+)\/(?:blob|raw)\/[^/]+\/(.+)$/i.exec(url);
|
|
93
|
+
if (!match) {
|
|
94
|
+
return null;
|
|
95
|
+
}
|
|
96
|
+
if (`${match[1]}/${match[2]}`.toLowerCase() !== `${owner}/${repo}`.toLowerCase()) {
|
|
97
|
+
return null;
|
|
98
|
+
}
|
|
99
|
+
return decodeURIComponent(match[3]);
|
|
100
|
+
}
|
|
101
|
+
/** True when a link's visible text is just a filename/path, not a category name. */
|
|
102
|
+
export function isFilenameLabel(label, path) {
|
|
103
|
+
const text = label.trim().toLowerCase();
|
|
104
|
+
if (!text) {
|
|
105
|
+
return true;
|
|
106
|
+
}
|
|
107
|
+
const file = path.slice(path.lastIndexOf('/') + 1).toLowerCase();
|
|
108
|
+
return (text === file ||
|
|
109
|
+
text === path.toLowerCase() ||
|
|
110
|
+
text.includes('/') ||
|
|
111
|
+
MARKDOWN_EXT.test(text));
|
|
112
|
+
}
|
package/dist/markdown.d.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { RepoInfoDetails } from './github.js';
|
|
1
|
+
import { RepoIdentifier, RepoInfoDetails } from './github.js';
|
|
2
2
|
import { Logger } from './logger.js';
|
|
3
3
|
import type { Heading, Link, List, Parent, Root } from 'mdast';
|
|
4
4
|
export interface JsonOutput {
|
|
@@ -18,12 +18,17 @@ export interface SortOptions {
|
|
|
18
18
|
}
|
|
19
19
|
export interface RepoInfo {
|
|
20
20
|
archived: boolean;
|
|
21
|
+
description?: null | string;
|
|
22
|
+
homepage?: null | string;
|
|
21
23
|
id: number;
|
|
22
24
|
language: null | string;
|
|
25
|
+
license?: null | string;
|
|
23
26
|
last_commit: null | string;
|
|
27
|
+
open_issues?: number;
|
|
24
28
|
owner: string;
|
|
25
29
|
repo: string;
|
|
26
30
|
stars: number;
|
|
31
|
+
topics?: string[];
|
|
27
32
|
}
|
|
28
33
|
/** The sole crossing from the API shape to the emitted one, so both agree on the renames. */
|
|
29
34
|
export declare function toRepoInfo(details: RepoInfoDetails): RepoInfo;
|
|
@@ -32,7 +37,7 @@ export interface JsonItem {
|
|
|
32
37
|
description: null | string;
|
|
33
38
|
first_seen?: string;
|
|
34
39
|
node_type: 'item';
|
|
35
|
-
repo_info
|
|
40
|
+
repo_info?: RepoInfo;
|
|
36
41
|
title: string;
|
|
37
42
|
}
|
|
38
43
|
export interface JsonGroup {
|
|
@@ -55,7 +60,9 @@ export interface JsonSection {
|
|
|
55
60
|
items: JsonNode[];
|
|
56
61
|
title: string;
|
|
57
62
|
}
|
|
58
|
-
export declare function
|
|
63
|
+
export declare function countItems(output: JsonOutput): number;
|
|
64
|
+
export declare function hollowsPrevious(previous: JsonOutput, next: JsonOutput): boolean;
|
|
65
|
+
export declare function processMarkdownContent(originalContent: string, token: string, replacements?: ReplacementRule[], sortOptions?: SortOptions, relativeLinkPrefix?: string, enhancedRepository?: string, enhancedRepositoryDescription?: string, originalRepositorySha?: string, originalRepositoryInfo?: null | RepoInfoDetails, sourceRepository?: RepoIdentifier, previousJson?: JsonOutput, now?: Date, log?: Logger): Promise<{
|
|
59
66
|
finalContent: string;
|
|
60
67
|
jsonData: JsonOutput;
|
|
61
68
|
}>;
|
package/dist/markdown.js
CHANGED
|
@@ -5,20 +5,39 @@ import remarkStringify from 'remark-stringify';
|
|
|
5
5
|
import { unified } from 'unified';
|
|
6
6
|
import { visit } from 'unist-util-visit';
|
|
7
7
|
import { firstSeenFor } from './first-seen.js';
|
|
8
|
-
import { formatRequestError, getRepoInfo as fetchRepoInfo, makeOctokit, parseGitHubUrl, } from './github.js';
|
|
8
|
+
import { formatRequestError, getRepoFileOrNull, getRepoInfo as fetchRepoInfo, makeOctokit, parseGitHubUrl, } from './github.js';
|
|
9
|
+
import { isFilenameLabel, isSkippedPath, MAX_FOLLOWED_FILE_BYTES, MAX_FOLLOWED_FILES, parseSameRepoBlobPath, resolveRepoPath, } from './internal-links.js';
|
|
9
10
|
import { consoleLog } from './logger.js';
|
|
10
11
|
/** The sole crossing from the API shape to the emitted one, so both agree on the renames. */
|
|
11
12
|
export function toRepoInfo(details) {
|
|
12
13
|
return {
|
|
13
14
|
archived: details.archived,
|
|
15
|
+
description: details.description,
|
|
16
|
+
homepage: details.homepage,
|
|
14
17
|
id: details.id,
|
|
15
18
|
language: details.language,
|
|
19
|
+
license: details.license,
|
|
16
20
|
last_commit: details.pushed_at,
|
|
21
|
+
open_issues: details.open_issues_count,
|
|
17
22
|
owner: details.owner,
|
|
18
23
|
repo: details.repo,
|
|
19
24
|
stars: details.stargazers_count,
|
|
25
|
+
topics: details.topics,
|
|
20
26
|
};
|
|
21
27
|
}
|
|
28
|
+
export function countItems(output) {
|
|
29
|
+
const count = (nodes) => nodes.reduce((sum, node) => sum + (node.node_type === 'item' ? 1 : 0) + count(node.children), 0);
|
|
30
|
+
return count(output.items.flatMap(section => section.items));
|
|
31
|
+
}
|
|
32
|
+
// A >95% item collapse against the previous run is a parse regression or a
|
|
33
|
+
// source restructure, never a real edit — the caller refuses to write rather
|
|
34
|
+
// than hollow the mirror (android-root committed a skeleton daily for three
|
|
35
|
+
// days before this guard existed).
|
|
36
|
+
const HOLLOW_SHRINKAGE = 0.05;
|
|
37
|
+
export function hollowsPrevious(previous, next) {
|
|
38
|
+
const before = countItems(previous);
|
|
39
|
+
return before > 0 && countItems(next) < before * HOLLOW_SHRINKAGE;
|
|
40
|
+
}
|
|
22
41
|
// The lookup's single throttled client coordinates rate limits across the whole
|
|
23
42
|
// pool, so this bounds in-flight targets rather than requests.
|
|
24
43
|
const FETCH_CONCURRENCY = 10;
|
|
@@ -57,9 +76,9 @@ function createRepoInfoLookup(token, log) {
|
|
|
57
76
|
* collapse to a single fetch inside the lookup, then fan back out to every alias
|
|
58
77
|
* here.
|
|
59
78
|
*
|
|
60
|
-
* Each target's failure is independent and non-fatal: a dead link
|
|
61
|
-
*
|
|
62
|
-
*
|
|
79
|
+
* Each target's failure is independent and non-fatal: a dead link drops its
|
|
80
|
+
* entry and lifts the children. Only the source README fetch in main.ts can
|
|
81
|
+
* fail the run.
|
|
63
82
|
*/
|
|
64
83
|
async function fetchTargetData(urls, repos) {
|
|
65
84
|
const log = repos.client.log;
|
|
@@ -80,15 +99,169 @@ async function fetchTargetData(urls, repos) {
|
|
|
80
99
|
log.info(`Target fetch: ${repoInfoMap.size}/${urls.size} repo-info ok in ${Date.now() - fetchStart}ms (concurrency ${FETCH_CONCURRENCY}).`);
|
|
81
100
|
return repoInfoMap;
|
|
82
101
|
}
|
|
83
|
-
|
|
102
|
+
// The same-repo markdown links of one document: resolved relative and
|
|
103
|
+
// root-absolute targets, plus same-repo absolute blob/raw URLs. Labels ride
|
|
104
|
+
// along because they become the spliced category's title.
|
|
105
|
+
function collectDocLinks(tree, baseDir, source) {
|
|
106
|
+
const links = [];
|
|
107
|
+
visit(tree, 'link', (node) => {
|
|
108
|
+
const path = resolveRepoPath(node.url, baseDir) ??
|
|
109
|
+
parseSameRepoBlobPath(node.url, source.owner, source.repo);
|
|
110
|
+
if (!path || isSkippedPath(path)) {
|
|
111
|
+
return;
|
|
112
|
+
}
|
|
113
|
+
links.push({ label: getInlineText(node.children), path });
|
|
114
|
+
});
|
|
115
|
+
return links;
|
|
116
|
+
}
|
|
117
|
+
const FRONTMATTER = /^---\r?\n[\s\S]*?\r?\n---\r?\n?/;
|
|
118
|
+
// VitePress/Docusaurus content files open with YAML frontmatter that remark
|
|
119
|
+
// would misread as a thematic break plus a setext heading.
|
|
120
|
+
function stripFrontmatter(content) {
|
|
121
|
+
return content.replace(FRONTMATTER, '');
|
|
122
|
+
}
|
|
123
|
+
// Removes the file's leading H1 — the injected category heading replaces it,
|
|
124
|
+
// so the document never carries the same title twice — and returns its text
|
|
125
|
+
// for the filename-label fallback.
|
|
126
|
+
function stripTitleHeading(tree) {
|
|
127
|
+
const first = tree.children[0];
|
|
128
|
+
if (first?.type === 'heading' && first.depth === 1) {
|
|
129
|
+
tree.children.shift();
|
|
130
|
+
return getNodeText(first);
|
|
131
|
+
}
|
|
132
|
+
return '';
|
|
133
|
+
}
|
|
134
|
+
// One level shallower than the file's shallowest remaining heading, so the
|
|
135
|
+
// file's own sections nest inside the category; a heading-less file takes the
|
|
136
|
+
// conventional section level of 2.
|
|
137
|
+
function categoryHeadingDepth(children) {
|
|
138
|
+
let shallowest = Infinity;
|
|
139
|
+
for (const node of children) {
|
|
140
|
+
if (node.type === 'heading') {
|
|
141
|
+
shallowest = Math.min(shallowest, node.depth);
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
// shallowest is 1..6, so the clamped result stays inside the depth union.
|
|
145
|
+
return shallowest === Infinity
|
|
146
|
+
? 2
|
|
147
|
+
: Math.max(1, shallowest - 1);
|
|
148
|
+
}
|
|
149
|
+
// Relative links and images resolve against the source repo's tree; in the
|
|
150
|
+
// mirror they dangle. Rewritten to GitHub URLs at HEAD.
|
|
151
|
+
function rewriteRelativeLinksToSource(tree, baseDir, source) {
|
|
152
|
+
const absolute = (url) => {
|
|
153
|
+
if (!url ||
|
|
154
|
+
url.startsWith('#') ||
|
|
155
|
+
url.includes('://') ||
|
|
156
|
+
url.startsWith('mailto:')) {
|
|
157
|
+
return url;
|
|
158
|
+
}
|
|
159
|
+
const path = resolveRepoPath(url, baseDir);
|
|
160
|
+
return path
|
|
161
|
+
? `https://github.com/${source.owner}/${source.repo}/blob/HEAD/${path}`
|
|
162
|
+
: url;
|
|
163
|
+
};
|
|
164
|
+
visit(tree, 'link', (node) => {
|
|
165
|
+
node.url = absolute(node.url);
|
|
166
|
+
});
|
|
167
|
+
visit(tree, 'image', (node) => {
|
|
168
|
+
node.url = absolute(node.url);
|
|
169
|
+
});
|
|
170
|
+
}
|
|
171
|
+
/**
|
|
172
|
+
* Appends the source's internal markdown files to the tree, each under a
|
|
173
|
+
* heading titled by its link's text — the README-is-a-landing-page shape.
|
|
174
|
+
* Breadth-first in document order with a visited set, so a file linked from
|
|
175
|
+
* both the README and an index page is fetched and spliced once, the README's
|
|
176
|
+
* label winning. A file joins the document only when it carries at least
|
|
177
|
+
* `minLinks` GitHub links; navigation pages are still parsed for their links,
|
|
178
|
+
* which is where this shape often keeps them.
|
|
179
|
+
*/
|
|
180
|
+
async function appendInternalDocs(tree, source, repos, minLinks, log) {
|
|
181
|
+
const processor = unified().use(remarkParse).use(remarkGfm);
|
|
182
|
+
const queue = collectDocLinks(tree, '', source);
|
|
183
|
+
const visited = new Set();
|
|
184
|
+
const labels = new Map();
|
|
185
|
+
let fetched = 0;
|
|
186
|
+
let appended = 0;
|
|
187
|
+
let processed = 0;
|
|
188
|
+
while (queue.length > 0 && processed < MAX_FOLLOWED_FILES) {
|
|
189
|
+
const link = queue.shift();
|
|
190
|
+
if (visited.has(link.path)) {
|
|
191
|
+
continue;
|
|
192
|
+
}
|
|
193
|
+
processed += 1;
|
|
194
|
+
visited.add(link.path);
|
|
195
|
+
if (link.label.trim() && !labels.has(link.path)) {
|
|
196
|
+
labels.set(link.path, link.label);
|
|
197
|
+
}
|
|
198
|
+
const raw = await getRepoFileOrNull(repos.client, source.owner, source.repo, link.path);
|
|
199
|
+
if (!raw) {
|
|
200
|
+
continue;
|
|
201
|
+
}
|
|
202
|
+
fetched += 1;
|
|
203
|
+
if (raw.length > MAX_FOLLOWED_FILE_BYTES) {
|
|
204
|
+
log.warn(`Skipping ${link.path}: over ${MAX_FOLLOWED_FILE_BYTES} bytes.`);
|
|
205
|
+
continue;
|
|
206
|
+
}
|
|
207
|
+
const fileTree = processor.parse(stripFrontmatter(raw));
|
|
208
|
+
const ownTitle = stripTitleHeading(fileTree);
|
|
209
|
+
const baseDir = link.path.includes('/')
|
|
210
|
+
? link.path.slice(0, link.path.lastIndexOf('/'))
|
|
211
|
+
: '';
|
|
212
|
+
for (const nested of collectDocLinks(fileTree, baseDir, source)) {
|
|
213
|
+
if (!visited.has(nested.path)) {
|
|
214
|
+
queue.push(nested);
|
|
215
|
+
}
|
|
216
|
+
}
|
|
217
|
+
if (countGitHubRepos(fileTree) < minLinks) {
|
|
218
|
+
continue;
|
|
219
|
+
}
|
|
220
|
+
const label = labels.get(link.path) ?? '';
|
|
221
|
+
const title = isFilenameLabel(label, link.path)
|
|
222
|
+
? ownTitle || label || link.path
|
|
223
|
+
: label;
|
|
224
|
+
const heading = {
|
|
225
|
+
type: 'heading',
|
|
226
|
+
depth: categoryHeadingDepth(fileTree.children),
|
|
227
|
+
children: [{ type: 'text', value: title }],
|
|
228
|
+
};
|
|
229
|
+
rewriteRelativeLinksToSource(fileTree, baseDir, source);
|
|
230
|
+
tree.children.push(heading, ...fileTree.children);
|
|
231
|
+
appended += 1;
|
|
232
|
+
}
|
|
233
|
+
rewriteRelativeLinksToSource(tree, '', source);
|
|
234
|
+
log.info(`README skeleton: followed ${fetched} internal file(s), appended ${appended} as categories.`);
|
|
235
|
+
}
|
|
236
|
+
export async function processMarkdownContent(originalContent, token, replacements = [], sortOptions = { by: '', minLinks: 2 }, relativeLinkPrefix = '', enhancedRepository, enhancedRepositoryDescription, originalRepositorySha, originalRepositoryInfo, sourceRepository, previousJson, now = new Date(), log = consoleLog) {
|
|
84
237
|
const repos = createRepoInfoLookup(token, log);
|
|
85
238
|
const brandingEnabled = replacements.some(rule => rule.type === 'branding');
|
|
86
239
|
const contentAfterReplacements = applyTextReplacements(originalContent, replacements.filter(rule => rule.type !== 'branding'), log);
|
|
87
240
|
const processor = unified().use(remarkParse).use(remarkGfm);
|
|
88
241
|
const tree = processor.parse(contentAfterReplacements);
|
|
89
242
|
normalizeGitHubUrls(tree);
|
|
243
|
+
// A README too thin to carry the list is a landing page: the content lives
|
|
244
|
+
// in the files it links. Splice them in before anything reads the tree, so
|
|
245
|
+
// the repo fetch, badges and the section walk all see the combined document.
|
|
246
|
+
if (sourceRepository && countGitHubRepos(tree) < sortOptions.minLinks) {
|
|
247
|
+
await appendInternalDocs(tree, sourceRepository, repos, sortOptions.minLinks, log);
|
|
248
|
+
normalizeGitHubUrls(tree);
|
|
249
|
+
}
|
|
90
250
|
const githubUrls = collectGitHubLinks(tree);
|
|
91
|
-
|
|
251
|
+
// Links into the source repository itself are navigation — its doc pages,
|
|
252
|
+
// its contributing links — never entries. android-root's index links every
|
|
253
|
+
// category page, and each of those links would borrow the source repo's
|
|
254
|
+
// identity and emit an item for it. Excluding the source from the target
|
|
255
|
+
// fetch leaves every emission path reading such a link as a dead one: no
|
|
256
|
+
// item, children lifted, the row left in markdown.
|
|
257
|
+
const targetUrls = sourceRepository &&
|
|
258
|
+
new Set([...githubUrls].filter(url => {
|
|
259
|
+
const id = parseGitHubUrl(url);
|
|
260
|
+
return !(id &&
|
|
261
|
+
`${id.owner}/${id.repo}`.toLowerCase() ===
|
|
262
|
+
`${sourceRepository.owner}/${sourceRepository.repo}`.toLowerCase());
|
|
263
|
+
}));
|
|
264
|
+
const repoInfoMap = await fetchTargetData(targetUrls ?? githubUrls, repos);
|
|
92
265
|
const firstSeen = firstSeenFor(previousJson, now);
|
|
93
266
|
const { sections, title: rawTitle, titleHeadingIndex, } = processTree(tree, repoInfoMap, sortOptions, originalRepositoryInfo, firstSeen);
|
|
94
267
|
// Single source of truth for the document title: brand it once and use the
|
|
@@ -176,8 +349,31 @@ function collectGitHubLinks(tree) {
|
|
|
176
349
|
urls.add(node.url);
|
|
177
350
|
}
|
|
178
351
|
});
|
|
352
|
+
// A details summary's anchor stays raw html (normalizeGitHubUrls leaves
|
|
353
|
+
// block html alone), so the identity it names joins the fetch set here;
|
|
354
|
+
// the summary promotion looks the repo up by the same href.
|
|
355
|
+
visit(tree, 'html', (node) => {
|
|
356
|
+
const identity = parseDetailsSummary(node.value)?.identity;
|
|
357
|
+
if (identity) {
|
|
358
|
+
urls.add(identity.url);
|
|
359
|
+
}
|
|
360
|
+
});
|
|
179
361
|
return urls;
|
|
180
362
|
}
|
|
363
|
+
// The number of distinct GitHub REPOS a document addresses — the census unit
|
|
364
|
+
// for both the skeleton trigger and the splice gate. Distinct URLs would
|
|
365
|
+
// overcount: a skeleton README badges the source repo and links its issues
|
|
366
|
+
// page, which is two URLs over one repo.
|
|
367
|
+
function countGitHubRepos(tree) {
|
|
368
|
+
const repos = new Set();
|
|
369
|
+
visit(tree, 'link', (node) => {
|
|
370
|
+
const id = parseGitHubUrl(node.url);
|
|
371
|
+
if (id) {
|
|
372
|
+
repos.add(`${id.owner}/${id.repo}`.toLowerCase());
|
|
373
|
+
}
|
|
374
|
+
});
|
|
375
|
+
return repos.size;
|
|
376
|
+
}
|
|
181
377
|
// Input normalization, same family as fixRelativeLinks: make GitHub repos a
|
|
182
378
|
// source expresses WITHOUT markdown links visible as real link nodes, so
|
|
183
379
|
// every downstream consumer — repo fetch, entry tests, gates, badges — sees
|
|
@@ -473,10 +669,14 @@ function processListRecursively(listNode, repoInfoMap, sortOptions, firstSeen, i
|
|
|
473
669
|
// The caller's section-scope gate decision (sectionGatePasses). Absent for
|
|
474
670
|
// nested lists (emitted under their parent regardless) and for top-level
|
|
475
671
|
// lists with no open container (preamble), which gate per list.
|
|
476
|
-
sectionGateOpen
|
|
672
|
+
sectionGateOpen,
|
|
673
|
+
// The promoted details-item this list sits directly inside: an entry
|
|
674
|
+
// restating that repo (the best-of "GitHub" bullet) is the entry the
|
|
675
|
+
// summary already emitted, not a second item — its children still lift,
|
|
676
|
+
// the way a dead link's do.
|
|
677
|
+
suppressRepo) {
|
|
477
678
|
if (!isNested) {
|
|
478
|
-
const gateOpen = sectionGateOpen ??
|
|
479
|
-
countLinkedItems(listNode) >= sortOptions.minLinks;
|
|
679
|
+
const gateOpen = sectionGateOpen ?? countLinkedItems(listNode) >= sortOptions.minLinks;
|
|
480
680
|
if (!gateOpen) {
|
|
481
681
|
return [];
|
|
482
682
|
}
|
|
@@ -498,6 +698,10 @@ sectionGateOpen) {
|
|
|
498
698
|
const childrenJson = nestedContent.flatMap(child => child.type === 'list'
|
|
499
699
|
? processListRecursively(child, repoInfoMap, sortOptions, firstSeen, true)
|
|
500
700
|
: processTableRows(child, repoInfoMap, true, firstSeen));
|
|
701
|
+
if (suppressRepo && repoInfo === suppressRepo) {
|
|
702
|
+
entries.push({ emitted: childrenJson, node: itemNode, repoInfo });
|
|
703
|
+
continue;
|
|
704
|
+
}
|
|
501
705
|
// Title/description split on the FIRST paragraph only — a paper-list
|
|
502
706
|
// entry's identity link may live in a later paragraph (findOwnGitHubLink)
|
|
503
707
|
// while its title text stays the leading one.
|
|
@@ -511,6 +715,7 @@ sectionGateOpen) {
|
|
|
511
715
|
// The shared title fallbacks (an inline-code link label carries no text
|
|
512
716
|
// nodes, so the split alone can leave an empty title).
|
|
513
717
|
entryText.title = entryTitle(entryText.title, ownLink, repoInfo);
|
|
718
|
+
entryText.description = entryDescription(entryText.description, repoInfo);
|
|
514
719
|
const emitted = emitEntryNodes(githubUrl, repoInfo, entryText, childrenJson, firstSeen);
|
|
515
720
|
entries.push({ emitted, node: itemNode, repoInfo });
|
|
516
721
|
}
|
|
@@ -539,6 +744,23 @@ function entryTitle(base, ownLink, repoInfo) {
|
|
|
539
744
|
}
|
|
540
745
|
return repoInfo ? `${repoInfo.owner}/${repoInfo.repo}` : base;
|
|
541
746
|
}
|
|
747
|
+
// Description with the fallback every entry source shares: the split's own
|
|
748
|
+
// trailing prose when it says something — minus a leading badge cluster —
|
|
749
|
+
// else — for a live repo link — owner/name. The own link's label is never
|
|
750
|
+
// consulted: in the corpus it only ever echoes the title back (":tada:
|
|
751
|
+
// Doom" already says "Doom"). A degenerate base with no live repo link (a
|
|
752
|
+
// group) keeps its text: nothing better exists.
|
|
753
|
+
function entryDescription(base, repoInfo) {
|
|
754
|
+
// The noise strip runs only behind a stripped cluster: a bare leading
|
|
755
|
+
// dash or colon can be a word's own (the table paths never noise-stripped
|
|
756
|
+
// their cells, and "-equivalent" / ":bird:" descriptions are real).
|
|
757
|
+
const withoutBadges = stripBadgeClusters(base);
|
|
758
|
+
const stripped = withoutBadges === base ? base : stripLeadingNoise(withoutBadges);
|
|
759
|
+
if (!isDegenerateDescription(stripped)) {
|
|
760
|
+
return stripped;
|
|
761
|
+
}
|
|
762
|
+
return repoInfo ? `${repoInfo.owner}/${repoInfo.repo}` : stripped;
|
|
763
|
+
}
|
|
542
764
|
function processTableRows(tableNode, repoInfoMap, gateOpen, firstSeen) {
|
|
543
765
|
if (!gateOpen) {
|
|
544
766
|
return [];
|
|
@@ -578,9 +800,10 @@ function processTableRows(tableNode, repoInfoMap, gateOpen, firstSeen) {
|
|
|
578
800
|
]
|
|
579
801
|
.join(' ')
|
|
580
802
|
.trim();
|
|
803
|
+
const title = entryTitle(titleText.title, ownLink, repoInfo);
|
|
581
804
|
items.push(...emitEntryNodes(ownLink.url, repoInfo, {
|
|
582
|
-
title
|
|
583
|
-
description,
|
|
805
|
+
title,
|
|
806
|
+
description: entryDescription(description, repoInfo),
|
|
584
807
|
}, [], firstSeen));
|
|
585
808
|
continue;
|
|
586
809
|
}
|
|
@@ -592,9 +815,10 @@ function processTableRows(tableNode, repoInfoMap, gateOpen, firstSeen) {
|
|
|
592
815
|
emittedUrls.add(ownLink.url);
|
|
593
816
|
const repoInfo = repoInfoMap.get(ownLink.url) ?? null;
|
|
594
817
|
const cellText = splitEntryText(row.children[cellIndex].children);
|
|
818
|
+
const title = entryTitle(cellText.title, ownLink, repoInfo);
|
|
595
819
|
items.push(...emitEntryNodes(ownLink.url, repoInfo, {
|
|
596
|
-
title
|
|
597
|
-
description: cellText.description,
|
|
820
|
+
title,
|
|
821
|
+
description: entryDescription(cellText.description, repoInfo),
|
|
598
822
|
}, [], firstSeen));
|
|
599
823
|
}
|
|
600
824
|
}
|
|
@@ -615,6 +839,13 @@ function isDegenerateTitle(title) {
|
|
|
615
839
|
function isMeaningfulLinkText(text) {
|
|
616
840
|
return text !== '' && !isDegenerateTitle(text);
|
|
617
841
|
}
|
|
842
|
+
// Description-only tag words beyond the title set: the second-link label
|
|
843
|
+
// families the corpus census measured ("website", "documentation", …).
|
|
844
|
+
const DESC_TAG_TEXT = /^(?:website|homepage|home\s+page|link|here|documentation|repository)$/i;
|
|
845
|
+
function isDegenerateDescription(description) {
|
|
846
|
+
const trimmed = description.trim();
|
|
847
|
+
return (trimmed === '' || DESC_TAG_TEXT.test(trimmed) || isDegenerateTitle(trimmed));
|
|
848
|
+
}
|
|
618
849
|
// Dated entry lines end their tag cluster with a publication date ("4 Feb
|
|
619
850
|
// 2023", "19 Apr 2022", "2023") — a real corpus family (dated paper/tutorial
|
|
620
851
|
// lists), not a sentence continuation.
|
|
@@ -730,9 +961,10 @@ function blockquoteEntries(blockquote) {
|
|
|
730
961
|
function entryNodesFor(ownLink, inlines, repoInfoMap, firstSeen) {
|
|
731
962
|
const repoInfo = repoInfoMap.get(ownLink.url) ?? null;
|
|
732
963
|
const entryText = splitEntryText(inlines);
|
|
964
|
+
const title = entryTitle(entryText.title, ownLink, repoInfo);
|
|
733
965
|
return emitEntryNodes(ownLink.url, repoInfo, {
|
|
734
|
-
title
|
|
735
|
-
description: entryText.description,
|
|
966
|
+
title,
|
|
967
|
+
description: entryDescription(entryText.description, repoInfo),
|
|
736
968
|
}, [], firstSeen);
|
|
737
969
|
}
|
|
738
970
|
// The <details><summary>…</summary> collapsible-section idiom: the summary
|
|
@@ -742,17 +974,54 @@ function entryNodesFor(ownLink, inlines, repoInfoMap, firstSeen) {
|
|
|
742
974
|
// no summary, or an empty one, delimits nothing.
|
|
743
975
|
const DETAILS_SUMMARY = /<details[^>]*>[\s\S]*?<summary[^>]*>([\s\S]*?)<\/summary>/i;
|
|
744
976
|
const DETAILS_CLOSE = /^\s*<\/details>/i;
|
|
745
|
-
|
|
977
|
+
// One summary's read of the walk: the summary's whole text (the container
|
|
978
|
+
// title when the block is not an entry), and — when the summary carries the
|
|
979
|
+
// entry's identity (the best-of generator's shape: repo anchor, badge
|
|
980
|
+
// cluster, curated sentence) — that anchor and the prose trailing it.
|
|
981
|
+
const SUMMARY_ANCHOR = /<a\s[^>]*href=(["'])([^"']*)\1[^>]*>([\s\S]*?)<\/a>/gi;
|
|
982
|
+
function summaryHtmlText(html) {
|
|
983
|
+
return html
|
|
984
|
+
.replace(/<[^>]+>/g, ' ')
|
|
985
|
+
.replace(/\s+/g, ' ')
|
|
986
|
+
.trim();
|
|
987
|
+
}
|
|
988
|
+
function parseDetailsSummary(htmlValue) {
|
|
746
989
|
const match = DETAILS_SUMMARY.exec(htmlValue);
|
|
747
990
|
if (!match) {
|
|
748
991
|
return null;
|
|
749
992
|
}
|
|
750
|
-
const
|
|
993
|
+
const inner = match[1];
|
|
994
|
+
const title = inner
|
|
751
995
|
.replace(/<kbd>[\s\S]*?<\/kbd>/gi, '')
|
|
752
996
|
.replace(/<[^>]+>/g, ' ')
|
|
753
997
|
.replace(/\s+/g, ' ')
|
|
754
998
|
.trim();
|
|
755
|
-
|
|
999
|
+
let identity = null;
|
|
1000
|
+
let prose = '';
|
|
1001
|
+
for (const anchor of inner.matchAll(SUMMARY_ANCHOR)) {
|
|
1002
|
+
if (!parseGitHubUrl(anchor[2])) {
|
|
1003
|
+
continue;
|
|
1004
|
+
}
|
|
1005
|
+
identity = { label: summaryHtmlText(anchor[3]), url: anchor[2] };
|
|
1006
|
+
prose = summaryHtmlText(inner.slice((anchor.index ?? 0) + anchor[0].length));
|
|
1007
|
+
break;
|
|
1008
|
+
}
|
|
1009
|
+
return { identity, prose, title: title || null };
|
|
1010
|
+
}
|
|
1011
|
+
function detailsSummaryTitle(htmlValue) {
|
|
1012
|
+
return parseDetailsSummary(htmlValue)?.title ?? null;
|
|
1013
|
+
}
|
|
1014
|
+
// A badge cluster from a generated summary: a parenthesized run of medals,
|
|
1015
|
+
// counts and size suffixes (🥈31 · ⭐ 1.9K) — symbols, digits and punctuation,
|
|
1016
|
+
// no words, at least one of them. A parenthetical with any other letter (an
|
|
1017
|
+
// aside in prose) or an empty pair () stays.
|
|
1018
|
+
const BADGE_CLUSTER = /^\([\p{P}\p{S}\p{N}\sKMG]+\)\s*/u;
|
|
1019
|
+
function stripBadgeClusters(text) {
|
|
1020
|
+
let out = text.trimStart();
|
|
1021
|
+
while (BADGE_CLUSTER.test(out)) {
|
|
1022
|
+
out = out.replace(BADGE_CLUSTER, '');
|
|
1023
|
+
}
|
|
1024
|
+
return out;
|
|
756
1025
|
}
|
|
757
1026
|
// The summary text of a <details><summary> block inside a list item, when the
|
|
758
1027
|
// item has no paragraph text of its own.
|
|
@@ -905,7 +1174,7 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepositoryInfo, fir
|
|
|
905
1174
|
// comparison holds even for aliased spellings.
|
|
906
1175
|
const rementionsOpenItem = (link) => {
|
|
907
1176
|
const repoInfo = repoInfoMap.get(link.url);
|
|
908
|
-
return !!repoInfo && stack.some(container => container.repoInfo === repoInfo);
|
|
1177
|
+
return (!!repoInfo && stack.some(container => container.repoInfo === repoInfo));
|
|
909
1178
|
};
|
|
910
1179
|
for (let i = 0; i < tree.children.length; i++) {
|
|
911
1180
|
const node = tree.children[i];
|
|
@@ -928,7 +1197,9 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepositoryInfo, fir
|
|
|
928
1197
|
else if (node.type === 'paragraph') {
|
|
929
1198
|
const text = getNodeText(node);
|
|
930
1199
|
// Boilerplate "back to top" lines are neither entries nor description.
|
|
931
|
-
const ownLink = BACK_TO_TOP.test(text)
|
|
1200
|
+
const ownLink = BACK_TO_TOP.test(text)
|
|
1201
|
+
? undefined
|
|
1202
|
+
: paragraphEntryLink(node);
|
|
932
1203
|
// An entry paragraph behaves like a one-item list; a failed gate, a
|
|
933
1204
|
// dead target, or a re-mention of the enclosing item leaves plain
|
|
934
1205
|
// description.
|
|
@@ -938,7 +1209,10 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepositoryInfo, fir
|
|
|
938
1209
|
else {
|
|
939
1210
|
const container = stack[stack.length - 1];
|
|
940
1211
|
// Avoid adding boilerplate "back to top" links to descriptions.
|
|
941
|
-
if (container &&
|
|
1212
|
+
if (container &&
|
|
1213
|
+
text &&
|
|
1214
|
+
!BACK_TO_TOP.test(text) &&
|
|
1215
|
+
collectsProse(container)) {
|
|
942
1216
|
container.description = container.description
|
|
943
1217
|
? `${container.description}\n${text}`
|
|
944
1218
|
: text;
|
|
@@ -961,7 +1235,10 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepositoryInfo, fir
|
|
|
961
1235
|
else {
|
|
962
1236
|
const container = stack[stack.length - 1];
|
|
963
1237
|
// Avoid adding boilerplate "back to top" links to descriptions.
|
|
964
|
-
if (container &&
|
|
1238
|
+
if (container &&
|
|
1239
|
+
text &&
|
|
1240
|
+
!BACK_TO_TOP.test(text) &&
|
|
1241
|
+
collectsProse(container)) {
|
|
965
1242
|
container.description = container.description
|
|
966
1243
|
? `${container.description}\n${text}`
|
|
967
1244
|
: text;
|
|
@@ -969,30 +1246,61 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepositoryInfo, fir
|
|
|
969
1246
|
}
|
|
970
1247
|
}
|
|
971
1248
|
else if (node.type === 'html') {
|
|
972
|
-
// A details-summary block
|
|
973
|
-
//
|
|
974
|
-
//
|
|
975
|
-
|
|
976
|
-
|
|
1249
|
+
// A details-summary block is the entry itself when its summary carries
|
|
1250
|
+
// the identity (the best-of generator's shape) — promoted exactly like
|
|
1251
|
+
// a link-bearing heading, so the entry nests in its real category with
|
|
1252
|
+
// the summary's curated sentence as its description and the badges
|
|
1253
|
+
// gone. Any other summary opens a container like a heading would: a
|
|
1254
|
+
// group under the open container (the JSON has no nested sections), a
|
|
1255
|
+
// section at document level. The close tag (or the next summary) ends
|
|
1256
|
+
// it — never the enclosing section, which keeps collecting after the
|
|
1257
|
+
// collapsible block.
|
|
1258
|
+
const summary = parseDetailsSummary(node.value);
|
|
1259
|
+
if (summary) {
|
|
977
1260
|
closeInnermostDetails(stack, sectionRecords, firstSeen);
|
|
978
1261
|
// Container depths never decrease going up the stack. When the open
|
|
979
1262
|
// containers sit deeper than sectionDepth (a mid-document H1 defines
|
|
980
|
-
// sectionDepth while the content sections run deeper), the
|
|
981
|
-
//
|
|
982
|
-
//
|
|
983
|
-
//
|
|
1263
|
+
// sectionDepth while the content sections run deeper), the container
|
|
1264
|
+
// joins at the current depth — pushing at the shallower sectionDepth
|
|
1265
|
+
// would invert the stack and strand the gate (which reads the stack
|
|
1266
|
+
// bottom) on a tiny outer section.
|
|
984
1267
|
const joinDepth = stack.length === 0
|
|
985
1268
|
? sectionDepth
|
|
986
1269
|
: Math.max(sectionDepth, stack[stack.length - 1].headingDepth);
|
|
987
|
-
|
|
988
|
-
|
|
989
|
-
|
|
990
|
-
|
|
991
|
-
|
|
992
|
-
|
|
993
|
-
|
|
994
|
-
|
|
995
|
-
|
|
1270
|
+
const repoInfo = summary.identity
|
|
1271
|
+
? (repoInfoMap.get(summary.identity.url) ?? null)
|
|
1272
|
+
: null;
|
|
1273
|
+
const promoted = summary.identity &&
|
|
1274
|
+
repoInfo &&
|
|
1275
|
+
isMeaningfulLinkText(summary.identity.label)
|
|
1276
|
+
? { identity: summary.identity, repoInfo }
|
|
1277
|
+
: null;
|
|
1278
|
+
if (promoted && stack.length === 0) {
|
|
1279
|
+
openSynthesizedSection(stack, i, sectionDepth);
|
|
1280
|
+
}
|
|
1281
|
+
if (promoted && gateForSection(stack[0])) {
|
|
1282
|
+
stack.push({
|
|
1283
|
+
children: [],
|
|
1284
|
+
description: entryDescription(summary.prose, promoted.repoInfo),
|
|
1285
|
+
headingDepth: joinDepth,
|
|
1286
|
+
headingIndex: i,
|
|
1287
|
+
kind: 'item',
|
|
1288
|
+
openedByDetails: true,
|
|
1289
|
+
repoInfo: promoted.repoInfo,
|
|
1290
|
+
title: promoted.identity.label,
|
|
1291
|
+
});
|
|
1292
|
+
}
|
|
1293
|
+
else {
|
|
1294
|
+
stack.push({
|
|
1295
|
+
children: [],
|
|
1296
|
+
description: '',
|
|
1297
|
+
headingDepth: joinDepth,
|
|
1298
|
+
headingIndex: i,
|
|
1299
|
+
kind: stack.length === 0 ? 'section' : 'group',
|
|
1300
|
+
openedByDetails: true,
|
|
1301
|
+
title: summary.title ?? '',
|
|
1302
|
+
});
|
|
1303
|
+
}
|
|
996
1304
|
}
|
|
997
1305
|
else if (DETAILS_CLOSE.test(node.value)) {
|
|
998
1306
|
closeInnermostDetails(stack, sectionRecords, firstSeen);
|
|
@@ -1008,8 +1316,13 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepositoryInfo, fir
|
|
|
1008
1316
|
}
|
|
1009
1317
|
// Every list inside the open container contributes items — a section is
|
|
1010
1318
|
// not closed by its first list — and the minLinks gate is decided per
|
|
1011
|
-
// section, against the whole section subtree.
|
|
1012
|
-
|
|
1319
|
+
// section, against the whole section subtree. A list directly inside a
|
|
1320
|
+
// promoted details-item suppresses the re-mention of that item's repo.
|
|
1321
|
+
const innermost = stack[stack.length - 1];
|
|
1322
|
+
const suppress = innermost.kind === 'item' && innermost.openedByDetails
|
|
1323
|
+
? innermost.repoInfo
|
|
1324
|
+
: undefined;
|
|
1325
|
+
const items = processListRecursively(node, repoInfoMap, sortOptions, firstSeen, false, gateForSection(stack[0]), suppress);
|
|
1013
1326
|
stack[stack.length - 1].children.push(...items);
|
|
1014
1327
|
}
|
|
1015
1328
|
else if (node.type === 'table') {
|
|
@@ -1029,6 +1342,12 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepositoryInfo, fir
|
|
|
1029
1342
|
.map(record => record.section);
|
|
1030
1343
|
return { sections, title: documentTitle, titleHeadingIndex };
|
|
1031
1344
|
}
|
|
1345
|
+
// A details-promoted item's description is the summary's curated sentence;
|
|
1346
|
+
// prose inside the collapsible body is the card's content, not more
|
|
1347
|
+
// description.
|
|
1348
|
+
function collectsProse(container) {
|
|
1349
|
+
return container.kind !== 'item' || !container.openedByDetails;
|
|
1350
|
+
}
|
|
1032
1351
|
// The heading depth that opens top-level sections: the shallowest structural
|
|
1033
1352
|
// heading in the document other than the title slot. H1s count — ~20% of
|
|
1034
1353
|
// mirror READMEs use `# Section` after the title H1, and hardcoding H2 would
|
|
@@ -1292,7 +1611,11 @@ function finalizeContainer(container, stack, sectionRecords, firstSeen) {
|
|
|
1292
1611
|
if (container.kind === 'section') {
|
|
1293
1612
|
sectionRecords.push({
|
|
1294
1613
|
headingIndex: container.headingIndex,
|
|
1295
|
-
section: {
|
|
1614
|
+
section: {
|
|
1615
|
+
description,
|
|
1616
|
+
items: container.children,
|
|
1617
|
+
title: container.title,
|
|
1618
|
+
},
|
|
1296
1619
|
});
|
|
1297
1620
|
return;
|
|
1298
1621
|
}
|
|
@@ -1323,8 +1646,7 @@ function finalizeContainer(container, stack, sectionRecords, firstSeen) {
|
|
|
1323
1646
|
// entry heading triggers must stop at the synthesized section wrapping its
|
|
1324
1647
|
// run — the next entry heading of the run lands back inside it.
|
|
1325
1648
|
function closeContainers(stack, depth, sectionRecords, firstSeen, stopAtSynthesized = false) {
|
|
1326
|
-
while (stack.length > 0 &&
|
|
1327
|
-
stack[stack.length - 1].headingDepth >= depth) {
|
|
1649
|
+
while (stack.length > 0 && stack[stack.length - 1].headingDepth >= depth) {
|
|
1328
1650
|
if (stopAtSynthesized && stack[stack.length - 1].openedBySynthesis) {
|
|
1329
1651
|
break;
|
|
1330
1652
|
}
|
package/dist/orchestrator.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { Logger } from './logger.js';
|
|
2
|
-
import type { RepoInfoDetails } from './github.js';
|
|
2
|
+
import type { RepoInfoDetails, RepoIdentifier } from './github.js';
|
|
3
3
|
import { JsonOutput, ReplacementRule } from './markdown.js';
|
|
4
4
|
export interface EnhanceOptions {
|
|
5
5
|
content: string;
|
|
@@ -14,6 +14,7 @@ export interface EnhanceOptions {
|
|
|
14
14
|
relativeLinkPrefix?: string;
|
|
15
15
|
replacements?: ReplacementRule[];
|
|
16
16
|
sortBy?: '' | 'last_commit' | 'stars';
|
|
17
|
+
sourceRepository?: RepoIdentifier;
|
|
17
18
|
token: string;
|
|
18
19
|
}
|
|
19
20
|
export interface EnhanceResult {
|
package/dist/orchestrator.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { processMarkdownContent, } from './markdown.js';
|
|
2
2
|
export async function enhance(options) {
|
|
3
|
-
const { content, disableBranding = false, log, now = new Date(), originalRepositoryInfo, originalRepositorySha, previousJson, relativeLinkPrefix = '', replacements = [], sortBy = '', enhancedRepository, enhancedRepositoryDescription, token, } = options;
|
|
3
|
+
const { content, disableBranding = false, log, now = new Date(), originalRepositoryInfo, originalRepositorySha, previousJson, relativeLinkPrefix = '', replacements = [], sortBy = '', sourceRepository, enhancedRepository, enhancedRepositoryDescription, token, } = options;
|
|
4
4
|
// Branding is an internal rule prepended to the caller's own; build a fresh
|
|
5
5
|
// array so the caller's `replacements` is never mutated.
|
|
6
6
|
const branding = { type: 'branding' };
|
|
@@ -9,7 +9,7 @@ export async function enhance(options) {
|
|
|
9
9
|
by: sortBy,
|
|
10
10
|
minLinks: 2,
|
|
11
11
|
};
|
|
12
|
-
const { finalContent, jsonData } = await processMarkdownContent(content, token, rules, sortOptions, relativeLinkPrefix, enhancedRepository, enhancedRepositoryDescription, originalRepositorySha, originalRepositoryInfo, previousJson, now, log);
|
|
12
|
+
const { finalContent, jsonData } = await processMarkdownContent(content, token, rules, sortOptions, relativeLinkPrefix, enhancedRepository, enhancedRepositoryDescription, originalRepositorySha, originalRepositoryInfo, sourceRepository, previousJson, now, log);
|
|
13
13
|
return {
|
|
14
14
|
finalContent,
|
|
15
15
|
jsonData,
|