@enhansome/core 1.11.0 → 1.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/first-seen.js +1 -1
- package/dist/github.d.ts +7 -0
- package/dist/github.js +34 -2
- package/dist/index.d.ts +2 -2
- package/dist/index.js +2 -2
- package/dist/internal-links.d.ts +21 -0
- package/dist/internal-links.js +112 -0
- package/dist/markdown.d.ts +10 -8
- package/dist/markdown.js +362 -45
- package/dist/orchestrator.d.ts +2 -1
- package/dist/orchestrator.js +2 -2
- package/package.json +1 -1
package/dist/first-seen.js
CHANGED
|
@@ -9,7 +9,7 @@ export function firstSeenFor(previous, now) {
|
|
|
9
9
|
}
|
|
10
10
|
function collect(nodes, index) {
|
|
11
11
|
for (const node of nodes) {
|
|
12
|
-
if (node.node_type === 'item' && node.first_seen) {
|
|
12
|
+
if (node.node_type === 'item' && node.repo_info && node.first_seen) {
|
|
13
13
|
const existing = index.get(node.repo_info.id);
|
|
14
14
|
if (existing === undefined || node.first_seen < existing) {
|
|
15
15
|
index.set(node.repo_info.id, node.first_seen);
|
package/dist/github.d.ts
CHANGED
|
@@ -86,6 +86,13 @@ export declare function getRepoInfo(octokit: GithubClient, owner: string, repo:
|
|
|
86
86
|
*/
|
|
87
87
|
export declare function getRepoInfoOrNull(octokit: GithubClient, owner: string, repo: string): Promise<null | RepoInfoDetails>;
|
|
88
88
|
export declare function getReadme(octokit: GithubClient, owner: string, repo: string, format?: 'html' | 'raw'): Promise<string>;
|
|
89
|
+
/**
|
|
90
|
+
* One in-repo markdown file, for following a README's links into the source's
|
|
91
|
+
* content files. Returns null (logged) instead of failing the run: a followed
|
|
92
|
+
* file is best-effort the way repo-info failures are — only the README fetch
|
|
93
|
+
* itself can fail the run.
|
|
94
|
+
*/
|
|
95
|
+
export declare function getRepoFileOrNull(octokit: GithubClient, owner: string, repo: string, path: string): Promise<null | string>;
|
|
89
96
|
/** Root file/directory names — the compile-manifest gate reads them to tell a
|
|
90
97
|
* directory of resources from a repo that IS the deliverable. A `path: ''`
|
|
91
98
|
* listing is always an array; the single-entry branch defends against a file
|
package/dist/github.js
CHANGED
|
@@ -110,6 +110,28 @@ export async function getReadme(octokit, owner, repo, format = 'raw') {
|
|
|
110
110
|
// still describe the JSON shape - cast to string.
|
|
111
111
|
return response.data;
|
|
112
112
|
}
|
|
113
|
+
/**
|
|
114
|
+
* One in-repo markdown file, for following a README's links into the source's
|
|
115
|
+
* content files. Returns null (logged) instead of failing the run: a followed
|
|
116
|
+
* file is best-effort the way repo-info failures are — only the README fetch
|
|
117
|
+
* itself can fail the run.
|
|
118
|
+
*/
|
|
119
|
+
export async function getRepoFileOrNull(octokit, owner, repo, path) {
|
|
120
|
+
try {
|
|
121
|
+
octokit.log.debug(`Fetching ${owner}/${repo}/${path}`);
|
|
122
|
+
const { data } = await octokit.rest.repos.getContent({
|
|
123
|
+
mediaType: { format: 'raw' },
|
|
124
|
+
owner,
|
|
125
|
+
path,
|
|
126
|
+
repo,
|
|
127
|
+
});
|
|
128
|
+
return data;
|
|
129
|
+
}
|
|
130
|
+
catch (error) {
|
|
131
|
+
octokit.log.warn(`Failed to fetch ${owner}/${repo}/${path}: ${formatRequestError(error)}`);
|
|
132
|
+
return null;
|
|
133
|
+
}
|
|
134
|
+
}
|
|
113
135
|
/** Root file/directory names — the compile-manifest gate reads them to tell a
|
|
114
136
|
* directory of resources from a repo that IS the deliverable. A `path: ''`
|
|
115
137
|
* listing is always an array; the single-entry branch defends against a file
|
|
@@ -153,7 +175,17 @@ export function parseOwnerRepo(value) {
|
|
|
153
175
|
if (parts.length !== 2) {
|
|
154
176
|
return null;
|
|
155
177
|
}
|
|
156
|
-
return {
|
|
178
|
+
return {
|
|
179
|
+
owner: stripZeroWidth(parts[0]),
|
|
180
|
+
repo: stripZeroWidth(parts[1]).replace(/\.git$/, ''),
|
|
181
|
+
};
|
|
182
|
+
}
|
|
183
|
+
// Zero-width characters glued to a repo name (awesome-computer-vision links
|
|
184
|
+
// neuraltalk with an invisible BOM) are not identity; URL.pathname keeps them
|
|
185
|
+
// percent-encoded, so both spellings are stripped before the name is used.
|
|
186
|
+
const ZERO_WIDTH = /(?:%EF%BB%BF|%E2%80%8[B-F]|%E2%81%A0)|[\u200B-\u200F\u2060\uFEFF]/gi;
|
|
187
|
+
function stripZeroWidth(segment) {
|
|
188
|
+
return segment.replace(ZERO_WIDTH, '');
|
|
157
189
|
}
|
|
158
190
|
export function parseGitHubUrl(url) {
|
|
159
191
|
try {
|
|
@@ -165,7 +197,7 @@ export function parseGitHubUrl(url) {
|
|
|
165
197
|
.split('/')
|
|
166
198
|
.filter(part => part.length > 0);
|
|
167
199
|
if (pathParts.length >= 2) {
|
|
168
|
-
const owner = pathParts[0], repo = pathParts[1].replace(/\.git$/, '');
|
|
200
|
+
const owner = stripZeroWidth(pathParts[0]), repo = stripZeroWidth(pathParts[1]).replace(/\.git$/, '');
|
|
169
201
|
return { owner, repo };
|
|
170
202
|
}
|
|
171
203
|
return null;
|
package/dist/index.d.ts
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
|
-
export { formatRequestError, getLatestCommitSha, getReadme, getRepoInfo, getRepoInfoOrNull, getRootEntryNames, makeOctokit, parseGitHubUrl, parseOwnerRepo, } from './github.js';
|
|
1
|
+
export { formatRequestError, getLatestCommitSha, getReadme, getRepoFileOrNull, getRepoInfo, getRepoInfoOrNull, getRootEntryNames, makeOctokit, parseGitHubUrl, parseOwnerRepo, } from './github.js';
|
|
2
2
|
export type { GithubClient, MakeOctokitOptions, RepoIdentifier, RepoInfoDetails, ThrottleOptions, } from './github.js';
|
|
3
3
|
export type { Logger } from './logger.js';
|
|
4
4
|
export { consoleLog, silentLog } from './logger.js';
|
|
5
|
-
export { toRepoInfo } from './markdown.js';
|
|
5
|
+
export { countItems, hollowsPrevious, toRepoInfo } from './markdown.js';
|
|
6
6
|
export type { JsonGroup, JsonItem, JsonMetadata, JsonNode, JsonOutput, JsonSection, ReplacementRule, RepoInfo, } from './markdown.js';
|
|
7
7
|
export { enhance } from './orchestrator.js';
|
|
8
8
|
export type { EnhanceOptions, EnhanceResult } from './orchestrator.js';
|
package/dist/index.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
export { formatRequestError, getLatestCommitSha, getReadme, getRepoInfo, getRepoInfoOrNull, getRootEntryNames, makeOctokit, parseGitHubUrl, parseOwnerRepo, } from './github.js';
|
|
1
|
+
export { formatRequestError, getLatestCommitSha, getReadme, getRepoFileOrNull, getRepoInfo, getRepoInfoOrNull, getRootEntryNames, makeOctokit, parseGitHubUrl, parseOwnerRepo, } from './github.js';
|
|
2
2
|
export { consoleLog, silentLog } from './logger.js';
|
|
3
|
-
export { toRepoInfo } from './markdown.js';
|
|
3
|
+
export { countItems, hollowsPrevious, toRepoInfo } from './markdown.js';
|
|
4
4
|
export { enhance } from './orchestrator.js';
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
/** Bounds for one follow run: files fetched, and bytes per file. */
|
|
2
|
+
export declare const MAX_FOLLOWED_FILES = 50;
|
|
3
|
+
export declare const MAX_FOLLOWED_FILE_BYTES = 1000000;
|
|
4
|
+
/** True for a resolved repo path that must never be followed or inlined. */
|
|
5
|
+
export declare function isSkippedPath(path: string): boolean;
|
|
6
|
+
/**
|
|
7
|
+
* Resolves one link target against the referring document's directory.
|
|
8
|
+
* Accepts repo-relative (`docs/x.md`), referring-file-relative
|
|
9
|
+
* (`../guides/y.md`) and repo-root-absolute (`/x.md`) forms. Returns null for
|
|
10
|
+
* anything that is not an in-repo path: external URLs, anchors, non-markdown
|
|
11
|
+
* files, or `..` escaping the repo root.
|
|
12
|
+
*/
|
|
13
|
+
export declare function resolveRepoPath(target: string, baseDir: string): null | string;
|
|
14
|
+
/**
|
|
15
|
+
* Extracts the in-repo markdown path from a same-repo absolute GitHub URL
|
|
16
|
+
* (`https://github.com/owner/repo/blob/main/docs/x.md`), or null for links to
|
|
17
|
+
* any other host/repo.
|
|
18
|
+
*/
|
|
19
|
+
export declare function parseSameRepoBlobPath(url: string, owner: string, repo: string): null | string;
|
|
20
|
+
/** True when a link's visible text is just a filename/path, not a category name. */
|
|
21
|
+
export declare function isFilenameLabel(label: string, path: string): boolean;
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
// Path rules for following a README's links into other markdown files of the
|
|
2
|
+
// SAME repo — the "content moved out of the README" shape (android-root moved
|
|
3
|
+
// 600+ entries into docs/apps-and-modules/*.md and left a landing page).
|
|
4
|
+
// Pure string/path logic only; fetching and AST splicing live in markdown.ts.
|
|
5
|
+
/** Bounds for one follow run: files fetched, and bytes per file. */
|
|
6
|
+
export const MAX_FOLLOWED_FILES = 50;
|
|
7
|
+
export const MAX_FOLLOWED_FILE_BYTES = 1_000_000;
|
|
8
|
+
const MARKDOWN_EXT = /\.(md|markdown)$/i;
|
|
9
|
+
// Repo-convention meta documents. Only honored at the repo root and under
|
|
10
|
+
// .github/ — elsewhere a same-named file is content: android-root's
|
|
11
|
+
// docs/apps-and-modules/security.md is the Security category, and a basename
|
|
12
|
+
// rule would silently drop it.
|
|
13
|
+
const META_BASENAMES = new Set([
|
|
14
|
+
'acknowledgements',
|
|
15
|
+
'acknowledgments',
|
|
16
|
+
'authors',
|
|
17
|
+
'changes',
|
|
18
|
+
'changelog',
|
|
19
|
+
'citation',
|
|
20
|
+
'code of conduct',
|
|
21
|
+
'code_of_conduct',
|
|
22
|
+
'code-of-conduct',
|
|
23
|
+
'conduct',
|
|
24
|
+
'contributing',
|
|
25
|
+
'contributors',
|
|
26
|
+
'funding',
|
|
27
|
+
'history',
|
|
28
|
+
'issue_template',
|
|
29
|
+
'license',
|
|
30
|
+
'licence',
|
|
31
|
+
'notice',
|
|
32
|
+
'pull_request_template',
|
|
33
|
+
'security',
|
|
34
|
+
'support',
|
|
35
|
+
'todo',
|
|
36
|
+
]);
|
|
37
|
+
// The root README and its translations/self-links are alternate views of the
|
|
38
|
+
// document being enhanced, never additional content — inlining one would
|
|
39
|
+
// duplicate the whole list. Directory READMEs (pages/RISH.md aside,
|
|
40
|
+
// CLI/README.md shapes) are content and stay followable.
|
|
41
|
+
const ROOT_README = /^readme[^/]*\.(md|markdown)$/i;
|
|
42
|
+
function baseName(path) {
|
|
43
|
+
const tail = path.slice(path.lastIndexOf('/') + 1);
|
|
44
|
+
const dot = tail.lastIndexOf('.');
|
|
45
|
+
return (dot > 0 ? tail.slice(0, dot) : tail).toLowerCase();
|
|
46
|
+
}
|
|
47
|
+
/** True for a resolved repo path that must never be followed or inlined. */
|
|
48
|
+
export function isSkippedPath(path) {
|
|
49
|
+
if (!MARKDOWN_EXT.test(path)) {
|
|
50
|
+
return true;
|
|
51
|
+
}
|
|
52
|
+
if (!path.includes('/')) {
|
|
53
|
+
return ROOT_README.test(path) || META_BASENAMES.has(baseName(path));
|
|
54
|
+
}
|
|
55
|
+
return path.startsWith('.github/') && META_BASENAMES.has(baseName(path));
|
|
56
|
+
}
|
|
57
|
+
/**
|
|
58
|
+
* Resolves one link target against the referring document's directory.
|
|
59
|
+
* Accepts repo-relative (`docs/x.md`), referring-file-relative
|
|
60
|
+
* (`../guides/y.md`) and repo-root-absolute (`/x.md`) forms. Returns null for
|
|
61
|
+
* anything that is not an in-repo path: external URLs, anchors, non-markdown
|
|
62
|
+
* files, or `..` escaping the repo root.
|
|
63
|
+
*/
|
|
64
|
+
export function resolveRepoPath(target, baseDir) {
|
|
65
|
+
const stripped = target.split('#')[0].split('?')[0].trim();
|
|
66
|
+
if (!stripped || stripped.includes('://') || stripped.startsWith('mailto:')) {
|
|
67
|
+
return null;
|
|
68
|
+
}
|
|
69
|
+
const segments = (stripped.startsWith('/') ? [] : baseDir.split('/')).concat(stripped.split('/'));
|
|
70
|
+
const stack = [];
|
|
71
|
+
for (const segment of segments) {
|
|
72
|
+
if (segment === '' || segment === '.') {
|
|
73
|
+
continue;
|
|
74
|
+
}
|
|
75
|
+
if (segment === '..') {
|
|
76
|
+
if (stack.length === 0) {
|
|
77
|
+
return null;
|
|
78
|
+
}
|
|
79
|
+
stack.pop();
|
|
80
|
+
continue;
|
|
81
|
+
}
|
|
82
|
+
stack.push(segment);
|
|
83
|
+
}
|
|
84
|
+
return stack.length > 0 ? stack.join('/') : null;
|
|
85
|
+
}
|
|
86
|
+
/**
|
|
87
|
+
* Extracts the in-repo markdown path from a same-repo absolute GitHub URL
|
|
88
|
+
* (`https://github.com/owner/repo/blob/main/docs/x.md`), or null for links to
|
|
89
|
+
* any other host/repo.
|
|
90
|
+
*/
|
|
91
|
+
export function parseSameRepoBlobPath(url, owner, repo) {
|
|
92
|
+
const match = /^https?:\/\/(?:www\.)?github\.com\/([A-Za-z0-9_.-]+)\/([A-Za-z0-9_.-]+)\/(?:blob|raw)\/[^/]+\/(.+)$/i.exec(url);
|
|
93
|
+
if (!match) {
|
|
94
|
+
return null;
|
|
95
|
+
}
|
|
96
|
+
if (`${match[1]}/${match[2]}`.toLowerCase() !== `${owner}/${repo}`.toLowerCase()) {
|
|
97
|
+
return null;
|
|
98
|
+
}
|
|
99
|
+
return decodeURIComponent(match[3]);
|
|
100
|
+
}
|
|
101
|
+
/** True when a link's visible text is just a filename/path, not a category name. */
|
|
102
|
+
export function isFilenameLabel(label, path) {
|
|
103
|
+
const text = label.trim().toLowerCase();
|
|
104
|
+
if (!text) {
|
|
105
|
+
return true;
|
|
106
|
+
}
|
|
107
|
+
const file = path.slice(path.lastIndexOf('/') + 1).toLowerCase();
|
|
108
|
+
return (text === file ||
|
|
109
|
+
text === path.toLowerCase() ||
|
|
110
|
+
text.includes('/') ||
|
|
111
|
+
MARKDOWN_EXT.test(text));
|
|
112
|
+
}
|
package/dist/markdown.d.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { RepoInfoDetails } from './github.js';
|
|
1
|
+
import { RepoIdentifier, RepoInfoDetails } from './github.js';
|
|
2
2
|
import { Logger } from './logger.js';
|
|
3
3
|
import type { Heading, Link, List, Parent, Root } from 'mdast';
|
|
4
4
|
export interface JsonOutput {
|
|
@@ -18,17 +18,17 @@ export interface SortOptions {
|
|
|
18
18
|
}
|
|
19
19
|
export interface RepoInfo {
|
|
20
20
|
archived: boolean;
|
|
21
|
-
description
|
|
22
|
-
homepage
|
|
21
|
+
description?: null | string;
|
|
22
|
+
homepage?: null | string;
|
|
23
23
|
id: number;
|
|
24
24
|
language: null | string;
|
|
25
|
-
license
|
|
25
|
+
license?: null | string;
|
|
26
26
|
last_commit: null | string;
|
|
27
|
-
open_issues
|
|
27
|
+
open_issues?: number;
|
|
28
28
|
owner: string;
|
|
29
29
|
repo: string;
|
|
30
30
|
stars: number;
|
|
31
|
-
topics
|
|
31
|
+
topics?: string[];
|
|
32
32
|
}
|
|
33
33
|
/** The sole crossing from the API shape to the emitted one, so both agree on the renames. */
|
|
34
34
|
export declare function toRepoInfo(details: RepoInfoDetails): RepoInfo;
|
|
@@ -37,7 +37,7 @@ export interface JsonItem {
|
|
|
37
37
|
description: null | string;
|
|
38
38
|
first_seen?: string;
|
|
39
39
|
node_type: 'item';
|
|
40
|
-
repo_info
|
|
40
|
+
repo_info?: RepoInfo;
|
|
41
41
|
title: string;
|
|
42
42
|
}
|
|
43
43
|
export interface JsonGroup {
|
|
@@ -60,7 +60,9 @@ export interface JsonSection {
|
|
|
60
60
|
items: JsonNode[];
|
|
61
61
|
title: string;
|
|
62
62
|
}
|
|
63
|
-
export declare function
|
|
63
|
+
export declare function countItems(output: JsonOutput): number;
|
|
64
|
+
export declare function hollowsPrevious(previous: JsonOutput, next: JsonOutput): boolean;
|
|
65
|
+
export declare function processMarkdownContent(originalContent: string, token: string, replacements?: ReplacementRule[], sortOptions?: SortOptions, relativeLinkPrefix?: string, enhancedRepository?: string, enhancedRepositoryDescription?: string, originalRepositorySha?: string, originalRepositoryInfo?: null | RepoInfoDetails, sourceRepository?: RepoIdentifier, previousJson?: JsonOutput, now?: Date, log?: Logger): Promise<{
|
|
64
66
|
finalContent: string;
|
|
65
67
|
jsonData: JsonOutput;
|
|
66
68
|
}>;
|
package/dist/markdown.js
CHANGED
|
@@ -5,7 +5,8 @@ import remarkStringify from 'remark-stringify';
|
|
|
5
5
|
import { unified } from 'unified';
|
|
6
6
|
import { visit } from 'unist-util-visit';
|
|
7
7
|
import { firstSeenFor } from './first-seen.js';
|
|
8
|
-
import { formatRequestError, getRepoInfo as fetchRepoInfo, makeOctokit, parseGitHubUrl, } from './github.js';
|
|
8
|
+
import { formatRequestError, getRepoFileOrNull, getRepoInfo as fetchRepoInfo, makeOctokit, parseGitHubUrl, } from './github.js';
|
|
9
|
+
import { isFilenameLabel, isSkippedPath, MAX_FOLLOWED_FILE_BYTES, MAX_FOLLOWED_FILES, parseSameRepoBlobPath, resolveRepoPath, } from './internal-links.js';
|
|
9
10
|
import { consoleLog } from './logger.js';
|
|
10
11
|
/** The sole crossing from the API shape to the emitted one, so both agree on the renames. */
|
|
11
12
|
export function toRepoInfo(details) {
|
|
@@ -24,6 +25,19 @@ export function toRepoInfo(details) {
|
|
|
24
25
|
topics: details.topics,
|
|
25
26
|
};
|
|
26
27
|
}
|
|
28
|
+
export function countItems(output) {
|
|
29
|
+
const count = (nodes) => nodes.reduce((sum, node) => sum + (node.node_type === 'item' ? 1 : 0) + count(node.children), 0);
|
|
30
|
+
return count(output.items.flatMap(section => section.items));
|
|
31
|
+
}
|
|
32
|
+
// A >95% item collapse against the previous run is a parse regression or a
|
|
33
|
+
// source restructure, never a real edit — the caller refuses to write rather
|
|
34
|
+
// than hollow the mirror (android-root committed a skeleton daily for three
|
|
35
|
+
// days before this guard existed).
|
|
36
|
+
const HOLLOW_SHRINKAGE = 0.05;
|
|
37
|
+
export function hollowsPrevious(previous, next) {
|
|
38
|
+
const before = countItems(previous);
|
|
39
|
+
return before > 0 && countItems(next) < before * HOLLOW_SHRINKAGE;
|
|
40
|
+
}
|
|
27
41
|
// The lookup's single throttled client coordinates rate limits across the whole
|
|
28
42
|
// pool, so this bounds in-flight targets rather than requests.
|
|
29
43
|
const FETCH_CONCURRENCY = 10;
|
|
@@ -62,9 +76,9 @@ function createRepoInfoLookup(token, log) {
|
|
|
62
76
|
* collapse to a single fetch inside the lookup, then fan back out to every alias
|
|
63
77
|
* here.
|
|
64
78
|
*
|
|
65
|
-
* Each target's failure is independent and non-fatal: a dead link
|
|
66
|
-
*
|
|
67
|
-
*
|
|
79
|
+
* Each target's failure is independent and non-fatal: a dead link drops its
|
|
80
|
+
* entry and lifts the children. Only the source README fetch in main.ts can
|
|
81
|
+
* fail the run.
|
|
68
82
|
*/
|
|
69
83
|
async function fetchTargetData(urls, repos) {
|
|
70
84
|
const log = repos.client.log;
|
|
@@ -85,15 +99,169 @@ async function fetchTargetData(urls, repos) {
|
|
|
85
99
|
log.info(`Target fetch: ${repoInfoMap.size}/${urls.size} repo-info ok in ${Date.now() - fetchStart}ms (concurrency ${FETCH_CONCURRENCY}).`);
|
|
86
100
|
return repoInfoMap;
|
|
87
101
|
}
|
|
88
|
-
|
|
102
|
+
// The same-repo markdown links of one document: resolved relative and
|
|
103
|
+
// root-absolute targets, plus same-repo absolute blob/raw URLs. Labels ride
|
|
104
|
+
// along because they become the spliced category's title.
|
|
105
|
+
function collectDocLinks(tree, baseDir, source) {
|
|
106
|
+
const links = [];
|
|
107
|
+
visit(tree, 'link', (node) => {
|
|
108
|
+
const path = resolveRepoPath(node.url, baseDir) ??
|
|
109
|
+
parseSameRepoBlobPath(node.url, source.owner, source.repo);
|
|
110
|
+
if (!path || isSkippedPath(path)) {
|
|
111
|
+
return;
|
|
112
|
+
}
|
|
113
|
+
links.push({ label: getInlineText(node.children), path });
|
|
114
|
+
});
|
|
115
|
+
return links;
|
|
116
|
+
}
|
|
117
|
+
const FRONTMATTER = /^---\r?\n[\s\S]*?\r?\n---\r?\n?/;
|
|
118
|
+
// VitePress/Docusaurus content files open with YAML frontmatter that remark
|
|
119
|
+
// would misread as a thematic break plus a setext heading.
|
|
120
|
+
function stripFrontmatter(content) {
|
|
121
|
+
return content.replace(FRONTMATTER, '');
|
|
122
|
+
}
|
|
123
|
+
// Removes the file's leading H1 — the injected category heading replaces it,
|
|
124
|
+
// so the document never carries the same title twice — and returns its text
|
|
125
|
+
// for the filename-label fallback.
|
|
126
|
+
function stripTitleHeading(tree) {
|
|
127
|
+
const first = tree.children[0];
|
|
128
|
+
if (first?.type === 'heading' && first.depth === 1) {
|
|
129
|
+
tree.children.shift();
|
|
130
|
+
return getNodeText(first);
|
|
131
|
+
}
|
|
132
|
+
return '';
|
|
133
|
+
}
|
|
134
|
+
// One level shallower than the file's shallowest remaining heading, so the
|
|
135
|
+
// file's own sections nest inside the category; a heading-less file takes the
|
|
136
|
+
// conventional section level of 2.
|
|
137
|
+
function categoryHeadingDepth(children) {
|
|
138
|
+
let shallowest = Infinity;
|
|
139
|
+
for (const node of children) {
|
|
140
|
+
if (node.type === 'heading') {
|
|
141
|
+
shallowest = Math.min(shallowest, node.depth);
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
// shallowest is 1..6, so the clamped result stays inside the depth union.
|
|
145
|
+
return shallowest === Infinity
|
|
146
|
+
? 2
|
|
147
|
+
: Math.max(1, shallowest - 1);
|
|
148
|
+
}
|
|
149
|
+
// Relative links and images resolve against the source repo's tree; in the
|
|
150
|
+
// mirror they dangle. Rewritten to GitHub URLs at HEAD.
|
|
151
|
+
function rewriteRelativeLinksToSource(tree, baseDir, source) {
|
|
152
|
+
const absolute = (url) => {
|
|
153
|
+
if (!url ||
|
|
154
|
+
url.startsWith('#') ||
|
|
155
|
+
url.includes('://') ||
|
|
156
|
+
url.startsWith('mailto:')) {
|
|
157
|
+
return url;
|
|
158
|
+
}
|
|
159
|
+
const path = resolveRepoPath(url, baseDir);
|
|
160
|
+
return path
|
|
161
|
+
? `https://github.com/${source.owner}/${source.repo}/blob/HEAD/${path}`
|
|
162
|
+
: url;
|
|
163
|
+
};
|
|
164
|
+
visit(tree, 'link', (node) => {
|
|
165
|
+
node.url = absolute(node.url);
|
|
166
|
+
});
|
|
167
|
+
visit(tree, 'image', (node) => {
|
|
168
|
+
node.url = absolute(node.url);
|
|
169
|
+
});
|
|
170
|
+
}
|
|
171
|
+
/**
|
|
172
|
+
* Appends the source's internal markdown files to the tree, each under a
|
|
173
|
+
* heading titled by its link's text — the README-is-a-landing-page shape.
|
|
174
|
+
* Breadth-first in document order with a visited set, so a file linked from
|
|
175
|
+
* both the README and an index page is fetched and spliced once, the README's
|
|
176
|
+
* label winning. A file joins the document only when it carries at least
|
|
177
|
+
* `minLinks` GitHub links; navigation pages are still parsed for their links,
|
|
178
|
+
* which is where this shape often keeps them.
|
|
179
|
+
*/
|
|
180
|
+
async function appendInternalDocs(tree, source, repos, minLinks, log) {
|
|
181
|
+
const processor = unified().use(remarkParse).use(remarkGfm);
|
|
182
|
+
const queue = collectDocLinks(tree, '', source);
|
|
183
|
+
const visited = new Set();
|
|
184
|
+
const labels = new Map();
|
|
185
|
+
let fetched = 0;
|
|
186
|
+
let appended = 0;
|
|
187
|
+
let processed = 0;
|
|
188
|
+
while (queue.length > 0 && processed < MAX_FOLLOWED_FILES) {
|
|
189
|
+
const link = queue.shift();
|
|
190
|
+
if (visited.has(link.path)) {
|
|
191
|
+
continue;
|
|
192
|
+
}
|
|
193
|
+
processed += 1;
|
|
194
|
+
visited.add(link.path);
|
|
195
|
+
if (link.label.trim() && !labels.has(link.path)) {
|
|
196
|
+
labels.set(link.path, link.label);
|
|
197
|
+
}
|
|
198
|
+
const raw = await getRepoFileOrNull(repos.client, source.owner, source.repo, link.path);
|
|
199
|
+
if (!raw) {
|
|
200
|
+
continue;
|
|
201
|
+
}
|
|
202
|
+
fetched += 1;
|
|
203
|
+
if (raw.length > MAX_FOLLOWED_FILE_BYTES) {
|
|
204
|
+
log.warn(`Skipping ${link.path}: over ${MAX_FOLLOWED_FILE_BYTES} bytes.`);
|
|
205
|
+
continue;
|
|
206
|
+
}
|
|
207
|
+
const fileTree = processor.parse(stripFrontmatter(raw));
|
|
208
|
+
const ownTitle = stripTitleHeading(fileTree);
|
|
209
|
+
const baseDir = link.path.includes('/')
|
|
210
|
+
? link.path.slice(0, link.path.lastIndexOf('/'))
|
|
211
|
+
: '';
|
|
212
|
+
for (const nested of collectDocLinks(fileTree, baseDir, source)) {
|
|
213
|
+
if (!visited.has(nested.path)) {
|
|
214
|
+
queue.push(nested);
|
|
215
|
+
}
|
|
216
|
+
}
|
|
217
|
+
if (countGitHubRepos(fileTree) < minLinks) {
|
|
218
|
+
continue;
|
|
219
|
+
}
|
|
220
|
+
const label = labels.get(link.path) ?? '';
|
|
221
|
+
const title = isFilenameLabel(label, link.path)
|
|
222
|
+
? ownTitle || label || link.path
|
|
223
|
+
: label;
|
|
224
|
+
const heading = {
|
|
225
|
+
type: 'heading',
|
|
226
|
+
depth: categoryHeadingDepth(fileTree.children),
|
|
227
|
+
children: [{ type: 'text', value: title }],
|
|
228
|
+
};
|
|
229
|
+
rewriteRelativeLinksToSource(fileTree, baseDir, source);
|
|
230
|
+
tree.children.push(heading, ...fileTree.children);
|
|
231
|
+
appended += 1;
|
|
232
|
+
}
|
|
233
|
+
rewriteRelativeLinksToSource(tree, '', source);
|
|
234
|
+
log.info(`README skeleton: followed ${fetched} internal file(s), appended ${appended} as categories.`);
|
|
235
|
+
}
|
|
236
|
+
export async function processMarkdownContent(originalContent, token, replacements = [], sortOptions = { by: '', minLinks: 2 }, relativeLinkPrefix = '', enhancedRepository, enhancedRepositoryDescription, originalRepositorySha, originalRepositoryInfo, sourceRepository, previousJson, now = new Date(), log = consoleLog) {
|
|
89
237
|
const repos = createRepoInfoLookup(token, log);
|
|
90
238
|
const brandingEnabled = replacements.some(rule => rule.type === 'branding');
|
|
91
239
|
const contentAfterReplacements = applyTextReplacements(originalContent, replacements.filter(rule => rule.type !== 'branding'), log);
|
|
92
240
|
const processor = unified().use(remarkParse).use(remarkGfm);
|
|
93
241
|
const tree = processor.parse(contentAfterReplacements);
|
|
94
242
|
normalizeGitHubUrls(tree);
|
|
243
|
+
// A README too thin to carry the list is a landing page: the content lives
|
|
244
|
+
// in the files it links. Splice them in before anything reads the tree, so
|
|
245
|
+
// the repo fetch, badges and the section walk all see the combined document.
|
|
246
|
+
if (sourceRepository && countGitHubRepos(tree) < sortOptions.minLinks) {
|
|
247
|
+
await appendInternalDocs(tree, sourceRepository, repos, sortOptions.minLinks, log);
|
|
248
|
+
normalizeGitHubUrls(tree);
|
|
249
|
+
}
|
|
95
250
|
const githubUrls = collectGitHubLinks(tree);
|
|
96
|
-
|
|
251
|
+
// Links into the source repository itself are navigation — its doc pages,
|
|
252
|
+
// its contributing links — never entries. android-root's index links every
|
|
253
|
+
// category page, and each of those links would borrow the source repo's
|
|
254
|
+
// identity and emit an item for it. Excluding the source from the target
|
|
255
|
+
// fetch leaves every emission path reading such a link as a dead one: no
|
|
256
|
+
// item, children lifted, the row left in markdown.
|
|
257
|
+
const targetUrls = sourceRepository &&
|
|
258
|
+
new Set([...githubUrls].filter(url => {
|
|
259
|
+
const id = parseGitHubUrl(url);
|
|
260
|
+
return !(id &&
|
|
261
|
+
`${id.owner}/${id.repo}`.toLowerCase() ===
|
|
262
|
+
`${sourceRepository.owner}/${sourceRepository.repo}`.toLowerCase());
|
|
263
|
+
}));
|
|
264
|
+
const repoInfoMap = await fetchTargetData(targetUrls ?? githubUrls, repos);
|
|
97
265
|
const firstSeen = firstSeenFor(previousJson, now);
|
|
98
266
|
const { sections, title: rawTitle, titleHeadingIndex, } = processTree(tree, repoInfoMap, sortOptions, originalRepositoryInfo, firstSeen);
|
|
99
267
|
// Single source of truth for the document title: brand it once and use the
|
|
@@ -181,8 +349,31 @@ function collectGitHubLinks(tree) {
|
|
|
181
349
|
urls.add(node.url);
|
|
182
350
|
}
|
|
183
351
|
});
|
|
352
|
+
// A details summary's anchor stays raw html (normalizeGitHubUrls leaves
|
|
353
|
+
// block html alone), so the identity it names joins the fetch set here;
|
|
354
|
+
// the summary promotion looks the repo up by the same href.
|
|
355
|
+
visit(tree, 'html', (node) => {
|
|
356
|
+
const identity = parseDetailsSummary(node.value)?.identity;
|
|
357
|
+
if (identity) {
|
|
358
|
+
urls.add(identity.url);
|
|
359
|
+
}
|
|
360
|
+
});
|
|
184
361
|
return urls;
|
|
185
362
|
}
|
|
363
|
+
// The number of distinct GitHub REPOS a document addresses — the census unit
|
|
364
|
+
// for both the skeleton trigger and the splice gate. Distinct URLs would
|
|
365
|
+
// overcount: a skeleton README badges the source repo and links its issues
|
|
366
|
+
// page, which is two URLs over one repo.
|
|
367
|
+
function countGitHubRepos(tree) {
|
|
368
|
+
const repos = new Set();
|
|
369
|
+
visit(tree, 'link', (node) => {
|
|
370
|
+
const id = parseGitHubUrl(node.url);
|
|
371
|
+
if (id) {
|
|
372
|
+
repos.add(`${id.owner}/${id.repo}`.toLowerCase());
|
|
373
|
+
}
|
|
374
|
+
});
|
|
375
|
+
return repos.size;
|
|
376
|
+
}
|
|
186
377
|
// Input normalization, same family as fixRelativeLinks: make GitHub repos a
|
|
187
378
|
// source expresses WITHOUT markdown links visible as real link nodes, so
|
|
188
379
|
// every downstream consumer — repo fetch, entry tests, gates, badges — sees
|
|
@@ -478,10 +669,14 @@ function processListRecursively(listNode, repoInfoMap, sortOptions, firstSeen, i
|
|
|
478
669
|
// The caller's section-scope gate decision (sectionGatePasses). Absent for
|
|
479
670
|
// nested lists (emitted under their parent regardless) and for top-level
|
|
480
671
|
// lists with no open container (preamble), which gate per list.
|
|
481
|
-
sectionGateOpen
|
|
672
|
+
sectionGateOpen,
|
|
673
|
+
// The promoted details-item this list sits directly inside: an entry
|
|
674
|
+
// restating that repo (the best-of "GitHub" bullet) is the entry the
|
|
675
|
+
// summary already emitted, not a second item — its children still lift,
|
|
676
|
+
// the way a dead link's do.
|
|
677
|
+
suppressRepo) {
|
|
482
678
|
if (!isNested) {
|
|
483
|
-
const gateOpen = sectionGateOpen ??
|
|
484
|
-
countLinkedItems(listNode) >= sortOptions.minLinks;
|
|
679
|
+
const gateOpen = sectionGateOpen ?? countLinkedItems(listNode) >= sortOptions.minLinks;
|
|
485
680
|
if (!gateOpen) {
|
|
486
681
|
return [];
|
|
487
682
|
}
|
|
@@ -503,6 +698,10 @@ sectionGateOpen) {
|
|
|
503
698
|
const childrenJson = nestedContent.flatMap(child => child.type === 'list'
|
|
504
699
|
? processListRecursively(child, repoInfoMap, sortOptions, firstSeen, true)
|
|
505
700
|
: processTableRows(child, repoInfoMap, true, firstSeen));
|
|
701
|
+
if (suppressRepo && repoInfo === suppressRepo) {
|
|
702
|
+
entries.push({ emitted: childrenJson, node: itemNode, repoInfo });
|
|
703
|
+
continue;
|
|
704
|
+
}
|
|
506
705
|
// Title/description split on the FIRST paragraph only — a paper-list
|
|
507
706
|
// entry's identity link may live in a later paragraph (findOwnGitHubLink)
|
|
508
707
|
// while its title text stays the leading one.
|
|
@@ -516,6 +715,7 @@ sectionGateOpen) {
|
|
|
516
715
|
// The shared title fallbacks (an inline-code link label carries no text
|
|
517
716
|
// nodes, so the split alone can leave an empty title).
|
|
518
717
|
entryText.title = entryTitle(entryText.title, ownLink, repoInfo);
|
|
718
|
+
entryText.description = entryDescription(entryText.description, repoInfo);
|
|
519
719
|
const emitted = emitEntryNodes(githubUrl, repoInfo, entryText, childrenJson, firstSeen);
|
|
520
720
|
entries.push({ emitted, node: itemNode, repoInfo });
|
|
521
721
|
}
|
|
@@ -544,6 +744,23 @@ function entryTitle(base, ownLink, repoInfo) {
|
|
|
544
744
|
}
|
|
545
745
|
return repoInfo ? `${repoInfo.owner}/${repoInfo.repo}` : base;
|
|
546
746
|
}
|
|
747
|
+
// Description with the fallback every entry source shares: the split's own
|
|
748
|
+
// trailing prose when it says something — minus a leading badge cluster —
|
|
749
|
+
// else — for a live repo link — owner/name. The own link's label is never
|
|
750
|
+
// consulted: in the corpus it only ever echoes the title back (":tada:
|
|
751
|
+
// Doom" already says "Doom"). A degenerate base with no live repo link (a
|
|
752
|
+
// group) keeps its text: nothing better exists.
|
|
753
|
+
function entryDescription(base, repoInfo) {
|
|
754
|
+
// The noise strip runs only behind a stripped cluster: a bare leading
|
|
755
|
+
// dash or colon can be a word's own (the table paths never noise-stripped
|
|
756
|
+
// their cells, and "-equivalent" / ":bird:" descriptions are real).
|
|
757
|
+
const withoutBadges = stripBadgeClusters(base);
|
|
758
|
+
const stripped = withoutBadges === base ? base : stripLeadingNoise(withoutBadges);
|
|
759
|
+
if (!isDegenerateDescription(stripped)) {
|
|
760
|
+
return stripped;
|
|
761
|
+
}
|
|
762
|
+
return repoInfo ? `${repoInfo.owner}/${repoInfo.repo}` : stripped;
|
|
763
|
+
}
|
|
547
764
|
function processTableRows(tableNode, repoInfoMap, gateOpen, firstSeen) {
|
|
548
765
|
if (!gateOpen) {
|
|
549
766
|
return [];
|
|
@@ -583,9 +800,10 @@ function processTableRows(tableNode, repoInfoMap, gateOpen, firstSeen) {
|
|
|
583
800
|
]
|
|
584
801
|
.join(' ')
|
|
585
802
|
.trim();
|
|
803
|
+
const title = entryTitle(titleText.title, ownLink, repoInfo);
|
|
586
804
|
items.push(...emitEntryNodes(ownLink.url, repoInfo, {
|
|
587
|
-
title
|
|
588
|
-
description,
|
|
805
|
+
title,
|
|
806
|
+
description: entryDescription(description, repoInfo),
|
|
589
807
|
}, [], firstSeen));
|
|
590
808
|
continue;
|
|
591
809
|
}
|
|
@@ -597,9 +815,10 @@ function processTableRows(tableNode, repoInfoMap, gateOpen, firstSeen) {
|
|
|
597
815
|
emittedUrls.add(ownLink.url);
|
|
598
816
|
const repoInfo = repoInfoMap.get(ownLink.url) ?? null;
|
|
599
817
|
const cellText = splitEntryText(row.children[cellIndex].children);
|
|
818
|
+
const title = entryTitle(cellText.title, ownLink, repoInfo);
|
|
600
819
|
items.push(...emitEntryNodes(ownLink.url, repoInfo, {
|
|
601
|
-
title
|
|
602
|
-
description: cellText.description,
|
|
820
|
+
title,
|
|
821
|
+
description: entryDescription(cellText.description, repoInfo),
|
|
603
822
|
}, [], firstSeen));
|
|
604
823
|
}
|
|
605
824
|
}
|
|
@@ -620,6 +839,13 @@ function isDegenerateTitle(title) {
|
|
|
620
839
|
function isMeaningfulLinkText(text) {
|
|
621
840
|
return text !== '' && !isDegenerateTitle(text);
|
|
622
841
|
}
|
|
842
|
+
// Description-only tag words beyond the title set: the second-link label
|
|
843
|
+
// families the corpus census measured ("website", "documentation", …).
|
|
844
|
+
const DESC_TAG_TEXT = /^(?:website|homepage|home\s+page|link|here|documentation|repository)$/i;
|
|
845
|
+
function isDegenerateDescription(description) {
|
|
846
|
+
const trimmed = description.trim();
|
|
847
|
+
return (trimmed === '' || DESC_TAG_TEXT.test(trimmed) || isDegenerateTitle(trimmed));
|
|
848
|
+
}
|
|
623
849
|
// Dated entry lines end their tag cluster with a publication date ("4 Feb
|
|
624
850
|
// 2023", "19 Apr 2022", "2023") — a real corpus family (dated paper/tutorial
|
|
625
851
|
// lists), not a sentence continuation.
|
|
@@ -735,9 +961,10 @@ function blockquoteEntries(blockquote) {
|
|
|
735
961
|
function entryNodesFor(ownLink, inlines, repoInfoMap, firstSeen) {
|
|
736
962
|
const repoInfo = repoInfoMap.get(ownLink.url) ?? null;
|
|
737
963
|
const entryText = splitEntryText(inlines);
|
|
964
|
+
const title = entryTitle(entryText.title, ownLink, repoInfo);
|
|
738
965
|
return emitEntryNodes(ownLink.url, repoInfo, {
|
|
739
|
-
title
|
|
740
|
-
description: entryText.description,
|
|
966
|
+
title,
|
|
967
|
+
description: entryDescription(entryText.description, repoInfo),
|
|
741
968
|
}, [], firstSeen);
|
|
742
969
|
}
|
|
743
970
|
// The <details><summary>…</summary> collapsible-section idiom: the summary
|
|
@@ -747,17 +974,54 @@ function entryNodesFor(ownLink, inlines, repoInfoMap, firstSeen) {
|
|
|
747
974
|
// no summary, or an empty one, delimits nothing.
|
|
748
975
|
const DETAILS_SUMMARY = /<details[^>]*>[\s\S]*?<summary[^>]*>([\s\S]*?)<\/summary>/i;
|
|
749
976
|
const DETAILS_CLOSE = /^\s*<\/details>/i;
|
|
750
|
-
|
|
977
|
+
// One summary's read of the walk: the summary's whole text (the container
|
|
978
|
+
// title when the block is not an entry), and — when the summary carries the
|
|
979
|
+
// entry's identity (the best-of generator's shape: repo anchor, badge
|
|
980
|
+
// cluster, curated sentence) — that anchor and the prose trailing it.
|
|
981
|
+
const SUMMARY_ANCHOR = /<a\s[^>]*href=(["'])([^"']*)\1[^>]*>([\s\S]*?)<\/a>/gi;
|
|
982
|
+
function summaryHtmlText(html) {
|
|
983
|
+
return html
|
|
984
|
+
.replace(/<[^>]+>/g, ' ')
|
|
985
|
+
.replace(/\s+/g, ' ')
|
|
986
|
+
.trim();
|
|
987
|
+
}
|
|
988
|
+
function parseDetailsSummary(htmlValue) {
|
|
751
989
|
const match = DETAILS_SUMMARY.exec(htmlValue);
|
|
752
990
|
if (!match) {
|
|
753
991
|
return null;
|
|
754
992
|
}
|
|
755
|
-
const
|
|
993
|
+
const inner = match[1];
|
|
994
|
+
const title = inner
|
|
756
995
|
.replace(/<kbd>[\s\S]*?<\/kbd>/gi, '')
|
|
757
996
|
.replace(/<[^>]+>/g, ' ')
|
|
758
997
|
.replace(/\s+/g, ' ')
|
|
759
998
|
.trim();
|
|
760
|
-
|
|
999
|
+
let identity = null;
|
|
1000
|
+
let prose = '';
|
|
1001
|
+
for (const anchor of inner.matchAll(SUMMARY_ANCHOR)) {
|
|
1002
|
+
if (!parseGitHubUrl(anchor[2])) {
|
|
1003
|
+
continue;
|
|
1004
|
+
}
|
|
1005
|
+
identity = { label: summaryHtmlText(anchor[3]), url: anchor[2] };
|
|
1006
|
+
prose = summaryHtmlText(inner.slice((anchor.index ?? 0) + anchor[0].length));
|
|
1007
|
+
break;
|
|
1008
|
+
}
|
|
1009
|
+
return { identity, prose, title: title || null };
|
|
1010
|
+
}
|
|
1011
|
+
function detailsSummaryTitle(htmlValue) {
|
|
1012
|
+
return parseDetailsSummary(htmlValue)?.title ?? null;
|
|
1013
|
+
}
|
|
1014
|
+
// A badge cluster from a generated summary: a parenthesized run of medals,
|
|
1015
|
+
// counts and size suffixes (🥈31 · ⭐ 1.9K) — symbols, digits and punctuation,
|
|
1016
|
+
// no words, at least one of them. A parenthetical with any other letter (an
|
|
1017
|
+
// aside in prose) or an empty pair () stays.
|
|
1018
|
+
const BADGE_CLUSTER = /^\([\p{P}\p{S}\p{N}\sKMG]+\)\s*/u;
|
|
1019
|
+
function stripBadgeClusters(text) {
|
|
1020
|
+
let out = text.trimStart();
|
|
1021
|
+
while (BADGE_CLUSTER.test(out)) {
|
|
1022
|
+
out = out.replace(BADGE_CLUSTER, '');
|
|
1023
|
+
}
|
|
1024
|
+
return out;
|
|
761
1025
|
}
|
|
762
1026
|
// The summary text of a <details><summary> block inside a list item, when the
|
|
763
1027
|
// item has no paragraph text of its own.
|
|
@@ -910,7 +1174,7 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepositoryInfo, fir
|
|
|
910
1174
|
// comparison holds even for aliased spellings.
|
|
911
1175
|
const rementionsOpenItem = (link) => {
|
|
912
1176
|
const repoInfo = repoInfoMap.get(link.url);
|
|
913
|
-
return !!repoInfo && stack.some(container => container.repoInfo === repoInfo);
|
|
1177
|
+
return (!!repoInfo && stack.some(container => container.repoInfo === repoInfo));
|
|
914
1178
|
};
|
|
915
1179
|
for (let i = 0; i < tree.children.length; i++) {
|
|
916
1180
|
const node = tree.children[i];
|
|
@@ -933,7 +1197,9 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepositoryInfo, fir
|
|
|
933
1197
|
else if (node.type === 'paragraph') {
|
|
934
1198
|
const text = getNodeText(node);
|
|
935
1199
|
// Boilerplate "back to top" lines are neither entries nor description.
|
|
936
|
-
const ownLink = BACK_TO_TOP.test(text)
|
|
1200
|
+
const ownLink = BACK_TO_TOP.test(text)
|
|
1201
|
+
? undefined
|
|
1202
|
+
: paragraphEntryLink(node);
|
|
937
1203
|
// An entry paragraph behaves like a one-item list; a failed gate, a
|
|
938
1204
|
// dead target, or a re-mention of the enclosing item leaves plain
|
|
939
1205
|
// description.
|
|
@@ -943,7 +1209,10 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepositoryInfo, fir
|
|
|
943
1209
|
else {
|
|
944
1210
|
const container = stack[stack.length - 1];
|
|
945
1211
|
// Avoid adding boilerplate "back to top" links to descriptions.
|
|
946
|
-
if (container &&
|
|
1212
|
+
if (container &&
|
|
1213
|
+
text &&
|
|
1214
|
+
!BACK_TO_TOP.test(text) &&
|
|
1215
|
+
collectsProse(container)) {
|
|
947
1216
|
container.description = container.description
|
|
948
1217
|
? `${container.description}\n${text}`
|
|
949
1218
|
: text;
|
|
@@ -966,7 +1235,10 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepositoryInfo, fir
|
|
|
966
1235
|
else {
|
|
967
1236
|
const container = stack[stack.length - 1];
|
|
968
1237
|
// Avoid adding boilerplate "back to top" links to descriptions.
|
|
969
|
-
if (container &&
|
|
1238
|
+
if (container &&
|
|
1239
|
+
text &&
|
|
1240
|
+
!BACK_TO_TOP.test(text) &&
|
|
1241
|
+
collectsProse(container)) {
|
|
970
1242
|
container.description = container.description
|
|
971
1243
|
? `${container.description}\n${text}`
|
|
972
1244
|
: text;
|
|
@@ -974,30 +1246,61 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepositoryInfo, fir
|
|
|
974
1246
|
}
|
|
975
1247
|
}
|
|
976
1248
|
else if (node.type === 'html') {
|
|
977
|
-
// A details-summary block
|
|
978
|
-
//
|
|
979
|
-
//
|
|
980
|
-
|
|
981
|
-
|
|
1249
|
+
// A details-summary block is the entry itself when its summary carries
|
|
1250
|
+
// the identity (the best-of generator's shape) — promoted exactly like
|
|
1251
|
+
// a link-bearing heading, so the entry nests in its real category with
|
|
1252
|
+
// the summary's curated sentence as its description and the badges
|
|
1253
|
+
// gone. Any other summary opens a container like a heading would: a
|
|
1254
|
+
// group under the open container (the JSON has no nested sections), a
|
|
1255
|
+
// section at document level. The close tag (or the next summary) ends
|
|
1256
|
+
// it — never the enclosing section, which keeps collecting after the
|
|
1257
|
+
// collapsible block.
|
|
1258
|
+
const summary = parseDetailsSummary(node.value);
|
|
1259
|
+
if (summary) {
|
|
982
1260
|
closeInnermostDetails(stack, sectionRecords, firstSeen);
|
|
983
1261
|
// Container depths never decrease going up the stack. When the open
|
|
984
1262
|
// containers sit deeper than sectionDepth (a mid-document H1 defines
|
|
985
|
-
// sectionDepth while the content sections run deeper), the
|
|
986
|
-
//
|
|
987
|
-
//
|
|
988
|
-
//
|
|
1263
|
+
// sectionDepth while the content sections run deeper), the container
|
|
1264
|
+
// joins at the current depth — pushing at the shallower sectionDepth
|
|
1265
|
+
// would invert the stack and strand the gate (which reads the stack
|
|
1266
|
+
// bottom) on a tiny outer section.
|
|
989
1267
|
const joinDepth = stack.length === 0
|
|
990
1268
|
? sectionDepth
|
|
991
1269
|
: Math.max(sectionDepth, stack[stack.length - 1].headingDepth);
|
|
992
|
-
|
|
993
|
-
|
|
994
|
-
|
|
995
|
-
|
|
996
|
-
|
|
997
|
-
|
|
998
|
-
|
|
999
|
-
|
|
1000
|
-
|
|
1270
|
+
const repoInfo = summary.identity
|
|
1271
|
+
? (repoInfoMap.get(summary.identity.url) ?? null)
|
|
1272
|
+
: null;
|
|
1273
|
+
const promoted = summary.identity &&
|
|
1274
|
+
repoInfo &&
|
|
1275
|
+
isMeaningfulLinkText(summary.identity.label)
|
|
1276
|
+
? { identity: summary.identity, repoInfo }
|
|
1277
|
+
: null;
|
|
1278
|
+
if (promoted && stack.length === 0) {
|
|
1279
|
+
openSynthesizedSection(stack, i, sectionDepth);
|
|
1280
|
+
}
|
|
1281
|
+
if (promoted && gateForSection(stack[0])) {
|
|
1282
|
+
stack.push({
|
|
1283
|
+
children: [],
|
|
1284
|
+
description: entryDescription(summary.prose, promoted.repoInfo),
|
|
1285
|
+
headingDepth: joinDepth,
|
|
1286
|
+
headingIndex: i,
|
|
1287
|
+
kind: 'item',
|
|
1288
|
+
openedByDetails: true,
|
|
1289
|
+
repoInfo: promoted.repoInfo,
|
|
1290
|
+
title: promoted.identity.label,
|
|
1291
|
+
});
|
|
1292
|
+
}
|
|
1293
|
+
else {
|
|
1294
|
+
stack.push({
|
|
1295
|
+
children: [],
|
|
1296
|
+
description: '',
|
|
1297
|
+
headingDepth: joinDepth,
|
|
1298
|
+
headingIndex: i,
|
|
1299
|
+
kind: stack.length === 0 ? 'section' : 'group',
|
|
1300
|
+
openedByDetails: true,
|
|
1301
|
+
title: summary.title ?? '',
|
|
1302
|
+
});
|
|
1303
|
+
}
|
|
1001
1304
|
}
|
|
1002
1305
|
else if (DETAILS_CLOSE.test(node.value)) {
|
|
1003
1306
|
closeInnermostDetails(stack, sectionRecords, firstSeen);
|
|
@@ -1013,8 +1316,13 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepositoryInfo, fir
|
|
|
1013
1316
|
}
|
|
1014
1317
|
// Every list inside the open container contributes items — a section is
|
|
1015
1318
|
// not closed by its first list — and the minLinks gate is decided per
|
|
1016
|
-
// section, against the whole section subtree.
|
|
1017
|
-
|
|
1319
|
+
// section, against the whole section subtree. A list directly inside a
|
|
1320
|
+
// promoted details-item suppresses the re-mention of that item's repo.
|
|
1321
|
+
const innermost = stack[stack.length - 1];
|
|
1322
|
+
const suppress = innermost.kind === 'item' && innermost.openedByDetails
|
|
1323
|
+
? innermost.repoInfo
|
|
1324
|
+
: undefined;
|
|
1325
|
+
const items = processListRecursively(node, repoInfoMap, sortOptions, firstSeen, false, gateForSection(stack[0]), suppress);
|
|
1018
1326
|
stack[stack.length - 1].children.push(...items);
|
|
1019
1327
|
}
|
|
1020
1328
|
else if (node.type === 'table') {
|
|
@@ -1034,6 +1342,12 @@ function processTree(tree, repoInfoMap, sortOptions, originalRepositoryInfo, fir
|
|
|
1034
1342
|
.map(record => record.section);
|
|
1035
1343
|
return { sections, title: documentTitle, titleHeadingIndex };
|
|
1036
1344
|
}
|
|
1345
|
+
// A details-promoted item's description is the summary's curated sentence;
|
|
1346
|
+
// prose inside the collapsible body is the card's content, not more
|
|
1347
|
+
// description.
|
|
1348
|
+
function collectsProse(container) {
|
|
1349
|
+
return container.kind !== 'item' || !container.openedByDetails;
|
|
1350
|
+
}
|
|
1037
1351
|
// The heading depth that opens top-level sections: the shallowest structural
|
|
1038
1352
|
// heading in the document other than the title slot. H1s count — ~20% of
|
|
1039
1353
|
// mirror READMEs use `# Section` after the title H1, and hardcoding H2 would
|
|
@@ -1297,7 +1611,11 @@ function finalizeContainer(container, stack, sectionRecords, firstSeen) {
|
|
|
1297
1611
|
if (container.kind === 'section') {
|
|
1298
1612
|
sectionRecords.push({
|
|
1299
1613
|
headingIndex: container.headingIndex,
|
|
1300
|
-
section: {
|
|
1614
|
+
section: {
|
|
1615
|
+
description,
|
|
1616
|
+
items: container.children,
|
|
1617
|
+
title: container.title,
|
|
1618
|
+
},
|
|
1301
1619
|
});
|
|
1302
1620
|
return;
|
|
1303
1621
|
}
|
|
@@ -1328,8 +1646,7 @@ function finalizeContainer(container, stack, sectionRecords, firstSeen) {
|
|
|
1328
1646
|
// entry heading triggers must stop at the synthesized section wrapping its
|
|
1329
1647
|
// run — the next entry heading of the run lands back inside it.
|
|
1330
1648
|
function closeContainers(stack, depth, sectionRecords, firstSeen, stopAtSynthesized = false) {
|
|
1331
|
-
while (stack.length > 0 &&
|
|
1332
|
-
stack[stack.length - 1].headingDepth >= depth) {
|
|
1649
|
+
while (stack.length > 0 && stack[stack.length - 1].headingDepth >= depth) {
|
|
1333
1650
|
if (stopAtSynthesized && stack[stack.length - 1].openedBySynthesis) {
|
|
1334
1651
|
break;
|
|
1335
1652
|
}
|
package/dist/orchestrator.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { Logger } from './logger.js';
|
|
2
|
-
import type { RepoInfoDetails } from './github.js';
|
|
2
|
+
import type { RepoInfoDetails, RepoIdentifier } from './github.js';
|
|
3
3
|
import { JsonOutput, ReplacementRule } from './markdown.js';
|
|
4
4
|
export interface EnhanceOptions {
|
|
5
5
|
content: string;
|
|
@@ -14,6 +14,7 @@ export interface EnhanceOptions {
|
|
|
14
14
|
relativeLinkPrefix?: string;
|
|
15
15
|
replacements?: ReplacementRule[];
|
|
16
16
|
sortBy?: '' | 'last_commit' | 'stars';
|
|
17
|
+
sourceRepository?: RepoIdentifier;
|
|
17
18
|
token: string;
|
|
18
19
|
}
|
|
19
20
|
export interface EnhanceResult {
|
package/dist/orchestrator.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { processMarkdownContent, } from './markdown.js';
|
|
2
2
|
export async function enhance(options) {
|
|
3
|
-
const { content, disableBranding = false, log, now = new Date(), originalRepositoryInfo, originalRepositorySha, previousJson, relativeLinkPrefix = '', replacements = [], sortBy = '', enhancedRepository, enhancedRepositoryDescription, token, } = options;
|
|
3
|
+
const { content, disableBranding = false, log, now = new Date(), originalRepositoryInfo, originalRepositorySha, previousJson, relativeLinkPrefix = '', replacements = [], sortBy = '', sourceRepository, enhancedRepository, enhancedRepositoryDescription, token, } = options;
|
|
4
4
|
// Branding is an internal rule prepended to the caller's own; build a fresh
|
|
5
5
|
// array so the caller's `replacements` is never mutated.
|
|
6
6
|
const branding = { type: 'branding' };
|
|
@@ -9,7 +9,7 @@ export async function enhance(options) {
|
|
|
9
9
|
by: sortBy,
|
|
10
10
|
minLinks: 2,
|
|
11
11
|
};
|
|
12
|
-
const { finalContent, jsonData } = await processMarkdownContent(content, token, rules, sortOptions, relativeLinkPrefix, enhancedRepository, enhancedRepositoryDescription, originalRepositorySha, originalRepositoryInfo, previousJson, now, log);
|
|
12
|
+
const { finalContent, jsonData } = await processMarkdownContent(content, token, rules, sortOptions, relativeLinkPrefix, enhancedRepository, enhancedRepositoryDescription, originalRepositorySha, originalRepositoryInfo, sourceRepository, previousJson, now, log);
|
|
13
13
|
return {
|
|
14
14
|
finalContent,
|
|
15
15
|
jsonData,
|