@enhansome/core 1.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,89 @@
1
+ import { Octokit } from '@octokit/core';
2
+ import { Logger } from './logger.js';
3
+ import type { ThrottlingOptions } from '@octokit/plugin-throttling';
4
+ declare const HardenedOctokit: typeof Octokit & import("@octokit/core/types").Constructor<import("@octokit/plugin-rest-endpoint-methods").Api & {
5
+ paginate: import("@octokit/plugin-paginate-rest").PaginateInterface;
6
+ } & import("@octokit/plugin-retry").RetryPlugin>;
7
+ export type GithubClient = InstanceType<typeof HardenedOctokit>;
8
+ export interface RepoInfoDetails {
9
+ archived: boolean;
10
+ description: null | string;
11
+ language: null | string;
12
+ open_issues_count: number;
13
+ owner: string;
14
+ pushed_at: null | string;
15
+ repo: string;
16
+ stargazers_count: number;
17
+ topics: string[];
18
+ }
19
+ export interface RepoIdentifier {
20
+ owner: string;
21
+ repo: string;
22
+ }
23
+ /**
24
+ * Exported for unit testing; wired into the throttling plugin by `makeOctokit`.
25
+ *
26
+ * `maxWaitSeconds` caps how long a single retry will wait for a rate-limit
27
+ * reset. It defaults to 300s: the Action must self-bound its run time so a
28
+ * workflow never hangs on a limit it cannot service. Longer-lived callers raise
29
+ * it to wait out a reset instead of aborting into an unrecoverable 403.
30
+ *
31
+ * `maxRetries` bounds the rate-limit retry budget reported here and matches the
32
+ * transport-retry count set in `makeOctokit`, so the two budgets stay aligned.
33
+ */
34
+ export declare function createRateLimitHandler(kind: 'primary' | 'secondary', log?: Logger, maxWaitSeconds?: number, maxRetries?: number): (retryAfter: number, reqOptions: {
35
+ method: string;
36
+ url: string;
37
+ }, _octokit: unknown, retryCount: number) => boolean;
38
+ type ThrottleGroup = NonNullable<ThrottlingOptions['write']>;
39
+ /**
40
+ * Pass-through overrides for `@octokit/plugin-throttling`. The plugin's groups
41
+ * are process-wide singletons by default; supplying your own detaches this
42
+ * client so its limits apply per-instance instead of being shared across every
43
+ * Octokit in the process.
44
+ */
45
+ export interface ThrottleOptions {
46
+ auth?: ThrottleGroup;
47
+ /** Secondary-rate-limit retry wait in seconds when the response carries no retry-after header (plugin default 60). */
48
+ fallbackSecondaryRateRetryAfter?: number;
49
+ global?: ThrottleGroup;
50
+ notifications?: ThrottleGroup;
51
+ search?: ThrottleGroup;
52
+ /** Per-request Bottleneck timeout in ms (plugin default 120_000). */
53
+ timeout?: number;
54
+ write?: ThrottleGroup;
55
+ }
56
+ /**
57
+ * Tuning for the hardened client built by `makeOctokit`. All fields optional;
58
+ * omitting the bag yields the Action defaults — `consoleLog`, 3 retries, a 300s
59
+ * rate-limit cap, and the throttling plugin's built-in limits.
60
+ */
61
+ export interface MakeOctokitOptions {
62
+ /** Sink for debug/warn/error; defaults to `consoleLog`. */
63
+ log?: Logger;
64
+ /** Transport and rate-limit retry budget (default 3). */
65
+ maxRetries?: number;
66
+ /** Rate-limit retry-wait cap in seconds (default 300); see `createRateLimitHandler`. */
67
+ maxWaitSeconds?: number;
68
+ /** Throttling overrides forwarded into the plugin's `throttle` config. */
69
+ throttle?: ThrottleOptions;
70
+ }
71
+ /**
72
+ * The client is where the sink lives: Octokit takes a `log` and exposes it as
73
+ * `octokit.log`, so every function below reaches it through the client it is
74
+ * already given, instead of taking a logger of its own.
75
+ */
76
+ export declare function makeOctokit(token: string, { log, maxRetries, maxWaitSeconds, throttle, }?: MakeOctokitOptions): GithubClient;
77
+ export declare function getRepoInfo(octokit: GithubClient, owner: string, repo: string): Promise<RepoInfoDetails>;
78
+ export declare function getReadme(octokit: GithubClient, owner: string, repo: string, format?: 'html' | 'raw'): Promise<string>;
79
+ /** Root file/directory names — the compile-manifest gate reads them to tell a
80
+ * directory of resources from a repo that IS the deliverable. A `path: ''`
81
+ * listing is always an array; the single-entry branch defends against a file
82
+ * response. */
83
+ export declare function getRootEntryNames(octokit: GithubClient, owner: string, repo: string): Promise<string[]>;
84
+ /** Source revision for the JSON output, so an enhanced list is traceable to its origin commit. */
85
+ export declare function getLatestCommitSha(octokit: GithubClient, owner: string, repo: string): Promise<null | string>;
86
+ export declare function parseOwnerRepo(value: string): null | RepoIdentifier;
87
+ export declare function parseGitHubUrl(url: string): null | RepoIdentifier;
88
+ export declare function formatRequestError(error: unknown): string;
89
+ export {};
package/dist/github.js ADDED
@@ -0,0 +1,172 @@
1
+ import { Octokit } from '@octokit/core';
2
+ import { paginateRest } from '@octokit/plugin-paginate-rest';
3
+ import { restEndpointMethods } from '@octokit/plugin-rest-endpoint-methods';
4
+ import { retry } from '@octokit/plugin-retry';
5
+ import { throttling } from '@octokit/plugin-throttling';
6
+ import { consoleLog } from './logger.js';
7
+ const DEFAULT_MAX_WAIT_TIME_SECONDS = 300, MAX_RETRIES = 3;
8
+ // `@actions/github`'s `GitHub` is `Octokit.plugin(restEndpointMethods,
9
+ // paginateRest).defaults(...)` — it pre-applies REST endpoint methods
10
+ // (`octokit.rest.*`) and pagination. We dropped `@actions/github` to keep core
11
+ // free of the Actions runtime, so re-apply those two plugins here alongside the
12
+ // retry/throttling hardening, on a bare `Octokit`. Note this drops
13
+ // `@actions/github`'s `.defaults` (GHE baseUrl + proxy agent/fetch) — acceptable
14
+ // for the github.com + GITHUB_TOKEN path the action uses; self-hosted runners
15
+ // behind a proxy or GHE installs would regress.
16
+ const HardenedOctokit = Octokit.plugin(restEndpointMethods, paginateRest, retry, throttling);
17
+ /**
18
+ * Exported for unit testing; wired into the throttling plugin by `makeOctokit`.
19
+ *
20
+ * `maxWaitSeconds` caps how long a single retry will wait for a rate-limit
21
+ * reset. It defaults to 300s: the Action must self-bound its run time so a
22
+ * workflow never hangs on a limit it cannot service. Longer-lived callers raise
23
+ * it to wait out a reset instead of aborting into an unrecoverable 403.
24
+ *
25
+ * `maxRetries` bounds the rate-limit retry budget reported here and matches the
26
+ * transport-retry count set in `makeOctokit`, so the two budgets stay aligned.
27
+ */
28
+ export function createRateLimitHandler(kind, log = consoleLog, maxWaitSeconds = DEFAULT_MAX_WAIT_TIME_SECONDS, maxRetries = MAX_RETRIES) {
29
+ return (retryAfter, reqOptions, _octokit, retryCount) => {
30
+ const where = `${reqOptions.method} ${reqOptions.url}`;
31
+ if (retryAfter > maxWaitSeconds) {
32
+ log.error(`${kind} rate limit retry-after (${retryAfter}s) exceeds the maximum wait time of ${maxWaitSeconds}s. Aborting retries for ${where}.`);
33
+ return false;
34
+ }
35
+ if (retryCount >= maxRetries) {
36
+ log.error(`Giving up on ${where} after ${maxRetries} ${kind} rate-limit retries.`);
37
+ return false;
38
+ }
39
+ log.warn(`${kind} rate limit hit for ${where}. Waiting ${retryAfter}s before retry ${retryCount + 1}/${maxRetries}.`);
40
+ return true;
41
+ };
42
+ }
43
+ /**
44
+ * The client is where the sink lives: Octokit takes a `log` and exposes it as
45
+ * `octokit.log`, so every function below reaches it through the client it is
46
+ * already given, instead of taking a logger of its own.
47
+ */
48
+ export function makeOctokit(token, { log = consoleLog, maxRetries = MAX_RETRIES, maxWaitSeconds, throttle, } = {}) {
49
+ const options = {
50
+ log,
51
+ // The throttling plugin owns rate-limit (403/429) retries, so keep them out
52
+ // of the retry plugin to avoid double-handling.
53
+ retry: {
54
+ doNotRetry: [400, 401, 403, 404, 410, 422, 429, 451],
55
+ retries: maxRetries,
56
+ },
57
+ // `throttle` spreads first so the rate-limit handlers below always win — a
58
+ // caller can tune Bottleneck, never replace the hardened retry policy.
59
+ throttle: {
60
+ ...throttle,
61
+ onRateLimit: createRateLimitHandler('primary', log, maxWaitSeconds, maxRetries),
62
+ onSecondaryRateLimit: createRateLimitHandler('secondary', log, maxWaitSeconds, maxRetries),
63
+ },
64
+ };
65
+ return token
66
+ ? new HardenedOctokit({ ...options, auth: token })
67
+ : new HardenedOctokit(options);
68
+ }
69
+ export async function getRepoInfo(octokit, owner, repo) {
70
+ octokit.log.debug(`Fetching repository info for ${owner}/${repo}`);
71
+ const { data } = await octokit.rest.repos.get({ owner, repo });
72
+ return {
73
+ archived: data.archived,
74
+ description: data.description ?? null,
75
+ language: data.language,
76
+ open_issues_count: data.open_issues_count,
77
+ owner: data.owner.login,
78
+ pushed_at: data.pushed_at,
79
+ repo: data.name,
80
+ stargazers_count: data.stargazers_count,
81
+ topics: data.topics ?? [],
82
+ };
83
+ }
84
+ export async function getReadme(octokit, owner, repo, format = 'raw') {
85
+ octokit.log.debug(`Fetching ${format} README for ${owner}/${repo}`);
86
+ const response = await octokit.rest.repos.getReadme({
87
+ mediaType: { format },
88
+ owner,
89
+ repo,
90
+ });
91
+ // With `raw`/`html` media types the body is a string, but the generated types
92
+ // still describe the JSON shape - cast to string.
93
+ return response.data;
94
+ }
95
+ /** Root file/directory names — the compile-manifest gate reads them to tell a
96
+ * directory of resources from a repo that IS the deliverable. A `path: ''`
97
+ * listing is always an array; the single-entry branch defends against a file
98
+ * response. */
99
+ export async function getRootEntryNames(octokit, owner, repo) {
100
+ octokit.log.debug(`Listing root directory for ${owner}/${repo}`);
101
+ const { data } = await octokit.rest.repos.getContent({
102
+ owner,
103
+ path: '',
104
+ repo,
105
+ });
106
+ const entries = Array.isArray(data) ? data : [data];
107
+ return entries.map(entry => entry.name);
108
+ }
109
+ /** Source revision for the JSON output, so an enhanced list is traceable to its origin commit. */
110
+ export async function getLatestCommitSha(octokit, owner, repo) {
111
+ try {
112
+ octokit.log.debug(`Fetching latest commit SHA for ${owner}/${repo}`);
113
+ const { data } = await octokit.rest.repos.listCommits({
114
+ owner,
115
+ per_page: 1,
116
+ repo,
117
+ });
118
+ return data[0]?.sha ?? null;
119
+ }
120
+ catch (error) {
121
+ octokit.log.error(`Failed to fetch latest commit for ${owner}/${repo}: ${formatRequestError(error)}`);
122
+ return null;
123
+ }
124
+ }
125
+ export function parseOwnerRepo(value) {
126
+ const trimmed = value.trim();
127
+ if (!trimmed) {
128
+ return null;
129
+ }
130
+ if (trimmed.includes('github.com')) {
131
+ const url = trimmed.startsWith('http') ? trimmed : `https://${trimmed}`;
132
+ return parseGitHubUrl(url);
133
+ }
134
+ const parts = trimmed.split('/').filter(part => part.length > 0);
135
+ if (parts.length !== 2) {
136
+ return null;
137
+ }
138
+ return { owner: parts[0], repo: parts[1].replace(/\.git$/, '') };
139
+ }
140
+ export function parseGitHubUrl(url) {
141
+ try {
142
+ const parsedUrl = new URL(url);
143
+ if (parsedUrl.hostname !== 'github.com') {
144
+ return null;
145
+ }
146
+ const pathParts = parsedUrl.pathname
147
+ .split('/')
148
+ .filter(part => part.length > 0);
149
+ if (pathParts.length >= 2) {
150
+ const owner = pathParts[0], repo = pathParts[1].replace(/\.git$/, '');
151
+ return { owner, repo };
152
+ }
153
+ return null;
154
+ }
155
+ catch {
156
+ // A relative or malformed href is simply not a GitHub repo link, which is
157
+ // the routine case for most links in a README — not a diagnostic.
158
+ return null;
159
+ }
160
+ }
161
+ export function formatRequestError(error) {
162
+ const message = error instanceof Error ? error.message : String(error);
163
+ const status = getErrorStatus(error);
164
+ return status === undefined ? message : `${status}: ${message}`;
165
+ }
166
+ function getErrorStatus(error) {
167
+ if (!!error && typeof error === 'object' && 'status' in error) {
168
+ const { status } = error;
169
+ return typeof status === 'number' && status > 0 ? status : undefined;
170
+ }
171
+ return undefined;
172
+ }
@@ -0,0 +1,8 @@
1
+ export { formatRequestError, getLatestCommitSha, getReadme, getRepoInfo, getRootEntryNames, makeOctokit, parseGitHubUrl, parseOwnerRepo, } from './github.js';
2
+ export type { GithubClient, MakeOctokitOptions, RepoIdentifier, RepoInfoDetails, ThrottleOptions, } from './github.js';
3
+ export type { Logger } from './logger.js';
4
+ export { consoleLog, silentLog } from './logger.js';
5
+ export { toRepoInfo } from './markdown.js';
6
+ export type { JsonGroup, JsonItem, JsonMetadata, JsonNode, JsonOutput, JsonSection, ReplacementRule, RepoInfo, } from './markdown.js';
7
+ export { enhance } from './orchestrator.js';
8
+ export type { EnhanceOptions, EnhanceResult } from './orchestrator.js';
package/dist/index.js ADDED
@@ -0,0 +1,4 @@
1
+ export { formatRequestError, getLatestCommitSha, getReadme, getRepoInfo, getRootEntryNames, makeOctokit, parseGitHubUrl, parseOwnerRepo, } from './github.js';
2
+ export { consoleLog, silentLog } from './logger.js';
3
+ export { toRepoInfo } from './markdown.js';
4
+ export { enhance } from './orchestrator.js';
@@ -0,0 +1,15 @@
1
+ /**
2
+ * Octokit's own log shape, which every client already carries as `octokit.log`.
3
+ * Reusing it — rather than inventing a second logger — is what lets the sink
4
+ * ride along with the client that is already passed to everything that logs.
5
+ */
6
+ export interface Logger {
7
+ debug: (message: string) => unknown;
8
+ error: (message: string) => unknown;
9
+ info: (message: string) => unknown;
10
+ warn: (message: string) => unknown;
11
+ }
12
+ /** Library default: routes diagnostics to the console without a runner dependency. */
13
+ export declare const consoleLog: Logger;
14
+ /** For an embedder that wants none of the above on its stdout. */
15
+ export declare const silentLog: Logger;
package/dist/logger.js ADDED
@@ -0,0 +1,22 @@
1
+ /** Library default: routes diagnostics to the console without a runner dependency. */
2
+ export const consoleLog = {
3
+ debug: message => {
4
+ console.debug(message);
5
+ },
6
+ error: message => {
7
+ console.error(message);
8
+ },
9
+ info: message => {
10
+ console.info(message);
11
+ },
12
+ warn: message => {
13
+ console.warn(message);
14
+ },
15
+ };
16
+ /** For an embedder that wants none of the above on its stdout. */
17
+ export const silentLog = {
18
+ debug: () => undefined,
19
+ error: () => undefined,
20
+ info: () => undefined,
21
+ warn: () => undefined,
22
+ };
@@ -0,0 +1,58 @@
1
+ import { RepoInfoDetails } from './github.js';
2
+ import { Logger } from './logger.js';
3
+ export interface JsonOutput {
4
+ items: JsonSection[];
5
+ metadata: JsonMetadata;
6
+ }
7
+ export type ReplacementRule = {
8
+ find: string;
9
+ replace: string;
10
+ type: 'literal' | 'regex';
11
+ } | {
12
+ type: 'branding';
13
+ };
14
+ export interface SortOptions {
15
+ by: '' | 'last_commit' | 'stars';
16
+ minLinks: number;
17
+ }
18
+ export interface RepoInfo {
19
+ archived: boolean;
20
+ language: null | string;
21
+ last_commit: null | string;
22
+ owner: string;
23
+ repo: string;
24
+ stars: number;
25
+ }
26
+ /** The sole crossing from the API shape to the emitted one, so both agree on the renames. */
27
+ export declare function toRepoInfo(details: RepoInfoDetails): RepoInfo;
28
+ export interface JsonItem {
29
+ children: JsonNode[];
30
+ description: null | string;
31
+ node_type: 'item';
32
+ repo_info?: RepoInfo;
33
+ title: string;
34
+ }
35
+ export interface JsonGroup {
36
+ children: JsonNode[];
37
+ description: null | string;
38
+ node_type: 'group';
39
+ title: string;
40
+ }
41
+ export type JsonNode = JsonGroup | JsonItem;
42
+ export interface JsonMetadata {
43
+ enhanced_repository: null | string;
44
+ enhanced_repository_description: null | string;
45
+ last_updated: string;
46
+ original_repository: string;
47
+ original_repository_sha: null | string;
48
+ title: string;
49
+ }
50
+ export interface JsonSection {
51
+ description: null | string;
52
+ items: JsonNode[];
53
+ title: string;
54
+ }
55
+ export declare function processMarkdownContent(originalContent: string, token: string, replacements: ReplacementRule[] | undefined, sortOptions: SortOptions | undefined, originalRepository: string, relativeLinkPrefix?: string, enhancedRepository?: string, enhancedRepositoryDescription?: string, originalRepositorySha?: string, now?: Date, log?: Logger): Promise<{
56
+ finalContent: string;
57
+ jsonData: JsonOutput;
58
+ }>;
@@ -0,0 +1,519 @@
1
+ import * as path from 'path';
2
+ import remarkGfm from 'remark-gfm';
3
+ import remarkParse from 'remark-parse';
4
+ import remarkStringify from 'remark-stringify';
5
+ import { unified } from 'unified';
6
+ import { visit } from 'unist-util-visit';
7
+ import { formatRequestError, getRepoInfo as fetchRepoInfo, makeOctokit, parseGitHubUrl, } from './github.js';
8
+ import { consoleLog } from './logger.js';
9
+ /** The sole crossing from the API shape to the emitted one, so both agree on the renames. */
10
+ export function toRepoInfo(details) {
11
+ return {
12
+ archived: details.archived,
13
+ language: details.language,
14
+ last_commit: details.pushed_at,
15
+ owner: details.owner,
16
+ repo: details.repo,
17
+ stars: details.stargazers_count,
18
+ };
19
+ }
20
+ // The lookup's single throttled client coordinates rate limits across the whole
21
+ // pool, so this bounds in-flight targets rather than requests.
22
+ const FETCH_CONCURRENCY = 10;
23
+ /** Task rejections propagate, so a caller that must not abort the batch on a single failure wraps its own per-item try/catch. */
24
+ async function forEachConcurrent(items, limit, task) {
25
+ const queue = Array.from(items);
26
+ async function worker() {
27
+ for (let item = queue.shift(); item !== undefined; item = queue.shift()) {
28
+ await task(item);
29
+ }
30
+ }
31
+ await Promise.all(Array.from({ length: limit }, () => worker()));
32
+ }
33
+ // One throttled client + a per-repo memo for the run: aliased refs (a bare link
34
+ // and a deep `/tree/...` link into the same repo) collapse to a single fetch,
35
+ // the same way a case-variant spelling of one repo costs one round-trip.
36
+ function createRepoInfoLookup(token, log) {
37
+ const client = makeOctokit(token, { log });
38
+ const cache = new Map();
39
+ return {
40
+ client,
41
+ getRepoInfo(ref) {
42
+ const key = `${ref.owner.toLowerCase()}/${ref.repo.toLowerCase()}`;
43
+ let pending = cache.get(key);
44
+ if (!pending) {
45
+ pending = fetchRepoInfo(client, ref.owner, ref.repo);
46
+ cache.set(key, pending);
47
+ }
48
+ return pending;
49
+ },
50
+ };
51
+ }
52
+ /**
53
+ * Fetches repo info for every GitHub link the tree addresses by URL, keyed by
54
+ * URL so badge insertion can look each one up. Aliased URLs pointing at one repo
55
+ * collapse to a single fetch inside the lookup, then fan back out to every alias
56
+ * here.
57
+ *
58
+ * Each target's failure is independent and non-fatal: a dead link is skipped
59
+ * with a warning and its item is still emitted (no repo_info). Only the source
60
+ * README fetch in main.ts can fail the run.
61
+ */
62
+ async function fetchTargetData(urls, repos) {
63
+ const log = repos.client.log;
64
+ const repoInfoMap = new Map();
65
+ const fetchStart = Date.now();
66
+ await forEachConcurrent(urls, FETCH_CONCURRENCY, async (url) => {
67
+ const details = parseGitHubUrl(url);
68
+ if (!details) {
69
+ return;
70
+ }
71
+ try {
72
+ repoInfoMap.set(url, await repos.getRepoInfo(details));
73
+ }
74
+ catch (error) {
75
+ log.warn(`Skipping repo info for ${url}: ${formatRequestError(error)}`);
76
+ }
77
+ });
78
+ log.info(`Target fetch: ${repoInfoMap.size}/${urls.size} repo-info ok in ${Date.now() - fetchStart}ms (concurrency ${FETCH_CONCURRENCY}).`);
79
+ return repoInfoMap;
80
+ }
81
+ export async function processMarkdownContent(originalContent, token, replacements = [], sortOptions = { by: '', minLinks: 2 }, originalRepository, relativeLinkPrefix = '', enhancedRepository, enhancedRepositoryDescription, originalRepositorySha, now = new Date(), log = consoleLog) {
82
+ const repos = createRepoInfoLookup(token, log);
83
+ const brandingEnabled = replacements.some(rule => rule.type === 'branding');
84
+ const contentAfterReplacements = applyTextReplacements(originalContent, replacements.filter(rule => rule.type !== 'branding'), log);
85
+ const processor = unified().use(remarkParse).use(remarkGfm);
86
+ const tree = processor.parse(contentAfterReplacements);
87
+ const githubUrls = collectGitHubLinks(tree);
88
+ const repoInfoMap = await fetchTargetData(githubUrls, repos);
89
+ // The title derives from the *source* repository (originalRepository), never
90
+ // the enhanced/mirror repo — otherwise the org name doubles into the title.
91
+ const { sections, title: rawTitle, titleHeadingIndex, } = processTree(tree, repoInfoMap, sortOptions, originalRepository);
92
+ // Single source of truth for the document title: brand it once and use the
93
+ // same value for the markdown H1 and metadata.title (parity).
94
+ const title = brandingEnabled ? brandTitle(rawTitle) : rawTitle;
95
+ if (brandingEnabled) {
96
+ applyBrandingToTree(tree, title, titleHeadingIndex);
97
+ }
98
+ const metadata = {
99
+ last_updated: now.toISOString(),
100
+ original_repository: originalRepository.trim(),
101
+ original_repository_sha: (originalRepositorySha?.trim() ?? '') || null,
102
+ enhanced_repository: (enhancedRepository?.trim() ?? '') || null,
103
+ enhanced_repository_description: (enhancedRepositoryDescription?.trim() ?? '') || null,
104
+ title,
105
+ };
106
+ const jsonData = {
107
+ items: sections,
108
+ metadata,
109
+ };
110
+ addInfoBadges(tree, repoInfoMap);
111
+ fixRelativeLinks(tree, relativeLinkPrefix);
112
+ let finalContent = serializeAst(tree, originalContent);
113
+ if (brandingEnabled) {
114
+ finalContent = appendEnhansomedFooter(finalContent, now);
115
+ }
116
+ return {
117
+ finalContent,
118
+ jsonData,
119
+ };
120
+ }
121
+ function addInfoBadges(tree, repoInfoMap) {
122
+ const modifications = new Map();
123
+ visit(tree, 'link', (node, index, parent) => {
124
+ if (index === undefined || !parent) {
125
+ return;
126
+ }
127
+ const repoInfo = repoInfoMap.get(node.url);
128
+ if (!repoInfo) {
129
+ return;
130
+ }
131
+ const badgeNode = {
132
+ type: 'text',
133
+ value: createBadgeText(repoInfo),
134
+ };
135
+ if (!modifications.has(parent)) {
136
+ modifications.set(parent, []);
137
+ }
138
+ modifications.get(parent)?.push({ index: index + 1, node: badgeNode });
139
+ });
140
+ for (const [parent, changes] of modifications.entries()) {
141
+ changes.sort((a, b) => b.index - a.index);
142
+ for (const { index, node } of changes) {
143
+ parent.children.splice(index, 0, node);
144
+ }
145
+ }
146
+ }
147
+ function applyTextReplacements(content, rules, log) {
148
+ let processedContent = content;
149
+ for (const rule of rules) {
150
+ if (rule.type === 'literal') {
151
+ log.debug(`Applying literal replacement: '${rule.find}' -> '${rule.replace}'`);
152
+ processedContent = processedContent.replaceAll(rule.find, rule.replace);
153
+ }
154
+ else if (rule.type === 'regex') {
155
+ try {
156
+ const regex = new RegExp(rule.find, 'gm');
157
+ log.debug(`Applying regex replacement: /${rule.find}/gm -> '${rule.replace}'`);
158
+ processedContent = processedContent.replace(regex, rule.replace);
159
+ }
160
+ catch (e) {
161
+ log.warn(`Skipping invalid regex pattern '${rule.find}': ${e instanceof Error ? e.message : e}`);
162
+ }
163
+ }
164
+ // 'branding' rules are applied to the AST/title downstream, not here.
165
+ }
166
+ return processedContent;
167
+ }
168
+ function collectGitHubLinks(tree) {
169
+ const urls = new Set();
170
+ visit(tree, 'link', (node) => {
171
+ if (parseGitHubUrl(node.url)) {
172
+ urls.add(node.url);
173
+ }
174
+ });
175
+ return urls;
176
+ }
177
+ function createBadgeText(info) {
178
+ if (info.archived) {
179
+ return ' ⚠️ Archived';
180
+ }
181
+ const parts = [
182
+ `⭐ ${info.stargazers_count.toLocaleString()}`,
183
+ `🐛 ${info.open_issues_count.toLocaleString()}`,
184
+ ];
185
+ if (info.language) {
186
+ parts.push(`🌐 ${info.language}`);
187
+ }
188
+ if (info.pushed_at) {
189
+ parts.push(`📅 ${formatDate(info.pushed_at)}`);
190
+ }
191
+ return ` ${parts.join(' | ')}`;
192
+ }
193
+ function findFirstGitHubLink(node) {
194
+ let linkUrl;
195
+ visit(node, 'link', (linkNode) => {
196
+ if (!linkUrl && parseGitHubUrl(linkNode.url)) {
197
+ linkUrl = linkNode.url;
198
+ }
199
+ });
200
+ return linkUrl;
201
+ }
202
+ // The GitHub link that represents a list item: the FIRST GitHub link in the
203
+ // item's OWN paragraph only. A nested-descendant link is deliberately ignored
204
+ // — it belongs to a child, not to this item. An item is a GitHub node iff its
205
+ // own paragraph links to a GitHub repo; an item with no own GitHub link but
206
+ // nested GitHub children is a group, not a node borrowing a child's identity.
207
+ //
208
+ // A GitHub link that is secondary within the paragraph (e.g.
209
+ // `[name](marketplace) … [On GitHub](github)`) is still the item's own link, so
210
+ // `findFirstGitHubLink` over the paragraph finds it correctly.
211
+ function findOwnGitHubLink(itemNode) {
212
+ const paragraph = itemNode.children.find((child) => child.type === 'paragraph');
213
+ return paragraph ? findFirstGitHubLink(paragraph) : undefined;
214
+ }
215
+ function fixRelativeLinks(tree, relativeLinkPrefix) {
216
+ if (!relativeLinkPrefix) {
217
+ return;
218
+ }
219
+ if (relativeLinkPrefix) {
220
+ visit(tree, 'link', node => {
221
+ if (!node.url.startsWith('http') &&
222
+ !node.url.startsWith('/') &&
223
+ !node.url.startsWith('#')) {
224
+ node.url = path.join(relativeLinkPrefix, node.url).replace(/\\/g, '/');
225
+ }
226
+ });
227
+ }
228
+ }
229
+ function formatDate(isoString) {
230
+ if (!isoString) {
231
+ return '';
232
+ }
233
+ return new Date(isoString).toISOString().split('T')[0];
234
+ }
235
+ function getNodeText(node) {
236
+ return getInlineText([node]);
237
+ }
238
+ // Concatenates text descendants, collapsing whitespace (incl. a lone soft-break
239
+ // newline) to one space. Takes a node slice so callers can isolate text
240
+ // before/after a specific child.
241
+ function getInlineText(nodes) {
242
+ let text = '';
243
+ for (const node of nodes) {
244
+ visit(node, 'text', (textNode) => {
245
+ text += textNode.value;
246
+ });
247
+ }
248
+ return text.replace(/\s+/g, ' ').trim();
249
+ }
250
+ // Strip one leading separator (dash, pipe, colon, middot) from the prose
251
+ // trailing a link, but stay tightly scoped so a meaningful leading character
252
+ // (e.g. "(" in "(deprecated) …") survives.
253
+ function stripLeadingNoise(text) {
254
+ return text.replace(/^\s*[-–—·|:]\s*/, '').trim();
255
+ }
256
+ // Missing repo info sinks below nodes that have it; two missing tie so a stable
257
+ // sort keeps source order. The single comparator shared by the JSON builder and
258
+ // the AST sorter, so both outputs agree on order.
259
+ function compareByRepoInfo(by, a, b) {
260
+ if (!by) {
261
+ return 0;
262
+ }
263
+ // Must precede the field access below (a/b may be null).
264
+ if (!a || !b) {
265
+ return a ? -1 : b ? 1 : 0;
266
+ }
267
+ if (by === 'stars') {
268
+ return b.stargazers_count - a.stargazers_count;
269
+ }
270
+ // `by` is narrowed to 'last_commit' here (the only remaining option).
271
+ const timeA = a.pushed_at ? new Date(a.pushed_at).getTime() : 0;
272
+ const timeB = b.pushed_at ? new Date(b.pushed_at).getTime() : 0;
273
+ return timeB - timeA;
274
+ }
275
+ function processListRecursively(listNode, repoInfoMap, sortOptions, isNested = false) {
276
+ // The section-sparsity gate counts items whose SUBTREE contains a GitHub
277
+ // link, not items with an own-paragraph link only. A category with no own
278
+ // link but nested GitHub children still counts (it becomes a group);
279
+ // switching to `findOwnGitHubLink` would silently drop purely categorical
280
+ // sections, so it deliberately diverges from the identity resolver below.
281
+ const itemsWithGitHubLinks = listNode.children.filter(item => !!findFirstGitHubLink(item));
282
+ if (!isNested && itemsWithGitHubLinks.length < sortOptions.minLinks) {
283
+ return [];
284
+ }
285
+ // Zip each item with its JSON node and repo info so one sort orders both the
286
+ // rendered AST and the emitted JSON. `json` is null only for non-GitHub
287
+ // leaves (no own link, no nested GitHub children): kept in the AST, dropped
288
+ // from JSON.
289
+ const entries = [];
290
+ for (const itemNode of listNode.children) {
291
+ const githubUrl = findOwnGitHubLink(itemNode);
292
+ const repoInfo = githubUrl ? (repoInfoMap.get(githubUrl) ?? null) : null;
293
+ const nestedLists = itemNode.children.filter((child) => child.type === 'list');
294
+ const childrenJson = nestedLists.flatMap(nestedList => processListRecursively(nestedList, repoInfoMap, sortOptions, true));
295
+ let title = '';
296
+ let description = '';
297
+ const paragraph = itemNode.children.find(p => p.type === 'paragraph');
298
+ if (paragraph) {
299
+ const linkIndex = paragraph.children.findIndex((c) => c.type === 'link');
300
+ if (linkIndex !== -1) {
301
+ // Title = leading text + the first link's text, so prefix tags like
302
+ // "[UPDATED]" survive; description is the prose trailing the link.
303
+ title = getInlineText(paragraph.children.slice(0, linkIndex + 1));
304
+ description = stripLeadingNoise(getInlineText(paragraph.children.slice(linkIndex + 1)));
305
+ }
306
+ else {
307
+ // No link: the whole paragraph is the title, so leave description empty
308
+ // (avoid echoing the title back).
309
+ title = getInlineText(paragraph.children);
310
+ description = '';
311
+ }
312
+ }
313
+ // No-own-link items with children become groups — NEVER give them a
314
+ // `repo_info`, that's the identity-borrowing bug. No-own-link, no-child
315
+ // items are non-GitHub leaves: kept in markdown, dropped from JSON.
316
+ // TODO(future): preserve non-GitHub leaves in a separate shape.
317
+ let jsonData = null;
318
+ if (githubUrl) {
319
+ const item = {
320
+ node_type: 'item',
321
+ title,
322
+ description: description || null,
323
+ children: childrenJson,
324
+ };
325
+ if (repoInfo) {
326
+ item.repo_info = toRepoInfo(repoInfo);
327
+ }
328
+ jsonData = item;
329
+ }
330
+ else if (childrenJson.length > 0) {
331
+ jsonData = {
332
+ node_type: 'group',
333
+ title,
334
+ description: description || null,
335
+ children: childrenJson,
336
+ };
337
+ }
338
+ entries.push({ json: jsonData, node: itemNode, repoInfo });
339
+ }
340
+ if (sortOptions.by) {
341
+ entries.sort((a, b) => compareByRepoInfo(sortOptions.by, a.repoInfo, b.repoInfo));
342
+ }
343
+ // Reorder the AST to match the sort so rendered markdown and JSON agree.
344
+ listNode.children = entries.map(entry => entry.node);
345
+ return entries
346
+ .map(entry => entry.json)
347
+ .filter((json) => json !== null);
348
+ }
349
+ const INVALID_TITLE_PATTERNS = [
350
+ /^contributing/i,
351
+ /^license$/i,
352
+ /^resources$/i,
353
+ /^contents$/i,
354
+ /^table of contents$/i,
355
+ /^other awesome/i,
356
+ /^other awesomeness$/i,
357
+ /^communities/i,
358
+ /^guides$/i,
359
+ /^tools$/i,
360
+ /^video$/i,
361
+ /^science/i,
362
+ ];
363
+ function repoNameFromIdentifier(identifier) {
364
+ const trimmed = identifier.trim();
365
+ if (!trimmed) {
366
+ return '';
367
+ }
368
+ if (/^https?:\/\//i.test(trimmed)) {
369
+ const parts = trimmed
370
+ .replace(/\.git$/i, '')
371
+ .replace(/[?#].*$/, '')
372
+ .split('/')
373
+ .filter(Boolean);
374
+ return parts[parts.length - 1] ?? '';
375
+ }
376
+ const slashIndex = trimmed.lastIndexOf('/');
377
+ return slashIndex === -1 ? trimmed : trimmed.slice(slashIndex + 1);
378
+ }
379
+ function formatRepoNameAsTitle(repoName) {
380
+ const cleaned = repoName.replace(/[-_]/g, ' ').replace(/\s+/g, ' ').trim();
381
+ if (/^awesome\s+/i.test(cleaned)) {
382
+ return cleaned.replace(/^awesome\s+/i, 'Awesome ');
383
+ }
384
+ return `Awesome ${cleaned}`;
385
+ }
386
+ function isValidTitle(title) {
387
+ if (!title || title.trim() === '') {
388
+ return false;
389
+ }
390
+ return !INVALID_TITLE_PATTERNS.some(pattern => pattern.test(title.trim()));
391
+ }
392
+ /** Never duplicates "Awesome": if the title already contains it, append the suffix verbatim; otherwise prefix first. */
393
+ function brandTitle(title) {
394
+ const trimmed = title.trim();
395
+ if (!trimmed) {
396
+ return trimmed;
397
+ }
398
+ if (/\bawesome\b/i.test(trimmed)) {
399
+ return `${trimmed} with stars`;
400
+ }
401
+ return `Awesome ${trimmed} with stars`;
402
+ }
403
+ // Shared by title extraction and branding so both act on the very same heading.
404
+ function findTitleHeadingIndex(tree) {
405
+ return tree.children.findIndex((node) => node.type === 'heading' &&
406
+ node.depth === 1 &&
407
+ isValidTitle(getNodeText(node)));
408
+ }
409
+ /**
410
+ * Makes the rendered H1 match metadata.title, replacing whichever heading
411
+ * occupies the title slot (a valid title H1, else the first generic H1, else a
412
+ * freshly injected one) so the branded title is always the sole title H1.
413
+ */
414
+ function applyBrandingToTree(tree, title, titleHeadingIndex) {
415
+ if (!title.trim()) {
416
+ return;
417
+ }
418
+ const heading = {
419
+ type: 'heading',
420
+ depth: 1,
421
+ children: [{ type: 'text', value: title }],
422
+ };
423
+ if (titleHeadingIndex !== -1) {
424
+ tree.children[titleHeadingIndex] = heading;
425
+ return;
426
+ }
427
+ // No valid title H1 anywhere: override the first H1 (the de-facto title slot)
428
+ // rather than unshifting alongside it, which would leave two H1s. Inject at
429
+ // the top only when the source has no H1 at all.
430
+ const firstH1Index = tree.children.findIndex((node) => node.type === 'heading' && node.depth === 1);
431
+ if (firstH1Index !== -1) {
432
+ tree.children[firstH1Index] = heading;
433
+ }
434
+ else {
435
+ tree.children.unshift(heading);
436
+ }
437
+ }
438
+ function processTree(tree, repoInfoMap, sortOptions, originalRepository) {
439
+ // Remember the title H1's index so branding replaces the exact same node;
440
+ // scope matches branding/section-building so they can't drift apart.
441
+ const titleHeadingIndex = findTitleHeadingIndex(tree);
442
+ let documentTitle = titleHeadingIndex === -1
443
+ ? ''
444
+ : getNodeText(tree.children[titleHeadingIndex]);
445
+ // Derive a subject from the *source* repository name when no valid H1 is
446
+ // present. Using the source — not the enhanced/mirror repo — keeps the org
447
+ // name out of the title.
448
+ if (documentTitle === '' && originalRepository) {
449
+ const repoName = repoNameFromIdentifier(originalRepository);
450
+ if (repoName) {
451
+ documentTitle = formatRepoNameAsTitle(repoName);
452
+ }
453
+ }
454
+ const sections = [];
455
+ let currentSection = null;
456
+ for (const node of tree.children) {
457
+ if (node.type === 'heading' && node.depth > 1) {
458
+ if (currentSection) {
459
+ sections.push(currentSection);
460
+ }
461
+ currentSection = {
462
+ description: '',
463
+ items: [],
464
+ title: getNodeText(node),
465
+ };
466
+ }
467
+ else if (currentSection) {
468
+ if (node.type === 'paragraph') {
469
+ const paragraphText = getNodeText(node);
470
+ // Avoid adding boilerplate "back to top" links to description
471
+ if (!paragraphText.includes('back to top')) {
472
+ if (currentSection.description) {
473
+ currentSection.description += `\n${paragraphText}`;
474
+ }
475
+ else {
476
+ currentSection.description = paragraphText;
477
+ }
478
+ }
479
+ }
480
+ else if (node.type === 'list') {
481
+ const items = processListRecursively(node, repoInfoMap, sortOptions);
482
+ if (items.length > 0) {
483
+ currentSection.items = items;
484
+ sections.push(currentSection);
485
+ }
486
+ currentSection = null;
487
+ }
488
+ }
489
+ else if (node.type === 'list') {
490
+ // No active section: not part of any JSON section, but still sort its AST
491
+ // so the rendered markdown matches.
492
+ processListRecursively(node, repoInfoMap, sortOptions);
493
+ }
494
+ }
495
+ if (currentSection) {
496
+ sections.push(currentSection);
497
+ }
498
+ return { sections, title: documentTitle, titleHeadingIndex };
499
+ }
500
+ function serializeAst(tree, originalContent) {
501
+ let finalContent = unified()
502
+ .use(remarkStringify)
503
+ .use(remarkGfm)
504
+ .stringify(tree);
505
+ const originalHadNewline = originalContent.endsWith('\n') || originalContent === '';
506
+ if (finalContent.endsWith('\n') && !originalHadNewline) {
507
+ finalContent = finalContent.slice(0, -1);
508
+ }
509
+ else if (!finalContent.endsWith('\n') && originalHadNewline) {
510
+ finalContent += '\n';
511
+ }
512
+ return finalContent;
513
+ }
514
+ function appendEnhansomedFooter(content, now) {
515
+ const date = now.toISOString().split('T')[0];
516
+ const footer = `***\n\n> _Enhansomed by [enhansome](https://github.com/enhansome) on ${date}._`;
517
+ const body = content.replace(/\n+$/, '');
518
+ return `${body}\n\n${footer}\n`;
519
+ }
@@ -0,0 +1,23 @@
1
+ import { Logger } from './logger.js';
2
+ import { JsonOutput, ReplacementRule } from './markdown.js';
3
+ export interface EnhanceOptions {
4
+ content: string;
5
+ disableBranding?: boolean;
6
+ enhancedRepository?: string;
7
+ enhancedRepositoryDescription?: string;
8
+ /** Defaults to the console sink; pass your own (e.g. an Actions workflow-command sink) to route diagnostics. */
9
+ log?: Logger;
10
+ now?: Date;
11
+ originalRepository: string;
12
+ originalRepositorySha?: string;
13
+ relativeLinkPrefix?: string;
14
+ /** Text substitutions applied to the source before it is parsed. */
15
+ replacements?: ReplacementRule[];
16
+ sortBy?: '' | 'last_commit' | 'stars';
17
+ token: string;
18
+ }
19
+ export interface EnhanceResult {
20
+ finalContent: string;
21
+ jsonData: JsonOutput;
22
+ }
23
+ export declare function enhance(options: EnhanceOptions): Promise<EnhanceResult>;
@@ -0,0 +1,17 @@
1
+ import { processMarkdownContent, } from './markdown.js';
2
+ export async function enhance(options) {
3
+ const { content, disableBranding = false, log, now = new Date(), originalRepository, originalRepositorySha, relativeLinkPrefix = '', replacements = [], sortBy = '', enhancedRepository, enhancedRepositoryDescription, token, } = options;
4
+ // Branding is an internal rule prepended to the caller's own; build a fresh
5
+ // array so the caller's `replacements` is never mutated.
6
+ const branding = { type: 'branding' };
7
+ const rules = disableBranding ? replacements : [branding, ...replacements];
8
+ const sortOptions = {
9
+ by: sortBy,
10
+ minLinks: 2,
11
+ };
12
+ const { finalContent, jsonData } = await processMarkdownContent(content, token, rules, sortOptions, originalRepository, relativeLinkPrefix, enhancedRepository, enhancedRepositoryDescription, originalRepositorySha, now, log);
13
+ return {
14
+ finalContent,
15
+ jsonData,
16
+ };
17
+ }
package/package.json ADDED
@@ -0,0 +1,52 @@
1
+ {
2
+ "name": "@enhansome/core",
3
+ "version": "1.6.1",
4
+ "description": "Library core for enhansome — enhance markdown with GitHub star counts.",
5
+ "type": "module",
6
+ "main": "dist/index.js",
7
+ "types": "dist/index.d.ts",
8
+ "exports": {
9
+ ".": {
10
+ "types": "./dist/index.d.ts",
11
+ "import": "./dist/index.js"
12
+ },
13
+ "./package.json": "./package.json"
14
+ },
15
+ "files": [
16
+ "dist"
17
+ ],
18
+ "publishConfig": {
19
+ "access": "public"
20
+ },
21
+ "engines": {
22
+ "node": ">=22"
23
+ },
24
+ "scripts": {
25
+ "build": "tsc -p tsconfig.build.json",
26
+ "typecheck": "tsc --noEmit",
27
+ "test": "vitest run",
28
+ "test:update-goldens": "UPDATE_GOLDENS=1 vitest run"
29
+ },
30
+ "dependencies": {
31
+ "@octokit/core": "^7.0.6",
32
+ "@octokit/plugin-paginate-rest": "^14.0.0",
33
+ "@octokit/plugin-rest-endpoint-methods": "^17.0.0",
34
+ "@octokit/plugin-retry": "^8.1.0",
35
+ "@octokit/plugin-throttling": "^11.0.3",
36
+ "remark-gfm": "^4.0.1",
37
+ "remark-parse": "^11.0.0",
38
+ "remark-stringify": "^11.0.0",
39
+ "unified": "^11.0.5",
40
+ "unist-util-visit": "^5.1.0"
41
+ },
42
+ "devDependencies": {
43
+ "@octokit/request-error": "^7.1.0",
44
+ "@types/mdast": "^4.0.4",
45
+ "@types/node": "^26.1.0",
46
+ "@types/unist": "^3.0.3",
47
+ "@vitest/coverage-v8": "^4.1.9",
48
+ "typescript": "^6.0.3",
49
+ "vitest": "^4.1.9"
50
+ },
51
+ "packageManager": "yarn@4.17.1+sha512.ccbfabf7d7b6b32075088be9386fb9a2e00bb6887ef07fa56effabc890a56d53da1ccc4128d62db245fcbd3961b236d75335bdf7d5320ed6eafb7588b7ad4697"
52
+ }