@writedocs/generator 0.7.2 → 0.7.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/astro.config.mjs +10 -4
- package/bin/writedocs.js +28 -7
- package/package.json +1 -1
- package/src/cli/build-auth.js +16 -9
- package/src/cli/build.js +3 -1
- package/src/cli/convert.js +6 -2
- package/src/cli/dev.js +3 -1
- package/src/cli/generate-api-pages.js +2 -2
- package/src/cli/init.js +3 -1
- package/src/cli/output.js +8 -1
- package/src/cli/preflight.js +63 -1
- package/src/cli/update.js +6 -1
- package/src/cli/write-redirects-file.js +2 -1
- package/src/components/ApiReferencePanel.astro +46 -12
- package/src/components/Steps.astro +2 -2
- package/src/layout/BaseLayout.astro +21 -6
- package/src/layout/styles/banner.css +1 -1
- package/src/lib/a11y-check.js +19 -42
- package/src/lib/agent-markdown.js +137 -0
- package/src/lib/asset-path.js +29 -0
- package/src/lib/color.js +49 -0
- package/src/lib/config-file.js +25 -0
- package/src/lib/config-schema.js +15 -12
- package/src/lib/config-schema.ts +20 -12
- package/src/lib/config.ts +18 -135
- package/src/lib/json-schema-descriptions.js +1 -1
- package/src/lib/llms-index.ts +88 -0
- package/src/lib/llms.js +200 -0
- package/src/lib/mintlify-convert.js +2 -1
- package/src/lib/pages.js +137 -11
- package/src/lib/styles-asset-integration.js +1 -18
- package/src/lib/writedocs-legacy-convert.js +2 -1
- package/src/pages/[...slug].astro +11 -1
- package/src/pages/[...slug].md.ts +15 -6
- package/src/pages/llms/[...path].md.ts +23 -0
- package/src/pages/llms-full.txt.ts +30 -9
- package/src/pages/llms.txt.ts +21 -125
- package/src/scripts/search.ts +39 -14
- package/writedocs.schema.json +2 -2
package/src/lib/config.ts
CHANGED
|
@@ -3,6 +3,7 @@ import path from 'node:path';
|
|
|
3
3
|
import { EXCLUDED_TOP_LEVEL_DIRS } from './pages.js';
|
|
4
4
|
import matter from 'gray-matter';
|
|
5
5
|
import { writedocsTempDir } from './writedocs-temp-dir.js';
|
|
6
|
+
import { readableTextOn } from './color.js';
|
|
6
7
|
import {
|
|
7
8
|
formatValidationIssues,
|
|
8
9
|
validateDocsConfig,
|
|
@@ -208,13 +209,12 @@ export function resolveNavbarColor(
|
|
|
208
209
|
return { background: value.background, accent: value.accent };
|
|
209
210
|
}
|
|
210
211
|
|
|
211
|
-
/** Picks black or white text for
|
|
212
|
-
*
|
|
213
|
-
*
|
|
214
|
-
*
|
|
215
|
-
* the
|
|
216
|
-
*
|
|
217
|
-
* one). Used two ways in BaseLayout.astro: for `--wd-navbar-accent-text`
|
|
212
|
+
/** Picks black or white text for `hexColor` - white unless white falls
|
|
213
|
+
* below 3:1 on it (readableTextOn() in lib/color.js), the same rule and
|
|
214
|
+
* WCAG math `writedocs a11y` checks with, so the check measures what the
|
|
215
|
+
* site renders. (This used a BT.601 brightness threshold.) Also
|
|
216
|
+
* gives `--wd-on-primary`, the text on step numbers, the info banner and
|
|
217
|
+
* the API playground's buttons. Used two ways in BaseLayout.astro: for `--wd-navbar-accent-text`
|
|
218
218
|
* (the active tab's own fill used to always be `var(--wd-primary)` with
|
|
219
219
|
* hardcoded `color: #fff`, which only actually read fine because every
|
|
220
220
|
* default/example primary color so far has been dark/saturated enough
|
|
@@ -224,19 +224,12 @@ export function resolveNavbarColor(
|
|
|
224
224
|
* hold unconditionally), and for `--wd-navbar-foreground` itself once
|
|
225
225
|
* `styles.navbar` is configured at all (the navbar's plain text/icon
|
|
226
226
|
* color - see `navbar`'s own schema comment for why that's always
|
|
227
|
-
* computed, never a writedocs.json value). Malformed input (not a
|
|
228
|
-
* `#rrggbb` hex) falls back to white rather than throwing - same "don't
|
|
227
|
+
* computed, never a writedocs.json value). Malformed input (not a
|
|
228
|
+
* `#rgb`/`#rrggbb` hex) falls back to white rather than throwing - same "don't
|
|
229
229
|
* fail a build over a cosmetic color value" posture every other color
|
|
230
230
|
* field here takes (none of them validate hex syntax either). */
|
|
231
231
|
export function contrastTextColor(hexColor: string): '#000000' | '#ffffff' {
|
|
232
|
-
|
|
233
|
-
if (!match) return '#ffffff';
|
|
234
|
-
const hex = match[1];
|
|
235
|
-
const r = parseInt(hex.slice(0, 2), 16);
|
|
236
|
-
const g = parseInt(hex.slice(2, 4), 16);
|
|
237
|
-
const b = parseInt(hex.slice(4, 6), 16);
|
|
238
|
-
const luminance = (299 * r + 587 * g + 114 * b) / 1000;
|
|
239
|
-
return luminance > 150 ? '#000000' : '#ffffff';
|
|
232
|
+
return readableTextOn(hexColor) as '#000000' | '#ffffff';
|
|
240
233
|
}
|
|
241
234
|
|
|
242
235
|
/** Whether an href points off-site - has an explicit scheme (`https:`,
|
|
@@ -481,122 +474,9 @@ export function loadDocsConfig(contentDir: string): DocsConfig {
|
|
|
481
474
|
// JavaScript, so `writedocs validate` can find a site's pages with plain Node.
|
|
482
475
|
export { findAllPages } from './pages.js';
|
|
483
476
|
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
* (bannerSchema and friends, above). Drop a file in, it loads;
|
|
488
|
-
* there's no field naming which ones to use, matching the same "just
|
|
489
|
-
* works" convention `docs/`'s own file discovery already follows (see
|
|
490
|
-
* findAllPages() above / `content-pipeline.mdx`) - a site author already
|
|
491
|
-
* drops content files in and expects them found, rather than also
|
|
492
|
-
* listing every one in writedocs.json.
|
|
493
|
-
*
|
|
494
|
-
* Two separate walks, because `public/` needs different treatment than
|
|
495
|
-
* everywhere else:
|
|
496
|
-
*
|
|
497
|
-
* - `css`/`js`: every `.css`/`.js` file outside `public/` (root, `docs/`,
|
|
498
|
-
* `snippets/`, any custom folder) - reused as the walk-with-exclusions
|
|
499
|
-
* shape from findAllPages() above, and its exact `EXCLUDED_TOP_LEVEL_DIRS`
|
|
500
|
-
* set (skipped only at the project root, same as there), so `dist/`,
|
|
501
|
-
* `node_modules/`, `.astro/`, `.writedocs/`, `.git/`, and (for this
|
|
502
|
-
* half only - see below) `public/` are never walked into. BaseLayout.astro
|
|
503
|
-
* reads each one's raw content and inlines it as a `<style>`/
|
|
504
|
-
* `<script is:inline>` tag.
|
|
505
|
-
* - `publicCss`/`publicJs`: every `.css`/`.js` file *inside* `public/`,
|
|
506
|
-
* walked separately (starting from `<contentDir>/public` rather than
|
|
507
|
-
* `contentDir` itself, so the same `EXCLUDED_TOP_LEVEL_DIRS` check
|
|
508
|
-
* doesn't apply here - there's no `public/public/` or `public/dist/`
|
|
509
|
-
* convention to guard against). Returned as public-URL-rooted hrefs
|
|
510
|
-
* (a leading `/`, no `public` segment - `public/custom.css` becomes
|
|
511
|
-
* `/custom.css`) rather than content-dir-relative paths, since these
|
|
512
|
-
* files are already served as static assets at exactly that URL once
|
|
513
|
-
* Astro copies `public/` into the build output. BaseLayout.astro
|
|
514
|
-
* renders these as ordinary `<link rel="stylesheet">`/`<script src>`
|
|
515
|
-
* tags pointing at that URL instead of inlining their content -
|
|
516
|
-
* inlining would duplicate every byte (once in the page's own HTML,
|
|
517
|
-
* once more as the independently-fetchable static file at that same
|
|
518
|
-
* URL) for no benefit, where a `<link>`/`<script src>` gets normal
|
|
519
|
-
* browser caching across pages instead of repeating the content on
|
|
520
|
-
* every single page's markup.
|
|
521
|
-
*
|
|
522
|
-
* Both halves are broader than they might sound - a stray `.js` file
|
|
523
|
-
* kept in `docs/`, `snippets/`, or `public/` for an unrelated reason
|
|
524
|
-
* (a snippet's own local helper, an image gallery's lightbox script
|
|
525
|
-
* someone dropped in `public/` to reference from a raw `<script src>`
|
|
526
|
-
* in an .mdx file, say) gets auto-injected sitewide the same as a
|
|
527
|
-
* deliberate one; there's no separate "this one's just tooling" signal
|
|
528
|
-
* to opt out of the convention short of renaming its extension.
|
|
529
|
-
*
|
|
530
|
-
* Both `css`/`js` and `publicCss`/`publicJs` are sorted alphabetically
|
|
531
|
-
* by their respective path for deterministic load order across rebuilds
|
|
532
|
-
* - same reasoning as llms.txt's own alphabetical-by-slug sort (see
|
|
533
|
-
* llms.txt.ts) - filesystem readdir order isn't guaranteed portable
|
|
534
|
-
* across OSes or directory-walk order otherwise.
|
|
535
|
-
*
|
|
536
|
-
* `css`/`js` return POSIX-separated paths relative to `contentDir`, not
|
|
537
|
-
* absolute paths or file contents - BaseLayout.astro (the sole caller)
|
|
538
|
-
* resolves and reads each one's content itself, right before inlining
|
|
539
|
-
* it, so a file's content is always current as of that specific
|
|
540
|
-
* request/build rather than cached here across a `writedocs dev`
|
|
541
|
-
* session. `publicCss`/`publicJs` return the public-URL hrefs described
|
|
542
|
-
* above - nothing to read, Astro's own static-file serving/copy already
|
|
543
|
-
* handles those. */
|
|
544
|
-
export function findRootAssets(
|
|
545
|
-
contentDir: string
|
|
546
|
-
): { css: string[]; js: string[]; publicCss: string[]; publicJs: string[] } {
|
|
547
|
-
const css: string[] = [];
|
|
548
|
-
const js: string[] = [];
|
|
549
|
-
function walk(dir: string, relBase: string) {
|
|
550
|
-
let entries: fs.Dirent[];
|
|
551
|
-
try {
|
|
552
|
-
entries = fs.readdirSync(dir, { withFileTypes: true });
|
|
553
|
-
} catch {
|
|
554
|
-
return;
|
|
555
|
-
}
|
|
556
|
-
for (const entry of entries) {
|
|
557
|
-
const rel = relBase ? `${relBase}/${entry.name}` : entry.name;
|
|
558
|
-
const abs = path.join(dir, entry.name);
|
|
559
|
-
if (entry.isDirectory()) {
|
|
560
|
-
if (relBase === '' && EXCLUDED_TOP_LEVEL_DIRS.has(entry.name)) continue;
|
|
561
|
-
walk(abs, rel);
|
|
562
|
-
continue;
|
|
563
|
-
}
|
|
564
|
-
if (!entry.isFile()) continue;
|
|
565
|
-
if (/\.css$/i.test(entry.name)) css.push(rel);
|
|
566
|
-
else if (/\.js$/i.test(entry.name)) js.push(rel);
|
|
567
|
-
}
|
|
568
|
-
}
|
|
569
|
-
walk(contentDir, '');
|
|
570
|
-
css.sort();
|
|
571
|
-
js.sort();
|
|
572
|
-
|
|
573
|
-
const publicCss: string[] = [];
|
|
574
|
-
const publicJs: string[] = [];
|
|
575
|
-
function walkPublic(dir: string, relBase: string) {
|
|
576
|
-
let entries: fs.Dirent[];
|
|
577
|
-
try {
|
|
578
|
-
entries = fs.readdirSync(dir, { withFileTypes: true });
|
|
579
|
-
} catch {
|
|
580
|
-
return;
|
|
581
|
-
}
|
|
582
|
-
for (const entry of entries) {
|
|
583
|
-
const rel = relBase ? `${relBase}/${entry.name}` : entry.name;
|
|
584
|
-
const abs = path.join(dir, entry.name);
|
|
585
|
-
if (entry.isDirectory()) {
|
|
586
|
-
walkPublic(abs, rel);
|
|
587
|
-
continue;
|
|
588
|
-
}
|
|
589
|
-
if (!entry.isFile()) continue;
|
|
590
|
-
if (/\.css$/i.test(entry.name)) publicCss.push(`/${rel}`);
|
|
591
|
-
else if (/\.js$/i.test(entry.name)) publicJs.push(`/${rel}`);
|
|
592
|
-
}
|
|
593
|
-
}
|
|
594
|
-
walkPublic(path.join(contentDir, 'public'), '');
|
|
595
|
-
publicCss.sort();
|
|
596
|
-
publicJs.sort();
|
|
597
|
-
|
|
598
|
-
return { css, js, publicCss, publicJs };
|
|
599
|
-
}
|
|
477
|
+
// findRootAssets() lives in lib/pages.js - plain JavaScript, so it can be
|
|
478
|
+
// tested with plain Node, next to findAllPages() whose exclusions it shares.
|
|
479
|
+
export { findRootAssets } from './pages.js';
|
|
600
480
|
|
|
601
481
|
/** The root-relative URL paths (leading `/`, e.g. `/images/hero.svg`)
|
|
602
482
|
* referenced by the writedocs.json/styles fields that point at a static asset
|
|
@@ -615,7 +495,7 @@ export function findRootAssets(
|
|
|
615
495
|
* `public/` was the one place `styles.background.images` (etc.) had to
|
|
616
496
|
* live, since Astro's own `publicDir` copy is the only thing that ever
|
|
617
497
|
* served them; everywhere else in this codebase's own "drop a file
|
|
618
|
-
* anywhere, it's found" convention (`findRootAssets()`
|
|
498
|
+
* anywhere, it's found" convention (`findRootAssets()` in lib/pages.js,
|
|
619
499
|
* `findAllPages()` for content) already worked project-wide. See that
|
|
620
500
|
* integration's own comment for the actual resolution mechanism (a dev-time
|
|
621
501
|
* middleware plus a post-build copy step, not a duplicated `publicDir`)
|
|
@@ -733,7 +613,10 @@ export function fileIdForEntry(
|
|
|
733
613
|
const absoluteContentDir = path.resolve(contentDir);
|
|
734
614
|
const absoluteFilePath = path.resolve(packageRoot, entry.filePath);
|
|
735
615
|
const relativeToContentDir = path.relative(absoluteContentDir, absoluteFilePath);
|
|
736
|
-
|
|
616
|
+
// Outside contentDir: `..`-prefixed, or - on Windows, when the file is on
|
|
617
|
+
// another drive (temp on C:, project on D:) - an absolute path, since
|
|
618
|
+
// path.relative() can't bridge drives.
|
|
619
|
+
if (relativeToContentDir.startsWith('..') || path.isAbsolute(relativeToContentDir)) return entry.id;
|
|
737
620
|
const posixRelative = relativeToContentDir.split(path.sep).join('/');
|
|
738
621
|
if (EXCLUDED_TOP_LEVEL_DIRS.has(posixRelative.split('/')[0])) return entry.id;
|
|
739
622
|
return posixRelative.replace(/\.mdx?$/i, '').replace(/\/index$/, '');
|
|
@@ -155,7 +155,7 @@ export const DESCRIPTIONS = {
|
|
|
155
155
|
...logo('footer.logo'),
|
|
156
156
|
|
|
157
157
|
api: 'The API playground.',
|
|
158
|
-
'api.proxy': '
|
|
158
|
+
'api.proxy': 'When the browser blocks a "Try it" request (the API doesn\'t allow calls from other sites - CORS), retry it through writedocs\' proxy. Requests always go to the API directly first. Default true.',
|
|
159
159
|
domain: 'The site\'s address, like "docs.example.com". Turns on sitemap.xml and absolute URLs in link previews.',
|
|
160
160
|
|
|
161
161
|
seo: 'Default metadata for every page. A page\'s own `seo` frontmatter overrides it field by field.',
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
// llms.txt and its child files (/llms/*.md) - the pages an AI tool can
|
|
2
|
+
// discover, in the site's navigation structure. Shared by llms.txt.ts (the
|
|
3
|
+
// root file) and llms/[...path].md.ts (the child files a big site's llms.txt
|
|
4
|
+
// links to), so both render from the same tree in the same run. The
|
|
5
|
+
// structure and the splitting live in lib/llms.js.
|
|
6
|
+
import { getCollection, type CollectionEntry } from 'astro:content';
|
|
7
|
+
import fs from 'node:fs';
|
|
8
|
+
import path from 'node:path';
|
|
9
|
+
import { loadDocsConfig, normalizeEntryId, findAllPages, resolveSiteUrl, fileIdForEntry } from './config';
|
|
10
|
+
import { buildLlmsTree, renderLlmsFiles } from './llms.js';
|
|
11
|
+
import { navigationPageOrder } from './pages.js';
|
|
12
|
+
import { writedocsTempDir } from './writedocs-temp-dir.js';
|
|
13
|
+
|
|
14
|
+
type DocsEntry = CollectionEntry<'pages'> | CollectionEntry<'generatedDocs'>;
|
|
15
|
+
|
|
16
|
+
// Mirrors Mintlify's own llms.txt behavior: a page's frontmatter
|
|
17
|
+
// `description` is cut at the first line break (a multi-paragraph
|
|
18
|
+
// description would blow out a one-line list entry) and at 300 characters,
|
|
19
|
+
// so every entry stays scannable.
|
|
20
|
+
const DESCRIPTION_MAX_CHARS = 300;
|
|
21
|
+
function truncateDescription(description: string | undefined): string | undefined {
|
|
22
|
+
if (!description) return undefined;
|
|
23
|
+
const firstLine = description.split('\n')[0].trim();
|
|
24
|
+
if (!firstLine) return undefined;
|
|
25
|
+
return firstLine.length > DESCRIPTION_MAX_CHARS ? firstLine.slice(0, DESCRIPTION_MAX_CHARS).trimEnd() + '…' : firstLine;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
// Most characters one file holds - llms.txt or a child file. The llms.txt
|
|
29
|
+
// spec sets no limit ("small enough to fit in context"); this is Mintlify's,
|
|
30
|
+
// about 25,000 tokens. Past it, sections move into child files instead of
|
|
31
|
+
// pages being dropped.
|
|
32
|
+
export const LLMS_MAX_CHARS = 100_000;
|
|
33
|
+
|
|
34
|
+
/** The generated files as Map<path, text> - 'llms.txt', then 'llms/<...>.md'
|
|
35
|
+
* when the site needs them - or null when the project has its own
|
|
36
|
+
* llms.txt, which replaces all of it. */
|
|
37
|
+
export async function llmsIndexFiles(): Promise<Map<string, string> | null> {
|
|
38
|
+
const contentDir = process.env.WRITEDOCS_CONTENT_DIR || process.cwd();
|
|
39
|
+
const packageRoot = process.env.WRITEDOCS_PACKAGE_ROOT || process.cwd();
|
|
40
|
+
// A hand-authored llms.txt at the project root (next to writedocs.json)
|
|
41
|
+
// wins outright - same override convention Mintlify documents.
|
|
42
|
+
if (fs.existsSync(path.join(contentDir, 'llms.txt'))) return null;
|
|
43
|
+
|
|
44
|
+
const config = loadDocsConfig(contentDir);
|
|
45
|
+
const siteUrl = resolveSiteUrl(config); // absolute origin if `domain` is set, else null
|
|
46
|
+
|
|
47
|
+
const hasGeneratedDocs = fs.existsSync(path.join(writedocsTempDir(contentDir), 'generated-docs'));
|
|
48
|
+
const hasPages = findAllPages(contentDir).length > 0;
|
|
49
|
+
const [pagesEntries, generatedDocsEntries] = await Promise.all([
|
|
50
|
+
hasPages ? getCollection('pages') : Promise.resolve([]),
|
|
51
|
+
hasGeneratedDocs ? getCollection('generatedDocs') : Promise.resolve([]),
|
|
52
|
+
]);
|
|
53
|
+
const entries: DocsEntry[] = [...pagesEntries, ...generatedDocsEntries];
|
|
54
|
+
|
|
55
|
+
const pages = new Map<string, { title: string; href: string; description?: string; slug: string }>();
|
|
56
|
+
for (const entry of entries) {
|
|
57
|
+
// A page that opted out of search-engine indexing (`seo.noindex`, also
|
|
58
|
+
// left out of sitemap.xml) is left out here too, and a frontmatter `url`
|
|
59
|
+
// page (Mintlify's external link) has no content of its own.
|
|
60
|
+
if (entry.data.seo?.noindex || entry.data.url) continue;
|
|
61
|
+
const slug = normalizeEntryId(entry.id);
|
|
62
|
+
// The .md route when it exists ([...slug].md.ts - with `contextMenu`,
|
|
63
|
+
// and never for an OpenAPI page, which renders from the spec), else the
|
|
64
|
+
// HTML page.
|
|
65
|
+
const hasMarkdownRoute = config.contextMenu && !entry.data.openapi;
|
|
66
|
+
const href = (siteUrl ?? '') + (hasMarkdownRoute ? `/${slug}.md` : slug === 'index' ? '/' : `/${slug}/`);
|
|
67
|
+
let description = truncateDescription(entry.data.description);
|
|
68
|
+
// Mirrors Mintlify: an OpenAPI operation page's description gets its
|
|
69
|
+
// "METHOD /path" appended - the page itself renders from the spec.
|
|
70
|
+
if (entry.data.openapi) description = description ? `${description} (${entry.data.openapi})` : entry.data.openapi;
|
|
71
|
+
pages.set(fileIdForEntry(contentDir, packageRoot, entry), { title: entry.data.title, href, description, slug });
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
const listed = new Set(navigationPageOrder(config.navigation));
|
|
75
|
+
const unlisted = [...pages.entries()]
|
|
76
|
+
.filter(([id]) => !listed.has(id))
|
|
77
|
+
.sort((a, b) => a[1].slug.localeCompare(b[1].slug))
|
|
78
|
+
.map(([id]) => id);
|
|
79
|
+
const tree = buildLlmsTree(config.navigation, (id: string) => pages.get(id) ?? null, unlisted);
|
|
80
|
+
|
|
81
|
+
return renderLlmsFiles({
|
|
82
|
+
name: config.name,
|
|
83
|
+
description: config.description,
|
|
84
|
+
tree,
|
|
85
|
+
maxChars: LLMS_MAX_CHARS,
|
|
86
|
+
urlFor: (file: string) => `${siteUrl ?? ''}/${file}`,
|
|
87
|
+
});
|
|
88
|
+
}
|
package/src/lib/llms.js
ADDED
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
// Shared by the llms routes (lib/llms-index.ts, llms-full.txt.ts): the order
|
|
2
|
+
// pages appear in, and llms.txt's structure. Plain JavaScript, so it can be
|
|
3
|
+
// tested with plain Node.
|
|
4
|
+
import { navigationPageOrder } from './pages.js';
|
|
5
|
+
|
|
6
|
+
/** `items` ({ fileId, slug }) in navigation order - the order a reader meets
|
|
7
|
+
* the pages in the sidebar - then every page the navigation doesn't list,
|
|
8
|
+
* by URL. */
|
|
9
|
+
export function orderByNavigation(items, navigation) {
|
|
10
|
+
const rank = new Map(navigationPageOrder(navigation).map((id, i) => [id, i]));
|
|
11
|
+
const at = (item) => rank.get(item.fileId) ?? Infinity;
|
|
12
|
+
return [...items].sort((a, b) => at(a) - at(b) || a.slug.localeCompare(b.slug));
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
// --- llms.txt as a tree -------------------------------------------------
|
|
16
|
+
//
|
|
17
|
+
// llms.txt lists every page under headings that follow the navigation -
|
|
18
|
+
// products, versions, languages, tabs, dropdowns, groups. A site too big for
|
|
19
|
+
// one file (MAX_CHARS) keeps llms.txt as a directory: its largest sections
|
|
20
|
+
// move into child files under /llms/, linked from where they were, and a
|
|
21
|
+
// child too big is split the same way. No page is ever left out - the scheme
|
|
22
|
+
// Mintlify moved to from cutting the list off
|
|
23
|
+
// (https://www.mintlify.com/blog/scaling-llms-txt).
|
|
24
|
+
|
|
25
|
+
const CONTAINERS = [
|
|
26
|
+
['tabs', 'tab'],
|
|
27
|
+
['versions', 'version'],
|
|
28
|
+
['languages', 'language'],
|
|
29
|
+
['dropdowns', 'dropdown'],
|
|
30
|
+
['products', 'product'],
|
|
31
|
+
];
|
|
32
|
+
|
|
33
|
+
/** A Section: { label, items: [page], sections: [Section] }. `pageFor(id)`
|
|
34
|
+
* gives a listed page ({ title, href, description }) or null for one that
|
|
35
|
+
* isn't listed (noindex, an external `url`, missing). A page shows up once,
|
|
36
|
+
* where the navigation first lists it; `unlisted` pages (not in the
|
|
37
|
+
* navigation at all) go in a trailing "Other pages" section. */
|
|
38
|
+
export function buildLlmsTree(navigation, pageFor, unlisted = []) {
|
|
39
|
+
const seen = new Set();
|
|
40
|
+
const page = (id) => {
|
|
41
|
+
if (seen.has(id)) return null;
|
|
42
|
+
const found = pageFor(id);
|
|
43
|
+
if (!found) return null;
|
|
44
|
+
seen.add(id);
|
|
45
|
+
return found;
|
|
46
|
+
};
|
|
47
|
+
const prune = (section) => section.items.length > 0 || section.sections.length > 0;
|
|
48
|
+
|
|
49
|
+
function fromList(list) {
|
|
50
|
+
const node = { items: [], sections: [] };
|
|
51
|
+
for (const item of list ?? []) {
|
|
52
|
+
if (typeof item === 'string') {
|
|
53
|
+
const p = page(item);
|
|
54
|
+
if (p) node.items.push(p);
|
|
55
|
+
} else if (item && typeof item === 'object' && typeof item.group === 'string' && Array.isArray(item.pages)) {
|
|
56
|
+
const own = typeof item.page === 'string' ? page(item.page) : null;
|
|
57
|
+
const inner = fromList(item.pages);
|
|
58
|
+
const section = { label: item.group, items: own ? [own, ...inner.items] : inner.items, sections: inner.sections };
|
|
59
|
+
if (prune(section)) node.sections.push(section);
|
|
60
|
+
}
|
|
61
|
+
// A link leaf ({ href }) isn't a page of this site.
|
|
62
|
+
}
|
|
63
|
+
return node;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
function fromContainer(container) {
|
|
67
|
+
if (Array.isArray(container)) return fromList(container);
|
|
68
|
+
if (!container || typeof container !== 'object') return { items: [], sections: [] };
|
|
69
|
+
if (Array.isArray(container.pages)) return fromList(container.pages);
|
|
70
|
+
for (const [key, labelKey] of CONTAINERS) {
|
|
71
|
+
if (!Array.isArray(container[key])) continue;
|
|
72
|
+
const sections = container[key]
|
|
73
|
+
.filter((child) => child && typeof child === 'object' && !('href' in child))
|
|
74
|
+
.map((child) => {
|
|
75
|
+
const label = labelKey === 'version' || labelKey === 'language' ? child.label ?? child[labelKey] : child[labelKey];
|
|
76
|
+
return { label: String(label), ...fromContainer(child) };
|
|
77
|
+
})
|
|
78
|
+
.filter(prune);
|
|
79
|
+
return { items: [], sections };
|
|
80
|
+
}
|
|
81
|
+
return { items: [], sections: [] };
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
const root = fromContainer(navigation);
|
|
85
|
+
for (const dropdown of (!Array.isArray(navigation) && navigation?.global?.dropdowns) || []) {
|
|
86
|
+
if ('href' in dropdown) continue;
|
|
87
|
+
const section = { label: String(dropdown.dropdown), ...fromContainer(dropdown) };
|
|
88
|
+
if (prune(section)) root.sections.push(section);
|
|
89
|
+
}
|
|
90
|
+
const rest = unlisted.map((id) => page(id)).filter(Boolean);
|
|
91
|
+
if (rest.length) root.sections.push({ label: 'Other pages', items: rest, sections: [] });
|
|
92
|
+
return root;
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
const pageLine = (p) => `- [${p.title}](${p.href})${p.description ? `: ${p.description}` : ''}`;
|
|
96
|
+
const countPages = (s) => s.items.length + s.sections.reduce((n, c) => n + countPages(c), 0);
|
|
97
|
+
const heading = (depth, label) => `${'#'.repeat(Math.min(depth, 6))} ${label}`;
|
|
98
|
+
|
|
99
|
+
function slugify(label) {
|
|
100
|
+
return String(label).toLowerCase().replace(/[^a-z0-9]+/g, '-').replace(/^-+|-+$/g, '') || 'section';
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
/** The body of `node` as Markdown lines: its pages, then each sub-section -
|
|
104
|
+
* under its own heading, or, when it's in `split`, as a link to its file. */
|
|
105
|
+
function renderBody(node, depth, split) {
|
|
106
|
+
const lines = [];
|
|
107
|
+
if (node.items.length) lines.push(...node.items.map(pageLine), '');
|
|
108
|
+
for (const section of node.sections) {
|
|
109
|
+
lines.push(heading(depth, section.label), '');
|
|
110
|
+
const file = split.get(section);
|
|
111
|
+
if (file) {
|
|
112
|
+
const n = countPages(section);
|
|
113
|
+
lines.push(`- [${section.label}](${file.url}): ${n} page${n === 1 ? '' : 's'}, listed in their own file`, '');
|
|
114
|
+
} else {
|
|
115
|
+
lines.push(...renderBody(section, depth + 1, split));
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
return lines;
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
const render = (header, node, depth, split) => [...header, ...renderBody(node, depth, split)].join('\n').replace(/\n{3,}/g, '\n\n').trimEnd() + '\n';
|
|
122
|
+
|
|
123
|
+
/** Every section inside `node` still rendered inline (not split, and not
|
|
124
|
+
* inside a split one), with the characters it takes. */
|
|
125
|
+
function inlineSections(node, depth, split, out = []) {
|
|
126
|
+
for (const section of node.sections) {
|
|
127
|
+
if (split.has(section)) continue;
|
|
128
|
+
out.push({ section, size: render([], { items: [], sections: [section] }, depth, split).length });
|
|
129
|
+
inlineSections(section, depth + 1, split, out);
|
|
130
|
+
}
|
|
131
|
+
return out;
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/** A section whose own page list alone is too big becomes parts, each a
|
|
135
|
+
* sub-section small enough for a file of its own. */
|
|
136
|
+
function chunkItems(node, maxChars, label) {
|
|
137
|
+
const budget = Math.max(1000, maxChars - 2000);
|
|
138
|
+
const parts = [];
|
|
139
|
+
let current = [];
|
|
140
|
+
let size = 0;
|
|
141
|
+
for (const p of node.items) {
|
|
142
|
+
const n = pageLine(p).length + 1;
|
|
143
|
+
if (current.length && size + n > budget) {
|
|
144
|
+
parts.push(current);
|
|
145
|
+
current = [];
|
|
146
|
+
size = 0;
|
|
147
|
+
}
|
|
148
|
+
current.push(p);
|
|
149
|
+
size += n;
|
|
150
|
+
}
|
|
151
|
+
if (current.length) parts.push(current);
|
|
152
|
+
// One part is the list as it was - nothing smaller to split into (a single
|
|
153
|
+
// page line longer than the budget). Leave it, rather than loop.
|
|
154
|
+
if (parts.length < 2) return [];
|
|
155
|
+
const sections = parts.map((items, i) => ({ label: `${label} (part ${i + 1} of ${parts.length})`, items, sections: [] }));
|
|
156
|
+
node.sections = [...sections, ...node.sections];
|
|
157
|
+
node.items = [];
|
|
158
|
+
return sections;
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
/** llms.txt, and any child files it needs, as Map<path, text>: 'llms.txt',
|
|
162
|
+
* then e.g. 'llms/core-platform.md'. `urlFor(path)` gives a file's URL. */
|
|
163
|
+
export function renderLlmsFiles({ name, description, tree, maxChars, urlFor }) {
|
|
164
|
+
const files = new Map();
|
|
165
|
+
const taken = new Set(['llms.txt']);
|
|
166
|
+
const rootUrl = urlFor('llms.txt');
|
|
167
|
+
|
|
168
|
+
function emit(filePath, header, node, depth, trail) {
|
|
169
|
+
const split = new Map();
|
|
170
|
+
const splitOut = (section) => {
|
|
171
|
+
const dir = filePath === 'llms.txt' ? 'llms' : filePath.replace(/\.md$/, '');
|
|
172
|
+
let childPath = `${dir}/${slugify(section.label)}.md`;
|
|
173
|
+
for (let n = 2; taken.has(childPath); n++) childPath = `${dir}/${slugify(section.label)}-${n}.md`;
|
|
174
|
+
taken.add(childPath);
|
|
175
|
+
split.set(section, { path: childPath, url: urlFor(childPath) });
|
|
176
|
+
};
|
|
177
|
+
// A page list too long for this file on its own becomes parts - all of
|
|
178
|
+
// them separate files, so this one is a plain list of the parts.
|
|
179
|
+
if (render(header, { items: node.items, sections: [] }, depth, split).length > maxChars) {
|
|
180
|
+
chunkItems(node, maxChars, trail.length ? trail[trail.length - 1] : 'Pages').forEach(splitOut);
|
|
181
|
+
}
|
|
182
|
+
// Then the biggest sections move out, one at a time, until this fits.
|
|
183
|
+
while (render(header, node, depth, split).length > maxChars) {
|
|
184
|
+
const candidates = inlineSections(node, depth, split);
|
|
185
|
+
if (!candidates.length) break;
|
|
186
|
+
splitOut(candidates.reduce((a, b) => (b.size > a.size ? b : a)).section);
|
|
187
|
+
}
|
|
188
|
+
files.set(filePath, render(header, node, depth, split));
|
|
189
|
+
for (const [section, file] of split) {
|
|
190
|
+
const childTrail = [...trail, section.label];
|
|
191
|
+
const childHeader = [`# ${name}: ${childTrail.join(' > ')}`, '', `> Part of the ${name} docs index: ${rootUrl}`, ''];
|
|
192
|
+
emit(file.path, childHeader, section, 2, childTrail);
|
|
193
|
+
}
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
const header = [`# ${name}`, ''];
|
|
197
|
+
if (description) header.push(`> ${description}`, '');
|
|
198
|
+
emit('llms.txt', header, tree, 2, []);
|
|
199
|
+
return files;
|
|
200
|
+
}
|
|
@@ -11,6 +11,7 @@
|
|
|
11
11
|
import fs from 'node:fs';
|
|
12
12
|
import path from 'node:path';
|
|
13
13
|
import { iconExists } from './icons.js';
|
|
14
|
+
import { readJsonText } from './config-file.js';
|
|
14
15
|
|
|
15
16
|
// ---------------------------------------------------------------------
|
|
16
17
|
// Reading docs.json
|
|
@@ -28,7 +29,7 @@ export function loadMintlifyConfig(file) {
|
|
|
28
29
|
if (seen.has(abs)) throw new Error(`Circular $ref: ${path.relative(root, abs)}`);
|
|
29
30
|
seen.add(abs);
|
|
30
31
|
try {
|
|
31
|
-
return resolve(JSON.parse(
|
|
32
|
+
return resolve(JSON.parse(readJsonText(abs)), path.dirname(abs));
|
|
32
33
|
} finally {
|
|
33
34
|
seen.delete(abs);
|
|
34
35
|
}
|