@writedocs/generator 0.7.3 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/astro.config.mjs +4 -0
- package/package.json +3 -2
- package/src/cli/build.js +8 -0
- package/src/cli/write-mcp-files.js +96 -0
- package/src/lib/agent-markdown.js +142 -0
- package/src/lib/config-file.js +1 -1
- package/src/lib/config-schema.js +5 -0
- package/src/lib/config-schema.ts +5 -0
- package/src/lib/json-schema-descriptions.js +1 -0
- package/src/lib/llms-index.ts +88 -0
- package/src/lib/llms.js +200 -0
- package/src/lib/mcp-dev-integration.js +60 -0
- package/src/lib/mcp-index.ts +101 -0
- package/src/lib/pages.js +20 -14
- package/src/mcp/server.js +262 -0
- package/src/pages/[...slug].md.ts +15 -6
- package/src/pages/[mcpIndex].json.ts +16 -0
- package/src/pages/llms/[...path].md.ts +23 -0
- package/src/pages/llms-full.txt.ts +30 -9
- package/src/pages/llms.txt.ts +21 -129
- package/writedocs.schema.json +6 -0
package/astro.config.mjs
CHANGED
|
@@ -36,6 +36,7 @@ import { remarkUnknownComponentFallback } from './src/lib/mdx-unknown-components
|
|
|
36
36
|
import { remarkExtractInlineReactComponents } from './src/lib/mdx-inline-react.js';
|
|
37
37
|
import { writedocsTempDir, writedocsBuildStagingDir } from './src/lib/writedocs-temp-dir.js';
|
|
38
38
|
import { stylesAssetFallback } from './src/lib/styles-asset-integration.js';
|
|
39
|
+
import { mcpDevServer } from './src/lib/mcp-dev-integration.js';
|
|
39
40
|
import { report } from './src/lib/cli-report.js';
|
|
40
41
|
import {
|
|
41
42
|
loadDocsConfig,
|
|
@@ -309,6 +310,9 @@ export default defineConfig({
|
|
|
309
310
|
// field list and styles-asset-integration.js for how it's actually
|
|
310
311
|
// served/copied.
|
|
311
312
|
stylesAssetFallback(contentDir),
|
|
313
|
+
// /mcp in `writedocs dev`, as the build's dist/_worker.js serves it -
|
|
314
|
+
// see src/lib/mcp-dev-integration.js.
|
|
315
|
+
mcpDevServer({ version: JSON.parse(fs.readFileSync(path.join(packageRoot, 'package.json'), 'utf-8')).version }),
|
|
312
316
|
],
|
|
313
317
|
output: 'static',
|
|
314
318
|
// Dual Shiki themes for fenced code blocks (```) in MDX content, so
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@writedocs/generator",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.8.0",
|
|
4
4
|
"description": "Static site generator for docs — a writedocs.json + MDX folder in, a static site out.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
@@ -9,7 +9,8 @@
|
|
|
9
9
|
"exports": {
|
|
10
10
|
"./components": "./src/components/index.ts",
|
|
11
11
|
"./config-schema": "./src/lib/config-schema.js",
|
|
12
|
-
"./writedocs.schema.json": "./writedocs.schema.json"
|
|
12
|
+
"./writedocs.schema.json": "./writedocs.schema.json",
|
|
13
|
+
"./mcp": "./src/mcp/server.js"
|
|
13
14
|
},
|
|
14
15
|
"files": [
|
|
15
16
|
"bin",
|
package/src/cli/build.js
CHANGED
|
@@ -5,6 +5,7 @@ import { runPagefind } from './run-pagefind.js';
|
|
|
5
5
|
import { preflightCheck, sameDriveCheck, writableInstallCheck } from './preflight.js';
|
|
6
6
|
import { generateApiPages } from './generate-api-pages.js';
|
|
7
7
|
import { writeRedirectsFile } from './write-redirects-file.js';
|
|
8
|
+
import { writeMcpFiles } from './write-mcp-files.js';
|
|
8
9
|
import { writedocsBuildStagingDir } from '../lib/writedocs-temp-dir.js';
|
|
9
10
|
import { log, step, plural, duration, displayPath, formatProblems, color, CliExit } from './output.js';
|
|
10
11
|
import { describeError, builtRoute, authorWarning, verboseLine } from './astro-output.js';
|
|
@@ -112,6 +113,13 @@ export async function runBuild({ contentDir, packageRoot, verbose = false }) {
|
|
|
112
113
|
// - this turns those into real instant edge redirects on hosts that read
|
|
113
114
|
// a `_redirects` file (Cloudflare Pages, Netlify), purely additively.
|
|
114
115
|
writeRedirectsFile(distDir, contentDir);
|
|
116
|
+
// The site's MCP server at /mcp: mcp-index.json is already in dist/ (an
|
|
117
|
+
// Astro route); this adds the Cloudflare _worker.js that serves it - see
|
|
118
|
+
// write-mcp-files.js for every host.
|
|
119
|
+
const { version: writedocsVersion } = JSON.parse(fs.readFileSync(path.join(packageRoot, 'package.json'), 'utf-8'));
|
|
120
|
+
const mcp = writeMcpFiles(distDir, contentDir, writedocsVersion);
|
|
121
|
+
if (mcp.written.length) notes.push('MCP server at /mcp - dist/_worker.js runs it on Cloudflare; mcp-index.json holds the pages.');
|
|
122
|
+
if (mcp.skipped) addWarning(null, `No _worker.js for the MCP server: ${mcp.skipped}.`);
|
|
115
123
|
|
|
116
124
|
const indexing = step('Indexing search');
|
|
117
125
|
try {
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
// After `writedocs build`: the files that make dist/ serve the site's MCP
|
|
2
|
+
// server at /mcp on Cloudflare, with no setup of the host's own - next to
|
|
3
|
+
// mcp-index.json, which the build itself writes (src/pages/[mcpIndex].json.ts).
|
|
4
|
+
//
|
|
5
|
+
// _worker.js the MCP server (src/mcp/server.js, inlined - no imports)
|
|
6
|
+
// plus a small fetch handler: /mcp goes to the server,
|
|
7
|
+
// everything else to the static files (env.ASSETS).
|
|
8
|
+
// - Cloudflare Pages runs it on its own ("advanced mode").
|
|
9
|
+
// - Cloudflare Workers with static assets: point `main` at
|
|
10
|
+
// dist/_worker.js, assets at dist/ with an ASSETS binding.
|
|
11
|
+
// _routes.json Pages: only /mcp runs the worker; every other path stays a
|
|
12
|
+
// plain static file (and `_redirects` keeps applying).
|
|
13
|
+
// .assetsignore Workers: don't publish _worker.js / _routes.json as files.
|
|
14
|
+
//
|
|
15
|
+
// Anywhere else - Netlify, Vercel, the WriteDocs platform's serving Worker -
|
|
16
|
+
// import handleMcpHttp from `@writedocs/generator/mcp` and give it
|
|
17
|
+
// mcp-index.json; a purely static host (GitHub Pages, S3) can't run /mcp.
|
|
18
|
+
//
|
|
19
|
+
// A project that brings its own _worker.js or _routes.json (in public/), or a
|
|
20
|
+
// Pages `functions/` folder (which Pages ignores once a _worker.js exists),
|
|
21
|
+
// keeps them: the worker isn't written, and the build says so.
|
|
22
|
+
import fs from 'node:fs';
|
|
23
|
+
import path from 'node:path';
|
|
24
|
+
import { fileURLToPath } from 'node:url';
|
|
25
|
+
|
|
26
|
+
const SERVER_SOURCE = path.join(path.dirname(fileURLToPath(import.meta.url)), '..', 'mcp', 'server.js');
|
|
27
|
+
const IGNORED_BY_WORKERS = ['_worker.js', '_routes.json'];
|
|
28
|
+
|
|
29
|
+
/** The `_worker.js` source: server.js without its `export`s, and the fetch
|
|
30
|
+
* handler. `version` is the writedocs version, reported by the server. */
|
|
31
|
+
export function mcpWorkerSource(version) {
|
|
32
|
+
const server = fs.readFileSync(SERVER_SOURCE, 'utf-8').replace(/^export (?=(?:const|let|async function|function) )/gm, '');
|
|
33
|
+
return `// Generated by writedocs ${version} - the site's MCP server at /mcp.
|
|
34
|
+
// Cloudflare Pages runs this file on its own; on Cloudflare Workers, use it as
|
|
35
|
+
// \`main\` with dist/ as the static assets (binding ASSETS). See
|
|
36
|
+
// https://github.com/writedocs/writedocs (src/cli/write-mcp-files.js).
|
|
37
|
+
|
|
38
|
+
${server}
|
|
39
|
+
const WRITEDOCS_VERSION = ${JSON.stringify(version)};
|
|
40
|
+
|
|
41
|
+
// The page index, read once per isolate from the site's own static files.
|
|
42
|
+
let indexPromise = null;
|
|
43
|
+
function loadIndex(request, env) {
|
|
44
|
+
if (!indexPromise) {
|
|
45
|
+
indexPromise = env.ASSETS.fetch(new URL('/mcp-index.json', request.url))
|
|
46
|
+
.then((res) => {
|
|
47
|
+
if (!res.ok) throw new Error('mcp-index.json: HTTP ' + res.status);
|
|
48
|
+
return res.json();
|
|
49
|
+
})
|
|
50
|
+
.catch((err) => {
|
|
51
|
+
indexPromise = null;
|
|
52
|
+
throw err;
|
|
53
|
+
});
|
|
54
|
+
}
|
|
55
|
+
return indexPromise;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
export default {
|
|
59
|
+
async fetch(request, env) {
|
|
60
|
+
const { pathname } = new URL(request.url);
|
|
61
|
+
if (pathname === '/mcp' || pathname === '/mcp/') {
|
|
62
|
+
return handleMcpHttp(request, { loadIndex: () => loadIndex(request, env), version: WRITEDOCS_VERSION });
|
|
63
|
+
}
|
|
64
|
+
return env.ASSETS.fetch(request);
|
|
65
|
+
},
|
|
66
|
+
};
|
|
67
|
+
`;
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/** Writes the files into `distDir`. Returns { written: [names], skipped:
|
|
71
|
+
* reason or null } - nothing at all when there's no mcp-index.json
|
|
72
|
+
* (writedocs.json `"mcp": false`). */
|
|
73
|
+
export function writeMcpFiles(distDir, contentDir, version) {
|
|
74
|
+
if (!fs.existsSync(path.join(distDir, 'mcp-index.json'))) return { written: [], skipped: null };
|
|
75
|
+
|
|
76
|
+
const own = ['_worker.js', '_routes.json'].filter((name) => fs.existsSync(path.join(distDir, name)));
|
|
77
|
+
if (own.length) {
|
|
78
|
+
return { written: [], skipped: `the project has its own ${own.join(' and ')} - /mcp needs its handler added there (see @writedocs/generator/mcp)` };
|
|
79
|
+
}
|
|
80
|
+
if (fs.existsSync(path.join(contentDir, 'functions'))) {
|
|
81
|
+
return { written: [], skipped: 'the project has a Cloudflare Pages functions/ folder, which a _worker.js would switch off - add /mcp there with @writedocs/generator/mcp' };
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
fs.writeFileSync(path.join(distDir, '_worker.js'), mcpWorkerSource(version));
|
|
85
|
+
fs.writeFileSync(path.join(distDir, '_routes.json'), JSON.stringify({ version: 1, include: ['/mcp', '/mcp/'], exclude: [] }, null, 2) + '\n');
|
|
86
|
+
|
|
87
|
+
const ignorePath = path.join(distDir, '.assetsignore');
|
|
88
|
+
const existing = fs.existsSync(ignorePath) ? fs.readFileSync(ignorePath, 'utf-8') : '';
|
|
89
|
+
const lines = existing.split(/\r?\n/).map((l) => l.trim());
|
|
90
|
+
const missing = IGNORED_BY_WORKERS.filter((name) => !lines.includes(name));
|
|
91
|
+
if (missing.length) {
|
|
92
|
+
const prefix = existing && !existing.endsWith('\n') ? `${existing}\n` : existing;
|
|
93
|
+
fs.writeFileSync(ignorePath, `${prefix}${missing.join('\n')}\n`);
|
|
94
|
+
}
|
|
95
|
+
return { written: ['_worker.js', '_routes.json', '.assetsignore'], skipped: null };
|
|
96
|
+
}
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
// The Markdown an AI agent reads for a page - llms-full.txt and the per-page
|
|
2
|
+
// .md route. It starts from the page's raw MDX, so everything the build does
|
|
3
|
+
// to that source on the way to HTML has to be redone here, or the agent reads
|
|
4
|
+
// something other than the page: `<Visibility>` blocks, imported .mdx
|
|
5
|
+
// snippets (inlined, as the page shows them) and writedocs.json `variables`.
|
|
6
|
+
//
|
|
7
|
+
// Code is left exactly as written - fenced blocks and inline `code` - the same
|
|
8
|
+
// way the build never substitutes inside code: a page documenting the
|
|
9
|
+
// `[[key]]` syntax, or showing `<Snippet />` in an example, must keep it.
|
|
10
|
+
// Plain JavaScript, so it can be tested with plain Node.
|
|
11
|
+
import fs from 'node:fs';
|
|
12
|
+
import path from 'node:path';
|
|
13
|
+
import matter from 'gray-matter';
|
|
14
|
+
import { applyVisibilityForAgents } from './visibility.js';
|
|
15
|
+
|
|
16
|
+
/** `text` with `transform` applied to everything outside fenced code blocks
|
|
17
|
+
* and inline code spans; code comes back untouched. */
|
|
18
|
+
export function mapOutsideCode(text, transform) {
|
|
19
|
+
const out = [];
|
|
20
|
+
let prose = [];
|
|
21
|
+
let fence = null; // the opening marker (``` or ~~~, maybe longer) while inside a block
|
|
22
|
+
const flush = () => {
|
|
23
|
+
if (prose.length) out.push(mapOutsideInlineCode(prose.join('\n'), transform));
|
|
24
|
+
prose = [];
|
|
25
|
+
};
|
|
26
|
+
for (const line of text.split('\n')) {
|
|
27
|
+
const marker = /^\s*(`{3,}|~{3,})/.exec(line)?.[1];
|
|
28
|
+
if (fence) {
|
|
29
|
+
out.push(line);
|
|
30
|
+
if (marker && marker[0] === fence[0] && marker.length >= fence.length && line.trim() === marker) fence = null;
|
|
31
|
+
} else if (marker) {
|
|
32
|
+
flush();
|
|
33
|
+
fence = marker;
|
|
34
|
+
out.push(line);
|
|
35
|
+
} else {
|
|
36
|
+
prose.push(line);
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
flush();
|
|
40
|
+
return out.join('\n');
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
function mapOutsideInlineCode(text, transform) {
|
|
44
|
+
// Backtick runs of equal length delimit a code span (CommonMark).
|
|
45
|
+
return text
|
|
46
|
+
.split(/(`+)([\s\S]*?)(\1)(?!`)/)
|
|
47
|
+
.reduce((acc, part, i, parts) => {
|
|
48
|
+
const k = i % 4;
|
|
49
|
+
if (k === 0) acc.push(transform(part));
|
|
50
|
+
else if (k === 1) acc.push(part + parts[i + 1] + parts[i + 2]);
|
|
51
|
+
return acc;
|
|
52
|
+
}, [])
|
|
53
|
+
.join('');
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
const PLACEHOLDER = /\[\[\s*([\w.-]+)\s*\]\]/g;
|
|
57
|
+
// A standalone `{key}` or `{{key}}` expression (Mintlify writes the latter;
|
|
58
|
+
// MDX reads it as `{key}` inside an expression) - not an attribute value
|
|
59
|
+
// (`prop={key}`), which the build leaves alone too. The double form first,
|
|
60
|
+
// or its outer braces would be left around the value.
|
|
61
|
+
const MINTLIFY_PLACEHOLDER = /(?<!=\s*)\{\s*\{\s*([A-Za-z_$][\w$]*)\s*\}\s*\}|(?<!=\s*)\{\s*([A-Za-z_$][\w$]*)\s*\}/g;
|
|
62
|
+
|
|
63
|
+
/** writedocs.json `variables` - `[[key]]`, and Mintlify's `{key}` - replaced
|
|
64
|
+
* the way the build replaces them (lib/mdx-substitute-variables.js): outside
|
|
65
|
+
* code, and only for keys that exist. */
|
|
66
|
+
export function substituteVariables(text, variables) {
|
|
67
|
+
if (!variables || Object.keys(variables).length === 0) return text;
|
|
68
|
+
const has = (key) => Object.prototype.hasOwnProperty.call(variables, key);
|
|
69
|
+
return mapOutsideCode(text, (prose) =>
|
|
70
|
+
prose
|
|
71
|
+
.replace(PLACEHOLDER, (match, key) => (has(key) ? String(variables[key]) : match))
|
|
72
|
+
.replace(MINTLIFY_PLACEHOLDER, (match, doubled, single) => {
|
|
73
|
+
const key = doubled ?? single;
|
|
74
|
+
return has(key) ? String(variables[key]) : match;
|
|
75
|
+
})
|
|
76
|
+
);
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
const IMPORT_LINE = /^[ \t]*import\s+([A-Za-z_$][\w$]*)\s+from\s+['"]([^'"]+\.mdx?)['"];?[ \t]*$/gm;
|
|
80
|
+
|
|
81
|
+
function resolveSnippet(spec, fromFile, contentDir) {
|
|
82
|
+
if (spec.startsWith('/snippets/')) return path.join(contentDir, 'snippets', spec.slice('/snippets/'.length));
|
|
83
|
+
if (spec.startsWith('./') || spec.startsWith('../')) return path.resolve(path.dirname(fromFile), spec);
|
|
84
|
+
if (spec.startsWith('/')) return path.join(contentDir, spec.slice(1));
|
|
85
|
+
return null;
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
function attributesOf(source) {
|
|
89
|
+
const props = {};
|
|
90
|
+
for (const m of source.matchAll(/([A-Za-z_$][\w$-]*)\s*=\s*(?:"([^"]*)"|'([^']*)'|\{\s*["']([^"']*)["']\s*\})/g)) {
|
|
91
|
+
props[m[1]] = m[2] ?? m[3] ?? m[4];
|
|
92
|
+
}
|
|
93
|
+
return props;
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
function fillProps(snippet, props, children) {
|
|
97
|
+
return snippet
|
|
98
|
+
.replace(/\{\s*props\.children\s*\}/g, children ?? '')
|
|
99
|
+
.replace(/\{\s*props\.([A-Za-z_$][\w$]*)\s*\}/g, (match, key) => (key in props ? props[key] : match));
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/** Imported .mdx snippets inlined where the page uses them - the snippet's
|
|
103
|
+
* own text, with `{props.x}` filled from the tag's string attributes, and
|
|
104
|
+
* its own imports inlined in turn. The import line goes too. A snippet that
|
|
105
|
+
* can't be read, or a .jsx/.tsx component, stays as written. */
|
|
106
|
+
export function inlineSnippets(text, { file, contentDir, depth = 0 }) {
|
|
107
|
+
if (depth > 5) return text;
|
|
108
|
+
const snippets = new Map();
|
|
109
|
+
let result = mapOutsideCode(text, (prose) =>
|
|
110
|
+
prose.replace(IMPORT_LINE, (line, name, spec) => {
|
|
111
|
+
const target = resolveSnippet(spec, file, contentDir);
|
|
112
|
+
let raw;
|
|
113
|
+
try {
|
|
114
|
+
raw = target && fs.readFileSync(target, 'utf-8');
|
|
115
|
+
} catch {
|
|
116
|
+
raw = null;
|
|
117
|
+
}
|
|
118
|
+
if (!raw) return line;
|
|
119
|
+
const body = inlineSnippets(matter(raw).content.trim(), { file: target, contentDir, depth: depth + 1 });
|
|
120
|
+
snippets.set(name, body);
|
|
121
|
+
return '';
|
|
122
|
+
})
|
|
123
|
+
);
|
|
124
|
+
for (const [name, body] of snippets) {
|
|
125
|
+
result = mapOutsideCode(result, (prose) =>
|
|
126
|
+
prose
|
|
127
|
+
.replace(new RegExp(`<${name}\\b([^>]*?)\\/>`, 'g'), (_, attrs) => fillProps(body, attributesOf(attrs)))
|
|
128
|
+
.replace(new RegExp(`<${name}\\b([^>]*)>([\\s\\S]*?)<\\/${name}>`, 'g'), (_, attrs, children) =>
|
|
129
|
+
fillProps(body, attributesOf(attrs), children.trim())
|
|
130
|
+
)
|
|
131
|
+
);
|
|
132
|
+
}
|
|
133
|
+
return result.replace(/\n{3,}/g, '\n\n');
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
/** A page's body as an agent should read it. `file` is the page's absolute
|
|
137
|
+
* path (to resolve relative snippet imports). */
|
|
138
|
+
export function markdownForAgents(body, { file, contentDir, variables }) {
|
|
139
|
+
let text = applyVisibilityForAgents(body ?? '');
|
|
140
|
+
if (file) text = inlineSnippets(text, { file, contentDir });
|
|
141
|
+
return substituteVariables(text, variables);
|
|
142
|
+
}
|
package/src/lib/config-file.js
CHANGED
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
//
|
|
5
5
|
// Windows editors - Notepad, PowerShell 5.1's `Out-File`/`Set-Content
|
|
6
6
|
// -Encoding utf8` - save UTF-8 with a byte order mark. JSON.parse rejects
|
|
7
|
-
// it ("Unexpected token
|
|
7
|
+
// it ("Unexpected token U+FEFF"), and the error then blames commas and
|
|
8
8
|
// brackets the file doesn't have. Every reader strips it here instead.
|
|
9
9
|
import fs from 'node:fs';
|
|
10
10
|
import path from 'node:path';
|
package/src/lib/config-schema.js
CHANGED
|
@@ -522,6 +522,11 @@ const docsConfigSchema = z.object({
|
|
|
522
522
|
// The "Copy page" dropdown - see contextMenuSchema above. Absent by
|
|
523
523
|
// default (no menu, no .md routes).
|
|
524
524
|
contextMenu: contextMenuSchema.optional(),
|
|
525
|
+
// The site's MCP server - an AI tool connects to `/mcp` and searches and
|
|
526
|
+
// reads the docs. On by default: `writedocs build` writes the page index
|
|
527
|
+
// (mcp-index.json) and a Cloudflare-ready `_worker.js` into dist/ (see
|
|
528
|
+
// src/cli/write-mcp-files.js). `false` leaves all of it out.
|
|
529
|
+
mcp: z.boolean().default(true),
|
|
525
530
|
// See redirectSchema above. Wired directly into Astro's own `redirects`
|
|
526
531
|
// config option in astro.config.mjs.
|
|
527
532
|
redirects: z.array(redirectSchema).default([]),
|
package/src/lib/config-schema.ts
CHANGED
|
@@ -999,6 +999,11 @@ export const docsConfigSchema = z.object({
|
|
|
999
999
|
// The "Copy page" dropdown - see contextMenuSchema above. Absent by
|
|
1000
1000
|
// default (no menu, no .md routes).
|
|
1001
1001
|
contextMenu: contextMenuSchema.optional(),
|
|
1002
|
+
// The site's MCP server - an AI tool connects to `/mcp` and searches and
|
|
1003
|
+
// reads the docs. On by default: `writedocs build` writes the page index
|
|
1004
|
+
// (mcp-index.json) and a Cloudflare-ready `_worker.js` into dist/ (see
|
|
1005
|
+
// src/cli/write-mcp-files.js). `false` leaves all of it out.
|
|
1006
|
+
mcp: z.boolean().default(true),
|
|
1002
1007
|
// See redirectSchema above. Wired directly into Astro's own `redirects`
|
|
1003
1008
|
// config option in astro.config.mjs.
|
|
1004
1009
|
redirects: z.array(redirectSchema).default([]),
|
|
@@ -166,6 +166,7 @@ export const DESCRIPTIONS = {
|
|
|
166
166
|
'seo.noindex': 'Ask search engines not to index pages, and leave them out of sitemap.xml.',
|
|
167
167
|
contextMenu: 'Adds a "Copy page" menu to pages - copy as Markdown, or open the page in an AI assistant - and a Markdown copy of each page at its address + ".md".',
|
|
168
168
|
'contextMenu.openIn': 'Which AI assistants the menu offers. Default: all three.',
|
|
169
|
+
mcp: 'An MCP server at /mcp, so AI tools can search and read the docs. The build adds its index and a Cloudflare-ready _worker.js to dist/. Default true; false leaves them out.',
|
|
169
170
|
redirects: 'Redirects from old addresses. Each matches one exact path.',
|
|
170
171
|
'redirects[].source': 'The old path, like "/old-page".',
|
|
171
172
|
'redirects[].destination': 'Where to send it, like "/docs/new-page/".',
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
// llms.txt and its child files (/llms/*.md) - the pages an AI tool can
|
|
2
|
+
// discover, in the site's navigation structure. Shared by llms.txt.ts (the
|
|
3
|
+
// root file) and llms/[...path].md.ts (the child files a big site's llms.txt
|
|
4
|
+
// links to), so both render from the same tree in the same run. The
|
|
5
|
+
// structure and the splitting live in lib/llms.js.
|
|
6
|
+
import { getCollection, type CollectionEntry } from 'astro:content';
|
|
7
|
+
import fs from 'node:fs';
|
|
8
|
+
import path from 'node:path';
|
|
9
|
+
import { loadDocsConfig, normalizeEntryId, findAllPages, resolveSiteUrl, fileIdForEntry } from './config';
|
|
10
|
+
import { buildLlmsTree, renderLlmsFiles } from './llms.js';
|
|
11
|
+
import { navigationPageOrder } from './pages.js';
|
|
12
|
+
import { writedocsTempDir } from './writedocs-temp-dir.js';
|
|
13
|
+
|
|
14
|
+
type DocsEntry = CollectionEntry<'pages'> | CollectionEntry<'generatedDocs'>;
|
|
15
|
+
|
|
16
|
+
// Mirrors Mintlify's own llms.txt behavior: a page's frontmatter
|
|
17
|
+
// `description` is cut at the first line break (a multi-paragraph
|
|
18
|
+
// description would blow out a one-line list entry) and at 300 characters,
|
|
19
|
+
// so every entry stays scannable.
|
|
20
|
+
const DESCRIPTION_MAX_CHARS = 300;
|
|
21
|
+
function truncateDescription(description: string | undefined): string | undefined {
|
|
22
|
+
if (!description) return undefined;
|
|
23
|
+
const firstLine = description.split('\n')[0].trim();
|
|
24
|
+
if (!firstLine) return undefined;
|
|
25
|
+
return firstLine.length > DESCRIPTION_MAX_CHARS ? firstLine.slice(0, DESCRIPTION_MAX_CHARS).trimEnd() + '…' : firstLine;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
// Most characters one file holds - llms.txt or a child file. The llms.txt
|
|
29
|
+
// spec sets no limit ("small enough to fit in context"); this is Mintlify's,
|
|
30
|
+
// about 25,000 tokens. Past it, sections move into child files instead of
|
|
31
|
+
// pages being dropped.
|
|
32
|
+
export const LLMS_MAX_CHARS = 100_000;
|
|
33
|
+
|
|
34
|
+
/** The generated files as Map<path, text> - 'llms.txt', then 'llms/<...>.md'
|
|
35
|
+
* when the site needs them - or null when the project has its own
|
|
36
|
+
* llms.txt, which replaces all of it. */
|
|
37
|
+
export async function llmsIndexFiles(): Promise<Map<string, string> | null> {
|
|
38
|
+
const contentDir = process.env.WRITEDOCS_CONTENT_DIR || process.cwd();
|
|
39
|
+
const packageRoot = process.env.WRITEDOCS_PACKAGE_ROOT || process.cwd();
|
|
40
|
+
// A hand-authored llms.txt at the project root (next to writedocs.json)
|
|
41
|
+
// wins outright - same override convention Mintlify documents.
|
|
42
|
+
if (fs.existsSync(path.join(contentDir, 'llms.txt'))) return null;
|
|
43
|
+
|
|
44
|
+
const config = loadDocsConfig(contentDir);
|
|
45
|
+
const siteUrl = resolveSiteUrl(config); // absolute origin if `domain` is set, else null
|
|
46
|
+
|
|
47
|
+
const hasGeneratedDocs = fs.existsSync(path.join(writedocsTempDir(contentDir), 'generated-docs'));
|
|
48
|
+
const hasPages = findAllPages(contentDir).length > 0;
|
|
49
|
+
const [pagesEntries, generatedDocsEntries] = await Promise.all([
|
|
50
|
+
hasPages ? getCollection('pages') : Promise.resolve([]),
|
|
51
|
+
hasGeneratedDocs ? getCollection('generatedDocs') : Promise.resolve([]),
|
|
52
|
+
]);
|
|
53
|
+
const entries: DocsEntry[] = [...pagesEntries, ...generatedDocsEntries];
|
|
54
|
+
|
|
55
|
+
const pages = new Map<string, { title: string; href: string; description?: string; slug: string }>();
|
|
56
|
+
for (const entry of entries) {
|
|
57
|
+
// A page that opted out of search-engine indexing (`seo.noindex`, also
|
|
58
|
+
// left out of sitemap.xml) is left out here too, and a frontmatter `url`
|
|
59
|
+
// page (Mintlify's external link) has no content of its own.
|
|
60
|
+
if (entry.data.seo?.noindex || entry.data.url) continue;
|
|
61
|
+
const slug = normalizeEntryId(entry.id);
|
|
62
|
+
// The .md route when it exists ([...slug].md.ts - with `contextMenu`,
|
|
63
|
+
// and never for an OpenAPI page, which renders from the spec), else the
|
|
64
|
+
// HTML page.
|
|
65
|
+
const hasMarkdownRoute = config.contextMenu && !entry.data.openapi;
|
|
66
|
+
const href = (siteUrl ?? '') + (hasMarkdownRoute ? `/${slug}.md` : slug === 'index' ? '/' : `/${slug}/`);
|
|
67
|
+
let description = truncateDescription(entry.data.description);
|
|
68
|
+
// Mirrors Mintlify: an OpenAPI operation page's description gets its
|
|
69
|
+
// "METHOD /path" appended - the page itself renders from the spec.
|
|
70
|
+
if (entry.data.openapi) description = description ? `${description} (${entry.data.openapi})` : entry.data.openapi;
|
|
71
|
+
pages.set(fileIdForEntry(contentDir, packageRoot, entry), { title: entry.data.title, href, description, slug });
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
const listed = new Set(navigationPageOrder(config.navigation));
|
|
75
|
+
const unlisted = [...pages.entries()]
|
|
76
|
+
.filter(([id]) => !listed.has(id))
|
|
77
|
+
.sort((a, b) => a[1].slug.localeCompare(b[1].slug))
|
|
78
|
+
.map(([id]) => id);
|
|
79
|
+
const tree = buildLlmsTree(config.navigation, (id: string) => pages.get(id) ?? null, unlisted);
|
|
80
|
+
|
|
81
|
+
return renderLlmsFiles({
|
|
82
|
+
name: config.name,
|
|
83
|
+
description: config.description,
|
|
84
|
+
tree,
|
|
85
|
+
maxChars: LLMS_MAX_CHARS,
|
|
86
|
+
urlFor: (file: string) => `${siteUrl ?? ''}/${file}`,
|
|
87
|
+
});
|
|
88
|
+
}
|
package/src/lib/llms.js
ADDED
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
// Shared by the llms routes (lib/llms-index.ts, llms-full.txt.ts): the order
|
|
2
|
+
// pages appear in, and llms.txt's structure. Plain JavaScript, so it can be
|
|
3
|
+
// tested with plain Node.
|
|
4
|
+
import { navigationPageOrder } from './pages.js';
|
|
5
|
+
|
|
6
|
+
/** `items` ({ fileId, slug }) in navigation order - the order a reader meets
|
|
7
|
+
* the pages in the sidebar - then every page the navigation doesn't list,
|
|
8
|
+
* by URL. */
|
|
9
|
+
export function orderByNavigation(items, navigation) {
|
|
10
|
+
const rank = new Map(navigationPageOrder(navigation).map((id, i) => [id, i]));
|
|
11
|
+
const at = (item) => rank.get(item.fileId) ?? Infinity;
|
|
12
|
+
return [...items].sort((a, b) => at(a) - at(b) || a.slug.localeCompare(b.slug));
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
// --- llms.txt as a tree -------------------------------------------------
|
|
16
|
+
//
|
|
17
|
+
// llms.txt lists every page under headings that follow the navigation -
|
|
18
|
+
// products, versions, languages, tabs, dropdowns, groups. A site too big for
|
|
19
|
+
// one file (MAX_CHARS) keeps llms.txt as a directory: its largest sections
|
|
20
|
+
// move into child files under /llms/, linked from where they were, and a
|
|
21
|
+
// child too big is split the same way. No page is ever left out - the scheme
|
|
22
|
+
// Mintlify moved to from cutting the list off
|
|
23
|
+
// (https://www.mintlify.com/blog/scaling-llms-txt).
|
|
24
|
+
|
|
25
|
+
const CONTAINERS = [
|
|
26
|
+
['tabs', 'tab'],
|
|
27
|
+
['versions', 'version'],
|
|
28
|
+
['languages', 'language'],
|
|
29
|
+
['dropdowns', 'dropdown'],
|
|
30
|
+
['products', 'product'],
|
|
31
|
+
];
|
|
32
|
+
|
|
33
|
+
/** A Section: { label, items: [page], sections: [Section] }. `pageFor(id)`
|
|
34
|
+
* gives a listed page ({ title, href, description }) or null for one that
|
|
35
|
+
* isn't listed (noindex, an external `url`, missing). A page shows up once,
|
|
36
|
+
* where the navigation first lists it; `unlisted` pages (not in the
|
|
37
|
+
* navigation at all) go in a trailing "Other pages" section. */
|
|
38
|
+
export function buildLlmsTree(navigation, pageFor, unlisted = []) {
|
|
39
|
+
const seen = new Set();
|
|
40
|
+
const page = (id) => {
|
|
41
|
+
if (seen.has(id)) return null;
|
|
42
|
+
const found = pageFor(id);
|
|
43
|
+
if (!found) return null;
|
|
44
|
+
seen.add(id);
|
|
45
|
+
return found;
|
|
46
|
+
};
|
|
47
|
+
const prune = (section) => section.items.length > 0 || section.sections.length > 0;
|
|
48
|
+
|
|
49
|
+
function fromList(list) {
|
|
50
|
+
const node = { items: [], sections: [] };
|
|
51
|
+
for (const item of list ?? []) {
|
|
52
|
+
if (typeof item === 'string') {
|
|
53
|
+
const p = page(item);
|
|
54
|
+
if (p) node.items.push(p);
|
|
55
|
+
} else if (item && typeof item === 'object' && typeof item.group === 'string' && Array.isArray(item.pages)) {
|
|
56
|
+
const own = typeof item.page === 'string' ? page(item.page) : null;
|
|
57
|
+
const inner = fromList(item.pages);
|
|
58
|
+
const section = { label: item.group, items: own ? [own, ...inner.items] : inner.items, sections: inner.sections };
|
|
59
|
+
if (prune(section)) node.sections.push(section);
|
|
60
|
+
}
|
|
61
|
+
// A link leaf ({ href }) isn't a page of this site.
|
|
62
|
+
}
|
|
63
|
+
return node;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
function fromContainer(container) {
|
|
67
|
+
if (Array.isArray(container)) return fromList(container);
|
|
68
|
+
if (!container || typeof container !== 'object') return { items: [], sections: [] };
|
|
69
|
+
if (Array.isArray(container.pages)) return fromList(container.pages);
|
|
70
|
+
for (const [key, labelKey] of CONTAINERS) {
|
|
71
|
+
if (!Array.isArray(container[key])) continue;
|
|
72
|
+
const sections = container[key]
|
|
73
|
+
.filter((child) => child && typeof child === 'object' && !('href' in child))
|
|
74
|
+
.map((child) => {
|
|
75
|
+
const label = labelKey === 'version' || labelKey === 'language' ? child.label ?? child[labelKey] : child[labelKey];
|
|
76
|
+
return { label: String(label), ...fromContainer(child) };
|
|
77
|
+
})
|
|
78
|
+
.filter(prune);
|
|
79
|
+
return { items: [], sections };
|
|
80
|
+
}
|
|
81
|
+
return { items: [], sections: [] };
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
const root = fromContainer(navigation);
|
|
85
|
+
for (const dropdown of (!Array.isArray(navigation) && navigation?.global?.dropdowns) || []) {
|
|
86
|
+
if ('href' in dropdown) continue;
|
|
87
|
+
const section = { label: String(dropdown.dropdown), ...fromContainer(dropdown) };
|
|
88
|
+
if (prune(section)) root.sections.push(section);
|
|
89
|
+
}
|
|
90
|
+
const rest = unlisted.map((id) => page(id)).filter(Boolean);
|
|
91
|
+
if (rest.length) root.sections.push({ label: 'Other pages', items: rest, sections: [] });
|
|
92
|
+
return root;
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
const pageLine = (p) => `- [${p.title}](${p.href})${p.description ? `: ${p.description}` : ''}`;
|
|
96
|
+
const countPages = (s) => s.items.length + s.sections.reduce((n, c) => n + countPages(c), 0);
|
|
97
|
+
const heading = (depth, label) => `${'#'.repeat(Math.min(depth, 6))} ${label}`;
|
|
98
|
+
|
|
99
|
+
function slugify(label) {
|
|
100
|
+
return String(label).toLowerCase().replace(/[^a-z0-9]+/g, '-').replace(/^-+|-+$/g, '') || 'section';
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
/** The body of `node` as Markdown lines: its pages, then each sub-section -
|
|
104
|
+
* under its own heading, or, when it's in `split`, as a link to its file. */
|
|
105
|
+
function renderBody(node, depth, split) {
|
|
106
|
+
const lines = [];
|
|
107
|
+
if (node.items.length) lines.push(...node.items.map(pageLine), '');
|
|
108
|
+
for (const section of node.sections) {
|
|
109
|
+
lines.push(heading(depth, section.label), '');
|
|
110
|
+
const file = split.get(section);
|
|
111
|
+
if (file) {
|
|
112
|
+
const n = countPages(section);
|
|
113
|
+
lines.push(`- [${section.label}](${file.url}): ${n} page${n === 1 ? '' : 's'}, listed in their own file`, '');
|
|
114
|
+
} else {
|
|
115
|
+
lines.push(...renderBody(section, depth + 1, split));
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
return lines;
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
const render = (header, node, depth, split) => [...header, ...renderBody(node, depth, split)].join('\n').replace(/\n{3,}/g, '\n\n').trimEnd() + '\n';
|
|
122
|
+
|
|
123
|
+
/** Every section inside `node` still rendered inline (not split, and not
|
|
124
|
+
* inside a split one), with the characters it takes. */
|
|
125
|
+
function inlineSections(node, depth, split, out = []) {
|
|
126
|
+
for (const section of node.sections) {
|
|
127
|
+
if (split.has(section)) continue;
|
|
128
|
+
out.push({ section, size: render([], { items: [], sections: [section] }, depth, split).length });
|
|
129
|
+
inlineSections(section, depth + 1, split, out);
|
|
130
|
+
}
|
|
131
|
+
return out;
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/** A section whose own page list alone is too big becomes parts, each a
|
|
135
|
+
* sub-section small enough for a file of its own. */
|
|
136
|
+
function chunkItems(node, maxChars, label) {
|
|
137
|
+
const budget = Math.max(1000, maxChars - 2000);
|
|
138
|
+
const parts = [];
|
|
139
|
+
let current = [];
|
|
140
|
+
let size = 0;
|
|
141
|
+
for (const p of node.items) {
|
|
142
|
+
const n = pageLine(p).length + 1;
|
|
143
|
+
if (current.length && size + n > budget) {
|
|
144
|
+
parts.push(current);
|
|
145
|
+
current = [];
|
|
146
|
+
size = 0;
|
|
147
|
+
}
|
|
148
|
+
current.push(p);
|
|
149
|
+
size += n;
|
|
150
|
+
}
|
|
151
|
+
if (current.length) parts.push(current);
|
|
152
|
+
// One part is the list as it was - nothing smaller to split into (a single
|
|
153
|
+
// page line longer than the budget). Leave it, rather than loop.
|
|
154
|
+
if (parts.length < 2) return [];
|
|
155
|
+
const sections = parts.map((items, i) => ({ label: `${label} (part ${i + 1} of ${parts.length})`, items, sections: [] }));
|
|
156
|
+
node.sections = [...sections, ...node.sections];
|
|
157
|
+
node.items = [];
|
|
158
|
+
return sections;
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
/** llms.txt, and any child files it needs, as Map<path, text>: 'llms.txt',
|
|
162
|
+
* then e.g. 'llms/core-platform.md'. `urlFor(path)` gives a file's URL. */
|
|
163
|
+
export function renderLlmsFiles({ name, description, tree, maxChars, urlFor }) {
|
|
164
|
+
const files = new Map();
|
|
165
|
+
const taken = new Set(['llms.txt']);
|
|
166
|
+
const rootUrl = urlFor('llms.txt');
|
|
167
|
+
|
|
168
|
+
function emit(filePath, header, node, depth, trail) {
|
|
169
|
+
const split = new Map();
|
|
170
|
+
const splitOut = (section) => {
|
|
171
|
+
const dir = filePath === 'llms.txt' ? 'llms' : filePath.replace(/\.md$/, '');
|
|
172
|
+
let childPath = `${dir}/${slugify(section.label)}.md`;
|
|
173
|
+
for (let n = 2; taken.has(childPath); n++) childPath = `${dir}/${slugify(section.label)}-${n}.md`;
|
|
174
|
+
taken.add(childPath);
|
|
175
|
+
split.set(section, { path: childPath, url: urlFor(childPath) });
|
|
176
|
+
};
|
|
177
|
+
// A page list too long for this file on its own becomes parts - all of
|
|
178
|
+
// them separate files, so this one is a plain list of the parts.
|
|
179
|
+
if (render(header, { items: node.items, sections: [] }, depth, split).length > maxChars) {
|
|
180
|
+
chunkItems(node, maxChars, trail.length ? trail[trail.length - 1] : 'Pages').forEach(splitOut);
|
|
181
|
+
}
|
|
182
|
+
// Then the biggest sections move out, one at a time, until this fits.
|
|
183
|
+
while (render(header, node, depth, split).length > maxChars) {
|
|
184
|
+
const candidates = inlineSections(node, depth, split);
|
|
185
|
+
if (!candidates.length) break;
|
|
186
|
+
splitOut(candidates.reduce((a, b) => (b.size > a.size ? b : a)).section);
|
|
187
|
+
}
|
|
188
|
+
files.set(filePath, render(header, node, depth, split));
|
|
189
|
+
for (const [section, file] of split) {
|
|
190
|
+
const childTrail = [...trail, section.label];
|
|
191
|
+
const childHeader = [`# ${name}: ${childTrail.join(' > ')}`, '', `> Part of the ${name} docs index: ${rootUrl}`, ''];
|
|
192
|
+
emit(file.path, childHeader, section, 2, childTrail);
|
|
193
|
+
}
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
const header = [`# ${name}`, ''];
|
|
197
|
+
if (description) header.push(`> ${description}`, '');
|
|
198
|
+
emit('llms.txt', header, tree, 2, []);
|
|
199
|
+
return files;
|
|
200
|
+
}
|