docspack 0.1.1 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +11 -4
- package/dist/build.d.ts +21 -0
- package/dist/build.d.ts.map +1 -1
- package/dist/build.js +226 -29
- package/dist/build.js.map +1 -1
- package/dist/cli.js +95 -124
- package/dist/cli.js.map +1 -1
- package/dist/config.d.ts +7 -2
- package/dist/config.d.ts.map +1 -1
- package/dist/config.js +20 -1
- package/dist/config.js.map +1 -1
- package/dist/db.d.ts +9 -0
- package/dist/db.d.ts.map +1 -1
- package/dist/db.js +17 -2
- package/dist/db.js.map +1 -1
- package/dist/discovery.d.ts.map +1 -1
- package/dist/discovery.js +47 -25
- package/dist/discovery.js.map +1 -1
- package/dist/doctor.d.ts +6 -0
- package/dist/doctor.d.ts.map +1 -1
- package/dist/doctor.js +8 -4
- package/dist/doctor.js.map +1 -1
- package/dist/eval.d.ts +52 -0
- package/dist/eval.d.ts.map +1 -0
- package/dist/eval.js +101 -0
- package/dist/eval.js.map +1 -0
- package/dist/help.d.ts +31 -0
- package/dist/help.d.ts.map +1 -0
- package/dist/help.js +342 -0
- package/dist/help.js.map +1 -0
- package/dist/index.d.ts +4 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +4 -1
- package/dist/index.js.map +1 -1
- package/dist/preview.d.ts.map +1 -1
- package/dist/preview.js +4 -2
- package/dist/preview.js.map +1 -1
- package/dist/search.d.ts +12 -0
- package/dist/search.d.ts.map +1 -1
- package/dist/search.js +36 -6
- package/dist/search.js.map +1 -1
- package/dist/spec.d.ts +23 -1
- package/dist/spec.d.ts.map +1 -1
- package/dist/spec.js +47 -6
- package/dist/spec.js.map +1 -1
- package/dist/stopwords.d.ts +14 -0
- package/dist/stopwords.d.ts.map +1 -0
- package/dist/stopwords.js +121 -0
- package/dist/stopwords.js.map +1 -0
- package/dist/style.d.ts.map +1 -1
- package/dist/style.js +9 -2
- package/dist/style.js.map +1 -1
- package/dist/sync.js +1 -1
- package/dist/sync.js.map +1 -1
- package/dist/verify.d.ts +8 -2
- package/dist/verify.d.ts.map +1 -1
- package/dist/verify.js +48 -15
- package/dist/verify.js.map +1 -1
- package/package.json +1 -1
- package/src/build.ts +274 -29
- package/src/cli.ts +114 -124
- package/src/config.ts +35 -5
- package/src/db.ts +17 -2
- package/src/discovery.ts +44 -22
- package/src/doctor.ts +14 -5
- package/src/eval.ts +149 -0
- package/src/help.ts +382 -0
- package/src/index.ts +21 -0
- package/src/preview.ts +4 -1
- package/src/search.ts +50 -6
- package/src/spec.ts +69 -7
- package/src/stopwords.ts +120 -0
- package/src/style.ts +12 -2
- package/src/sync.ts +1 -1
- package/src/verify.ts +69 -19
package/src/build.ts
CHANGED
|
@@ -11,11 +11,14 @@ import { fetchableLinks, parseLlmsTxt } from "./llms-txt.js";
|
|
|
11
11
|
import {
|
|
12
12
|
CHUNKS_DIR,
|
|
13
13
|
type ChunkSpec,
|
|
14
|
+
type DocumentedLibrary,
|
|
14
15
|
estimateTokens,
|
|
15
16
|
LLMS_DIR,
|
|
16
17
|
MANIFEST_FILE,
|
|
18
|
+
parseDocumentedLibrary,
|
|
17
19
|
serializeManifest,
|
|
18
20
|
} from "./spec.js";
|
|
21
|
+
import { STOPWORDS } from "./stopwords.js";
|
|
19
22
|
|
|
20
23
|
export interface BuildOptions {
|
|
21
24
|
/** Directory the docs package is written to. */
|
|
@@ -30,6 +33,14 @@ export interface BuildOptions {
|
|
|
30
33
|
readonly source?: string;
|
|
31
34
|
readonly pages?: number;
|
|
32
35
|
readonly maxChunkTokens?: number;
|
|
36
|
+
/**
|
|
37
|
+
* Merge adjacent sections until a chunk reaches this size, never crossing `maxChunkTokens`.
|
|
38
|
+
* Off unless set: splitting at every heading is right for prose, and wrong only for generated
|
|
39
|
+
* reference, where a heading is a field name.
|
|
40
|
+
*/
|
|
41
|
+
readonly minChunkTokens?: number;
|
|
42
|
+
/** Libraries this package documents, each `name` or `name@version`. Recorded in the manifest. */
|
|
43
|
+
readonly documents?: readonly string[];
|
|
33
44
|
readonly http?: HttpClient;
|
|
34
45
|
readonly onProgress?: (message: string) => void;
|
|
35
46
|
}
|
|
@@ -76,7 +87,14 @@ export async function buildPackage(input: BuildOptions): Promise<BuildResult> {
|
|
|
76
87
|
}
|
|
77
88
|
|
|
78
89
|
const maxTokens = options.maxChunkTokens ?? DEFAULT_MAX_CHUNK_TOKENS;
|
|
79
|
-
const
|
|
90
|
+
const minTokens = options.minChunkTokens;
|
|
91
|
+
if (minTokens !== undefined && minTokens > maxTokens) {
|
|
92
|
+
throw new DocspackError(
|
|
93
|
+
`minChunkTokens (${minTokens}) is larger than maxChunkTokens (${maxTokens})`,
|
|
94
|
+
{ hint: "A chunk cannot be required to be bigger than it is allowed to be." },
|
|
95
|
+
);
|
|
96
|
+
}
|
|
97
|
+
const { chunks, files } = chunkDocuments(documents, maxTokens, minTokens);
|
|
80
98
|
|
|
81
99
|
const llmsDir = join(options.out, LLMS_DIR);
|
|
82
100
|
await rm(llmsDir, { recursive: true, force: true });
|
|
@@ -85,9 +103,15 @@ export async function buildPackage(input: BuildOptions): Promise<BuildResult> {
|
|
|
85
103
|
for (const file of files) {
|
|
86
104
|
await writeFile(join(llmsDir, file.path), file.contents, "utf8");
|
|
87
105
|
}
|
|
106
|
+
const libraries: DocumentedLibrary[] = (options.documents ?? []).map(parseDocumentedLibrary);
|
|
88
107
|
await writeFile(
|
|
89
108
|
join(llmsDir, MANIFEST_FILE),
|
|
90
|
-
serializeManifest({
|
|
109
|
+
serializeManifest({
|
|
110
|
+
name: identity.name,
|
|
111
|
+
version: identity.version,
|
|
112
|
+
...(libraries.length === 0 ? {} : { documents: libraries }),
|
|
113
|
+
chunks,
|
|
114
|
+
}),
|
|
91
115
|
"utf8",
|
|
92
116
|
);
|
|
93
117
|
await writeFile(join(options.out, "llms.txt"), renderLlmsTxt(identity.name, chunks), "utf8");
|
|
@@ -114,7 +138,7 @@ export async function buildPackage(input: BuildOptions): Promise<BuildResult> {
|
|
|
114
138
|
version: identity.version,
|
|
115
139
|
dir: options.out,
|
|
116
140
|
chunks: chunks.length,
|
|
117
|
-
tokens: chunks.reduce((total, chunk) => total + chunk.tokens, 0),
|
|
141
|
+
tokens: chunks.reduce((total, chunk) => total + (chunk.tokens ?? 0), 0),
|
|
118
142
|
warnings,
|
|
119
143
|
};
|
|
120
144
|
}
|
|
@@ -142,6 +166,12 @@ async function withConfig(options: BuildOptions): Promise<BuildOptions> {
|
|
|
142
166
|
...(options.maxChunkTokens === undefined && config.maxChunkTokens !== undefined
|
|
143
167
|
? { maxChunkTokens: config.maxChunkTokens }
|
|
144
168
|
: {}),
|
|
169
|
+
...(options.minChunkTokens === undefined && config.minChunkTokens !== undefined
|
|
170
|
+
? { minChunkTokens: config.minChunkTokens }
|
|
171
|
+
: {}),
|
|
172
|
+
...(options.documents === undefined && config.documents !== undefined
|
|
173
|
+
? { documents: config.documents }
|
|
174
|
+
: {}),
|
|
145
175
|
...(options.pages === undefined && config.pages !== undefined ? { pages: config.pages } : {}),
|
|
146
176
|
};
|
|
147
177
|
}
|
|
@@ -396,20 +426,24 @@ function renderOperation(
|
|
|
396
426
|
function chunkDocuments(
|
|
397
427
|
documents: readonly SourceDocument[],
|
|
398
428
|
maxTokens: number,
|
|
429
|
+
minTokens: number | undefined,
|
|
399
430
|
): { chunks: ChunkSpec[]; files: { path: string; contents: string }[] } {
|
|
400
431
|
const chunks: ChunkSpec[] = [];
|
|
401
432
|
const files: { path: string; contents: string }[] = [];
|
|
402
433
|
const taken = new Set<string>();
|
|
403
434
|
|
|
404
435
|
for (const document of documents) {
|
|
436
|
+
const text = collapseTablePadding(document.text);
|
|
405
437
|
const sections =
|
|
406
438
|
document.atomic === true
|
|
407
|
-
? [{ heading: document.title, body:
|
|
408
|
-
: splitMarkdown(
|
|
439
|
+
? [{ heading: document.title, body: text, level: 0 }]
|
|
440
|
+
: splitMarkdown(text, document.title, maxTokens, minTokens);
|
|
409
441
|
|
|
410
442
|
for (const section of sections) {
|
|
411
|
-
const
|
|
412
|
-
const
|
|
443
|
+
const directives = readDirectives(section.body);
|
|
444
|
+
const body = withoutDuplicateHeading(directives.body, section.heading);
|
|
445
|
+
const id = uniqueId(chunkSlug(document.title, section.heading), taken);
|
|
446
|
+
const contents = `# ${section.heading}\n\n<!-- docspack: from ${document.origin} -->\n\n${body}\n`;
|
|
413
447
|
const file = `${CHUNKS_DIR}/${id}.md`;
|
|
414
448
|
|
|
415
449
|
files.push({ path: file, contents });
|
|
@@ -417,8 +451,18 @@ function chunkDocuments(
|
|
|
417
451
|
id,
|
|
418
452
|
file,
|
|
419
453
|
tokens: estimateTokens(contents),
|
|
420
|
-
tags:
|
|
421
|
-
|
|
454
|
+
tags: chunkTags(directives.tags, [
|
|
455
|
+
document.title,
|
|
456
|
+
...(section.headings ?? [section.heading]),
|
|
457
|
+
...(document.tags ?? []),
|
|
458
|
+
]),
|
|
459
|
+
entities: [
|
|
460
|
+
...new Set([
|
|
461
|
+
...(document.entities ?? []),
|
|
462
|
+
...directives.entities,
|
|
463
|
+
...extractEntities(body),
|
|
464
|
+
]),
|
|
465
|
+
],
|
|
422
466
|
});
|
|
423
467
|
}
|
|
424
468
|
}
|
|
@@ -426,17 +470,35 @@ function chunkDocuments(
|
|
|
426
470
|
return { chunks, files };
|
|
427
471
|
}
|
|
428
472
|
|
|
473
|
+
/** One piece of a document, before it becomes a chunk. */
|
|
474
|
+
interface Section {
|
|
475
|
+
readonly heading: string;
|
|
476
|
+
readonly body: string;
|
|
477
|
+
/** Depth of the heading in the source, or 0 when it was inherited rather than written. */
|
|
478
|
+
readonly level: number;
|
|
479
|
+
/**
|
|
480
|
+
* Headings to index as tags, when they are not just `heading`. A packed chunk answers for every
|
|
481
|
+
* section in it, so all of their headings are search terms for it.
|
|
482
|
+
*/
|
|
483
|
+
readonly headings?: readonly string[];
|
|
484
|
+
}
|
|
485
|
+
|
|
429
486
|
/**
|
|
430
487
|
* Splits Markdown at `##` headings, then `###`, then paragraphs, until every piece fits the token
|
|
431
488
|
* budget. Retrieval works best when a chunk answers one question.
|
|
489
|
+
*
|
|
490
|
+
* With `minTokens` set, adjacent pieces are then packed back together up to the budget. Every
|
|
491
|
+
* API generator emits pages whose headings are field names — `Category` is one line, `Sizes` is
|
|
492
|
+
* three — and one chunk per heading there is hundreds of chunks too small to answer anything.
|
|
432
493
|
*/
|
|
433
494
|
function splitMarkdown(
|
|
434
495
|
text: string,
|
|
435
496
|
title: string,
|
|
436
497
|
maxTokens: number,
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
const
|
|
498
|
+
minTokens: number | undefined,
|
|
499
|
+
): Section[] {
|
|
500
|
+
const sections = splitAtLevel(text, 2, title, 0);
|
|
501
|
+
const result: Section[] = [];
|
|
440
502
|
|
|
441
503
|
for (const section of sections) {
|
|
442
504
|
if (estimateTokens(section.body) <= maxTokens) {
|
|
@@ -444,7 +506,7 @@ function splitMarkdown(
|
|
|
444
506
|
continue;
|
|
445
507
|
}
|
|
446
508
|
|
|
447
|
-
const deeper = splitAtLevel(section.body, 3, section.heading);
|
|
509
|
+
const deeper = splitAtLevel(section.body, 3, section.heading, section.level);
|
|
448
510
|
for (const part of deeper) {
|
|
449
511
|
if (estimateTokens(part.body) <= maxTokens) {
|
|
450
512
|
if (part.body.trim().length > 0) result.push(part);
|
|
@@ -454,22 +516,26 @@ function splitMarkdown(
|
|
|
454
516
|
...splitParagraphs(part.body, maxTokens).map((body, index) => ({
|
|
455
517
|
heading: index === 0 ? part.heading : `${part.heading} (${index + 1})`,
|
|
456
518
|
body,
|
|
519
|
+
// A continuation carries no heading of its own, so packing must not re-emit one.
|
|
520
|
+
level: index === 0 ? part.level : 0,
|
|
457
521
|
})),
|
|
458
522
|
);
|
|
459
523
|
}
|
|
460
524
|
}
|
|
461
525
|
|
|
462
|
-
|
|
526
|
+
const pieces = result.length > 0 ? result : [{ heading: title, body: text, level: 0 }];
|
|
527
|
+
return minTokens === undefined ? pieces : packSections(pieces, title, minTokens, maxTokens);
|
|
463
528
|
}
|
|
464
529
|
|
|
465
530
|
function splitAtLevel(
|
|
466
531
|
text: string,
|
|
467
532
|
level: number,
|
|
468
533
|
fallbackHeading: string,
|
|
469
|
-
|
|
534
|
+
fallbackLevel: number,
|
|
535
|
+
): Section[] {
|
|
470
536
|
const marker = new RegExp(`^#{${level}}\\s+(.+?)\\s*$`);
|
|
471
|
-
const sections: { heading: string; body: string[] }[] = [];
|
|
472
|
-
let current = { heading: fallbackHeading, body: [] as string[] };
|
|
537
|
+
const sections: { heading: string; body: string[]; level: number }[] = [];
|
|
538
|
+
let current = { heading: fallbackHeading, body: [] as string[], level: fallbackLevel };
|
|
473
539
|
let inFence = false;
|
|
474
540
|
|
|
475
541
|
for (const line of text.split("\n")) {
|
|
@@ -478,7 +544,7 @@ function splitAtLevel(
|
|
|
478
544
|
const match = inFence ? null : marker.exec(line);
|
|
479
545
|
if (match?.[1] !== undefined) {
|
|
480
546
|
if (current.body.join("\n").trim().length > 0) sections.push(current);
|
|
481
|
-
current = { heading: match[1], body: [] };
|
|
547
|
+
current = { heading: match[1], body: [], level };
|
|
482
548
|
continue;
|
|
483
549
|
}
|
|
484
550
|
current.body.push(line);
|
|
@@ -487,10 +553,57 @@ function splitAtLevel(
|
|
|
487
553
|
|
|
488
554
|
return sections.map((section) => ({
|
|
489
555
|
heading: section.heading,
|
|
556
|
+
level: section.level,
|
|
490
557
|
body: section.body.join("\n").trim(),
|
|
491
558
|
}));
|
|
492
559
|
}
|
|
493
560
|
|
|
561
|
+
/**
|
|
562
|
+
* Merges adjacent sections until a group reaches `minTokens`, never crossing `maxTokens`. A
|
|
563
|
+
* merged group is headed by the document title and keeps each section's own heading inside it,
|
|
564
|
+
* so a whole component lands in one chunk without losing the structure a reader needs.
|
|
565
|
+
*/
|
|
566
|
+
function packSections(
|
|
567
|
+
sections: readonly Section[],
|
|
568
|
+
title: string,
|
|
569
|
+
minTokens: number,
|
|
570
|
+
maxTokens: number,
|
|
571
|
+
): Section[] {
|
|
572
|
+
const packed: Section[] = [];
|
|
573
|
+
let group: Section[] = [];
|
|
574
|
+
let tokens = 0;
|
|
575
|
+
|
|
576
|
+
const flush = (): void => {
|
|
577
|
+
if (group.length === 1 && group[0] !== undefined) packed.push(group[0]);
|
|
578
|
+
else if (group.length > 1) {
|
|
579
|
+
packed.push({
|
|
580
|
+
heading: title,
|
|
581
|
+
level: 0,
|
|
582
|
+
body: group.map(withHeading).join("\n\n"),
|
|
583
|
+
headings: group.map((section) => section.heading),
|
|
584
|
+
});
|
|
585
|
+
}
|
|
586
|
+
group = [];
|
|
587
|
+
tokens = 0;
|
|
588
|
+
};
|
|
589
|
+
|
|
590
|
+
for (const section of sections) {
|
|
591
|
+
const size = estimateTokens(withHeading(section));
|
|
592
|
+
if (group.length > 0 && (tokens >= minTokens || tokens + size > maxTokens)) flush();
|
|
593
|
+
group.push(section);
|
|
594
|
+
tokens += size;
|
|
595
|
+
}
|
|
596
|
+
flush();
|
|
597
|
+
|
|
598
|
+
return packed;
|
|
599
|
+
}
|
|
600
|
+
|
|
601
|
+
/** A section as it appeared in the source, heading included when it had one. */
|
|
602
|
+
function withHeading(section: Section): string {
|
|
603
|
+
if (section.level === 0) return section.body;
|
|
604
|
+
return `${"#".repeat(section.level)} ${section.heading}\n\n${section.body}`;
|
|
605
|
+
}
|
|
606
|
+
|
|
494
607
|
function splitParagraphs(text: string, maxTokens: number): string[] {
|
|
495
608
|
const parts: string[] = [];
|
|
496
609
|
let buffer: string[] = [];
|
|
@@ -508,13 +621,90 @@ function splitParagraphs(text: string, maxTokens: number): string[] {
|
|
|
508
621
|
return parts;
|
|
509
622
|
}
|
|
510
623
|
|
|
511
|
-
/**
|
|
624
|
+
/** A member access in inline code, e.g. `Stripe.setApiKey`. */
|
|
625
|
+
const DOTTED_ENTITY = /`([A-Za-z_$][\w$]*(?:\.[A-Za-z_$][\w$]*)+)\(?\)?`/g;
|
|
626
|
+
/**
|
|
627
|
+
* An inline-code token that is an identifier on its own: `TwoColumn`, `useSlide`,
|
|
628
|
+
* `text-accent`. Component libraries name most of their surface without a dot, so requiring
|
|
629
|
+
* one made `verify` blind to them. Two words are required — a single capitalised word in a
|
|
630
|
+
* code span is as often a value as an API name.
|
|
631
|
+
*/
|
|
632
|
+
const BARE_ENTITY =
|
|
633
|
+
/`([A-Za-z][a-z0-9]*(?:[A-Z][A-Za-z0-9]*)+|[a-z][a-z0-9]*(?:-[a-z0-9]+)+)\(?\)?`/g;
|
|
634
|
+
|
|
635
|
+
const MAX_ENTITIES = 20;
|
|
636
|
+
|
|
637
|
+
/**
|
|
638
|
+
* Pulls the identifiers a section names out of its inline code.
|
|
639
|
+
*
|
|
640
|
+
* Member accesses and camel or Pascal names come first. A reference page often enumerates
|
|
641
|
+
* dozens of hyphenated variants — `token-0`, `token-1` — and letting those fill the cap would
|
|
642
|
+
* drop the component the page is actually about.
|
|
643
|
+
*/
|
|
512
644
|
function extractEntities(body: string): string[] {
|
|
513
|
-
const
|
|
514
|
-
|
|
515
|
-
|
|
645
|
+
const dotted = matchAll(body, DOTTED_ENTITY);
|
|
646
|
+
const bare = matchAll(body, BARE_ENTITY);
|
|
647
|
+
const ordered = [
|
|
648
|
+
...dotted,
|
|
649
|
+
...bare.filter((name) => !name.includes("-")),
|
|
650
|
+
...bare.filter((name) => name.includes("-")),
|
|
651
|
+
];
|
|
652
|
+
return [...new Set(ordered)].slice(0, MAX_ENTITIES);
|
|
653
|
+
}
|
|
654
|
+
|
|
655
|
+
function matchAll(text: string, pattern: RegExp): string[] {
|
|
656
|
+
const found: string[] = [];
|
|
657
|
+
for (const match of text.matchAll(pattern)) {
|
|
658
|
+
if (match[1] !== undefined) found.push(match[1]);
|
|
516
659
|
}
|
|
517
|
-
return
|
|
660
|
+
return found;
|
|
661
|
+
}
|
|
662
|
+
|
|
663
|
+
/** One authored override under a heading: `<!-- docspack: tags=grid,columns -->`. */
|
|
664
|
+
const DIRECTIVE = /^[ \t]*<!--[ \t]*docspack:[ \t]*(tags|entities)[ \t]*=([^>]*?)-->[ \t]*$/gim;
|
|
665
|
+
|
|
666
|
+
/**
|
|
667
|
+
* Reads per-section directives and removes them from the body. Front matter belongs to a whole
|
|
668
|
+
* document, so a page of eighty sections cannot aim any one of them; this is that lever.
|
|
669
|
+
*/
|
|
670
|
+
function readDirectives(body: string): { body: string; tags: string[]; entities: string[] } {
|
|
671
|
+
const tags: string[] = [];
|
|
672
|
+
const entities: string[] = [];
|
|
673
|
+
|
|
674
|
+
const text = body.replace(DIRECTIVE, (_line, key: string, value: string) => {
|
|
675
|
+
const values = value
|
|
676
|
+
.split(",")
|
|
677
|
+
.map((item) => item.trim())
|
|
678
|
+
.filter((item) => item.length > 0);
|
|
679
|
+
(key.toLowerCase() === "tags" ? tags : entities).push(...values);
|
|
680
|
+
return "";
|
|
681
|
+
});
|
|
682
|
+
|
|
683
|
+
return { body: text.replace(/\n{3,}/g, "\n\n").trim(), tags, entities };
|
|
684
|
+
}
|
|
685
|
+
|
|
686
|
+
/**
|
|
687
|
+
* Drops a source `# Title` that repeats the heading being written above it. A single-`H1`
|
|
688
|
+
* document has no `##` to split at, so its heading falls back to the title and the body would
|
|
689
|
+
* otherwise open with the same line twice.
|
|
690
|
+
*/
|
|
691
|
+
function withoutDuplicateHeading(body: string, heading: string): string {
|
|
692
|
+
const [first = "", ...rest] = body.split("\n");
|
|
693
|
+
const match = /^#\s+(.+?)\s*$/.exec(first.trim());
|
|
694
|
+
if (match?.[1] === undefined) return body;
|
|
695
|
+
if (match[1].trim().toLowerCase() !== heading.trim().toLowerCase()) return body;
|
|
696
|
+
return rest.join("\n").trim();
|
|
697
|
+
}
|
|
698
|
+
|
|
699
|
+
/**
|
|
700
|
+
* The chunk id for a section. When the section heading is the document title — every
|
|
701
|
+
* single-`H1` page — concatenating the two stutters: `colours-colours`.
|
|
702
|
+
*/
|
|
703
|
+
function chunkSlug(title: string, heading: string): string {
|
|
704
|
+
const titleSlug = slugify(title, "chunk");
|
|
705
|
+
return slugify(heading, "chunk") === titleSlug
|
|
706
|
+
? titleSlug
|
|
707
|
+
: slugify(`${title}-${heading}`, "chunk");
|
|
518
708
|
}
|
|
519
709
|
|
|
520
710
|
/** Reduces a heading to a chunk id: lowercase, hyphen-separated, no path separators. */
|
|
@@ -529,14 +719,69 @@ function slugify(input: string, fallback: string): string {
|
|
|
529
719
|
return slug.length > 0 ? slug : fallback;
|
|
530
720
|
}
|
|
531
721
|
|
|
532
|
-
|
|
533
|
-
|
|
722
|
+
const MAX_TAGS = 24;
|
|
723
|
+
|
|
724
|
+
/**
|
|
725
|
+
* The words indexed alongside a chunk.
|
|
726
|
+
*
|
|
727
|
+
* Directive tags are taken as the author wrote them — those are deliberate, and one of them
|
|
728
|
+
* being a function word is the author's decision. Words lifted out of a title or a heading are
|
|
729
|
+
* filtered: nobody chose them as search terms, and the tag field is weighted 3× in the ranking,
|
|
730
|
+
* which is what made "How to use it" the answer to every question beginning "how do I".
|
|
731
|
+
*/
|
|
732
|
+
function chunkTags(authored: readonly string[], derived: readonly string[]): string[] {
|
|
733
|
+
const tags = new Set(words(authored));
|
|
734
|
+
for (const word of words(derived)) {
|
|
735
|
+
if (!STOPWORDS.has(word)) tags.add(word);
|
|
736
|
+
}
|
|
737
|
+
return [...tags].slice(0, MAX_TAGS);
|
|
738
|
+
}
|
|
739
|
+
|
|
740
|
+
function words(inputs: readonly string[]): string[] {
|
|
741
|
+
const found: string[] = [];
|
|
534
742
|
for (const input of inputs) {
|
|
535
|
-
|
|
536
|
-
words.add(word);
|
|
537
|
-
}
|
|
743
|
+
found.push(...(input.toLowerCase().match(/[\p{L}\p{N}]{2,}/gu) ?? []));
|
|
538
744
|
}
|
|
539
|
-
return
|
|
745
|
+
return found;
|
|
746
|
+
}
|
|
747
|
+
|
|
748
|
+
/** A row of a Markdown table: `| Name | Type |`, and the `|---|---:|` rule under it. */
|
|
749
|
+
const TABLE_ROW = /^\|.*\|/;
|
|
750
|
+
const TABLE_RULE = /^[|\-: \t]+$/;
|
|
751
|
+
|
|
752
|
+
/**
|
|
753
|
+
* Collapses the alignment padding out of Markdown tables.
|
|
754
|
+
*
|
|
755
|
+
* Every API generator pads table cells into columns. That padding is whitespace to a reader and
|
|
756
|
+
* to a renderer, but tokens are estimated from characters, so a generated props table is
|
|
757
|
+
* measured at several times its real weight — which spends the response budget, trips
|
|
758
|
+
* `chunk-too-large`, and forces splits that leave fragments competing for the same query. One
|
|
759
|
+
* chart's props table measured 4,094 tokens padded and 1,245 collapsed.
|
|
760
|
+
*
|
|
761
|
+
* Runs inside a code fence are left alone; a fenced table is example text, and its alignment is
|
|
762
|
+
* the thing being shown.
|
|
763
|
+
*/
|
|
764
|
+
export function collapseTablePadding(text: string): string {
|
|
765
|
+
let inFence = false;
|
|
766
|
+
|
|
767
|
+
return text
|
|
768
|
+
.split("\n")
|
|
769
|
+
.map((line) => {
|
|
770
|
+
if (/^\s*(?:```|~~~)/.test(line)) {
|
|
771
|
+
inFence = !inFence;
|
|
772
|
+
return line;
|
|
773
|
+
}
|
|
774
|
+
if (inFence) return line;
|
|
775
|
+
|
|
776
|
+
const indent = /^[ \t]*/.exec(line)?.[0] ?? "";
|
|
777
|
+
const rest = line.slice(indent.length);
|
|
778
|
+
// Four spaces or more is an indented code block, whatever the line looks like.
|
|
779
|
+
if (indent.length >= 4 || !TABLE_ROW.test(rest)) return line;
|
|
780
|
+
|
|
781
|
+
const collapsed = rest.replace(/[ \t]{2,}/g, " ");
|
|
782
|
+
return indent + (TABLE_RULE.test(rest) ? collapsed.replace(/-{4,}/g, "---") : collapsed);
|
|
783
|
+
})
|
|
784
|
+
.join("\n");
|
|
540
785
|
}
|
|
541
786
|
|
|
542
787
|
function uniqueId(base: string, taken: Set<string>): string {
|