docspack 0.2.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +9 -3
- package/dist/build.d.ts +21 -0
- package/dist/build.d.ts.map +1 -1
- package/dist/build.js +194 -31
- package/dist/build.js.map +1 -1
- package/dist/cli.js +74 -131
- package/dist/cli.js.map +1 -1
- package/dist/config.d.ts +7 -2
- package/dist/config.d.ts.map +1 -1
- package/dist/config.js +20 -1
- package/dist/config.js.map +1 -1
- package/dist/db.d.ts +8 -4
- package/dist/db.d.ts.map +1 -1
- package/dist/db.js +13 -7
- package/dist/db.js.map +1 -1
- package/dist/doctor.d.ts.map +1 -1
- package/dist/doctor.js +47 -4
- package/dist/doctor.js.map +1 -1
- package/dist/document.d.ts +2 -0
- package/dist/document.d.ts.map +1 -1
- package/dist/document.js +7 -3
- package/dist/document.js.map +1 -1
- package/dist/eval.d.ts +52 -0
- package/dist/eval.d.ts.map +1 -0
- package/dist/eval.js +101 -0
- package/dist/eval.js.map +1 -0
- package/dist/help.d.ts +31 -0
- package/dist/help.d.ts.map +1 -0
- package/dist/help.js +342 -0
- package/dist/help.js.map +1 -0
- package/dist/index.d.ts +4 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +4 -1
- package/dist/index.js.map +1 -1
- package/dist/preview.d.ts.map +1 -1
- package/dist/preview.js +4 -2
- package/dist/preview.js.map +1 -1
- package/dist/search.d.ts +7 -0
- package/dist/search.d.ts.map +1 -1
- package/dist/search.js +33 -3
- package/dist/search.js.map +1 -1
- package/dist/spec.d.ts +27 -1
- package/dist/spec.d.ts.map +1 -1
- package/dist/spec.js +53 -5
- package/dist/spec.js.map +1 -1
- package/dist/stopwords.d.ts +14 -0
- package/dist/stopwords.d.ts.map +1 -0
- package/dist/stopwords.js +121 -0
- package/dist/stopwords.js.map +1 -0
- package/dist/sync.js +1 -1
- package/dist/sync.js.map +1 -1
- package/dist/verify.d.ts +8 -2
- package/dist/verify.d.ts.map +1 -1
- package/dist/verify.js +84 -23
- package/dist/verify.js.map +1 -1
- package/package.json +1 -1
- package/src/build.ts +261 -33
- package/src/cli.ts +91 -138
- package/src/config.ts +35 -5
- package/src/db.ts +13 -7
- package/src/doctor.ts +49 -5
- package/src/document.ts +13 -4
- package/src/eval.ts +149 -0
- package/src/help.ts +382 -0
- package/src/index.ts +20 -0
- package/src/preview.ts +4 -1
- package/src/search.ts +41 -3
- package/src/spec.ts +84 -6
- package/src/stopwords.ts +120 -0
- package/src/sync.ts +1 -1
- package/src/verify.ts +108 -28
package/src/build.ts
CHANGED
|
@@ -11,11 +11,16 @@ import { fetchableLinks, parseLlmsTxt } from "./llms-txt.js";
|
|
|
11
11
|
import {
|
|
12
12
|
CHUNKS_DIR,
|
|
13
13
|
type ChunkSpec,
|
|
14
|
+
type DocumentedLibrary,
|
|
14
15
|
estimateTokens,
|
|
15
16
|
LLMS_DIR,
|
|
16
17
|
MANIFEST_FILE,
|
|
18
|
+
type PackageManifest,
|
|
19
|
+
parseDocumentedLibrary,
|
|
20
|
+
parseManifest,
|
|
17
21
|
serializeManifest,
|
|
18
22
|
} from "./spec.js";
|
|
23
|
+
import { STOPWORDS } from "./stopwords.js";
|
|
19
24
|
|
|
20
25
|
export interface BuildOptions {
|
|
21
26
|
/** Directory the docs package is written to. */
|
|
@@ -30,6 +35,14 @@ export interface BuildOptions {
|
|
|
30
35
|
readonly source?: string;
|
|
31
36
|
readonly pages?: number;
|
|
32
37
|
readonly maxChunkTokens?: number;
|
|
38
|
+
/**
|
|
39
|
+
* Merge adjacent sections until a chunk reaches this size, never crossing `maxChunkTokens`.
|
|
40
|
+
* Off unless set: splitting at every heading is right for prose, and wrong only for generated
|
|
41
|
+
* reference, where a heading is a field name.
|
|
42
|
+
*/
|
|
43
|
+
readonly minChunkTokens?: number;
|
|
44
|
+
/** Libraries this package documents, each `name` or `name@version`. Recorded in the manifest. */
|
|
45
|
+
readonly documents?: readonly string[];
|
|
33
46
|
readonly http?: HttpClient;
|
|
34
47
|
readonly onProgress?: (message: string) => void;
|
|
35
48
|
}
|
|
@@ -53,6 +66,8 @@ interface SourceDocument {
|
|
|
53
66
|
readonly atomic?: boolean;
|
|
54
67
|
readonly tags?: readonly string[];
|
|
55
68
|
readonly entities?: readonly string[];
|
|
69
|
+
/** Libraries this document describes, when that is narrower than the package's `documents`. */
|
|
70
|
+
readonly documents?: readonly string[];
|
|
56
71
|
}
|
|
57
72
|
|
|
58
73
|
const DEFAULT_MAX_CHUNK_TOKENS = 800;
|
|
@@ -76,18 +91,34 @@ export async function buildPackage(input: BuildOptions): Promise<BuildResult> {
|
|
|
76
91
|
}
|
|
77
92
|
|
|
78
93
|
const maxTokens = options.maxChunkTokens ?? DEFAULT_MAX_CHUNK_TOKENS;
|
|
79
|
-
const
|
|
94
|
+
const minTokens = options.minChunkTokens;
|
|
95
|
+
if (minTokens !== undefined && minTokens > maxTokens) {
|
|
96
|
+
throw new DocspackError(
|
|
97
|
+
`minChunkTokens (${minTokens}) is larger than maxChunkTokens (${maxTokens})`,
|
|
98
|
+
{ hint: "A chunk cannot be required to be bigger than it is allowed to be." },
|
|
99
|
+
);
|
|
100
|
+
}
|
|
101
|
+
const { chunks, files, collisions } = chunkDocuments(documents, maxTokens, minTokens);
|
|
80
102
|
|
|
81
103
|
const llmsDir = join(options.out, LLMS_DIR);
|
|
104
|
+
// Read before the payload is removed: a chunk id is what `feedback add --chunk` pins and what
|
|
105
|
+
// an answer is headed with, so a rebuild that renames one silently orphans both.
|
|
106
|
+
warnings.push(...idWarnings(collisions, await departedChunkIds(llmsDir, chunks)));
|
|
82
107
|
await rm(llmsDir, { recursive: true, force: true });
|
|
83
108
|
await mkdir(join(llmsDir, CHUNKS_DIR), { recursive: true });
|
|
84
109
|
|
|
85
110
|
for (const file of files) {
|
|
86
111
|
await writeFile(join(llmsDir, file.path), file.contents, "utf8");
|
|
87
112
|
}
|
|
113
|
+
const libraries: DocumentedLibrary[] = (options.documents ?? []).map(parseDocumentedLibrary);
|
|
88
114
|
await writeFile(
|
|
89
115
|
join(llmsDir, MANIFEST_FILE),
|
|
90
|
-
serializeManifest({
|
|
116
|
+
serializeManifest({
|
|
117
|
+
name: identity.name,
|
|
118
|
+
version: identity.version,
|
|
119
|
+
...(libraries.length === 0 ? {} : { documents: libraries }),
|
|
120
|
+
chunks,
|
|
121
|
+
}),
|
|
91
122
|
"utf8",
|
|
92
123
|
);
|
|
93
124
|
await writeFile(join(options.out, "llms.txt"), renderLlmsTxt(identity.name, chunks), "utf8");
|
|
@@ -114,7 +145,7 @@ export async function buildPackage(input: BuildOptions): Promise<BuildResult> {
|
|
|
114
145
|
version: identity.version,
|
|
115
146
|
dir: options.out,
|
|
116
147
|
chunks: chunks.length,
|
|
117
|
-
tokens: chunks.reduce((total, chunk) => total + chunk.tokens, 0),
|
|
148
|
+
tokens: chunks.reduce((total, chunk) => total + (chunk.tokens ?? 0), 0),
|
|
118
149
|
warnings,
|
|
119
150
|
};
|
|
120
151
|
}
|
|
@@ -142,6 +173,12 @@ async function withConfig(options: BuildOptions): Promise<BuildOptions> {
|
|
|
142
173
|
...(options.maxChunkTokens === undefined && config.maxChunkTokens !== undefined
|
|
143
174
|
? { maxChunkTokens: config.maxChunkTokens }
|
|
144
175
|
: {}),
|
|
176
|
+
...(options.minChunkTokens === undefined && config.minChunkTokens !== undefined
|
|
177
|
+
? { minChunkTokens: config.minChunkTokens }
|
|
178
|
+
: {}),
|
|
179
|
+
...(options.documents === undefined && config.documents !== undefined
|
|
180
|
+
? { documents: config.documents }
|
|
181
|
+
: {}),
|
|
145
182
|
...(options.pages === undefined && config.pages !== undefined ? { pages: config.pages } : {}),
|
|
146
183
|
};
|
|
147
184
|
}
|
|
@@ -218,6 +255,7 @@ async function collectFromDirectory(dir: string): Promise<SourceDocument[]> {
|
|
|
218
255
|
origin: relativePath,
|
|
219
256
|
text: cleaned.text,
|
|
220
257
|
...(cleaned.tags.length === 0 ? {} : { tags: cleaned.tags }),
|
|
258
|
+
...(cleaned.documents.length === 0 ? {} : { documents: cleaned.documents }),
|
|
221
259
|
});
|
|
222
260
|
}
|
|
223
261
|
return documents;
|
|
@@ -396,21 +434,31 @@ function renderOperation(
|
|
|
396
434
|
function chunkDocuments(
|
|
397
435
|
documents: readonly SourceDocument[],
|
|
398
436
|
maxTokens: number,
|
|
399
|
-
|
|
437
|
+
minTokens: number | undefined,
|
|
438
|
+
): {
|
|
439
|
+
chunks: ChunkSpec[];
|
|
440
|
+
files: { path: string; contents: string }[];
|
|
441
|
+
collisions: string[];
|
|
442
|
+
} {
|
|
400
443
|
const chunks: ChunkSpec[] = [];
|
|
401
444
|
const files: { path: string; contents: string }[] = [];
|
|
445
|
+
const collisions: string[] = [];
|
|
402
446
|
const taken = new Set<string>();
|
|
403
447
|
|
|
404
448
|
for (const document of documents) {
|
|
449
|
+
const text = collapseTablePadding(document.text);
|
|
405
450
|
const sections =
|
|
406
451
|
document.atomic === true
|
|
407
|
-
? [{ heading: document.title, body:
|
|
408
|
-
: splitMarkdown(
|
|
452
|
+
? [{ heading: document.title, body: text, level: 0 }]
|
|
453
|
+
: splitMarkdown(text, document.title, maxTokens, minTokens);
|
|
409
454
|
|
|
410
455
|
for (const section of sections) {
|
|
411
456
|
const directives = readDirectives(section.body);
|
|
412
457
|
const body = withoutDuplicateHeading(directives.body, section.heading);
|
|
413
|
-
const
|
|
458
|
+
const derived = chunkSlug(document.title, section.heading);
|
|
459
|
+
const id = uniqueId(derived, taken);
|
|
460
|
+
if (id !== derived) collisions.push(derived);
|
|
461
|
+
const libraries = directives.documents.length > 0 ? directives.documents : document.documents;
|
|
414
462
|
const contents = `# ${section.heading}\n\n<!-- docspack: from ${document.origin} -->\n\n${body}\n`;
|
|
415
463
|
const file = `${CHUNKS_DIR}/${id}.md`;
|
|
416
464
|
|
|
@@ -419,11 +467,9 @@ function chunkDocuments(
|
|
|
419
467
|
id,
|
|
420
468
|
file,
|
|
421
469
|
tokens: estimateTokens(contents),
|
|
422
|
-
|
|
423
|
-
tags: uniqueWords([
|
|
424
|
-
...directives.tags,
|
|
470
|
+
tags: chunkTags(directives.tags, [
|
|
425
471
|
document.title,
|
|
426
|
-
section.heading,
|
|
472
|
+
...(section.headings ?? [section.heading]),
|
|
427
473
|
...(document.tags ?? []),
|
|
428
474
|
]),
|
|
429
475
|
entities: [
|
|
@@ -433,24 +479,90 @@ function chunkDocuments(
|
|
|
433
479
|
...extractEntities(body),
|
|
434
480
|
]),
|
|
435
481
|
],
|
|
482
|
+
...(libraries === undefined || libraries.length === 0
|
|
483
|
+
? {}
|
|
484
|
+
: { documents: libraries.map(parseDocumentedLibrary) }),
|
|
436
485
|
});
|
|
437
486
|
}
|
|
438
487
|
}
|
|
439
488
|
|
|
440
|
-
return { chunks, files };
|
|
489
|
+
return { chunks, files, collisions };
|
|
490
|
+
}
|
|
491
|
+
|
|
492
|
+
/**
|
|
493
|
+
* The two ways a chunk id moves without anyone deciding it should. Both are reported as one
|
|
494
|
+
* line each, however many ids are involved: a large package collides in dozens of places, and a
|
|
495
|
+
* warning per id is a wall of text nobody reads.
|
|
496
|
+
*/
|
|
497
|
+
function idWarnings(collisions: readonly string[], gone: readonly string[]): string[] {
|
|
498
|
+
const warnings: string[] = [];
|
|
499
|
+
if (collisions.length > 0) {
|
|
500
|
+
warnings.push(
|
|
501
|
+
`${plural(collisions.length, "derived chunk id")} collided and ${collisions.length === 1 ? "was" : "were"} given a numeric suffix: ${list(collisions)} — which section keeps the bare id depends on the order the sources were read in`,
|
|
502
|
+
);
|
|
503
|
+
}
|
|
504
|
+
if (gone.length > 0) {
|
|
505
|
+
warnings.push(
|
|
506
|
+
`${plural(gone.length, "chunk id")} the previous build published ${gone.length === 1 ? "is" : "are"} gone: ${list(gone)} — recorded feedback and published links pinned to ${gone.length === 1 ? "it" : "them"} no longer resolve`,
|
|
507
|
+
);
|
|
508
|
+
}
|
|
509
|
+
return warnings;
|
|
510
|
+
}
|
|
511
|
+
|
|
512
|
+
const plural = (count: number, noun: string): string => `${count} ${noun}${count === 1 ? "" : "s"}`;
|
|
513
|
+
|
|
514
|
+
const list = (ids: readonly string[]): string =>
|
|
515
|
+
`${ids.slice(0, 5).join(", ")}${ids.length > 5 ? `, and ${ids.length - 5} more` : ""}`;
|
|
516
|
+
|
|
517
|
+
/**
|
|
518
|
+
* Chunk ids the payload on disk carries that this build does not, or nothing when there is no
|
|
519
|
+
* previous payload to compare with. An unreadable manifest is not a finding: `build` is what
|
|
520
|
+
* replaces it.
|
|
521
|
+
*/
|
|
522
|
+
async function departedChunkIds(llmsDir: string, chunks: readonly ChunkSpec[]): Promise<string[]> {
|
|
523
|
+
let previous: PackageManifest;
|
|
524
|
+
try {
|
|
525
|
+
previous = parseManifest(
|
|
526
|
+
JSON.parse(await readFile(join(llmsDir, MANIFEST_FILE), "utf8")),
|
|
527
|
+
`${LLMS_DIR}/${MANIFEST_FILE}`,
|
|
528
|
+
);
|
|
529
|
+
} catch {
|
|
530
|
+
return [];
|
|
531
|
+
}
|
|
532
|
+
|
|
533
|
+
const current = new Set(chunks.map((chunk) => chunk.id));
|
|
534
|
+
return previous.chunks.map((chunk) => chunk.id).filter((id) => !current.has(id));
|
|
535
|
+
}
|
|
536
|
+
|
|
537
|
+
/** One piece of a document, before it becomes a chunk. */
|
|
538
|
+
interface Section {
|
|
539
|
+
readonly heading: string;
|
|
540
|
+
readonly body: string;
|
|
541
|
+
/** Depth of the heading in the source, or 0 when it was inherited rather than written. */
|
|
542
|
+
readonly level: number;
|
|
543
|
+
/**
|
|
544
|
+
* Headings to index as tags, when they are not just `heading`. A packed chunk answers for every
|
|
545
|
+
* section in it, so all of their headings are search terms for it.
|
|
546
|
+
*/
|
|
547
|
+
readonly headings?: readonly string[];
|
|
441
548
|
}
|
|
442
549
|
|
|
443
550
|
/**
|
|
444
551
|
* Splits Markdown at `##` headings, then `###`, then paragraphs, until every piece fits the token
|
|
445
552
|
* budget. Retrieval works best when a chunk answers one question.
|
|
553
|
+
*
|
|
554
|
+
* With `minTokens` set, adjacent pieces are then packed back together up to the budget. Every
|
|
555
|
+
* API generator emits pages whose headings are field names — `Category` is one line, `Sizes` is
|
|
556
|
+
* three — and one chunk per heading there is hundreds of chunks too small to answer anything.
|
|
446
557
|
*/
|
|
447
558
|
function splitMarkdown(
|
|
448
559
|
text: string,
|
|
449
560
|
title: string,
|
|
450
561
|
maxTokens: number,
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
const
|
|
562
|
+
minTokens: number | undefined,
|
|
563
|
+
): Section[] {
|
|
564
|
+
const sections = splitAtLevel(text, 2, title, 0);
|
|
565
|
+
const result: Section[] = [];
|
|
454
566
|
|
|
455
567
|
for (const section of sections) {
|
|
456
568
|
if (estimateTokens(section.body) <= maxTokens) {
|
|
@@ -458,7 +570,7 @@ function splitMarkdown(
|
|
|
458
570
|
continue;
|
|
459
571
|
}
|
|
460
572
|
|
|
461
|
-
const deeper = splitAtLevel(section.body, 3, section.heading);
|
|
573
|
+
const deeper = splitAtLevel(section.body, 3, section.heading, section.level);
|
|
462
574
|
for (const part of deeper) {
|
|
463
575
|
if (estimateTokens(part.body) <= maxTokens) {
|
|
464
576
|
if (part.body.trim().length > 0) result.push(part);
|
|
@@ -468,22 +580,26 @@ function splitMarkdown(
|
|
|
468
580
|
...splitParagraphs(part.body, maxTokens).map((body, index) => ({
|
|
469
581
|
heading: index === 0 ? part.heading : `${part.heading} (${index + 1})`,
|
|
470
582
|
body,
|
|
583
|
+
// A continuation carries no heading of its own, so packing must not re-emit one.
|
|
584
|
+
level: index === 0 ? part.level : 0,
|
|
471
585
|
})),
|
|
472
586
|
);
|
|
473
587
|
}
|
|
474
588
|
}
|
|
475
589
|
|
|
476
|
-
|
|
590
|
+
const pieces = result.length > 0 ? result : [{ heading: title, body: text, level: 0 }];
|
|
591
|
+
return minTokens === undefined ? pieces : packSections(pieces, title, minTokens, maxTokens);
|
|
477
592
|
}
|
|
478
593
|
|
|
479
594
|
function splitAtLevel(
|
|
480
595
|
text: string,
|
|
481
596
|
level: number,
|
|
482
597
|
fallbackHeading: string,
|
|
483
|
-
|
|
598
|
+
fallbackLevel: number,
|
|
599
|
+
): Section[] {
|
|
484
600
|
const marker = new RegExp(`^#{${level}}\\s+(.+?)\\s*$`);
|
|
485
|
-
const sections: { heading: string; body: string[] }[] = [];
|
|
486
|
-
let current = { heading: fallbackHeading, body: [] as string[] };
|
|
601
|
+
const sections: { heading: string; body: string[]; level: number }[] = [];
|
|
602
|
+
let current = { heading: fallbackHeading, body: [] as string[], level: fallbackLevel };
|
|
487
603
|
let inFence = false;
|
|
488
604
|
|
|
489
605
|
for (const line of text.split("\n")) {
|
|
@@ -492,7 +608,7 @@ function splitAtLevel(
|
|
|
492
608
|
const match = inFence ? null : marker.exec(line);
|
|
493
609
|
if (match?.[1] !== undefined) {
|
|
494
610
|
if (current.body.join("\n").trim().length > 0) sections.push(current);
|
|
495
|
-
current = { heading: match[1], body: [] };
|
|
611
|
+
current = { heading: match[1], body: [], level };
|
|
496
612
|
continue;
|
|
497
613
|
}
|
|
498
614
|
current.body.push(line);
|
|
@@ -501,10 +617,57 @@ function splitAtLevel(
|
|
|
501
617
|
|
|
502
618
|
return sections.map((section) => ({
|
|
503
619
|
heading: section.heading,
|
|
620
|
+
level: section.level,
|
|
504
621
|
body: section.body.join("\n").trim(),
|
|
505
622
|
}));
|
|
506
623
|
}
|
|
507
624
|
|
|
625
|
+
/**
|
|
626
|
+
* Merges adjacent sections until a group reaches `minTokens`, never crossing `maxTokens`. A
|
|
627
|
+
* merged group is headed by the document title and keeps each section's own heading inside it,
|
|
628
|
+
* so a whole component lands in one chunk without losing the structure a reader needs.
|
|
629
|
+
*/
|
|
630
|
+
function packSections(
|
|
631
|
+
sections: readonly Section[],
|
|
632
|
+
title: string,
|
|
633
|
+
minTokens: number,
|
|
634
|
+
maxTokens: number,
|
|
635
|
+
): Section[] {
|
|
636
|
+
const packed: Section[] = [];
|
|
637
|
+
let group: Section[] = [];
|
|
638
|
+
let tokens = 0;
|
|
639
|
+
|
|
640
|
+
const flush = (): void => {
|
|
641
|
+
if (group.length === 1 && group[0] !== undefined) packed.push(group[0]);
|
|
642
|
+
else if (group.length > 1) {
|
|
643
|
+
packed.push({
|
|
644
|
+
heading: title,
|
|
645
|
+
level: 0,
|
|
646
|
+
body: group.map(withHeading).join("\n\n"),
|
|
647
|
+
headings: group.map((section) => section.heading),
|
|
648
|
+
});
|
|
649
|
+
}
|
|
650
|
+
group = [];
|
|
651
|
+
tokens = 0;
|
|
652
|
+
};
|
|
653
|
+
|
|
654
|
+
for (const section of sections) {
|
|
655
|
+
const size = estimateTokens(withHeading(section));
|
|
656
|
+
if (group.length > 0 && (tokens >= minTokens || tokens + size > maxTokens)) flush();
|
|
657
|
+
group.push(section);
|
|
658
|
+
tokens += size;
|
|
659
|
+
}
|
|
660
|
+
flush();
|
|
661
|
+
|
|
662
|
+
return packed;
|
|
663
|
+
}
|
|
664
|
+
|
|
665
|
+
/** A section as it appeared in the source, heading included when it had one. */
|
|
666
|
+
function withHeading(section: Section): string {
|
|
667
|
+
if (section.level === 0) return section.body;
|
|
668
|
+
return `${"#".repeat(section.level)} ${section.heading}\n\n${section.body}`;
|
|
669
|
+
}
|
|
670
|
+
|
|
508
671
|
function splitParagraphs(text: string, maxTokens: number): string[] {
|
|
509
672
|
const parts: string[] = [];
|
|
510
673
|
let buffer: string[] = [];
|
|
@@ -562,26 +725,36 @@ function matchAll(text: string, pattern: RegExp): string[] {
|
|
|
562
725
|
}
|
|
563
726
|
|
|
564
727
|
/** One authored override under a heading: `<!-- docspack: tags=grid,columns -->`. */
|
|
565
|
-
const DIRECTIVE =
|
|
728
|
+
const DIRECTIVE =
|
|
729
|
+
/^[ \t]*<!--[ \t]*docspack:[ \t]*(tags|entities|documents)[ \t]*=([^>]*?)-->[ \t]*$/gim;
|
|
566
730
|
|
|
567
731
|
/**
|
|
568
732
|
* Reads per-section directives and removes them from the body. Front matter belongs to a whole
|
|
569
733
|
* document, so a page of eighty sections cannot aim any one of them; this is that lever.
|
|
570
734
|
*/
|
|
571
|
-
function readDirectives(body: string): {
|
|
572
|
-
|
|
573
|
-
|
|
735
|
+
function readDirectives(body: string): {
|
|
736
|
+
body: string;
|
|
737
|
+
tags: string[];
|
|
738
|
+
entities: string[];
|
|
739
|
+
documents: string[];
|
|
740
|
+
} {
|
|
741
|
+
const collected: Record<string, string[]> = { tags: [], entities: [], documents: [] };
|
|
574
742
|
|
|
575
743
|
const text = body.replace(DIRECTIVE, (_line, key: string, value: string) => {
|
|
576
744
|
const values = value
|
|
577
745
|
.split(",")
|
|
578
746
|
.map((item) => item.trim())
|
|
579
747
|
.filter((item) => item.length > 0);
|
|
580
|
-
|
|
748
|
+
collected[key.toLowerCase()]?.push(...values);
|
|
581
749
|
return "";
|
|
582
750
|
});
|
|
583
751
|
|
|
584
|
-
return {
|
|
752
|
+
return {
|
|
753
|
+
body: text.replace(/\n{3,}/g, "\n\n").trim(),
|
|
754
|
+
tags: collected.tags ?? [],
|
|
755
|
+
entities: collected.entities ?? [],
|
|
756
|
+
documents: collected.documents ?? [],
|
|
757
|
+
};
|
|
585
758
|
}
|
|
586
759
|
|
|
587
760
|
/**
|
|
@@ -620,14 +793,69 @@ function slugify(input: string, fallback: string): string {
|
|
|
620
793
|
return slug.length > 0 ? slug : fallback;
|
|
621
794
|
}
|
|
622
795
|
|
|
623
|
-
|
|
624
|
-
|
|
796
|
+
const MAX_TAGS = 24;
|
|
797
|
+
|
|
798
|
+
/**
|
|
799
|
+
* The words indexed alongside a chunk.
|
|
800
|
+
*
|
|
801
|
+
* Directive tags are taken as the author wrote them — those are deliberate, and one of them
|
|
802
|
+
* being a function word is the author's decision. Words lifted out of a title or a heading are
|
|
803
|
+
* filtered: nobody chose them as search terms, and the tag field is weighted 3× in the ranking,
|
|
804
|
+
* which is what made "How to use it" the answer to every question beginning "how do I".
|
|
805
|
+
*/
|
|
806
|
+
function chunkTags(authored: readonly string[], derived: readonly string[]): string[] {
|
|
807
|
+
const tags = new Set(words(authored));
|
|
808
|
+
for (const word of words(derived)) {
|
|
809
|
+
if (!STOPWORDS.has(word)) tags.add(word);
|
|
810
|
+
}
|
|
811
|
+
return [...tags].slice(0, MAX_TAGS);
|
|
812
|
+
}
|
|
813
|
+
|
|
814
|
+
function words(inputs: readonly string[]): string[] {
|
|
815
|
+
const found: string[] = [];
|
|
625
816
|
for (const input of inputs) {
|
|
626
|
-
|
|
627
|
-
words.add(word);
|
|
628
|
-
}
|
|
817
|
+
found.push(...(input.toLowerCase().match(/[\p{L}\p{N}]{2,}/gu) ?? []));
|
|
629
818
|
}
|
|
630
|
-
return
|
|
819
|
+
return found;
|
|
820
|
+
}
|
|
821
|
+
|
|
822
|
+
/** A row of a Markdown table: `| Name | Type |`, and the `|---|---:|` rule under it. */
|
|
823
|
+
const TABLE_ROW = /^\|.*\|/;
|
|
824
|
+
const TABLE_RULE = /^[|\-: \t]+$/;
|
|
825
|
+
|
|
826
|
+
/**
|
|
827
|
+
* Collapses the alignment padding out of Markdown tables.
|
|
828
|
+
*
|
|
829
|
+
* Every API generator pads table cells into columns. That padding is whitespace to a reader and
|
|
830
|
+
* to a renderer, but tokens are estimated from characters, so a generated props table is
|
|
831
|
+
* measured at several times its real weight — which spends the response budget, trips
|
|
832
|
+
* `chunk-too-large`, and forces splits that leave fragments competing for the same query. One
|
|
833
|
+
* chart's props table measured 4,094 tokens padded and 1,245 collapsed.
|
|
834
|
+
*
|
|
835
|
+
* Runs inside a code fence are left alone; a fenced table is example text, and its alignment is
|
|
836
|
+
* the thing being shown.
|
|
837
|
+
*/
|
|
838
|
+
export function collapseTablePadding(text: string): string {
|
|
839
|
+
let inFence = false;
|
|
840
|
+
|
|
841
|
+
return text
|
|
842
|
+
.split("\n")
|
|
843
|
+
.map((line) => {
|
|
844
|
+
if (/^\s*(?:```|~~~)/.test(line)) {
|
|
845
|
+
inFence = !inFence;
|
|
846
|
+
return line;
|
|
847
|
+
}
|
|
848
|
+
if (inFence) return line;
|
|
849
|
+
|
|
850
|
+
const indent = /^[ \t]*/.exec(line)?.[0] ?? "";
|
|
851
|
+
const rest = line.slice(indent.length);
|
|
852
|
+
// Four spaces or more is an indented code block, whatever the line looks like.
|
|
853
|
+
if (indent.length >= 4 || !TABLE_ROW.test(rest)) return line;
|
|
854
|
+
|
|
855
|
+
const collapsed = rest.replace(/[ \t]{2,}/g, " ");
|
|
856
|
+
return indent + (TABLE_RULE.test(rest) ? collapsed.replace(/-{4,}/g, "---") : collapsed);
|
|
857
|
+
})
|
|
858
|
+
.join("\n");
|
|
631
859
|
}
|
|
632
860
|
|
|
633
861
|
function uniqueId(base: string, taken: Set<string>): string {
|