docspack 0.0.1 → 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +59 -0
- package/bin/docspack.js +25 -0
- package/dist/build.d.ts +31 -0
- package/dist/build.d.ts.map +1 -0
- package/dist/build.js +435 -0
- package/dist/build.js.map +1 -0
- package/dist/cli.d.ts +3 -0
- package/dist/cli.d.ts.map +1 -0
- package/dist/cli.js +763 -0
- package/dist/cli.js.map +1 -0
- package/dist/config.d.ts +41 -0
- package/dist/config.d.ts.map +1 -0
- package/dist/config.js +118 -0
- package/dist/config.js.map +1 -0
- package/dist/db.d.ts +60 -0
- package/dist/db.d.ts.map +1 -0
- package/dist/db.js +204 -0
- package/dist/db.js.map +1 -0
- package/dist/discovery.d.ts +31 -0
- package/dist/discovery.d.ts.map +1 -0
- package/dist/discovery.js +126 -0
- package/dist/discovery.js.map +1 -0
- package/dist/doctor.d.ts +25 -0
- package/dist/doctor.d.ts.map +1 -0
- package/dist/doctor.js +276 -0
- package/dist/doctor.js.map +1 -0
- package/dist/document.d.ts +13 -0
- package/dist/document.d.ts.map +1 -0
- package/dist/document.js +47 -0
- package/dist/document.js.map +1 -0
- package/dist/errors.d.ts +9 -0
- package/dist/errors.d.ts.map +1 -0
- package/dist/errors.js +10 -0
- package/dist/errors.js.map +1 -0
- package/dist/exports.d.ts +20 -0
- package/dist/exports.d.ts.map +1 -0
- package/dist/exports.js +100 -0
- package/dist/exports.js.map +1 -0
- package/dist/feedback.d.ts +68 -0
- package/dist/feedback.d.ts.map +1 -0
- package/dist/feedback.js +0 -0
- package/dist/feedback.js.map +1 -0
- package/dist/html.d.ts +4 -0
- package/dist/html.d.ts.map +1 -0
- package/dist/html.js +23 -0
- package/dist/html.js.map +1 -0
- package/dist/http.d.ts +30 -0
- package/dist/http.d.ts.map +1 -0
- package/dist/http.js +144 -0
- package/dist/http.js.map +1 -0
- package/dist/index.d.ts +25 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +25 -0
- package/dist/index.js.map +1 -0
- package/dist/init/detect.d.ts +16 -0
- package/dist/init/detect.d.ts.map +1 -0
- package/dist/init/detect.js +120 -0
- package/dist/init/detect.js.map +1 -0
- package/dist/init/plan.d.ts +43 -0
- package/dist/init/plan.d.ts.map +1 -0
- package/dist/init/plan.js +145 -0
- package/dist/init/plan.js.map +1 -0
- package/dist/init/run.d.ts +28 -0
- package/dist/init/run.d.ts.map +1 -0
- package/dist/init/run.js +96 -0
- package/dist/init/run.js.map +1 -0
- package/dist/init/templates.d.ts +24 -0
- package/dist/init/templates.d.ts.map +1 -0
- package/dist/init/templates.js +181 -0
- package/dist/init/templates.js.map +1 -0
- package/dist/init/write.d.ts +20 -0
- package/dist/init/write.d.ts.map +1 -0
- package/dist/init/write.js +56 -0
- package/dist/init/write.js.map +1 -0
- package/dist/kinds.d.ts +14 -0
- package/dist/kinds.d.ts.map +1 -0
- package/dist/kinds.js +15 -0
- package/dist/kinds.js.map +1 -0
- package/dist/llms-txt.d.ts +25 -0
- package/dist/llms-txt.d.ts.map +1 -0
- package/dist/llms-txt.js +94 -0
- package/dist/llms-txt.js.map +1 -0
- package/dist/mcp.d.ts +15 -0
- package/dist/mcp.d.ts.map +1 -0
- package/dist/mcp.js +158 -0
- package/dist/mcp.js.map +1 -0
- package/dist/preview.d.ts +18 -0
- package/dist/preview.d.ts.map +1 -0
- package/dist/preview.js +72 -0
- package/dist/preview.js.map +1 -0
- package/dist/prompt.d.ts +27 -0
- package/dist/prompt.d.ts.map +1 -0
- package/dist/prompt.js +79 -0
- package/dist/prompt.js.map +1 -0
- package/dist/search.d.ts +41 -0
- package/dist/search.d.ts.map +1 -0
- package/dist/search.js +60 -0
- package/dist/search.js.map +1 -0
- package/dist/snippet.d.ts +20 -0
- package/dist/snippet.d.ts.map +1 -0
- package/dist/snippet.js +29 -0
- package/dist/snippet.js.map +1 -0
- package/dist/spec.d.ts +38 -0
- package/dist/spec.d.ts.map +1 -0
- package/dist/spec.js +105 -0
- package/dist/spec.js.map +1 -0
- package/dist/style.d.ts +33 -0
- package/dist/style.d.ts.map +1 -0
- package/dist/style.js +94 -0
- package/dist/style.js.map +1 -0
- package/dist/submit.d.ts +61 -0
- package/dist/submit.d.ts.map +1 -0
- package/dist/submit.js +111 -0
- package/dist/submit.js.map +1 -0
- package/dist/sync.d.ts +29 -0
- package/dist/sync.d.ts.map +1 -0
- package/dist/sync.js +73 -0
- package/dist/sync.js.map +1 -0
- package/dist/verify.d.ts +44 -0
- package/dist/verify.d.ts.map +1 -0
- package/dist/verify.js +291 -0
- package/dist/verify.js.map +1 -0
- package/package.json +60 -5
- package/src/build.ts +572 -0
- package/src/cli.ts +883 -0
- package/src/config.ts +158 -0
- package/src/db.ts +261 -0
- package/src/discovery.ts +161 -0
- package/src/doctor.ts +344 -0
- package/src/document.ts +59 -0
- package/src/errors.ts +10 -0
- package/src/exports.ts +120 -0
- package/src/feedback.ts +0 -0
- package/src/html.ts +24 -0
- package/src/http.ts +190 -0
- package/src/index.ts +132 -0
- package/src/init/detect.ts +142 -0
- package/src/init/plan.ts +215 -0
- package/src/init/run.ts +142 -0
- package/src/init/templates.ts +200 -0
- package/src/init/write.ts +83 -0
- package/src/kinds.ts +17 -0
- package/src/llms-txt.ts +116 -0
- package/src/mcp.ts +196 -0
- package/src/preview.ts +98 -0
- package/src/prompt.ts +103 -0
- package/src/search.ts +96 -0
- package/src/snippet.ts +30 -0
- package/src/spec.ts +138 -0
- package/src/style.ts +111 -0
- package/src/submit.ts +189 -0
- package/src/sync.ts +112 -0
- package/src/verify.ts +355 -0
- package/bin/cli.js +0 -2
package/src/style.ts
ADDED
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Style rules for documentation written to be read by a model.
|
|
3
|
+
*
|
|
4
|
+
* The goal is density, not brevity for its own sake. Stripping prose wholesale measurably hurts
|
|
5
|
+
* retrieval: the index matches on the words a user would type, and those live in sentences, not
|
|
6
|
+
* in headings. What is free to remove is filler — narration and marketing that carries neither a
|
|
7
|
+
* fact nor a word anyone would search for.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
/** Narration and marketing. None of these words appear in a question someone asks an agent. */
|
|
11
|
+
const FILLER = [
|
|
12
|
+
/\bin this (guide|section|article|tutorial|post|chapter)\b/gi,
|
|
13
|
+
/\b(we|you)('ll| will) (now |then )?(learn|see|explore|cover|look at|dive into|walk through)\b/gi,
|
|
14
|
+
/\blet'?s (get started|begin|dive|take a look|explore)\b/gi,
|
|
15
|
+
/\bas you can see\b/gi,
|
|
16
|
+
/\bas (we )?(mentioned|discussed|noted) (above|earlier|before)\b/gi,
|
|
17
|
+
/\b(welcome to|congratulations|don'?t worry|no worries)\b/gi,
|
|
18
|
+
/\b(simply|just simply|easily) (call|use|add|run|pass|set)\b/gi,
|
|
19
|
+
/\b(blazing[- ]fast|lightning[- ]fast|seamlessly|effortlessly|cutting[- ]edge|world[- ]class|state[- ]of[- ]the[- ]art)\b/gi,
|
|
20
|
+
/\b(super|incredibly|extremely|amazingly) (fast|simple|easy|powerful)\b/gi,
|
|
21
|
+
/\bfirst things first\b/gi,
|
|
22
|
+
];
|
|
23
|
+
|
|
24
|
+
/** Boilerplate a documentation site adds around the content, never part of it. */
|
|
25
|
+
const BOILERPLATE = [
|
|
26
|
+
/^\s*\[!\[[^\]]*\]\([^)]*\)\]\([^)]*\)\s*$/, // badge wrapped in a link
|
|
27
|
+
/^\s*\[!\[[^\]]*\]\([^)]*\)\s*$/, // bare badge
|
|
28
|
+
/^\s*(was this (page )?helpful\??|edit this page|on this page|table of contents)\s*$/i,
|
|
29
|
+
/^\s*(previous|next)\s*[:|]\s*\S/i,
|
|
30
|
+
/^\s*<!--[\s\S]*?-->\s*$/,
|
|
31
|
+
];
|
|
32
|
+
|
|
33
|
+
const SENTENCE_SPLIT = /(?<=[.!?])\s+/;
|
|
34
|
+
/** A heading or list item starts a new unit, whatever the previous line ended with. */
|
|
35
|
+
const BLOCK_START = /\n(?=\s*(?:[-*+]\s|\d+\.\s|#{1,6}\s|\|))/;
|
|
36
|
+
const LONG_SENTENCE_WORDS = 40;
|
|
37
|
+
|
|
38
|
+
export interface FillerHit {
|
|
39
|
+
readonly phrase: string;
|
|
40
|
+
readonly count: number;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/** Filler phrases found in a document, with how often each occurs. */
|
|
44
|
+
export function findFiller(text: string): FillerHit[] {
|
|
45
|
+
const prose = withoutCode(text);
|
|
46
|
+
const found = new Map<string, number>();
|
|
47
|
+
|
|
48
|
+
for (const pattern of FILLER) {
|
|
49
|
+
for (const match of prose.matchAll(pattern)) {
|
|
50
|
+
const phrase = match[0].toLowerCase().replace(/\s+/g, " ");
|
|
51
|
+
found.set(phrase, (found.get(phrase) ?? 0) + 1);
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
return [...found]
|
|
56
|
+
.map(([phrase, count]) => ({ phrase, count }))
|
|
57
|
+
.sort((a, b) => b.count - a.count || a.phrase.localeCompare(b.phrase));
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/**
|
|
61
|
+
* Sentences long enough that a reader — human or model — loses the thread. Headings, list items
|
|
62
|
+
* and paragraphs are separate units: a heading followed by a bulleted list is not one sentence,
|
|
63
|
+
* however little punctuation it contains.
|
|
64
|
+
*/
|
|
65
|
+
export function findLongSentences(text: string, maxWords = LONG_SENTENCE_WORDS): string[] {
|
|
66
|
+
return withoutCode(text)
|
|
67
|
+
.split(/\n{2,}/)
|
|
68
|
+
.flatMap((block) => block.split(BLOCK_START))
|
|
69
|
+
.flatMap((block) => block.split(SENTENCE_SPLIT))
|
|
70
|
+
.map((sentence) => sentence.replace(/\s+/g, " ").trim())
|
|
71
|
+
.filter((sentence) => sentence.split(" ").filter(Boolean).length > maxWords);
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/**
|
|
75
|
+
* True when a chunk contains something concrete: a code block, an inline identifier, a list or a
|
|
76
|
+
* table. A chunk with none of those is usually narration about documentation rather than
|
|
77
|
+
* documentation.
|
|
78
|
+
*/
|
|
79
|
+
export function hasConcreteContent(text: string): boolean {
|
|
80
|
+
return (
|
|
81
|
+
/```|~~~/.test(text) ||
|
|
82
|
+
/`[^`\n]+`/.test(text) ||
|
|
83
|
+
/^\s*[-*+]\s|^\s*\d+\.\s/m.test(text) ||
|
|
84
|
+
/^\s*\|.*\|/m.test(text)
|
|
85
|
+
);
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/** Normalized form used to spot two chunks that say the same thing. */
|
|
89
|
+
export function contentFingerprint(text: string): string {
|
|
90
|
+
return text
|
|
91
|
+
.toLowerCase()
|
|
92
|
+
.replace(/<!--[\s\S]*?-->/g, "")
|
|
93
|
+
.replace(/[^a-z0-9]+/g, " ")
|
|
94
|
+
.trim();
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
/** Removes lines a documentation site wraps around its content. */
|
|
98
|
+
export function stripBoilerplate(text: string): string {
|
|
99
|
+
return text
|
|
100
|
+
.split("\n")
|
|
101
|
+
.filter((line) => !BOILERPLATE.some((pattern) => pattern.test(line)))
|
|
102
|
+
.join("\n");
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/** Text with fenced and inline code removed, so prose rules never fire on a code sample. */
|
|
106
|
+
export function withoutCode(text: string): string {
|
|
107
|
+
return text
|
|
108
|
+
.replace(/```[\s\S]*?```/g, " ")
|
|
109
|
+
.replace(/~~~[\s\S]*?~~~/g, " ")
|
|
110
|
+
.replace(/`[^`\n]*`/g, " ");
|
|
111
|
+
}
|
package/src/submit.ts
ADDED
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
import type { FeedbackChannel } from "./config.js";
|
|
2
|
+
import { readBuildConfig } from "./config.js";
|
|
3
|
+
import { discoverPackages } from "./discovery.js";
|
|
4
|
+
import { type FeedbackFinding, readFeedback } from "./feedback.js";
|
|
5
|
+
import { KINDS } from "./kinds.js";
|
|
6
|
+
import { packageId } from "./spec.js";
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* A finding that a vendor has asked to receive, rendered and ready for a human to file.
|
|
10
|
+
*
|
|
11
|
+
* `url` is a prefilled GitHub issue link. docspack does not open it, does not follow it, and
|
|
12
|
+
* has no code that could: a human clicks it, reads what is there, and files it under their own
|
|
13
|
+
* name. That is the whole transmission mechanism.
|
|
14
|
+
*/
|
|
15
|
+
export interface RoutedFinding {
|
|
16
|
+
readonly fingerprint: string;
|
|
17
|
+
readonly chunkId: string;
|
|
18
|
+
readonly title: string;
|
|
19
|
+
readonly body: string;
|
|
20
|
+
readonly labels: readonly string[];
|
|
21
|
+
/** Undefined when the rendered report is too long to survive a URL. */
|
|
22
|
+
readonly url?: string;
|
|
23
|
+
/** True when the report contains a reproduction, which may quote the user's own code. */
|
|
24
|
+
readonly containsCode: boolean;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
export type BlockedReason = "no-channel" | "not-accepted";
|
|
28
|
+
|
|
29
|
+
export interface BlockedFinding {
|
|
30
|
+
readonly fingerprint: string;
|
|
31
|
+
readonly chunkId: string;
|
|
32
|
+
readonly kind: string;
|
|
33
|
+
readonly reason: BlockedReason;
|
|
34
|
+
/** What the vendor does accept, when they accept something narrower. */
|
|
35
|
+
readonly accepts?: readonly string[];
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
export interface SubmitReport {
|
|
39
|
+
readonly routed: readonly RoutedFinding[];
|
|
40
|
+
readonly blocked: readonly BlockedFinding[];
|
|
41
|
+
/** Channels involved, so the caller can show each vendor's policy once. */
|
|
42
|
+
readonly channels: readonly { readonly packageId: string; readonly channel: FeedbackChannel }[];
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
export interface SubmitOptions {
|
|
46
|
+
readonly cwd: string;
|
|
47
|
+
readonly docspackVersion: string;
|
|
48
|
+
/** Only findings recorded against this package. */
|
|
49
|
+
readonly packageFilter?: string;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* A GitHub issue URL has to survive a browser address bar, and a reproduction can be long.
|
|
54
|
+
* Past this the report is printed for the human to paste instead of silently truncated.
|
|
55
|
+
*/
|
|
56
|
+
const MAX_URL = 6000;
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* Prepares recorded findings for filing. Nothing is sent, and nothing here can send: this
|
|
60
|
+
* function returns text.
|
|
61
|
+
*/
|
|
62
|
+
export async function prepareSubmission(options: SubmitOptions): Promise<SubmitReport> {
|
|
63
|
+
const findings = await readFeedback(options.cwd);
|
|
64
|
+
const channels = await channelsFor(options.cwd);
|
|
65
|
+
|
|
66
|
+
const routed: RoutedFinding[] = [];
|
|
67
|
+
const blocked: BlockedFinding[] = [];
|
|
68
|
+
const used = new Map<string, FeedbackChannel>();
|
|
69
|
+
|
|
70
|
+
for (const finding of findings) {
|
|
71
|
+
const id = packageId(finding.package, finding.version);
|
|
72
|
+
if (options.packageFilter !== undefined && !finding.package.includes(options.packageFilter)) {
|
|
73
|
+
continue;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
const channel = channels.get(id);
|
|
77
|
+
if (channel === undefined) {
|
|
78
|
+
blocked.push({ ...identity(finding), reason: "no-channel" });
|
|
79
|
+
continue;
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
// An omitted `accepts` means the vendor takes everything; declaring the channel was the
|
|
83
|
+
// opt-in. Narrowing it is how a vendor drowning in noise turns the volume down.
|
|
84
|
+
const accepts = channel.accepts ?? KINDS;
|
|
85
|
+
if (!accepts.includes(finding.kind)) {
|
|
86
|
+
blocked.push({ ...identity(finding), reason: "not-accepted", accepts });
|
|
87
|
+
continue;
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
used.set(id, channel);
|
|
91
|
+
routed.push(render(finding, channel, options.docspackVersion));
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
return {
|
|
95
|
+
routed,
|
|
96
|
+
blocked,
|
|
97
|
+
channels: [...used].map(([id, channel]) => ({ packageId: id, channel })),
|
|
98
|
+
};
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
function identity(finding: FeedbackFinding): {
|
|
102
|
+
fingerprint: string;
|
|
103
|
+
chunkId: string;
|
|
104
|
+
kind: string;
|
|
105
|
+
} {
|
|
106
|
+
return { fingerprint: finding.fingerprint, chunkId: finding.chunkId, kind: finding.kind };
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
/** The declared channel of every docs package installed here, by package id. */
|
|
110
|
+
async function channelsFor(cwd: string): Promise<Map<string, FeedbackChannel>> {
|
|
111
|
+
const { packages } = await discoverPackages(cwd);
|
|
112
|
+
const channels = new Map<string, FeedbackChannel>();
|
|
113
|
+
|
|
114
|
+
for (const pkg of packages) {
|
|
115
|
+
// A package whose config cannot be read simply has no channel; a broken vendor manifest
|
|
116
|
+
// must not stop a human from reviewing the rest of their findings.
|
|
117
|
+
const config = await readBuildConfig(pkg.dir).catch(() => ({ feedback: undefined }));
|
|
118
|
+
if (config.feedback !== undefined) channels.set(pkg.id, config.feedback);
|
|
119
|
+
}
|
|
120
|
+
return channels;
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
function render(
|
|
124
|
+
finding: FeedbackFinding,
|
|
125
|
+
channel: FeedbackChannel,
|
|
126
|
+
docspackVersion: string,
|
|
127
|
+
): RoutedFinding {
|
|
128
|
+
const title = `docspack: ${finding.evidence}`.slice(0, 120);
|
|
129
|
+
const body = renderBody(finding, docspackVersion);
|
|
130
|
+
const labels = channel.labels ?? [];
|
|
131
|
+
|
|
132
|
+
const url = issueUrl(channel.github, title, body, labels);
|
|
133
|
+
return {
|
|
134
|
+
fingerprint: finding.fingerprint,
|
|
135
|
+
chunkId: finding.chunkId,
|
|
136
|
+
title,
|
|
137
|
+
body,
|
|
138
|
+
labels,
|
|
139
|
+
...(url === undefined ? {} : { url }),
|
|
140
|
+
containsCode: finding.repro !== undefined,
|
|
141
|
+
};
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
/**
|
|
145
|
+
* The report a maintainer reads. Provenance is stated rather than hidden: maintainers have
|
|
146
|
+
* asked repeatedly to be able to tell machine-written reports apart, and concealing it is
|
|
147
|
+
* what makes them toxic.
|
|
148
|
+
*/
|
|
149
|
+
export function renderBody(finding: FeedbackFinding, docspackVersion: string): string {
|
|
150
|
+
const lines = [
|
|
151
|
+
`**Chunk**: \`${finding.chunkId}\``,
|
|
152
|
+
`**Kind**: ${finding.kind}${finding.kind === "drift" ? " — a name the docs use that the package does not declare" : ""}`,
|
|
153
|
+
`**Seen**: ${finding.seen} ${finding.seen === 1 ? "time" : "times"} in one project`,
|
|
154
|
+
"",
|
|
155
|
+
finding.evidence,
|
|
156
|
+
];
|
|
157
|
+
|
|
158
|
+
if (finding.expected !== undefined || finding.actual !== undefined) {
|
|
159
|
+
lines.push("", `Expected: ${finding.expected ?? "—"}`, `Actual: ${finding.actual ?? "—"}`);
|
|
160
|
+
}
|
|
161
|
+
if (finding.repro !== undefined) {
|
|
162
|
+
lines.push("", "Reproduction:", "", "```", finding.repro, "```");
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
lines.push(
|
|
166
|
+
"",
|
|
167
|
+
"---",
|
|
168
|
+
`Recorded with docspack ${docspackVersion}, read and filed by a human. docspack does not`,
|
|
169
|
+
"send reports; this was written locally and someone chose to bring it here.",
|
|
170
|
+
);
|
|
171
|
+
return lines.join("\n");
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
/**
|
|
175
|
+
* A prefilled issue URL, using GitHub's documented `title`, `body` and `labels` query
|
|
176
|
+
* parameters. Returns nothing when the result is too long to be usable as a link.
|
|
177
|
+
*/
|
|
178
|
+
export function issueUrl(
|
|
179
|
+
repo: string,
|
|
180
|
+
title: string,
|
|
181
|
+
body: string,
|
|
182
|
+
labels: readonly string[],
|
|
183
|
+
): string | undefined {
|
|
184
|
+
const query = new URLSearchParams({ title, body });
|
|
185
|
+
if (labels.length > 0) query.set("labels", labels.join(","));
|
|
186
|
+
|
|
187
|
+
const url = `https://github.com/${repo}/issues/new?${query.toString()}`;
|
|
188
|
+
return url.length > MAX_URL ? undefined : url;
|
|
189
|
+
}
|
package/src/sync.ts
ADDED
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
import { readFile } from "node:fs/promises";
|
|
2
|
+
import type { IndexedChunk, Store } from "./db.js";
|
|
3
|
+
import { type DiscoveredPackage, discoverPackages } from "./discovery.js";
|
|
4
|
+
import { chunkId, estimateTokens, resolveChunkFile } from "./spec.js";
|
|
5
|
+
|
|
6
|
+
export interface SyncOptions {
|
|
7
|
+
readonly cwd: string;
|
|
8
|
+
readonly store: Store;
|
|
9
|
+
/** Re-index packages that are already in the store. */
|
|
10
|
+
readonly force?: boolean;
|
|
11
|
+
readonly onProgress?: (message: string) => void;
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
export type SyncStatus = "indexed" | "cached";
|
|
15
|
+
|
|
16
|
+
export interface SyncedPackage {
|
|
17
|
+
readonly id: string;
|
|
18
|
+
readonly name: string;
|
|
19
|
+
readonly version: string;
|
|
20
|
+
readonly chunks: number;
|
|
21
|
+
readonly tokens: number;
|
|
22
|
+
readonly status: SyncStatus;
|
|
23
|
+
readonly trusted: boolean;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
export interface SyncResult {
|
|
27
|
+
readonly packages: readonly SyncedPackage[];
|
|
28
|
+
readonly problems: readonly string[];
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* Indexes every documentation package this project depends on. Packages already present in the
|
|
33
|
+
* global store are left alone, so the same `@stripe/docspack@2025.4.1` is only read once per
|
|
34
|
+
* machine rather than once per project.
|
|
35
|
+
*/
|
|
36
|
+
export async function syncProject(options: SyncOptions): Promise<SyncResult> {
|
|
37
|
+
const { packages, problems: discoveryProblems } = await discoverPackages(options.cwd);
|
|
38
|
+
const problems = [...discoveryProblems];
|
|
39
|
+
const synced: SyncedPackage[] = [];
|
|
40
|
+
|
|
41
|
+
for (const pkg of packages) {
|
|
42
|
+
if (options.force !== true && options.store.hasPackage(pkg.id)) {
|
|
43
|
+
synced.push({
|
|
44
|
+
id: pkg.id,
|
|
45
|
+
name: pkg.name,
|
|
46
|
+
version: pkg.version,
|
|
47
|
+
chunks: options.store.countChunks(pkg.id),
|
|
48
|
+
tokens: 0,
|
|
49
|
+
status: "cached",
|
|
50
|
+
trusted: pkg.trusted,
|
|
51
|
+
});
|
|
52
|
+
continue;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
options.onProgress?.(`indexing ${pkg.id}`);
|
|
56
|
+
const { chunks, problems: chunkProblems } = await readChunks(pkg);
|
|
57
|
+
problems.push(...chunkProblems);
|
|
58
|
+
|
|
59
|
+
if (chunks.length === 0) {
|
|
60
|
+
problems.push(`${pkg.name}: no readable chunks, nothing indexed`);
|
|
61
|
+
continue;
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
options.store.indexPackage({ id: pkg.id, name: pkg.name, version: pkg.version }, chunks);
|
|
65
|
+
synced.push({
|
|
66
|
+
id: pkg.id,
|
|
67
|
+
name: pkg.name,
|
|
68
|
+
version: pkg.version,
|
|
69
|
+
chunks: chunks.length,
|
|
70
|
+
tokens: chunks.reduce((total, chunk) => total + chunk.tokens, 0),
|
|
71
|
+
status: "indexed",
|
|
72
|
+
trusted: pkg.trusted,
|
|
73
|
+
});
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
return { packages: synced, problems };
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
async function readChunks(
|
|
80
|
+
pkg: DiscoveredPackage,
|
|
81
|
+
): Promise<{ chunks: IndexedChunk[]; problems: string[] }> {
|
|
82
|
+
const chunks: IndexedChunk[] = [];
|
|
83
|
+
const problems: string[] = [];
|
|
84
|
+
|
|
85
|
+
for (const chunk of pkg.manifest.chunks) {
|
|
86
|
+
let content: string;
|
|
87
|
+
try {
|
|
88
|
+
content = await readFile(resolveChunkFile(pkg.llmsDir, chunk.file), "utf8");
|
|
89
|
+
} catch (error) {
|
|
90
|
+
problems.push(
|
|
91
|
+
`${pkg.name} chunk "${chunk.id}": ${error instanceof Error ? error.message : String(error)}`,
|
|
92
|
+
);
|
|
93
|
+
continue;
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
const text = content.trim();
|
|
97
|
+
if (text.length === 0) {
|
|
98
|
+
problems.push(`${pkg.name} chunk "${chunk.id}" is empty`);
|
|
99
|
+
continue;
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
chunks.push({
|
|
103
|
+
chunkId: chunkId(pkg.id, chunk.id),
|
|
104
|
+
filePath: chunk.file,
|
|
105
|
+
tokens: chunk.tokens > 0 ? chunk.tokens : estimateTokens(text),
|
|
106
|
+
content: text,
|
|
107
|
+
tags: [...chunk.tags, ...chunk.entities],
|
|
108
|
+
});
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
return { chunks, problems };
|
|
112
|
+
}
|