create-zudo-doc 3.3.0 → 4.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,430 +0,0 @@
1
- #!/usr/bin/env tsx
2
- /**
3
- * tags:suggest — optional local-LLM tag suggester.
4
- *
5
- * Reads one or more doc files, asks a local Ollama instance for up to 3 tag
6
- * ids from the project vocabulary, and either opens an interactive approval
7
- * prompt (default) or appends suggestions to `.tag-suggestions.jsonl`
8
- * (`--batch`, also used automatically when stdout is not a TTY).
9
- *
10
- * Entirely developer-opt-in. Never runs in CI; never wired into b4push.
11
- */
12
- import { appendFile, readFile, writeFile } from "node:fs/promises";
13
- import { existsSync } from "node:fs";
14
- import { isAbsolute, relative, resolve } from "node:path";
15
- import { parseArgs } from "node:util";
16
- import matter from "gray-matter";
17
-
18
- import { tagVocabulary } from "../src/config/tag-vocabulary";
19
- // Minimal-scaffold cutover (epic zudolab/zudo-doc#2651): the standalone
20
- // `tag-vocabulary-types.ts` shim is gone — the type ships from the package.
21
- import type { TagVocabularyEntry } from "@takazudo/zudo-doc/settings";
22
-
23
- const DEFAULT_HOST = "http://localhost:11434";
24
- const DEFAULT_MODEL = "qwen2.5:7b";
25
- const BODY_CHAR_LIMIT = 1500;
26
- const REQUEST_TIMEOUT_MS = 60_000;
27
- const BATCH_FILE = ".tag-suggestions.jsonl";
28
-
29
- interface Args {
30
- files: string[];
31
- host: string;
32
- model: string;
33
- batch: boolean;
34
- help: boolean;
35
- }
36
-
37
- interface Suggestion {
38
- file: string;
39
- current: string[];
40
- suggested: string[];
41
- }
42
-
43
- function printHelp(): void {
44
- process.stdout.write(`Usage: pnpm tags:suggest [options] <file...>
45
-
46
- Ask a local Ollama LLM to suggest up to 3 tag ids from the project
47
- vocabulary for each doc file. Opt-in developer tool — never runs in CI.
48
-
49
- Options:
50
- --host <url> Ollama endpoint (default: ${DEFAULT_HOST})
51
- --model <id> Ollama model id (default: ${DEFAULT_MODEL})
52
- --batch Append suggestions to ${BATCH_FILE} instead of prompting.
53
- Auto-enabled when stdout is not a TTY.
54
- --help Show this help.
55
-
56
- Ollama setup:
57
- 1. Install from https://ollama.com/
58
- 2. Pull a model: \`ollama pull ${DEFAULT_MODEL}\`
59
- 3. Ensure the daemon is running at ${DEFAULT_HOST}
60
-
61
- Model trade-offs:
62
- - \`qwen2.5:7b\` (default) — balanced quality/speed; ~5 GB download.
63
- - \`llama3.1:8b\` — similar size, stronger English reasoning.
64
- - \`qwen2.5:3b\` / \`llama3.2:3b\` — smaller/faster, noisier output.
65
- Pass \`--model\` to override.
66
-
67
- Exit codes:
68
- 0 success
69
- 1 usage error
70
- 2 Ollama unreachable or returned unusable output
71
- `);
72
- }
73
-
74
- function parseCliArgs(argv: string[]): Args {
75
- let parsed;
76
- try {
77
- parsed = parseArgs({
78
- args: argv,
79
- allowPositionals: true,
80
- options: {
81
- host: { type: "string", default: DEFAULT_HOST },
82
- model: { type: "string", default: DEFAULT_MODEL },
83
- batch: { type: "boolean", default: false },
84
- help: { type: "boolean", default: false },
85
- },
86
- });
87
- } catch (err) {
88
- const msg = err instanceof Error ? err.message : String(err);
89
- process.stderr.write(`tags:suggest: ${msg}\n`);
90
- process.exit(1);
91
- }
92
- return {
93
- files: parsed.positionals,
94
- host: String(parsed.values.host ?? DEFAULT_HOST),
95
- model: String(parsed.values.model ?? DEFAULT_MODEL),
96
- batch: Boolean(parsed.values.batch),
97
- help: Boolean(parsed.values.help),
98
- };
99
- }
100
-
101
- function activeVocabulary(): TagVocabularyEntry[] {
102
- return tagVocabulary.filter((entry) => {
103
- const d = entry.deprecated;
104
- // Exclude fully-retired tags. Redirect-style deprecation still points
105
- // at a live canonical id, which the model may surface if relevant.
106
- if (d === true) return false;
107
- if (typeof d === "object" && d !== null && !("redirect" in d)) return false;
108
- return true;
109
- });
110
- }
111
-
112
- function buildPrompt(
113
- entries: TagVocabularyEntry[],
114
- title: string,
115
- body: string,
116
- ): string {
117
- const vocabLines = entries
118
- .map((e) => {
119
- const label = e.label ?? e.id;
120
- const desc = e.description ?? "";
121
- const group = e.group ? ` [${e.group}]` : "";
122
- return `- ${e.id}${group} — ${label}: ${desc}`.trim();
123
- })
124
- .join("\n");
125
- const snippet = body.slice(0, BODY_CHAR_LIMIT);
126
- return `You are tagging a documentation page.
127
-
128
- Vocabulary (pick ONLY from these ids):
129
- ${vocabLines}
130
-
131
- Document title: ${title}
132
-
133
- Document excerpt:
134
- """
135
- ${snippet}
136
- """
137
-
138
- Return a JSON array of AT MOST 3 tag ids that best fit this page. The
139
- array must contain only strings that exactly match ids from the
140
- vocabulary above. Respond with the JSON array and nothing else.`;
141
- }
142
-
143
- interface OllamaGenerateResponse {
144
- response?: string;
145
- error?: string;
146
- }
147
-
148
- async function callOllama(
149
- host: string,
150
- model: string,
151
- prompt: string,
152
- ): Promise<string> {
153
- const url = `${host.replace(/\/$/, "")}/api/generate`;
154
- const controller = new AbortController();
155
- const timer = setTimeout(() => controller.abort(), REQUEST_TIMEOUT_MS);
156
-
157
- let res: Response;
158
- try {
159
- res = await fetch(url, {
160
- method: "POST",
161
- headers: { "content-type": "application/json" },
162
- body: JSON.stringify({
163
- model,
164
- prompt,
165
- stream: false,
166
- format: "json",
167
- }),
168
- signal: controller.signal,
169
- });
170
- } catch (err) {
171
- clearTimeout(timer);
172
- if ((err as Error)?.name === "AbortError") {
173
- throw new OllamaError(
174
- `Ollama request timed out after ${REQUEST_TIMEOUT_MS / 1000}s at ${host}. Is \`ollama serve\` running?`,
175
- );
176
- }
177
- // Any other fetch rejection (ECONNREFUSED, UND_ERR_CONNECT_TIMEOUT,
178
- // DNS failure, bad port, …) means we could not talk to Ollama. Collapse
179
- // them into one actionable message — the full stack isn't useful to
180
- // the doc author running this CLI.
181
- throw new OllamaUnreachableError(host, model);
182
- }
183
- clearTimeout(timer);
184
-
185
- if (!res.ok) {
186
- let detail = "";
187
- try {
188
- const body = (await res.json()) as OllamaGenerateResponse;
189
- if (body?.error) detail = ` — ${body.error}`;
190
- } catch {
191
- // ignore; keep status line only
192
- }
193
- if (res.status === 404) {
194
- throw new OllamaError(
195
- `Ollama responded 404 at ${host}. Pull the model first: \`ollama pull ${model}\`.${detail}`,
196
- );
197
- }
198
- throw new OllamaError(
199
- `Ollama responded HTTP ${res.status} at ${host}.${detail}`,
200
- );
201
- }
202
-
203
- let json: OllamaGenerateResponse;
204
- try {
205
- json = (await res.json()) as OllamaGenerateResponse;
206
- } catch {
207
- throw new OllamaError(`Ollama returned non-JSON response at ${host}.`);
208
- }
209
- if (typeof json.response !== "string") {
210
- throw new OllamaError(`Ollama response missing 'response' field.`);
211
- }
212
- return json.response;
213
- }
214
-
215
- class OllamaError extends Error {
216
- readonly friendly = true;
217
- }
218
- class OllamaUnreachableError extends OllamaError {
219
- constructor(host: string, model: string) {
220
- super(
221
- `Ollama not reachable at ${host}. Install from https://ollama.com/ and run \`ollama pull ${model}\`.`,
222
- );
223
- }
224
- }
225
-
226
- function parseSuggestions(
227
- raw: string,
228
- allowedIds: Set<string>,
229
- ): string[] {
230
- const trimmed = raw.trim();
231
- // Accept either a bare JSON array or a JSON object that wraps one.
232
- // Some models return `{"tags": [...]}` even when asked for an array.
233
- let parsed: unknown;
234
- try {
235
- parsed = JSON.parse(trimmed);
236
- } catch {
237
- // Last-ditch: try to extract the first JSON array substring.
238
- const match = trimmed.match(/\[[\s\S]*\]/);
239
- if (!match) {
240
- throw new OllamaError(
241
- `Model output was not valid JSON. Raw: ${trimmed.slice(0, 200)}`,
242
- );
243
- }
244
- parsed = JSON.parse(match[0]);
245
- }
246
- let arr: unknown[];
247
- if (Array.isArray(parsed)) {
248
- arr = parsed;
249
- } else if (
250
- parsed &&
251
- typeof parsed === "object" &&
252
- Array.isArray((parsed as { tags?: unknown }).tags)
253
- ) {
254
- arr = (parsed as { tags: unknown[] }).tags;
255
- } else {
256
- throw new OllamaError(`Model output is not a JSON array.`);
257
- }
258
- const ids = arr
259
- .filter((v): v is string => typeof v === "string")
260
- .map((v) => v.trim())
261
- .filter((v) => v.length > 0 && allowedIds.has(v));
262
- // Dedupe, cap at 3.
263
- return Array.from(new Set(ids)).slice(0, 3);
264
- }
265
-
266
- function asStringArray(v: unknown): string[] {
267
- if (!Array.isArray(v)) return [];
268
- return v.filter((x): x is string => typeof x === "string");
269
- }
270
-
271
- async function writeFrontmatterTags(
272
- filePath: string,
273
- parsed: matter.GrayMatterFile<string>,
274
- nextTags: string[],
275
- ): Promise<void> {
276
- const data = { ...parsed.data, tags: nextTags };
277
- const rebuilt = matter.stringify(parsed.content, data);
278
- await writeFile(filePath, rebuilt, "utf-8");
279
- }
280
-
281
- async function appendBatch(
282
- repoRoot: string,
283
- entry: Suggestion,
284
- ): Promise<void> {
285
- const outPath = resolve(repoRoot, BATCH_FILE);
286
- await appendFile(outPath, JSON.stringify(entry) + "\n", "utf-8");
287
- }
288
-
289
- class UsageError extends Error {}
290
-
291
- function resolveFiles(repoRoot: string, inputs: string[]): string[] {
292
- const out: string[] = [];
293
- for (const raw of inputs) {
294
- const abs = isAbsolute(raw) ? raw : resolve(repoRoot, raw);
295
- if (!existsSync(abs)) {
296
- throw new UsageError(`file not found: ${raw}`);
297
- }
298
- out.push(abs);
299
- }
300
- return out;
301
- }
302
-
303
- async function main(): Promise<void> {
304
- const args = parseCliArgs(process.argv.slice(2));
305
- if (args.help) {
306
- printHelp();
307
- return;
308
- }
309
- if (args.files.length === 0) {
310
- process.stderr.write(
311
- "tags:suggest: no input files. Pass one or more `.mdx` paths, or run with --help.\n",
312
- );
313
- process.exit(1);
314
- }
315
-
316
- const repoRoot = process.cwd();
317
- const files = resolveFiles(repoRoot, args.files);
318
- const vocab = activeVocabulary();
319
- const allowedIds = new Set(vocab.map((e) => e.id));
320
-
321
- const isTty = Boolean(process.stdout.isTTY);
322
- const batchMode = args.batch || !isTty;
323
-
324
- // Dynamic import so the script's `--help` and arg parsing work even if
325
- // `@inquirer/prompts` is momentarily unavailable in an odd environment.
326
- type CheckboxFn = <V extends string>(config: {
327
- message: string;
328
- choices: { name: string; value: V; checked?: boolean }[];
329
- loop?: boolean;
330
- }) => Promise<V[]>;
331
- let checkbox: CheckboxFn | null = null;
332
- if (!batchMode) {
333
- const mod = (await import("@inquirer/prompts")) as {
334
- checkbox: CheckboxFn;
335
- };
336
- checkbox = mod.checkbox;
337
- }
338
-
339
- for (const filePath of files) {
340
- const rel = relative(repoRoot, filePath);
341
- const source = await readFile(filePath, "utf-8");
342
- const parsed = matter(source);
343
- const title =
344
- typeof parsed.data.title === "string" ? parsed.data.title : rel;
345
- const current = asStringArray(parsed.data.tags);
346
-
347
- let suggested: string[];
348
- try {
349
- const prompt = buildPrompt(vocab, title, parsed.content);
350
- const raw = await callOllama(args.host, args.model, prompt);
351
- suggested = parseSuggestions(raw, allowedIds);
352
- } catch (err) {
353
- if (err instanceof OllamaError) {
354
- process.stderr.write(`tags:suggest: ${err.message}\n`);
355
- process.exit(2);
356
- }
357
- throw err;
358
- }
359
-
360
- if (suggested.length === 0) {
361
- process.stdout.write(`${rel}: no vocabulary-matching suggestions\n`);
362
- continue;
363
- }
364
-
365
- if (batchMode) {
366
- await appendBatch(repoRoot, {
367
- file: rel,
368
- current,
369
- suggested,
370
- });
371
- process.stdout.write(
372
- `${rel}: recorded ${suggested.length} suggestion(s) to ${BATCH_FILE}\n`,
373
- );
374
- continue;
375
- }
376
-
377
- // Interactive approval: pre-check suggestions, show current tags for context.
378
- const combined = Array.from(new Set([...current, ...suggested]));
379
- const choices = combined.map((id) => {
380
- const inSuggestion = suggested.includes(id);
381
- const inCurrent = current.includes(id);
382
- const label =
383
- inSuggestion && inCurrent
384
- ? `${id} (current, also suggested)`
385
- : inSuggestion
386
- ? `${id} (suggested)`
387
- : `${id} (current)`;
388
- return {
389
- name: label,
390
- value: id,
391
- checked: inSuggestion || inCurrent,
392
- };
393
- });
394
- if (!checkbox) {
395
- // Should not happen — we set it above when !batchMode.
396
- throw new Error("interactive prompt loader missing");
397
- }
398
- const picked = await checkbox<string>({
399
- message: `${rel} — pick tags to write:`,
400
- choices,
401
- loop: false,
402
- });
403
- const nextTags = picked.slice();
404
- if (
405
- nextTags.length === current.length &&
406
- nextTags.every((t, i) => t === current[i])
407
- ) {
408
- process.stdout.write(`${rel}: unchanged\n`);
409
- continue;
410
- }
411
- await writeFrontmatterTags(filePath, parsed, nextTags);
412
- process.stdout.write(
413
- `${rel}: wrote tags [${nextTags.join(", ")}]\n`,
414
- );
415
- }
416
- }
417
-
418
- main().catch((err) => {
419
- if (err instanceof OllamaError) {
420
- process.stderr.write(`tags:suggest: ${err.message}\n`);
421
- process.exit(2);
422
- }
423
- if (err instanceof UsageError) {
424
- process.stderr.write(`tags:suggest: ${err.message}\n`);
425
- process.exit(1);
426
- }
427
- const msg = err instanceof Error ? err.message : String(err);
428
- process.stderr.write(`tags:suggest: ${msg}\n`);
429
- process.exit(1);
430
- });