docspack 0.4.0 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. package/README.md +31 -0
  2. package/dist/agent.d.ts +48 -0
  3. package/dist/agent.d.ts.map +1 -0
  4. package/dist/agent.js +243 -0
  5. package/dist/agent.js.map +1 -0
  6. package/dist/artifact.d.ts +32 -0
  7. package/dist/artifact.d.ts.map +1 -0
  8. package/dist/artifact.js +78 -0
  9. package/dist/artifact.js.map +1 -0
  10. package/dist/build.d.ts +23 -0
  11. package/dist/build.d.ts.map +1 -1
  12. package/dist/build.js +201 -96
  13. package/dist/build.js.map +1 -1
  14. package/dist/changed.d.ts +31 -0
  15. package/dist/changed.d.ts.map +1 -0
  16. package/dist/changed.js +71 -0
  17. package/dist/changed.js.map +1 -0
  18. package/dist/cli.js +222 -12
  19. package/dist/cli.js.map +1 -1
  20. package/dist/coverage.d.ts +35 -0
  21. package/dist/coverage.d.ts.map +1 -0
  22. package/dist/coverage.js +64 -0
  23. package/dist/coverage.js.map +1 -0
  24. package/dist/db.d.ts +75 -2
  25. package/dist/db.d.ts.map +1 -1
  26. package/dist/db.js +168 -6
  27. package/dist/db.js.map +1 -1
  28. package/dist/discovery.d.ts +14 -0
  29. package/dist/discovery.d.ts.map +1 -1
  30. package/dist/discovery.js +31 -6
  31. package/dist/discovery.js.map +1 -1
  32. package/dist/doctor.d.ts.map +1 -1
  33. package/dist/doctor.js +60 -9
  34. package/dist/doctor.js.map +1 -1
  35. package/dist/endpoints.d.ts +45 -0
  36. package/dist/endpoints.d.ts.map +1 -0
  37. package/dist/endpoints.js +155 -0
  38. package/dist/endpoints.js.map +1 -0
  39. package/dist/help.d.ts.map +1 -1
  40. package/dist/help.js +120 -5
  41. package/dist/help.js.map +1 -1
  42. package/dist/index.d.ts +10 -4
  43. package/dist/index.d.ts.map +1 -1
  44. package/dist/index.js +10 -4
  45. package/dist/index.js.map +1 -1
  46. package/dist/local.d.ts +67 -0
  47. package/dist/local.d.ts.map +1 -0
  48. package/dist/local.js +242 -0
  49. package/dist/local.js.map +1 -0
  50. package/dist/mcp.d.ts.map +1 -1
  51. package/dist/mcp.js +3 -0
  52. package/dist/mcp.js.map +1 -1
  53. package/dist/preview.d.ts.map +1 -1
  54. package/dist/preview.js +26 -7
  55. package/dist/preview.js.map +1 -1
  56. package/dist/search.d.ts +22 -1
  57. package/dist/search.d.ts.map +1 -1
  58. package/dist/search.js +147 -22
  59. package/dist/search.js.map +1 -1
  60. package/dist/surface.d.ts +45 -0
  61. package/dist/surface.d.ts.map +1 -0
  62. package/dist/surface.js +208 -0
  63. package/dist/surface.js.map +1 -0
  64. package/dist/sync.d.ts +8 -1
  65. package/dist/sync.d.ts.map +1 -1
  66. package/dist/sync.js +50 -1
  67. package/dist/sync.js.map +1 -1
  68. package/dist/verify.d.ts +9 -0
  69. package/dist/verify.d.ts.map +1 -1
  70. package/dist/verify.js +1 -1
  71. package/dist/verify.js.map +1 -1
  72. package/package.json +7 -5
  73. package/src/agent.ts +308 -0
  74. package/src/artifact.ts +110 -0
  75. package/src/build.ts +240 -111
  76. package/src/changed.ts +99 -0
  77. package/src/cli.ts +263 -12
  78. package/src/coverage.ts +96 -0
  79. package/src/db.ts +250 -7
  80. package/src/discovery.ts +40 -5
  81. package/src/doctor.ts +68 -8
  82. package/src/endpoints.ts +184 -0
  83. package/src/help.ts +120 -5
  84. package/src/index.ts +52 -1
  85. package/src/local.ts +335 -0
  86. package/src/mcp.ts +3 -0
  87. package/src/preview.ts +38 -13
  88. package/src/search.ts +203 -23
  89. package/src/surface.ts +265 -0
  90. package/src/sync.ts +66 -2
  91. package/src/verify.ts +1 -1
package/src/build.ts CHANGED
@@ -1,6 +1,16 @@
1
+ import { createHash } from "node:crypto";
1
2
  import type { Dirent } from "node:fs";
2
3
  import { mkdir, readdir, readFile, rm, writeFile } from "node:fs/promises";
3
- import { join, relative, sep } from "node:path";
4
+ import { basename, join, relative, sep } from "node:path";
5
+ import { readApiFile } from "@docspack/lapis/read";
6
+ import {
7
+ type ApiDocument,
8
+ OpenApiError,
9
+ operationDigest,
10
+ operationEntities,
11
+ operationTags,
12
+ operationTitle,
13
+ } from "@docspack/openapi";
4
14
  import { findEntry } from "@docspack/registry";
5
15
  import { readBuildConfig } from "./config.js";
6
16
  import { cleanDocument } from "./document.js";
@@ -33,6 +43,17 @@ export interface BuildOptions {
33
43
  readonly openapi?: string;
34
44
  /** Registry id or llms.txt URL to fetch and package. */
35
45
  readonly source?: string;
46
+ /**
47
+ * JSON records to package, or `-` for standard input. One chunk per record, so anything a query
48
+ * can produce rows for — a table, an export, an API response — reaches the index without this
49
+ * tool needing a driver for it.
50
+ */
51
+ readonly json?: string;
52
+ /**
53
+ * Build a working corpus rather than a publishable package: identity is derived instead of
54
+ * demanded, and no `package.json` is written.
55
+ */
56
+ readonly local?: boolean;
36
57
  readonly pages?: number;
37
58
  readonly maxChunkTokens?: number;
38
59
  /**
@@ -58,6 +79,16 @@ export interface BuildResult {
58
79
 
59
80
  /** A document before it is split into chunks. */
60
81
  interface SourceDocument {
82
+ /**
83
+ * Chunk id to use instead of one derived from the title. Atomic documents only.
84
+ *
85
+ * A derived id is a slug of the prose, which is right for a Markdown page — the heading is the
86
+ * stable thing about it. It is wrong for a generated reference, where the title carries a summary
87
+ * somebody will reword: `POST /v1/charges — Create a charge` becoming `…— Creates a charge`
88
+ * renames the chunk, and a chunk id is what `feedback add --chunk` pins and what a published link
89
+ * resolves to.
90
+ */
91
+ readonly id?: string;
61
92
  readonly title: string;
62
93
  /** File path or URL the text came from, recorded in the chunk for provenance. */
63
94
  readonly origin: string;
@@ -71,8 +102,8 @@ interface SourceDocument {
71
102
  }
72
103
 
73
104
  const DEFAULT_MAX_CHUNK_TOKENS = 800;
74
- const MARKDOWN = /\.(md|mdx|markdown)$/i;
75
- const HTTP_METHODS = ["get", "post", "put", "patch", "delete", "head", "options"] as const;
105
+ /** The source extensions a corpus is collected from. */
106
+ export const MARKDOWN = /\.(md|mdx|markdown)$/i;
76
107
 
77
108
  /**
78
109
  * Generates a docs package: `.llms/manifest.json`, `.llms/chunks/*.md` and an `llms.txt` table of
@@ -202,6 +233,16 @@ async function resolveIdentity(
202
233
  const version =
203
234
  options.version ?? (typeof existing.version === "string" ? existing.version : undefined);
204
235
 
236
+ // A working corpus has no publisher to name it and no release to version, so demanding both is
237
+ // friction with nothing on the other side of it. A publishable package still fails below.
238
+ if (options.local === true) {
239
+ return {
240
+ name: name ?? localPackageName(options),
241
+ version: version ?? LOCAL_VERSION,
242
+ createPackageJson: false,
243
+ };
244
+ }
245
+
205
246
  if (name === undefined) {
206
247
  throw new DocspackError("The package needs a name", {
207
248
  hint: "Pass --name @vendor/docspack, or run inside a directory that has a package.json.",
@@ -214,17 +255,134 @@ async function resolveIdentity(
214
255
  return { name, version, createPackageJson: !found };
215
256
  }
216
257
 
258
+ /**
259
+ * A stable version for a working corpus, rather than a hash of its contents.
260
+ *
261
+ * A content hash would look like the careful choice and is the wrong one: it changes the store key
262
+ * on every edit, so each save leaves the previous index behind as garbage that nothing evicts.
263
+ * Freshness is a property of each source, tracked per source, not of the corpus's name.
264
+ */
265
+ export const LOCAL_VERSION = "0.0.0";
266
+
267
+ /** Named after where the text came from, which is the only thing that distinguishes one corpus. */
268
+ export function localPackageName(options: Pick<BuildOptions, "from" | "json">): string {
269
+ const source =
270
+ options.from ?? (options.json !== undefined && options.json !== "-" ? options.json : undefined);
271
+ const base = source === undefined ? "corpus" : basename(source).replace(/\.[^.]+$/, "");
272
+ return `@local/${slugify(base, "corpus")}`;
273
+ }
274
+
275
+ /**
276
+ * Records as JSON, from a file or standard input: `[{ "id", "title", "text", "tags", "entities" }]`.
277
+ *
278
+ * This is the whole of the non-file input, and it is deliberately not a database adapter. A driver
279
+ * would put a native dependency into a CLI whose startup cost is the reason it is usable from an
280
+ * agent at all, and every store worth indexing can already emit JSON — `psql --json`,
281
+ * `sqlite3 -json`, an API response saved to a file. The conversion belongs to whoever owns the
282
+ * query.
283
+ *
284
+ * A record carrying an `id` is atomic: its id is its key, and splitting it would either duplicate
285
+ * that key across sections or discard it. A record without one is split like any other document.
286
+ */
287
+ async function collectFromJson(source: string): Promise<SourceDocument[]> {
288
+ const origin = source === "-" ? "stdin" : source;
289
+ const raw = source === "-" ? await readStdin() : await readFile(source, "utf8");
290
+
291
+ let parsed: unknown;
292
+ try {
293
+ parsed = JSON.parse(raw);
294
+ } catch (error) {
295
+ throw new DocspackError(`${origin} is not valid JSON`, {
296
+ hint: 'Expected an array of records: [{ "title": "…", "text": "…" }].',
297
+ cause: error,
298
+ });
299
+ }
300
+ if (!Array.isArray(parsed)) {
301
+ throw new DocspackError(`${origin} is not a JSON array`, {
302
+ hint: 'Expected [{ "title": "…", "text": "…" }], one entry per record.',
303
+ });
304
+ }
305
+ if (parsed.length === 0) {
306
+ throw new DocspackError(`${origin} contains no records`, {
307
+ hint: "A query that returned nothing indexes nothing; check the query first.",
308
+ });
309
+ }
310
+
311
+ return parsed.map((entry, index): SourceDocument => {
312
+ const where = `${origin} record ${index}`;
313
+ if (typeof entry !== "object" || entry === null || Array.isArray(entry)) {
314
+ throw new DocspackError(`${where} is not an object`);
315
+ }
316
+ const record = entry as Record<string, unknown>;
317
+ const title = field(record.title, `${where}.title`);
318
+ const text = field(record.text, `${where}.text`);
319
+ const id = record.id === undefined ? undefined : field(record.id, `${where}.id`);
320
+
321
+ return {
322
+ title,
323
+ text,
324
+ origin,
325
+ ...(id === undefined ? {} : { id: slugify(id, `record-${index}`), atomic: true }),
326
+ ...(record.tags === undefined ? {} : { tags: stringList(record.tags, `${where}.tags`) }),
327
+ ...(record.entities === undefined
328
+ ? {}
329
+ : { entities: stringList(record.entities, `${where}.entities`) }),
330
+ };
331
+ });
332
+ }
333
+
334
+ function field(value: unknown, where: string): string {
335
+ if (typeof value !== "string" || value.trim().length === 0) {
336
+ throw new DocspackError(`${where} must be a non-empty string`);
337
+ }
338
+ return value;
339
+ }
340
+
341
+ function stringList(value: unknown, where: string): string[] {
342
+ if (!Array.isArray(value) || value.some((entry) => typeof entry !== "string")) {
343
+ throw new DocspackError(`${where} must be an array of strings`);
344
+ }
345
+ return value as string[];
346
+ }
347
+
348
+ async function readStdin(): Promise<string> {
349
+ const chunks: Buffer[] = [];
350
+ for await (const chunk of process.stdin) chunks.push(Buffer.from(chunk));
351
+ return Buffer.concat(chunks).toString("utf8");
352
+ }
353
+
354
+ /**
355
+ * Prose and an API document, in that order, or a mirror on its own.
356
+ *
357
+ * `--from` and `--openapi` combine, because a library with an HTTP API usually has both and
358
+ * documents them in the same package — refusing the combination meant a vendor had to pick which
359
+ * half of their documentation to publish. A mirror does not combine with either: it is somebody
360
+ * else's whole documentation set, fetched, and merging our own prose into it would produce a package
361
+ * claiming to be a mirror of something it is not.
362
+ *
363
+ * Operations come first so that their ids win a collision. A prose chunk's id is derived and may
364
+ * take a numeric suffix without anything breaking; an operation's is pinned to its endpoint, and a
365
+ * suffix there is the stable address moving.
366
+ */
217
367
  async function collect(options: BuildOptions, warnings: string[]): Promise<SourceDocument[]> {
218
- const chosen = [options.from, options.openapi, options.source].filter(
219
- (value) => value !== undefined,
220
- );
221
- if (chosen.length > 1) {
222
- throw new DocspackError("Choose one input: --from, --openapi or a source");
368
+ if (options.source !== undefined) {
369
+ if (options.from !== undefined || options.openapi !== undefined || options.json !== undefined) {
370
+ throw new DocspackError("A source cannot be combined with --from, --openapi or --from-json", {
371
+ hint: "A mirror is a whole documentation set on its own.",
372
+ });
373
+ }
374
+ return collectFromSource(options, warnings);
223
375
  }
224
376
 
225
- if (options.openapi !== undefined) return collectFromOpenApi(options.openapi);
226
- if (options.source !== undefined) return collectFromSource(options, warnings);
227
- return collectFromDirectory(options.from ?? join(options.out, "docs"));
377
+ const documents: SourceDocument[] = [];
378
+ if (options.openapi !== undefined) documents.push(...(await collectFromOpenApi(options.openapi)));
379
+ if (options.json !== undefined) documents.push(...(await collectFromJson(options.json)));
380
+ // The default `docs/` directory is only read when nothing else was named: a build given only an
381
+ // OpenAPI document or a set of records should not fail on a `docs/` directory that does not exist.
382
+ if (options.from !== undefined || (options.openapi === undefined && options.json === undefined)) {
383
+ documents.push(...(await collectFromDirectory(options.from ?? join(options.out, "docs"))));
384
+ }
385
+ return documents;
228
386
  }
229
387
 
230
388
  async function collectFromDirectory(dir: string): Promise<SourceDocument[]> {
@@ -281,6 +439,13 @@ async function collectFromSource(
281
439
 
282
440
  const documents: SourceDocument[] = [];
283
441
  const links = fetchableLinks(parsed).slice(0, options.pages ?? 50);
442
+ // Two links can name the same document. `fetchableLinks` already drops a repeated URL and a
443
+ // repeated fragment, but a site whose index links anchors as a query — `/v4?id=codecs` — gets
444
+ // one entry per anchor, and each one fetches the whole page again. Left alone that packages
445
+ // the same text dozens of times: an index mostly made of duplicates, which is slower to search
446
+ // and ranks worse, because copies of one page crowd out the rest of the corpus.
447
+ const seen = new Set<string>();
448
+ let duplicates = 0;
284
449
 
285
450
  for (const link of links) {
286
451
  options.onProgress?.(`fetching ${link.url}`);
@@ -293,6 +458,12 @@ async function collectFromSource(
293
458
  warnings.push(`skipped ${link.url}: document was empty`);
294
459
  continue;
295
460
  }
461
+ const fingerprint = createHash("sha256").update(text).digest("hex");
462
+ if (seen.has(fingerprint)) {
463
+ duplicates += 1;
464
+ continue;
465
+ }
466
+ seen.add(fingerprint);
296
467
  documents.push({
297
468
  title: link.title.length > 0 ? link.title : (firstHeading(text) ?? link.url),
298
469
  origin: link.url,
@@ -306,6 +477,11 @@ async function collectFromSource(
306
477
  }
307
478
  }
308
479
 
480
+ if (duplicates > 0) {
481
+ warnings.push(
482
+ `skipped ${duplicates} of ${links.length} linked documents whose text repeated one already packaged`,
483
+ );
484
+ }
309
485
  if (fetchableLinks(parsed).length > links.length) {
310
486
  warnings.push(
311
487
  `packaged ${links.length} of ${fetchableLinks(parsed).length} linked documents (raise with --pages)`,
@@ -314,121 +490,69 @@ async function collectFromSource(
314
490
  return documents;
315
491
  }
316
492
 
493
+ /**
494
+ * One chunk per operation, plus the document's own overview.
495
+ *
496
+ * What each chunk carries is `@docspack/openapi`'s digest: the base URL, the credential, the inputs
497
+ * with their types, the body shape, the response, the failures, the types it reaches, and a runnable
498
+ * cURL call — in LAPIS notation, which is an open format for describing an API to a model. That
499
+ * package's README has the measurement; the short version is that a digest is about 72% smaller than
500
+ * the smallest correct slice of OpenAPI JSON for the same operation, and a rounding error against
501
+ * the whole document, which is what an agent without retrieval is handed instead.
502
+ *
503
+ * This is deliberately not a summary. An agent that retrieved the chunk for `POST /v1/charges` can
504
+ * make the call from it without reading anything else, which is the only test a reference chunk has
505
+ * to pass.
506
+ */
317
507
  async function collectFromOpenApi(file: string): Promise<SourceDocument[]> {
318
- let parsed: unknown;
319
- try {
320
- parsed = JSON.parse(await readFile(file, "utf8"));
321
- } catch (error) {
322
- throw new DocspackError(`Could not read the OpenAPI document ${file}`, {
323
- hint: "Only JSON is supported. Convert YAML with a tool such as `yq -o json`.",
324
- cause: error,
325
- });
326
- }
327
-
328
- const spec = parsed as {
329
- info?: { title?: unknown; description?: unknown; version?: unknown };
330
- paths?: Record<string, unknown>;
331
- };
508
+ // JSON or YAML, decided by what the file contains rather than by its extension. The reader is a
509
+ // separate entry point of `@docspack/lapis` so that only the callers who need a YAML parser load
510
+ // one; this is a CLI, so it is one of them.
511
+ const document = await readOpenApi(file);
332
512
  const documents: SourceDocument[] = [];
333
- const apiTitle = typeof spec.info?.title === "string" ? spec.info.title : "API";
334
513
 
335
- if (typeof spec.info?.description === "string" && spec.info.description.trim().length > 0) {
514
+ if (document.description.length > 0) {
336
515
  documents.push({
337
- title: `${apiTitle} overview`,
516
+ id: "api-overview",
517
+ title: `${document.title} overview`,
338
518
  origin: file,
339
- text: spec.info.description.trim(),
519
+ text: document.description,
340
520
  atomic: true,
341
- tags: ["overview", "introduction"],
521
+ tags: ["overview", "introduction", "api"],
342
522
  });
343
523
  }
344
524
 
345
- for (const [path, item] of Object.entries(spec.paths ?? {})) {
346
- if (typeof item !== "object" || item === null) continue;
347
- const pathItem = item as Record<string, unknown>;
348
-
349
- for (const method of HTTP_METHODS) {
350
- const operation = pathItem[method];
351
- if (typeof operation !== "object" || operation === null) continue;
352
- documents.push(renderOperation(method, path, operation as Record<string, unknown>, file));
353
- }
354
- }
355
-
356
- if (documents.length === 0) {
357
- throw new DocspackError(`No operations found in ${file}`, {
358
- hint: "The document needs a `paths` object with at least one operation.",
525
+ for (const operation of document.operations) {
526
+ documents.push({
527
+ // The endpoint, not the operation id: a vendor renames `createCharge` to `postCharges` and
528
+ // the endpoint is still `POST /v1/charges`. Both are recorded as entities, so either spelling
529
+ // still finds the chunk.
530
+ id: slugify(`${operation.method} ${operation.path}`, "operation"),
531
+ title: operationTitle(operation),
532
+ origin: `${file}#${operation.method} ${operation.path}`,
533
+ text: operationDigest(document, operation),
534
+ atomic: true,
535
+ tags: operationTags(operation),
536
+ entities: operationEntities(operation),
359
537
  });
360
538
  }
539
+
361
540
  return documents;
362
541
  }
363
542
 
364
- function renderOperation(
365
- method: string,
366
- path: string,
367
- operation: Record<string, unknown>,
368
- file: string,
369
- ): SourceDocument {
370
- const summary = typeof operation.summary === "string" ? operation.summary : "";
371
- const description = typeof operation.description === "string" ? operation.description : "";
372
- const operationId = typeof operation.operationId === "string" ? operation.operationId : undefined;
373
- const tags = Array.isArray(operation.tags)
374
- ? operation.tags.filter((tag): tag is string => typeof tag === "string")
375
- : [];
376
-
377
- const lines = [`${method.toUpperCase()} ${path}`, ""];
378
- if (summary.length > 0) lines.push(summary, "");
379
- if (description.length > 0) lines.push(description, "");
380
- if (operationId !== undefined) lines.push(`Operation: \`${operationId}\``, "");
381
-
382
- const parameters = Array.isArray(operation.parameters) ? operation.parameters : [];
383
- if (parameters.length > 0) {
384
- lines.push("## Parameters", "");
385
- for (const raw of parameters) {
386
- if (typeof raw !== "object" || raw === null) continue;
387
- const parameter = raw as {
388
- name?: unknown;
389
- in?: unknown;
390
- required?: unknown;
391
- description?: unknown;
392
- };
393
- if (typeof parameter.name !== "string") continue;
394
- const location = typeof parameter.in === "string" ? parameter.in : "query";
395
- const required = parameter.required === true ? ", required" : "";
396
- const note =
397
- typeof parameter.description === "string" && parameter.description.length > 0
398
- ? ` — ${parameter.description}`
399
- : "";
400
- lines.push(`- \`${parameter.name}\` (${location}${required})${note}`);
401
- }
402
- lines.push("");
403
- }
404
-
405
- const responses = operation.responses;
406
- if (typeof responses === "object" && responses !== null) {
407
- lines.push("## Responses", "");
408
- for (const [status, raw] of Object.entries(responses as Record<string, unknown>)) {
409
- const detail =
410
- typeof raw === "object" &&
411
- raw !== null &&
412
- typeof (raw as { description?: unknown }).description === "string"
413
- ? ` — ${(raw as { description: string }).description}`
414
- : "";
415
- lines.push(`- \`${status}\`${detail}`);
543
+ /** The reader's failures, re-reported as the CLI's own error so the hint survives. */
544
+ async function readOpenApi(file: string): Promise<ApiDocument> {
545
+ try {
546
+ return (await readApiFile(file)).document;
547
+ } catch (error) {
548
+ if (error instanceof OpenApiError) {
549
+ throw new DocspackError(error.message, {
550
+ ...(error.hint === undefined ? {} : { hint: error.hint }),
551
+ cause: error,
552
+ });
416
553
  }
417
- lines.push("");
554
+ throw error;
418
555
  }
419
-
420
- return {
421
- title: summary.length > 0 ? summary : `${method.toUpperCase()} ${path}`,
422
- origin: `${file}#${method} ${path}`,
423
- text: lines.join("\n").trim(),
424
- atomic: true,
425
- tags: [
426
- method,
427
- ...tags,
428
- ...path.split("/").filter((part) => part.length > 0 && !part.startsWith("{")),
429
- ],
430
- ...(operationId === undefined ? {} : { entities: [operationId] }),
431
- };
432
556
  }
433
557
 
434
558
  function chunkDocuments(
@@ -455,7 +579,12 @@ function chunkDocuments(
455
579
  for (const section of sections) {
456
580
  const directives = readDirectives(section.body);
457
581
  const body = withoutDuplicateHeading(directives.body, section.heading);
458
- const derived = chunkSlug(document.title, section.heading);
582
+ // Only for an atomic document: a pinned id on a document that splits would give every
583
+ // section the same id.
584
+ const derived =
585
+ document.atomic === true && document.id !== undefined
586
+ ? document.id
587
+ : chunkSlug(document.title, section.heading);
459
588
  const id = uniqueId(derived, taken);
460
589
  if (id !== derived) collisions.push(derived);
461
590
  const libraries = directives.documents.length > 0 ? directives.documents : document.documents;
package/src/changed.ts ADDED
@@ -0,0 +1,99 @@
1
+ import { identifiers } from "./coverage.js";
2
+ import type { Store } from "./db.js";
3
+ import { discoverLibraries, discoverPackages } from "./discovery.js";
4
+ import { DocspackError } from "./errors.js";
5
+ import { packageId } from "./spec.js";
6
+
7
+ export interface SurfaceChange {
8
+ readonly library: string;
9
+ readonly from: string;
10
+ readonly to: string;
11
+ /** Names the newer release exports that the older one did not. */
12
+ readonly added: readonly string[];
13
+ /** Names the older release exported that the newer one does not. */
14
+ readonly removed: readonly string[];
15
+ /**
16
+ * Added names that no documentation package installed here mentions. These are the ones a
17
+ * model cannot know: too new for its training data, and absent from what the vendor published.
18
+ */
19
+ readonly undocumented: readonly string[];
20
+ }
21
+
22
+ export interface ChangedOptions {
23
+ readonly cwd: string;
24
+ readonly store: Store;
25
+ /** Library name, optionally with the version to compare against: `hono` or `hono@4.0.0`. */
26
+ readonly library: string;
27
+ }
28
+
29
+ /**
30
+ * What changed in a library's exported surface between two versions in the store.
31
+ *
32
+ * Both versions come from the local index, never the network: the store is global, so a version
33
+ * indexed for another project on this machine is already here. Measured across two libraries,
34
+ * upgrades are overwhelmingly additive — hono 3.12.12 to 4.13.3 removed three names and added
35
+ * 243 — so the useful report is what exists now that an older release did not have.
36
+ */
37
+ export async function changedSurface(options: ChangedOptions): Promise<SurfaceChange> {
38
+ const [name, requested] = split(options.library);
39
+
40
+ const installed = (await discoverLibraries(options.cwd)).find((library) => library.name === name);
41
+ const indexed = options.store.versionsOf(name).filter((pkg) => pkg.kind === "artifact");
42
+ if (indexed.length === 0) {
43
+ throw new DocspackError(`No version of ${name} is indexed`, {
44
+ hint:
45
+ installed === undefined
46
+ ? `${name} is not a dependency of this project.`
47
+ : "Run `docspack sync` to index the installed version, then ask again.",
48
+ });
49
+ }
50
+
51
+ const to = installed?.version ?? indexed[indexed.length - 1]?.version;
52
+ if (to === undefined) throw new DocspackError(`No version of ${name} is indexed`);
53
+
54
+ const others = indexed.map((pkg) => pkg.version).filter((version) => version !== to);
55
+ const from = requested ?? others[others.length - 1];
56
+ if (from === undefined) {
57
+ throw new DocspackError(`Only ${name}@${to} is indexed, so there is nothing to compare`, {
58
+ hint: "Index another version by running `docspack sync` in a project that installs it. The store is shared across projects on this machine.",
59
+ });
60
+ }
61
+ if (!options.store.hasPackage(packageId(name, from))) {
62
+ throw new DocspackError(`${name}@${from} is not in the store`, {
63
+ hint: `Indexed versions: ${indexed.map((pkg) => pkg.version).join(", ")}`,
64
+ });
65
+ }
66
+
67
+ const before = new Set(options.store.symbolsOf(packageId(name, from)));
68
+ const after = new Set(options.store.symbolsOf(packageId(name, to)));
69
+ const added = [...after].filter((symbol) => !before.has(symbol)).sort();
70
+ const removed = [...before].filter((symbol) => !after.has(symbol)).sort();
71
+
72
+ const documented = await documentedNames(options);
73
+ return {
74
+ library: name,
75
+ from,
76
+ to,
77
+ added,
78
+ removed,
79
+ undocumented: added.filter((symbol) => !documented.has(symbol)),
80
+ };
81
+ }
82
+
83
+ /** Every identifier any documentation package installed here mentions. */
84
+ async function documentedNames(options: ChangedOptions): Promise<Set<string>> {
85
+ const names = new Set<string>();
86
+ for (const pkg of (await discoverPackages(options.cwd)).packages) {
87
+ for (const chunk of options.store.chunksOf(pkg.id)) {
88
+ for (const identifier of identifiers(chunk.content)) names.add(identifier);
89
+ }
90
+ }
91
+ return names;
92
+ }
93
+
94
+ /** `hono@4.0.0` into its parts, keeping a scope's leading `@` out of it. */
95
+ function split(spec: string): [name: string, version: string | undefined] {
96
+ const at = spec.lastIndexOf("@");
97
+ if (at <= 0) return [spec, undefined];
98
+ return [spec.slice(0, at), spec.slice(at + 1)];
99
+ }