@wgtechlabs/mdd-engine 0.1.0-pr.08f3b80 → 0.1.0-pr.55ddedf

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -85,6 +85,10 @@ import { search } from '@wgtechlabs/mdd-engine/search';
85
85
  const index = createSearchIndex(result.site);
86
86
  const restored = JSON.parse(JSON.stringify(index));
87
87
  console.log(search(restored, 'installation', { limit: 10 }));
88
+ // Multiple heading hits, breadcrumbs, and ranges for UI highlights.
89
+ console.log(search(restored, 'instalation', {
90
+ mode: 'sections', fuzzy: true, limit: 8,
91
+ }));
88
92
  // [{ title, url, section?, excerpt, score }]
89
93
  ```
90
94
 
@@ -129,6 +133,20 @@ Use GitHub-style alerts with `NOTE`, `TIP`, `IMPORTANT`, `WARNING`, or `CAUTION`
129
133
 
130
134
  `:::details[More information]` remains supported and becomes `details`/`summary`; its label is optional, attributes are errors, and its readable Markdown uses a blockquote with a bold label. The old `:::note`, `:::tip`, and `:::warning` directives now fail with a `REMOVED_COMPONENT` migration diagnostic. See [alerts and migration](docs/ALERTS.md) for all five types, theme hooks, and how to preserve custom titles and bodies.
131
135
 
136
+ Document an API endpoint with a leaf directive, then use ordinary Markdown for parameters and request/response examples:
137
+
138
+ ```markdown
139
+ ## Retrieve a widget
140
+
141
+ ::endpoint{method="GET" path="/v1/widgets/{id}"}
142
+
143
+ | Parameter | Type | Description |
144
+ | --- | --- | --- |
145
+ | `id` | string | Widget identifier. |
146
+ ```
147
+
148
+ The required `method` accepts GET, HEAD, POST, PUT, PATCH, DELETE, OPTIONS, TRACE, or CONNECT, normalizing lowercase/mixed case to uppercase. The required `path` starts with a single `/` and contains no whitespace or control characters; braces and query punctuation remain literal text. Labels, extra attributes, and inline/container forms are errors. The engine emits a `div.mdd-endpoint` containing `strong.mdd-endpoint-method.mdd-method-get` (or the corresponding lowercase method) and `code.mdd-endpoint-path`. Normalized Markdown contains `**GET**` followed by the code-formatted path, so the signature stays readable and searchable. Endpoint paths are not rewritten with the documentation base path. This is documentation only: no HTTP requests run. MDD and its themes provide presentation.
149
+
132
150
  Title precedence is frontmatter title, first H1, then readable filename. Navigation uses `navTitle` when supplied. Explicit `order` sorts first; remaining siblings sort deterministically by label and path.
133
151
 
134
152
  Local `.md` links, extensionless routes, reference links, and images resolve from their source document. A leading slash addresses the documentation root. `basePath` prefixes public links, including `/docs/` and `/repository/docs/`. When a file-style URL matches both an existing supported asset and a page route, the asset wins: `chart.png` selects the image, while `/chart.png/` explicitly selects the page. If no regular asset exists, dotted page routes still resolve. External links are preserved without network requests. Headings have `mdd-`-prefixed GitHub-style slugs; author links such as `#installation` are rewritten to `#mdd-installation`. Duplicate headings receive `-1`, `-2`, and subsequent suffixes.
@@ -0,0 +1,3 @@
1
+ import type { Nodes, Paragraph } from "mdast";
2
+ /** Endpoint signatures are documentation text, never links or executable requests. */
3
+ export declare function endpointParagraph(node: Nodes, source: string, report: (message: string) => void): Paragraph | undefined;
@@ -0,0 +1,68 @@
1
+ const methods = new Set([
2
+ "GET",
3
+ "HEAD",
4
+ "POST",
5
+ "PUT",
6
+ "PATCH",
7
+ "DELETE",
8
+ "OPTIONS",
9
+ "TRACE",
10
+ "CONNECT",
11
+ ]);
12
+ /** Endpoint signatures are documentation text, never links or executable requests. */
13
+ export function endpointParagraph(node, source, report) {
14
+ if (node.type !== "leafDirective") {
15
+ report('Use the leaf form ::endpoint{method="GET" path="/example"}.');
16
+ return;
17
+ }
18
+ const offset = node.position?.start.offset;
19
+ if (node.children.length ||
20
+ (offset !== undefined && source.startsWith("::endpoint[", offset))) {
21
+ report("The endpoint component does not support a label.");
22
+ return;
23
+ }
24
+ const attributes = node.attributes ?? {};
25
+ if (Object.keys(attributes).some((key) => key !== "method" && key !== "path")) {
26
+ report("The endpoint component supports only method and path attributes.");
27
+ return;
28
+ }
29
+ const method = (attributes.method ?? "").toUpperCase();
30
+ if (!methods.has(method)) {
31
+ report("The endpoint method must be GET, HEAD, POST, PUT, PATCH, DELETE, OPTIONS, TRACE, or CONNECT.");
32
+ return;
33
+ }
34
+ const path = attributes.path ?? "";
35
+ if (!path.startsWith("/") ||
36
+ path.startsWith("//") ||
37
+ // biome-ignore lint/suspicious/noControlCharactersInRegex: Endpoint text cannot contain invisible controls or whitespace.
38
+ /[\s\u0000-\u001f\u007f-\u009f]/u.test(path)) {
39
+ report("The endpoint path must start with a single / and contain no whitespace or control characters.");
40
+ return;
41
+ }
42
+ return {
43
+ type: "paragraph",
44
+ position: node.position,
45
+ // Tight lists unwrap ordinary paragraphs; retain the signature's block hook.
46
+ data: { hName: "div", hProperties: { className: ["mdd-endpoint"] } },
47
+ children: [
48
+ {
49
+ type: "strong",
50
+ data: {
51
+ hProperties: {
52
+ className: [
53
+ "mdd-endpoint-method",
54
+ `mdd-method-${method.toLowerCase()}`,
55
+ ],
56
+ },
57
+ },
58
+ children: [{ type: "text", value: method }],
59
+ },
60
+ { type: "text", value: " " },
61
+ {
62
+ type: "inlineCode",
63
+ value: path,
64
+ data: { hProperties: { className: ["mdd-endpoint-path"] } },
65
+ },
66
+ ],
67
+ };
68
+ }
package/dist/index.d.ts CHANGED
@@ -1,5 +1,5 @@
1
1
  import type { CompileOptions, CompileResult } from "./types.js";
2
- export { type SearchIndex, type SearchOptions, type SearchPage, type SearchResult, type SearchSection, search, validateSearchIndex, } from "./search.js";
2
+ export { type SearchIndex, type SearchMatch, type SearchOptions, type SearchPage, type SearchResult, type SearchSection, search, validateSearchIndex, } from "./search.js";
3
3
  export { createSearchIndex } from "./search-index.js";
4
4
  export type { Asset, CompileOptions, CompileResult, Diagnostic, Footer, Heading, Metadata, NavigationItem, Page, Site, SocialLink, Theme, } from "./types.js";
5
5
  /** Compile a local checkout; never fetch repositories, execute author code, or emit files. */
package/dist/markdown.js CHANGED
@@ -13,6 +13,7 @@ import { unified } from "unified";
13
13
  import { SKIP, visit } from "unist-util-visit";
14
14
  import { parseDocument as parseYaml } from "yaml";
15
15
  import { transformAlerts } from "./alerts.js";
16
+ import { endpointParagraph } from "./endpoints.js";
16
17
  const parser = unified()
17
18
  .use(remarkParse)
18
19
  .use(remarkGfm)
@@ -29,10 +30,20 @@ const htmlRenderer = unified()
29
30
  ...defaultSchema.attributes,
30
31
  aside: [["className", /^mdd-/]],
31
32
  details: [["className", "mdd-details"]],
33
+ div: [["className", "mdd-endpoint"]],
32
34
  p: [
33
35
  ...(defaultSchema.attributes?.p ?? []),
34
36
  ["className", "mdd-component-label"],
35
37
  ],
38
+ strong: [
39
+ ...(defaultSchema.attributes?.strong ?? []),
40
+ [
41
+ "className",
42
+ "mdd-endpoint-method",
43
+ /^mdd-method-(get|head|post|put|patch|delete|options|trace|connect)$/,
44
+ ],
45
+ ],
46
+ code: [["className", /^language-./, "mdd-endpoint-path"]],
36
47
  summary: [
37
48
  ...(defaultSchema.attributes?.summary ?? []),
38
49
  ["className", "mdd-component-label"],
@@ -206,8 +217,16 @@ export function parseDocument(source, file, diagnostics) {
206
217
  report("REMOVED_COMPONENT", `The ${node.name} directive was removed; use > [!${node.name.toUpperCase()}] followed by quoted body lines. Keep any optional custom title as bold text in the alert body.`, node);
207
218
  return;
208
219
  }
220
+ if (node.name === "endpoint") {
221
+ const paragraph = endpointParagraph(node, source, (message) => report("INVALID_COMPONENT", message, node));
222
+ if (paragraph && parent && index !== undefined) {
223
+ parent.children[index] = paragraph;
224
+ return [SKIP, index];
225
+ }
226
+ return;
227
+ }
209
228
  if (node.type !== "containerDirective" || node.name !== "details") {
210
- report("UNKNOWN_COMPONENT", `Unsupported component ${node.name}; use a GitHub alert or a details container.`, node);
229
+ report("UNKNOWN_COMPONENT", `Unsupported component ${node.name}; use a GitHub alert, details container, or endpoint leaf.`, node);
211
230
  return;
212
231
  }
213
232
  if (Object.keys(node.attributes ?? {}).length) {
@@ -7,8 +7,9 @@ import { transformAlerts } from "./alerts.js";
7
7
  import { validateSearchIndex, } from "./search.js";
8
8
  const parser = unified().use(remarkParse).use(remarkGfm);
9
9
  const blockTypes = new Set(["paragraph", "code", "tableCell", "listItem"]);
10
- function indexPage(page) {
10
+ function indexPage(page, breadcrumbs) {
11
11
  const sections = [{ title: "", url: page.url, text: "" }];
12
+ const ancestors = [];
12
13
  let headingIndex = 0;
13
14
  const chunks = [[]];
14
15
  const tree = parser.parse(page.markdown);
@@ -23,11 +24,20 @@ function indexPage(page) {
23
24
  heading.depth !== node.depth) {
24
25
  throw new TypeError("Search indexing requires matching compiled Markdown and headings.");
25
26
  }
27
+ while (ancestors.length &&
28
+ (ancestors.at(-1)?.depth ?? 0) >= heading.depth)
29
+ ancestors.pop();
26
30
  sections.push({
27
31
  title: heading.text,
28
32
  url: `${page.url}#${encodeURIComponent(heading.id)}`,
29
33
  text: "",
34
+ breadcrumbs: ancestors
35
+ .filter((ancestor) => !(ancestor.depth === 1 &&
36
+ ancestor.text.normalize("NFKC").toLowerCase() ===
37
+ page.title.normalize("NFKC").toLowerCase()))
38
+ .map((ancestor) => ancestor.text),
30
39
  });
40
+ ancestors.push(heading);
31
41
  chunks.push([]);
32
42
  return SKIP;
33
43
  }
@@ -55,12 +65,33 @@ function indexPage(page) {
55
65
  url: page.url,
56
66
  title: page.title,
57
67
  description: page.description ?? "",
68
+ breadcrumbs,
58
69
  sections,
59
70
  };
60
71
  }
61
72
  /** Build only from successfully compiled pages; no source paths or HTML are indexed. */
62
73
  export function createSearchIndex(site) {
63
- const index = { version: 1, pages: site.pages.map(indexPage) };
74
+ const navigation = new Map();
75
+ const pending = site.navigation.map((item) => ({
76
+ item,
77
+ ancestors: [],
78
+ }));
79
+ while (pending.length) {
80
+ const entry = pending.pop();
81
+ if (!entry)
82
+ break;
83
+ const { item, ancestors } = entry;
84
+ if (ancestors.length > 64)
85
+ throw new TypeError("Search navigation exceeds 64 ancestor labels.");
86
+ if (item.url)
87
+ navigation.set(item.url, ancestors);
88
+ for (const child of item.children ?? [])
89
+ pending.push({ item: child, ancestors: [...ancestors, item.title] });
90
+ }
91
+ const index = {
92
+ version: 1,
93
+ pages: site.pages.map((page) => indexPage(page, navigation.get(page.url) ?? [])),
94
+ };
64
95
  index.pages.sort((a, b) => (a.url < b.url ? -1 : a.url > b.url ? 1 : 0));
65
96
  validateSearchIndex(index);
66
97
  return index;
package/dist/search.d.ts CHANGED
@@ -7,6 +7,8 @@ export interface SearchPage {
7
7
  url: string;
8
8
  title: string;
9
9
  description: string;
10
+ /** Navigation ancestor labels, excluding this page's own label. */
11
+ breadcrumbs?: string[];
10
12
  sections: SearchSection[];
11
13
  }
12
14
  export interface SearchSection {
@@ -14,20 +16,37 @@ export interface SearchSection {
14
16
  title: string;
15
17
  url: string;
16
18
  text: string;
19
+ /** Ancestor headings, excluding this heading and a duplicate page-title H1. */
20
+ breadcrumbs?: string[];
17
21
  }
18
22
  export interface SearchOptions {
19
23
  /** Maximum results, from 0 to 100. Defaults to 10. */
20
24
  limit?: number;
25
+ /** Defaults to one result per page; sections returns independent heading hits. */
26
+ mode?: "pages" | "sections";
27
+ /** Allow one single-edit query correction per result. Requires sections mode. */
28
+ fuzzy?: boolean;
21
29
  }
30
+ /** UTF-16 offsets into the original displayed field; end is exclusive. */
31
+ export type SearchMatch = [start: number, end: number];
22
32
  export interface SearchResult {
33
+ kind: "page" | "section";
23
34
  title: string;
24
35
  url: string;
36
+ pageUrl: string;
37
+ breadcrumbs: string[];
25
38
  /** Present when the destination is a compiled heading. */
26
39
  section?: string;
27
40
  excerpt: string;
28
41
  score: number;
42
+ matches: {
43
+ title: SearchMatch[];
44
+ section: SearchMatch[];
45
+ excerpt: SearchMatch[];
46
+ breadcrumbs: SearchMatch[][];
47
+ };
29
48
  }
30
49
  /** Validate deserialized data before any result can become a link. */
31
50
  export declare function validateSearchIndex(value: unknown): asserts value is SearchIndex;
32
- /** Return at most one hit per page, without filesystem, network, or DOM access. */
51
+ /** Query without filesystem, network, or DOM access; page mode preserves legacy ranking. */
33
52
  export declare function search(index: SearchIndex, query: string, options?: SearchOptions): SearchResult[];
package/dist/search.js CHANGED
@@ -1,6 +1,17 @@
1
1
  function record(value) {
2
2
  return value !== null && typeof value === "object" && !Array.isArray(value);
3
3
  }
4
+ function breadcrumbs(value, maximum) {
5
+ if (value === undefined)
6
+ return true;
7
+ if (!Array.isArray(value) || value.length > maximum)
8
+ return false;
9
+ // Iterate holes too: a sparse array does not contain only string labels.
10
+ for (const label of value)
11
+ if (typeof label !== "string")
12
+ return false;
13
+ return true;
14
+ }
4
15
  function pageUrl(value) {
5
16
  if (typeof value !== "string" ||
6
17
  !value.startsWith("/") ||
@@ -48,6 +59,7 @@ export function validateSearchIndex(value) {
48
59
  urls.has(page.url) ||
49
60
  typeof page.title !== "string" ||
50
61
  typeof page.description !== "string" ||
62
+ !breadcrumbs(page.breadcrumbs, 64) ||
51
63
  !Array.isArray(page.sections))
52
64
  throw new TypeError("Invalid search index page.");
53
65
  urls.add(page.url);
@@ -57,7 +69,8 @@ export function validateSearchIndex(value) {
57
69
  !sectionUrl(section.url, page.url) ||
58
70
  sections.has(section.url) ||
59
71
  typeof section.title !== "string" ||
60
- typeof section.text !== "string")
72
+ typeof section.text !== "string" ||
73
+ !breadcrumbs(section.breadcrumbs, 6))
61
74
  throw new TypeError("Invalid search index section.");
62
75
  sections.add(section.url);
63
76
  }
@@ -70,29 +83,152 @@ function normalize(value) {
70
83
  .replace(/[^\p{L}\p{N}\p{M}_]+/gu, " ")
71
84
  .trim();
72
85
  }
73
- function excerpt(value) {
74
- const characters = Array.from(value.replace(/\s+/gu, " ").trim());
75
- return characters.length > 160
76
- ? `${characters.slice(0, 159).join("")}…`
77
- : characters.join("");
86
+ function contains(value, pattern) {
87
+ return pattern.whole
88
+ ? ` ${value} `.includes(` ${pattern.text} `)
89
+ : value.includes(pattern.text);
78
90
  }
79
- /** Return at most one hit per page, without filesystem, network, or DOM access. */
80
- export function search(index, query, options = {}) {
81
- validateSearchIndex(index);
82
- const limit = options.limit ?? 10;
83
- if (!Number.isInteger(limit) || limit < 0 || limit > 100) {
84
- throw new RangeError("Search limit must be an integer from 0 to 100.");
91
+ function mergeRanges(ranges) {
92
+ ranges.sort((a, b) => a[0] - b[0] || a[1] - b[1]);
93
+ const merged = [];
94
+ for (const range of ranges) {
95
+ const previous = merged.at(-1);
96
+ if (previous && range[0] <= previous[1])
97
+ previous[1] = Math.max(previous[1], range[1]);
98
+ else
99
+ merged.push([...range]);
85
100
  }
86
- if (typeof query !== "string")
87
- throw new TypeError("Search query must be a string.");
88
- if (query.length > 512)
89
- throw new RangeError("Search query must not exceed 512 characters.");
90
- const terms = [...new Set(normalize(query).split(" ").filter(Boolean))];
91
- const phrase = terms.join(" ");
92
- if (terms.length > 32)
93
- throw new RangeError("Search query must not exceed 32 unique terms.");
94
- if (!terms.length || !limit)
101
+ return merged;
102
+ }
103
+ /** Map normalized matches to complete original graphemes, including NFKC expansions. */
104
+ function matchRanges(value, patterns, firstOnly = false) {
105
+ const normalized = value.normalize("NFKC").toLowerCase();
106
+ const found = [];
107
+ for (const pattern of patterns) {
108
+ let offset = 0;
109
+ while (offset < normalized.length) {
110
+ const start = normalized.indexOf(pattern.text, offset);
111
+ if (start < 0)
112
+ break;
113
+ const end = start + pattern.text.length;
114
+ // One neighboring code point needs at most two UTF-16 units. Avoid scanning
115
+ // a growing prefix for each occurrence, or any boundary work for substrings.
116
+ if (!pattern.whole ||
117
+ (!/[\p{L}\p{N}\p{M}_]$/u.test(normalized.slice(Math.max(0, start - 2), start)) &&
118
+ !/^[\p{L}\p{N}\p{M}_]/u.test(normalized.slice(end, end + 2)))) {
119
+ found.push([start, end]);
120
+ if (firstOnly)
121
+ break;
122
+ }
123
+ offset = start + 1;
124
+ }
125
+ }
126
+ let ranges = mergeRanges(found);
127
+ if (!ranges.length)
95
128
  return [];
129
+ if (firstOnly)
130
+ ranges = ranges.slice(0, 1);
131
+ const groups = [];
132
+ for (const { segment, index } of new Intl.Segmenter(undefined, {
133
+ granularity: "grapheme",
134
+ }).segment(value)) {
135
+ let group = {
136
+ text: segment.normalize("NFKC"),
137
+ start: index,
138
+ end: index + segment.length,
139
+ };
140
+ // NFKC can combine adjacent graphemes, e.g. compatibility Hangul ㄱㅏ → 가.
141
+ while (groups.length) {
142
+ const previous = groups.at(-1);
143
+ if (!previous)
144
+ break;
145
+ const joined = previous.text + group.text;
146
+ const combined = joined.normalize("NFKC");
147
+ if (combined === joined)
148
+ break;
149
+ groups.pop();
150
+ group = { text: combined, start: previous.start, end: group.end };
151
+ }
152
+ groups.push(group);
153
+ }
154
+ // Whole-field lowercase retains contextual letters such as final sigma.
155
+ // Per-group lowercase supplies only lengths, including expansions like İ.
156
+ const mapped = [];
157
+ let offset = 0;
158
+ let rangeIndex = 0;
159
+ let originalStart = 0;
160
+ for (const group of groups) {
161
+ const next = offset + group.text.toLowerCase().length;
162
+ let range = ranges[rangeIndex];
163
+ while (range && range[0] < next) {
164
+ if (range[0] >= offset)
165
+ originalStart = group.start;
166
+ if (range[1] > next)
167
+ break;
168
+ mapped.push([originalStart, group.end]);
169
+ range = ranges[++rangeIndex];
170
+ }
171
+ if (!range)
172
+ break;
173
+ offset = next;
174
+ }
175
+ return mergeRanges(mapped);
176
+ }
177
+ function excerpt(value, patterns) {
178
+ const text = value.replace(/\s+/gu, " ").trim();
179
+ const characters = Array.from(text);
180
+ if (characters.length <= 160)
181
+ return text;
182
+ const match = matchRanges(text, patterns, true)[0];
183
+ const matchStart = match ? Array.from(text.slice(0, match[0])).length : 0;
184
+ const matchLength = match
185
+ ? Array.from(text.slice(match[0], match[1])).length
186
+ : 0;
187
+ // Reserve room for both ellipses and the matched term before adding context.
188
+ const context = Math.min(40, Math.max(0, 158 - matchLength));
189
+ const start = Math.min(Math.max(0, matchStart - context), characters.length - 159);
190
+ let end = Math.min(characters.length, start + 160 - (start > 0 ? 1 : 0));
191
+ if (end < characters.length)
192
+ end--;
193
+ const startOffset = characters.slice(0, start).join("").length;
194
+ const endOffset = characters.slice(0, end).join("").length;
195
+ // Move clipped edges inward to complete graphemes without exceeding the cap.
196
+ let safeStart = text.length;
197
+ let safeEnd = 0;
198
+ for (const { segment, index } of new Intl.Segmenter(undefined, {
199
+ granularity: "grapheme",
200
+ }).segment(text)) {
201
+ if (index >= startOffset && safeStart === text.length)
202
+ safeStart = index;
203
+ if (index + segment.length <= endOffset)
204
+ safeEnd = index + segment.length;
205
+ if (index >= endOffset)
206
+ break;
207
+ }
208
+ return `${safeStart > 0 ? "…" : ""}${text.slice(safeStart, safeEnd)}${safeEnd < text.length ? "…" : ""}`;
209
+ }
210
+ function candidateUrl(candidate) {
211
+ return candidate.destination?.url ?? candidate.page.url;
212
+ }
213
+ function compareCandidates(a, b) {
214
+ const aUrl = candidateUrl(a);
215
+ const bUrl = candidateUrl(b);
216
+ return (b.rank - a.rank ||
217
+ b.score - a.score ||
218
+ (aUrl < bUrl ? -1 : aUrl > bUrl ? 1 : 0));
219
+ }
220
+ function sourceText(sources, patterns, fallback) {
221
+ return (sources.find((text) => {
222
+ const normalized = normalize(text);
223
+ return patterns.some((pattern) => contains(normalized, pattern));
224
+ }) ??
225
+ sources.find((text) => text) ??
226
+ fallback);
227
+ }
228
+ /** Preserve page-mode weighting, cross-section coverage, and destination selection. */
229
+ function pageCandidates(index, terms) {
230
+ const phrase = terms.join(" ");
231
+ const patterns = terms.map((text) => ({ text }));
96
232
  const results = [];
97
233
  for (const page of index.pages) {
98
234
  const title = normalize(page.title);
@@ -131,23 +267,219 @@ export function search(index, query, options = {}) {
131
267
  // A stable sort preserves document order when multiple sections tie.
132
268
  matches.sort((a, b) => b.score - a.score);
133
269
  const best = matches[0]?.section;
134
- const destination = metadataMatch ? undefined : best;
135
270
  results.push({
136
- title: page.title,
137
- url: destination?.url ?? page.url,
138
- ...(destination?.title ? { section: destination.title } : {}),
139
- excerpt: excerpt(best?.text ||
140
- page.description ||
141
- page.sections.find((section) => section.text)?.text ||
142
- page.title),
271
+ page,
272
+ destination: metadataMatch ? undefined : best,
273
+ source: sourceText([
274
+ best?.text ?? "",
275
+ page.description,
276
+ ...page.sections.map((section) => section.text),
277
+ ], patterns, page.title),
143
278
  score: weights.reduce((sum, weight) => sum + weight, 0) +
144
279
  (title === phrase
145
280
  ? 24
146
281
  : sections.some((section) => section.title === phrase)
147
282
  ? 12
148
283
  : 0),
284
+ rank: 0,
285
+ patterns,
149
286
  });
150
287
  }
151
- results.sort((a, b) => b.score - a.score || (a.url < b.url ? -1 : a.url > b.url ? 1 : 0));
152
- return results.slice(0, limit);
288
+ return results;
289
+ }
290
+ /** Single insertion, deletion, substitution, or adjacent transposition, on code points. */
291
+ function singleEdit(query, token) {
292
+ if (token.length > 66)
293
+ return false;
294
+ const text = Array.from(token);
295
+ if (Math.abs(query.length - text.length) > 1)
296
+ return false;
297
+ let start = 0;
298
+ while (start < query.length && query[start] === text[start])
299
+ start++;
300
+ if (start === Math.min(query.length, text.length))
301
+ return true;
302
+ if (query.length === text.length) {
303
+ if (query.slice(start + 1).join("") === text.slice(start + 1).join(""))
304
+ return true;
305
+ return (query[start] === text[start + 1] &&
306
+ query[start + 1] === text[start] &&
307
+ query.slice(start + 2).join("") === text.slice(start + 2).join(""));
308
+ }
309
+ return query.length > text.length
310
+ ? query.slice(start + 1).join("") === text.slice(start).join("")
311
+ : query.slice(start).join("") === text.slice(start + 1).join("");
312
+ }
313
+ function fieldScore(fields, pattern) {
314
+ return Math.max(0, ...fields
315
+ .filter((field) => contains(field.text, pattern))
316
+ .map((field) => field.weight));
317
+ }
318
+ function matchFields(fields, terms, fuzzy, required = []) {
319
+ const patterns = terms.map((text) => ({ text }));
320
+ const weights = patterns.map((pattern) => fieldScore(fields, pattern));
321
+ const missing = weights.flatMap((weight, index) => (weight ? [] : [index]));
322
+ if (!missing.length)
323
+ return {
324
+ patterns,
325
+ score: weights.reduce((a, b) => a + b, 0),
326
+ corrected: false,
327
+ };
328
+ const missingIndex = missing[0];
329
+ if (!fuzzy || missing.length !== 1 || missingIndex === undefined)
330
+ return;
331
+ const query = Array.from(terms[missingIndex] ?? "");
332
+ if (query.length < 4 || query.length > 32)
333
+ return;
334
+ const needsOwnCorrection = required.length > 0 &&
335
+ !patterns.some((pattern) => fieldScore(required, pattern));
336
+ let correction;
337
+ let correctionWeight = 0;
338
+ for (const field of fields) {
339
+ if (field.weight < correctionWeight)
340
+ continue;
341
+ for (const token of field.text.split(" ")) {
342
+ if (singleEdit(query, token) &&
343
+ (!needsOwnCorrection ||
344
+ fieldScore(required, { text: token, whole: true }) > 0) &&
345
+ (field.weight > correctionWeight ||
346
+ correction === undefined ||
347
+ token < correction)) {
348
+ correction = token;
349
+ correctionWeight = field.weight;
350
+ }
351
+ }
352
+ }
353
+ if (!correction)
354
+ return;
355
+ patterns[missingIndex] = { text: correction, whole: true };
356
+ weights[missingIndex] = correctionWeight;
357
+ return {
358
+ patterns,
359
+ score: weights.reduce((a, b) => a + b, 0),
360
+ corrected: true,
361
+ };
362
+ }
363
+ /** Score only page metadata plus a single coherent section, never sibling bodies. */
364
+ function sectionCandidates(index, terms, fuzzy) {
365
+ const phrase = terms.join(" ");
366
+ const results = [];
367
+ for (const page of index.pages) {
368
+ const title = normalize(page.title);
369
+ const metadata = [
370
+ { text: title, weight: 8 },
371
+ { text: normalize(page.description), weight: 2 },
372
+ ];
373
+ const pageMatch = matchFields(metadata, terms, fuzzy);
374
+ let pageCandidate;
375
+ if (pageMatch) {
376
+ const exact = title === phrase;
377
+ pageCandidate = {
378
+ page,
379
+ source: sourceText([page.description, ...page.sections.map((section) => section.text)], pageMatch.patterns, page.title),
380
+ score: pageMatch.score + (exact ? 24 : 0),
381
+ rank: pageMatch.corrected ? 0 : exact ? 2 : 1,
382
+ patterns: pageMatch.patterns,
383
+ };
384
+ }
385
+ for (const section of page.sections) {
386
+ const heading = normalize(section.title);
387
+ const body = normalize(section.text);
388
+ const ownFields = [
389
+ { text: heading, weight: 4 },
390
+ { text: body, weight: 1 },
391
+ ];
392
+ const match = matchFields([...metadata, ...ownFields], terms, fuzzy, ownFields);
393
+ if (!match?.patterns.some((pattern) => fieldScore(ownFields, pattern)))
394
+ continue;
395
+ // Repeated page headings without their own matching body add no useful destination.
396
+ if (pageMatch &&
397
+ heading === title &&
398
+ !match.patterns.some((pattern) => contains(body, pattern)))
399
+ continue;
400
+ const exact = heading === phrase;
401
+ const candidate = {
402
+ page,
403
+ destination: section.url === page.url ? undefined : section,
404
+ source: sourceText([section.text, page.description], match.patterns, section.title || page.title),
405
+ score: match.score + (exact ? 12 : 0),
406
+ rank: match.corrected ? 0 : exact ? 2 : 1,
407
+ patterns: match.patterns,
408
+ };
409
+ if (candidate.destination)
410
+ results.push(candidate);
411
+ else if (!pageCandidate ||
412
+ compareCandidates(candidate, pageCandidate) < 0)
413
+ pageCandidate = candidate;
414
+ }
415
+ if (pageCandidate)
416
+ results.push(pageCandidate);
417
+ }
418
+ return results;
419
+ }
420
+ // Greater than the largest raw score: 32 terms × weight 8 + title bonus 24.
421
+ const sectionTierWeight = 281;
422
+ function displayResult(candidate) {
423
+ const { page, destination, patterns } = candidate;
424
+ const isSection = destination !== undefined && destination.url !== page.url;
425
+ const trail = isSection
426
+ ? [
427
+ ...(page.breadcrumbs ?? []),
428
+ page.title,
429
+ ...(destination.breadcrumbs ?? []),
430
+ ]
431
+ : [...(page.breadcrumbs ?? [])];
432
+ const snippet = excerpt(candidate.source, patterns);
433
+ return {
434
+ kind: isSection ? "section" : "page",
435
+ title: page.title,
436
+ pageUrl: page.url,
437
+ url: candidateUrl(candidate),
438
+ ...(destination?.title ? { section: destination.title } : {}),
439
+ breadcrumbs: trail,
440
+ excerpt: snippet,
441
+ score: candidate.rank * sectionTierWeight + candidate.score,
442
+ matches: {
443
+ title: matchRanges(page.title, patterns),
444
+ section: matchRanges(destination?.title ?? "", patterns),
445
+ excerpt: matchRanges(snippet, patterns),
446
+ breadcrumbs: trail.map((label) => matchRanges(label, patterns)),
447
+ },
448
+ };
449
+ }
450
+ /** Query without filesystem, network, or DOM access; page mode preserves legacy ranking. */
451
+ export function search(index, query, options = {}) {
452
+ validateSearchIndex(index);
453
+ if (options === null || typeof options !== "object" || Array.isArray(options))
454
+ throw new TypeError("Search options must be an object.");
455
+ const limit = options.limit ?? 10;
456
+ const mode = options.mode === undefined ? "pages" : options.mode;
457
+ const fuzzy = options.fuzzy === undefined ? false : options.fuzzy;
458
+ if (!Number.isInteger(limit) || limit < 0 || limit > 100)
459
+ throw new RangeError("Search limit must be an integer from 0 to 100.");
460
+ if (mode !== "pages" && mode !== "sections")
461
+ throw new RangeError("Search mode must be pages or sections.");
462
+ if (typeof fuzzy !== "boolean")
463
+ throw new TypeError("Search fuzzy option must be a boolean.");
464
+ if (fuzzy && mode !== "sections")
465
+ throw new RangeError("Fuzzy search requires sections mode.");
466
+ if (typeof query !== "string")
467
+ throw new TypeError("Search query must be a string.");
468
+ if (query.length > 512)
469
+ throw new RangeError("Search query must not exceed 512 characters.");
470
+ const terms = [...new Set(normalize(query).split(" ").filter(Boolean))];
471
+ if (terms.length > 32)
472
+ throw new RangeError("Search query must not exceed 32 unique terms.");
473
+ if (!terms.length || !limit)
474
+ return [];
475
+ const results = mode === "sections"
476
+ ? sectionCandidates(index, terms, false)
477
+ : pageCandidates(index, terms);
478
+ if (fuzzy && results.length < limit) {
479
+ const literalUrls = new Set(results.map(candidateUrl));
480
+ results.push(...sectionCandidates(index, terms, true).filter((candidate) => candidate.rank === 0 && !literalUrls.has(candidateUrl(candidate))));
481
+ }
482
+ results.sort(compareCandidates);
483
+ // Original-text mapping and excerpt windows are computed only for final hits.
484
+ return results.slice(0, limit).map(displayResult);
153
485
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@wgtechlabs/mdd-engine",
3
- "version": "0.1.0-pr.08f3b80",
3
+ "version": "0.1.0-pr.55ddedf",
4
4
  "description": "Headless Markdown documentation compiler for mdd",
5
5
  "type": "module",
6
6
  "license": "MIT",