@wgtechlabs/mdd-engine 0.1.0-pr.0cb765a → 0.1.0-pr.55ddedf

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -85,6 +85,10 @@ import { search } from '@wgtechlabs/mdd-engine/search';
85
85
  const index = createSearchIndex(result.site);
86
86
  const restored = JSON.parse(JSON.stringify(index));
87
87
  console.log(search(restored, 'installation', { limit: 10 }));
88
+ // Multiple heading hits, breadcrumbs, and ranges for UI highlights.
89
+ console.log(search(restored, 'instalation', {
90
+ mode: 'sections', fuzzy: true, limit: 8,
91
+ }));
88
92
  // [{ title, url, section?, excerpt, score }]
89
93
  ```
90
94
 
@@ -129,6 +133,20 @@ Use GitHub-style alerts with `NOTE`, `TIP`, `IMPORTANT`, `WARNING`, or `CAUTION`
129
133
 
130
134
  `:::details[More information]` remains supported and becomes `details`/`summary`; its label is optional, attributes are errors, and its readable Markdown uses a blockquote with a bold label. The old `:::note`, `:::tip`, and `:::warning` directives now fail with a `REMOVED_COMPONENT` migration diagnostic. See [alerts and migration](docs/ALERTS.md) for all five types, theme hooks, and how to preserve custom titles and bodies.
131
135
 
136
+ Document an API endpoint with a leaf directive, then use ordinary Markdown for parameters and request/response examples:
137
+
138
+ ```markdown
139
+ ## Retrieve a widget
140
+
141
+ ::endpoint{method="GET" path="/v1/widgets/{id}"}
142
+
143
+ | Parameter | Type | Description |
144
+ | --- | --- | --- |
145
+ | `id` | string | Widget identifier. |
146
+ ```
147
+
148
+ The required `method` accepts GET, HEAD, POST, PUT, PATCH, DELETE, OPTIONS, TRACE, or CONNECT, normalizing lowercase/mixed case to uppercase. The required `path` starts with a single `/` and contains no whitespace or control characters; braces and query punctuation remain literal text. Labels, extra attributes, and inline/container forms are errors. The engine emits a `div.mdd-endpoint` containing `strong.mdd-endpoint-method.mdd-method-get` (or the corresponding lowercase method) and `code.mdd-endpoint-path`. Normalized Markdown contains `**GET**` followed by the code-formatted path, so the signature stays readable and searchable. Endpoint paths are not rewritten with the documentation base path. This is documentation only: no HTTP requests run. MDD and its themes provide presentation.
149
+
132
150
  Title precedence is frontmatter title, first H1, then readable filename. Navigation uses `navTitle` when supplied. Explicit `order` sorts first; remaining siblings sort deterministically by label and path.
133
151
 
134
152
  Local `.md` links, extensionless routes, reference links, and images resolve from their source document. A leading slash addresses the documentation root. `basePath` prefixes public links, including `/docs/` and `/repository/docs/`. When a file-style URL matches both an existing supported asset and a page route, the asset wins: `chart.png` selects the image, while `/chart.png/` explicitly selects the page. If no regular asset exists, dotted page routes still resolve. External links are preserved without network requests. Headings have `mdd-`-prefixed GitHub-style slugs; author links such as `#installation` are rewritten to `#mdd-installation`. Duplicate headings receive `-1`, `-2`, and subsequent suffixes.
@@ -0,0 +1,3 @@
1
+ import type { Nodes, Paragraph } from "mdast";
2
+ /** Endpoint signatures are documentation text, never links or executable requests. */
3
+ export declare function endpointParagraph(node: Nodes, source: string, report: (message: string) => void): Paragraph | undefined;
@@ -0,0 +1,68 @@
1
+ const methods = new Set([
2
+ "GET",
3
+ "HEAD",
4
+ "POST",
5
+ "PUT",
6
+ "PATCH",
7
+ "DELETE",
8
+ "OPTIONS",
9
+ "TRACE",
10
+ "CONNECT",
11
+ ]);
12
+ /** Endpoint signatures are documentation text, never links or executable requests. */
13
+ export function endpointParagraph(node, source, report) {
14
+ if (node.type !== "leafDirective") {
15
+ report('Use the leaf form ::endpoint{method="GET" path="/example"}.');
16
+ return;
17
+ }
18
+ const offset = node.position?.start.offset;
19
+ if (node.children.length ||
20
+ (offset !== undefined && source.startsWith("::endpoint[", offset))) {
21
+ report("The endpoint component does not support a label.");
22
+ return;
23
+ }
24
+ const attributes = node.attributes ?? {};
25
+ if (Object.keys(attributes).some((key) => key !== "method" && key !== "path")) {
26
+ report("The endpoint component supports only method and path attributes.");
27
+ return;
28
+ }
29
+ const method = (attributes.method ?? "").toUpperCase();
30
+ if (!methods.has(method)) {
31
+ report("The endpoint method must be GET, HEAD, POST, PUT, PATCH, DELETE, OPTIONS, TRACE, or CONNECT.");
32
+ return;
33
+ }
34
+ const path = attributes.path ?? "";
35
+ if (!path.startsWith("/") ||
36
+ path.startsWith("//") ||
37
+ // biome-ignore lint/suspicious/noControlCharactersInRegex: Endpoint text cannot contain invisible controls or whitespace.
38
+ /[\s\u0000-\u001f\u007f-\u009f]/u.test(path)) {
39
+ report("The endpoint path must start with a single / and contain no whitespace or control characters.");
40
+ return;
41
+ }
42
+ return {
43
+ type: "paragraph",
44
+ position: node.position,
45
+ // Tight lists unwrap ordinary paragraphs; retain the signature's block hook.
46
+ data: { hName: "div", hProperties: { className: ["mdd-endpoint"] } },
47
+ children: [
48
+ {
49
+ type: "strong",
50
+ data: {
51
+ hProperties: {
52
+ className: [
53
+ "mdd-endpoint-method",
54
+ `mdd-method-${method.toLowerCase()}`,
55
+ ],
56
+ },
57
+ },
58
+ children: [{ type: "text", value: method }],
59
+ },
60
+ { type: "text", value: " " },
61
+ {
62
+ type: "inlineCode",
63
+ value: path,
64
+ data: { hProperties: { className: ["mdd-endpoint-path"] } },
65
+ },
66
+ ],
67
+ };
68
+ }
package/dist/index.d.ts CHANGED
@@ -1,5 +1,5 @@
1
1
  import type { CompileOptions, CompileResult } from "./types.js";
2
- export { type SearchIndex, type SearchOptions, type SearchPage, type SearchResult, type SearchSection, search, validateSearchIndex, } from "./search.js";
2
+ export { type SearchIndex, type SearchMatch, type SearchOptions, type SearchPage, type SearchResult, type SearchSection, search, validateSearchIndex, } from "./search.js";
3
3
  export { createSearchIndex } from "./search-index.js";
4
4
  export type { Asset, CompileOptions, CompileResult, Diagnostic, Footer, Heading, Metadata, NavigationItem, Page, Site, SocialLink, Theme, } from "./types.js";
5
5
  /** Compile a local checkout; never fetch repositories, execute author code, or emit files. */
package/dist/markdown.js CHANGED
@@ -13,6 +13,7 @@ import { unified } from "unified";
13
13
  import { SKIP, visit } from "unist-util-visit";
14
14
  import { parseDocument as parseYaml } from "yaml";
15
15
  import { transformAlerts } from "./alerts.js";
16
+ import { endpointParagraph } from "./endpoints.js";
16
17
  const parser = unified()
17
18
  .use(remarkParse)
18
19
  .use(remarkGfm)
@@ -29,10 +30,20 @@ const htmlRenderer = unified()
29
30
  ...defaultSchema.attributes,
30
31
  aside: [["className", /^mdd-/]],
31
32
  details: [["className", "mdd-details"]],
33
+ div: [["className", "mdd-endpoint"]],
32
34
  p: [
33
35
  ...(defaultSchema.attributes?.p ?? []),
34
36
  ["className", "mdd-component-label"],
35
37
  ],
38
+ strong: [
39
+ ...(defaultSchema.attributes?.strong ?? []),
40
+ [
41
+ "className",
42
+ "mdd-endpoint-method",
43
+ /^mdd-method-(get|head|post|put|patch|delete|options|trace|connect)$/,
44
+ ],
45
+ ],
46
+ code: [["className", /^language-./, "mdd-endpoint-path"]],
36
47
  summary: [
37
48
  ...(defaultSchema.attributes?.summary ?? []),
38
49
  ["className", "mdd-component-label"],
@@ -206,8 +217,16 @@ export function parseDocument(source, file, diagnostics) {
206
217
  report("REMOVED_COMPONENT", `The ${node.name} directive was removed; use > [!${node.name.toUpperCase()}] followed by quoted body lines. Keep any optional custom title as bold text in the alert body.`, node);
207
218
  return;
208
219
  }
220
+ if (node.name === "endpoint") {
221
+ const paragraph = endpointParagraph(node, source, (message) => report("INVALID_COMPONENT", message, node));
222
+ if (paragraph && parent && index !== undefined) {
223
+ parent.children[index] = paragraph;
224
+ return [SKIP, index];
225
+ }
226
+ return;
227
+ }
209
228
  if (node.type !== "containerDirective" || node.name !== "details") {
210
- report("UNKNOWN_COMPONENT", `Unsupported component ${node.name}; use a GitHub alert or a details container.`, node);
229
+ report("UNKNOWN_COMPONENT", `Unsupported component ${node.name}; use a GitHub alert, details container, or endpoint leaf.`, node);
211
230
  return;
212
231
  }
213
232
  if (Object.keys(node.attributes ?? {}).length) {
@@ -7,8 +7,9 @@ import { transformAlerts } from "./alerts.js";
7
7
  import { validateSearchIndex, } from "./search.js";
8
8
  const parser = unified().use(remarkParse).use(remarkGfm);
9
9
  const blockTypes = new Set(["paragraph", "code", "tableCell", "listItem"]);
10
- function indexPage(page) {
10
+ function indexPage(page, breadcrumbs) {
11
11
  const sections = [{ title: "", url: page.url, text: "" }];
12
+ const ancestors = [];
12
13
  let headingIndex = 0;
13
14
  const chunks = [[]];
14
15
  const tree = parser.parse(page.markdown);
@@ -23,11 +24,20 @@ function indexPage(page) {
23
24
  heading.depth !== node.depth) {
24
25
  throw new TypeError("Search indexing requires matching compiled Markdown and headings.");
25
26
  }
27
+ while (ancestors.length &&
28
+ (ancestors.at(-1)?.depth ?? 0) >= heading.depth)
29
+ ancestors.pop();
26
30
  sections.push({
27
31
  title: heading.text,
28
32
  url: `${page.url}#${encodeURIComponent(heading.id)}`,
29
33
  text: "",
34
+ breadcrumbs: ancestors
35
+ .filter((ancestor) => !(ancestor.depth === 1 &&
36
+ ancestor.text.normalize("NFKC").toLowerCase() ===
37
+ page.title.normalize("NFKC").toLowerCase()))
38
+ .map((ancestor) => ancestor.text),
30
39
  });
40
+ ancestors.push(heading);
31
41
  chunks.push([]);
32
42
  return SKIP;
33
43
  }
@@ -55,12 +65,33 @@ function indexPage(page) {
55
65
  url: page.url,
56
66
  title: page.title,
57
67
  description: page.description ?? "",
68
+ breadcrumbs,
58
69
  sections,
59
70
  };
60
71
  }
61
72
  /** Build only from successfully compiled pages; no source paths or HTML are indexed. */
62
73
  export function createSearchIndex(site) {
63
- const index = { version: 1, pages: site.pages.map(indexPage) };
74
+ const navigation = new Map();
75
+ const pending = site.navigation.map((item) => ({
76
+ item,
77
+ ancestors: [],
78
+ }));
79
+ while (pending.length) {
80
+ const entry = pending.pop();
81
+ if (!entry)
82
+ break;
83
+ const { item, ancestors } = entry;
84
+ if (ancestors.length > 64)
85
+ throw new TypeError("Search navigation exceeds 64 ancestor labels.");
86
+ if (item.url)
87
+ navigation.set(item.url, ancestors);
88
+ for (const child of item.children ?? [])
89
+ pending.push({ item: child, ancestors: [...ancestors, item.title] });
90
+ }
91
+ const index = {
92
+ version: 1,
93
+ pages: site.pages.map((page) => indexPage(page, navigation.get(page.url) ?? [])),
94
+ };
64
95
  index.pages.sort((a, b) => (a.url < b.url ? -1 : a.url > b.url ? 1 : 0));
65
96
  validateSearchIndex(index);
66
97
  return index;
package/dist/search.d.ts CHANGED
@@ -7,6 +7,8 @@ export interface SearchPage {
7
7
  url: string;
8
8
  title: string;
9
9
  description: string;
10
+ /** Navigation ancestor labels, excluding this page's own label. */
11
+ breadcrumbs?: string[];
10
12
  sections: SearchSection[];
11
13
  }
12
14
  export interface SearchSection {
@@ -14,20 +16,37 @@ export interface SearchSection {
14
16
  title: string;
15
17
  url: string;
16
18
  text: string;
19
+ /** Ancestor headings, excluding this heading and a duplicate page-title H1. */
20
+ breadcrumbs?: string[];
17
21
  }
18
22
  export interface SearchOptions {
19
23
  /** Maximum results, from 0 to 100. Defaults to 10. */
20
24
  limit?: number;
25
+ /** Defaults to one result per page; sections returns independent heading hits. */
26
+ mode?: "pages" | "sections";
27
+ /** Allow one single-edit query correction per result. Requires sections mode. */
28
+ fuzzy?: boolean;
21
29
  }
30
+ /** UTF-16 offsets into the original displayed field; end is exclusive. */
31
+ export type SearchMatch = [start: number, end: number];
22
32
  export interface SearchResult {
33
+ kind: "page" | "section";
23
34
  title: string;
24
35
  url: string;
36
+ pageUrl: string;
37
+ breadcrumbs: string[];
25
38
  /** Present when the destination is a compiled heading. */
26
39
  section?: string;
27
40
  excerpt: string;
28
41
  score: number;
42
+ matches: {
43
+ title: SearchMatch[];
44
+ section: SearchMatch[];
45
+ excerpt: SearchMatch[];
46
+ breadcrumbs: SearchMatch[][];
47
+ };
29
48
  }
30
49
  /** Validate deserialized data before any result can become a link. */
31
50
  export declare function validateSearchIndex(value: unknown): asserts value is SearchIndex;
32
- /** Return at most one hit per page, without filesystem, network, or DOM access. */
51
+ /** Query without filesystem, network, or DOM access; page mode preserves legacy ranking. */
33
52
  export declare function search(index: SearchIndex, query: string, options?: SearchOptions): SearchResult[];
package/dist/search.js CHANGED
@@ -1,6 +1,17 @@
1
1
  function record(value) {
2
2
  return value !== null && typeof value === "object" && !Array.isArray(value);
3
3
  }
4
+ function breadcrumbs(value, maximum) {
5
+ if (value === undefined)
6
+ return true;
7
+ if (!Array.isArray(value) || value.length > maximum)
8
+ return false;
9
+ // Iterate holes too: a sparse array does not contain only string labels.
10
+ for (const label of value)
11
+ if (typeof label !== "string")
12
+ return false;
13
+ return true;
14
+ }
4
15
  function pageUrl(value) {
5
16
  if (typeof value !== "string" ||
6
17
  !value.startsWith("/") ||
@@ -48,6 +59,7 @@ export function validateSearchIndex(value) {
48
59
  urls.has(page.url) ||
49
60
  typeof page.title !== "string" ||
50
61
  typeof page.description !== "string" ||
62
+ !breadcrumbs(page.breadcrumbs, 64) ||
51
63
  !Array.isArray(page.sections))
52
64
  throw new TypeError("Invalid search index page.");
53
65
  urls.add(page.url);
@@ -57,7 +69,8 @@ export function validateSearchIndex(value) {
57
69
  !sectionUrl(section.url, page.url) ||
58
70
  sections.has(section.url) ||
59
71
  typeof section.title !== "string" ||
60
- typeof section.text !== "string")
72
+ typeof section.text !== "string" ||
73
+ !breadcrumbs(section.breadcrumbs, 6))
61
74
  throw new TypeError("Invalid search index section.");
62
75
  sections.add(section.url);
63
76
  }
@@ -70,23 +83,51 @@ function normalize(value) {
70
83
  .replace(/[^\p{L}\p{N}\p{M}_]+/gu, " ")
71
84
  .trim();
72
85
  }
73
- /** Map a normalized match back to the original text, including NFKC expansions. */
74
- function matchRange(value, terms) {
86
+ function contains(value, pattern) {
87
+ return pattern.whole
88
+ ? ` ${value} `.includes(` ${pattern.text} `)
89
+ : value.includes(pattern.text);
90
+ }
91
+ function mergeRanges(ranges) {
92
+ ranges.sort((a, b) => a[0] - b[0] || a[1] - b[1]);
93
+ const merged = [];
94
+ for (const range of ranges) {
95
+ const previous = merged.at(-1);
96
+ if (previous && range[0] <= previous[1])
97
+ previous[1] = Math.max(previous[1], range[1]);
98
+ else
99
+ merged.push([...range]);
100
+ }
101
+ return merged;
102
+ }
103
+ /** Map normalized matches to complete original graphemes, including NFKC expansions. */
104
+ function matchRanges(value, patterns, firstOnly = false) {
75
105
  const normalized = value.normalize("NFKC").toLowerCase();
76
- let start = -1;
77
- let end = -1;
78
- for (const term of terms) {
79
- const found = normalized.indexOf(term);
80
- if (found >= 0 &&
81
- (start < 0 ||
82
- found < start ||
83
- (found === start && found + term.length > end))) {
84
- start = found;
85
- end = found + term.length;
106
+ const found = [];
107
+ for (const pattern of patterns) {
108
+ let offset = 0;
109
+ while (offset < normalized.length) {
110
+ const start = normalized.indexOf(pattern.text, offset);
111
+ if (start < 0)
112
+ break;
113
+ const end = start + pattern.text.length;
114
+ // One neighboring code point needs at most two UTF-16 units. Avoid scanning
115
+ // a growing prefix for each occurrence, or any boundary work for substrings.
116
+ if (!pattern.whole ||
117
+ (!/[\p{L}\p{N}\p{M}_]$/u.test(normalized.slice(Math.max(0, start - 2), start)) &&
118
+ !/^[\p{L}\p{N}\p{M}_]/u.test(normalized.slice(end, end + 2)))) {
119
+ found.push([start, end]);
120
+ if (firstOnly)
121
+ break;
122
+ }
123
+ offset = start + 1;
86
124
  }
87
125
  }
88
- if (start < 0)
89
- return;
126
+ let ranges = mergeRanges(found);
127
+ if (!ranges.length)
128
+ return [];
129
+ if (firstOnly)
130
+ ranges = ranges.slice(0, 1);
90
131
  const groups = [];
91
132
  for (const { segment, index } of new Intl.Segmenter(undefined, {
92
133
  granularity: "grapheme",
@@ -110,25 +151,35 @@ function matchRange(value, terms) {
110
151
  }
111
152
  groups.push(group);
112
153
  }
113
- // Match against whole-field lowercase for contextual letters such as sigma.
114
- // Per-group lowercase is used only for lengths (including expansions like İ).
154
+ // Whole-field lowercase retains contextual letters such as final sigma.
155
+ // Per-group lowercase supplies only lengths, including expansions like İ.
156
+ const mapped = [];
115
157
  let offset = 0;
158
+ let rangeIndex = 0;
116
159
  let originalStart = 0;
117
160
  for (const group of groups) {
118
161
  const next = offset + group.text.toLowerCase().length;
119
- if (offset <= start && start < next)
120
- originalStart = group.start;
121
- if (offset < end && end <= next)
122
- return [originalStart, group.end];
162
+ let range = ranges[rangeIndex];
163
+ while (range && range[0] < next) {
164
+ if (range[0] >= offset)
165
+ originalStart = group.start;
166
+ if (range[1] > next)
167
+ break;
168
+ mapped.push([originalStart, group.end]);
169
+ range = ranges[++rangeIndex];
170
+ }
171
+ if (!range)
172
+ break;
123
173
  offset = next;
124
174
  }
175
+ return mergeRanges(mapped);
125
176
  }
126
- function excerpt(value, terms) {
177
+ function excerpt(value, patterns) {
127
178
  const text = value.replace(/\s+/gu, " ").trim();
128
179
  const characters = Array.from(text);
129
180
  if (characters.length <= 160)
130
181
  return text;
131
- const match = matchRange(text, terms);
182
+ const match = matchRanges(text, patterns, true)[0];
132
183
  const matchStart = match ? Array.from(text.slice(0, match[0])).length : 0;
133
184
  const matchLength = match
134
185
  ? Array.from(text.slice(match[0], match[1])).length
@@ -139,25 +190,45 @@ function excerpt(value, terms) {
139
190
  let end = Math.min(characters.length, start + 160 - (start > 0 ? 1 : 0));
140
191
  if (end < characters.length)
141
192
  end--;
142
- return `${start > 0 ? "…" : ""}${characters.slice(start, end).join("")}${end < characters.length ? "…" : ""}`;
143
- }
144
- /** Return at most one hit per page, without filesystem, network, or DOM access. */
145
- export function search(index, query, options = {}) {
146
- validateSearchIndex(index);
147
- const limit = options.limit ?? 10;
148
- if (!Number.isInteger(limit) || limit < 0 || limit > 100) {
149
- throw new RangeError("Search limit must be an integer from 0 to 100.");
193
+ const startOffset = characters.slice(0, start).join("").length;
194
+ const endOffset = characters.slice(0, end).join("").length;
195
+ // Move clipped edges inward to complete graphemes without exceeding the cap.
196
+ let safeStart = text.length;
197
+ let safeEnd = 0;
198
+ for (const { segment, index } of new Intl.Segmenter(undefined, {
199
+ granularity: "grapheme",
200
+ }).segment(text)) {
201
+ if (index >= startOffset && safeStart === text.length)
202
+ safeStart = index;
203
+ if (index + segment.length <= endOffset)
204
+ safeEnd = index + segment.length;
205
+ if (index >= endOffset)
206
+ break;
150
207
  }
151
- if (typeof query !== "string")
152
- throw new TypeError("Search query must be a string.");
153
- if (query.length > 512)
154
- throw new RangeError("Search query must not exceed 512 characters.");
155
- const terms = [...new Set(normalize(query).split(" ").filter(Boolean))];
208
+ return `${safeStart > 0 ? "…" : ""}${text.slice(safeStart, safeEnd)}${safeEnd < text.length ? "…" : ""}`;
209
+ }
210
+ function candidateUrl(candidate) {
211
+ return candidate.destination?.url ?? candidate.page.url;
212
+ }
213
+ function compareCandidates(a, b) {
214
+ const aUrl = candidateUrl(a);
215
+ const bUrl = candidateUrl(b);
216
+ return (b.rank - a.rank ||
217
+ b.score - a.score ||
218
+ (aUrl < bUrl ? -1 : aUrl > bUrl ? 1 : 0));
219
+ }
220
+ function sourceText(sources, patterns, fallback) {
221
+ return (sources.find((text) => {
222
+ const normalized = normalize(text);
223
+ return patterns.some((pattern) => contains(normalized, pattern));
224
+ }) ??
225
+ sources.find((text) => text) ??
226
+ fallback);
227
+ }
228
+ /** Preserve page-mode weighting, cross-section coverage, and destination selection. */
229
+ function pageCandidates(index, terms) {
156
230
  const phrase = terms.join(" ");
157
- if (terms.length > 32)
158
- throw new RangeError("Search query must not exceed 32 unique terms.");
159
- if (!terms.length || !limit)
160
- return [];
231
+ const patterns = terms.map((text) => ({ text }));
161
232
  const results = [];
162
233
  for (const page of index.pages) {
163
234
  const title = normalize(page.title);
@@ -196,35 +267,219 @@ export function search(index, query, options = {}) {
196
267
  // A stable sort preserves document order when multiple sections tie.
197
268
  matches.sort((a, b) => b.score - a.score);
198
269
  const best = matches[0]?.section;
199
- const destination = metadataMatch ? undefined : best;
200
- const sources = [
201
- best?.text ?? "",
202
- page.description,
203
- ...page.sections.map((section) => section.text),
204
- ];
205
- const source = sources.find((text) => {
206
- const normalized = normalize(text);
207
- return terms.some((term) => normalized.includes(term));
208
- }) ??
209
- sources.find((text) => text) ??
210
- page.title;
211
270
  results.push({
212
- title: page.title,
213
- url: destination?.url ?? page.url,
214
- ...(destination?.title ? { section: destination.title } : {}),
215
- // Keep source text until ranking so only returned hits need a snippet.
216
- excerpt: source,
271
+ page,
272
+ destination: metadataMatch ? undefined : best,
273
+ source: sourceText([
274
+ best?.text ?? "",
275
+ page.description,
276
+ ...page.sections.map((section) => section.text),
277
+ ], patterns, page.title),
217
278
  score: weights.reduce((sum, weight) => sum + weight, 0) +
218
279
  (title === phrase
219
280
  ? 24
220
281
  : sections.some((section) => section.title === phrase)
221
282
  ? 12
222
283
  : 0),
284
+ rank: 0,
285
+ patterns,
223
286
  });
224
287
  }
225
- results.sort((a, b) => b.score - a.score || (a.url < b.url ? -1 : a.url > b.url ? 1 : 0));
226
- return results.slice(0, limit).map((result) => ({
227
- ...result,
228
- excerpt: excerpt(result.excerpt, terms),
229
- }));
288
+ return results;
289
+ }
290
+ /** Single insertion, deletion, substitution, or adjacent transposition, on code points. */
291
+ function singleEdit(query, token) {
292
+ if (token.length > 66)
293
+ return false;
294
+ const text = Array.from(token);
295
+ if (Math.abs(query.length - text.length) > 1)
296
+ return false;
297
+ let start = 0;
298
+ while (start < query.length && query[start] === text[start])
299
+ start++;
300
+ if (start === Math.min(query.length, text.length))
301
+ return true;
302
+ if (query.length === text.length) {
303
+ if (query.slice(start + 1).join("") === text.slice(start + 1).join(""))
304
+ return true;
305
+ return (query[start] === text[start + 1] &&
306
+ query[start + 1] === text[start] &&
307
+ query.slice(start + 2).join("") === text.slice(start + 2).join(""));
308
+ }
309
+ return query.length > text.length
310
+ ? query.slice(start + 1).join("") === text.slice(start).join("")
311
+ : query.slice(start).join("") === text.slice(start + 1).join("");
312
+ }
313
+ function fieldScore(fields, pattern) {
314
+ return Math.max(0, ...fields
315
+ .filter((field) => contains(field.text, pattern))
316
+ .map((field) => field.weight));
317
+ }
318
+ function matchFields(fields, terms, fuzzy, required = []) {
319
+ const patterns = terms.map((text) => ({ text }));
320
+ const weights = patterns.map((pattern) => fieldScore(fields, pattern));
321
+ const missing = weights.flatMap((weight, index) => (weight ? [] : [index]));
322
+ if (!missing.length)
323
+ return {
324
+ patterns,
325
+ score: weights.reduce((a, b) => a + b, 0),
326
+ corrected: false,
327
+ };
328
+ const missingIndex = missing[0];
329
+ if (!fuzzy || missing.length !== 1 || missingIndex === undefined)
330
+ return;
331
+ const query = Array.from(terms[missingIndex] ?? "");
332
+ if (query.length < 4 || query.length > 32)
333
+ return;
334
+ const needsOwnCorrection = required.length > 0 &&
335
+ !patterns.some((pattern) => fieldScore(required, pattern));
336
+ let correction;
337
+ let correctionWeight = 0;
338
+ for (const field of fields) {
339
+ if (field.weight < correctionWeight)
340
+ continue;
341
+ for (const token of field.text.split(" ")) {
342
+ if (singleEdit(query, token) &&
343
+ (!needsOwnCorrection ||
344
+ fieldScore(required, { text: token, whole: true }) > 0) &&
345
+ (field.weight > correctionWeight ||
346
+ correction === undefined ||
347
+ token < correction)) {
348
+ correction = token;
349
+ correctionWeight = field.weight;
350
+ }
351
+ }
352
+ }
353
+ if (!correction)
354
+ return;
355
+ patterns[missingIndex] = { text: correction, whole: true };
356
+ weights[missingIndex] = correctionWeight;
357
+ return {
358
+ patterns,
359
+ score: weights.reduce((a, b) => a + b, 0),
360
+ corrected: true,
361
+ };
362
+ }
363
+ /** Score only page metadata plus a single coherent section, never sibling bodies. */
364
+ function sectionCandidates(index, terms, fuzzy) {
365
+ const phrase = terms.join(" ");
366
+ const results = [];
367
+ for (const page of index.pages) {
368
+ const title = normalize(page.title);
369
+ const metadata = [
370
+ { text: title, weight: 8 },
371
+ { text: normalize(page.description), weight: 2 },
372
+ ];
373
+ const pageMatch = matchFields(metadata, terms, fuzzy);
374
+ let pageCandidate;
375
+ if (pageMatch) {
376
+ const exact = title === phrase;
377
+ pageCandidate = {
378
+ page,
379
+ source: sourceText([page.description, ...page.sections.map((section) => section.text)], pageMatch.patterns, page.title),
380
+ score: pageMatch.score + (exact ? 24 : 0),
381
+ rank: pageMatch.corrected ? 0 : exact ? 2 : 1,
382
+ patterns: pageMatch.patterns,
383
+ };
384
+ }
385
+ for (const section of page.sections) {
386
+ const heading = normalize(section.title);
387
+ const body = normalize(section.text);
388
+ const ownFields = [
389
+ { text: heading, weight: 4 },
390
+ { text: body, weight: 1 },
391
+ ];
392
+ const match = matchFields([...metadata, ...ownFields], terms, fuzzy, ownFields);
393
+ if (!match?.patterns.some((pattern) => fieldScore(ownFields, pattern)))
394
+ continue;
395
+ // Repeated page headings without their own matching body add no useful destination.
396
+ if (pageMatch &&
397
+ heading === title &&
398
+ !match.patterns.some((pattern) => contains(body, pattern)))
399
+ continue;
400
+ const exact = heading === phrase;
401
+ const candidate = {
402
+ page,
403
+ destination: section.url === page.url ? undefined : section,
404
+ source: sourceText([section.text, page.description], match.patterns, section.title || page.title),
405
+ score: match.score + (exact ? 12 : 0),
406
+ rank: match.corrected ? 0 : exact ? 2 : 1,
407
+ patterns: match.patterns,
408
+ };
409
+ if (candidate.destination)
410
+ results.push(candidate);
411
+ else if (!pageCandidate ||
412
+ compareCandidates(candidate, pageCandidate) < 0)
413
+ pageCandidate = candidate;
414
+ }
415
+ if (pageCandidate)
416
+ results.push(pageCandidate);
417
+ }
418
+ return results;
419
+ }
420
+ // Greater than the largest raw score: 32 terms × weight 8 + title bonus 24.
421
+ const sectionTierWeight = 281;
422
+ function displayResult(candidate) {
423
+ const { page, destination, patterns } = candidate;
424
+ const isSection = destination !== undefined && destination.url !== page.url;
425
+ const trail = isSection
426
+ ? [
427
+ ...(page.breadcrumbs ?? []),
428
+ page.title,
429
+ ...(destination.breadcrumbs ?? []),
430
+ ]
431
+ : [...(page.breadcrumbs ?? [])];
432
+ const snippet = excerpt(candidate.source, patterns);
433
+ return {
434
+ kind: isSection ? "section" : "page",
435
+ title: page.title,
436
+ pageUrl: page.url,
437
+ url: candidateUrl(candidate),
438
+ ...(destination?.title ? { section: destination.title } : {}),
439
+ breadcrumbs: trail,
440
+ excerpt: snippet,
441
+ score: candidate.rank * sectionTierWeight + candidate.score,
442
+ matches: {
443
+ title: matchRanges(page.title, patterns),
444
+ section: matchRanges(destination?.title ?? "", patterns),
445
+ excerpt: matchRanges(snippet, patterns),
446
+ breadcrumbs: trail.map((label) => matchRanges(label, patterns)),
447
+ },
448
+ };
449
+ }
450
+ /** Query without filesystem, network, or DOM access; page mode preserves legacy ranking. */
451
+ export function search(index, query, options = {}) {
452
+ validateSearchIndex(index);
453
+ if (options === null || typeof options !== "object" || Array.isArray(options))
454
+ throw new TypeError("Search options must be an object.");
455
+ const limit = options.limit ?? 10;
456
+ const mode = options.mode === undefined ? "pages" : options.mode;
457
+ const fuzzy = options.fuzzy === undefined ? false : options.fuzzy;
458
+ if (!Number.isInteger(limit) || limit < 0 || limit > 100)
459
+ throw new RangeError("Search limit must be an integer from 0 to 100.");
460
+ if (mode !== "pages" && mode !== "sections")
461
+ throw new RangeError("Search mode must be pages or sections.");
462
+ if (typeof fuzzy !== "boolean")
463
+ throw new TypeError("Search fuzzy option must be a boolean.");
464
+ if (fuzzy && mode !== "sections")
465
+ throw new RangeError("Fuzzy search requires sections mode.");
466
+ if (typeof query !== "string")
467
+ throw new TypeError("Search query must be a string.");
468
+ if (query.length > 512)
469
+ throw new RangeError("Search query must not exceed 512 characters.");
470
+ const terms = [...new Set(normalize(query).split(" ").filter(Boolean))];
471
+ if (terms.length > 32)
472
+ throw new RangeError("Search query must not exceed 32 unique terms.");
473
+ if (!terms.length || !limit)
474
+ return [];
475
+ const results = mode === "sections"
476
+ ? sectionCandidates(index, terms, false)
477
+ : pageCandidates(index, terms);
478
+ if (fuzzy && results.length < limit) {
479
+ const literalUrls = new Set(results.map(candidateUrl));
480
+ results.push(...sectionCandidates(index, terms, true).filter((candidate) => candidate.rank === 0 && !literalUrls.has(candidateUrl(candidate))));
481
+ }
482
+ results.sort(compareCandidates);
483
+ // Original-text mapping and excerpt windows are computed only for final hits.
484
+ return results.slice(0, limit).map(displayResult);
230
485
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@wgtechlabs/mdd-engine",
3
- "version": "0.1.0-pr.0cb765a",
3
+ "version": "0.1.0-pr.55ddedf",
4
4
  "description": "Headless Markdown documentation compiler for mdd",
5
5
  "type": "module",
6
6
  "license": "MIT",