@wgtechlabs/mdd-engine 0.1.0-pr.08f3b80 → 0.1.0-pr.55ddedf
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +18 -0
- package/dist/endpoints.d.ts +3 -0
- package/dist/endpoints.js +68 -0
- package/dist/index.d.ts +1 -1
- package/dist/markdown.js +20 -1
- package/dist/search-index.js +33 -2
- package/dist/search.d.ts +20 -1
- package/dist/search.js +363 -31
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -85,6 +85,10 @@ import { search } from '@wgtechlabs/mdd-engine/search';
|
|
|
85
85
|
const index = createSearchIndex(result.site);
|
|
86
86
|
const restored = JSON.parse(JSON.stringify(index));
|
|
87
87
|
console.log(search(restored, 'installation', { limit: 10 }));
|
|
88
|
+
// Multiple heading hits, breadcrumbs, and ranges for UI highlights.
|
|
89
|
+
console.log(search(restored, 'instalation', {
|
|
90
|
+
mode: 'sections', fuzzy: true, limit: 8,
|
|
91
|
+
}));
|
|
88
92
|
// [{ title, url, section?, excerpt, score }]
|
|
89
93
|
```
|
|
90
94
|
|
|
@@ -129,6 +133,20 @@ Use GitHub-style alerts with `NOTE`, `TIP`, `IMPORTANT`, `WARNING`, or `CAUTION`
|
|
|
129
133
|
|
|
130
134
|
`:::details[More information]` remains supported and becomes `details`/`summary`; its label is optional, attributes are errors, and its readable Markdown uses a blockquote with a bold label. The old `:::note`, `:::tip`, and `:::warning` directives now fail with a `REMOVED_COMPONENT` migration diagnostic. See [alerts and migration](docs/ALERTS.md) for all five types, theme hooks, and how to preserve custom titles and bodies.
|
|
131
135
|
|
|
136
|
+
Document an API endpoint with a leaf directive, then use ordinary Markdown for parameters and request/response examples:
|
|
137
|
+
|
|
138
|
+
```markdown
|
|
139
|
+
## Retrieve a widget
|
|
140
|
+
|
|
141
|
+
::endpoint{method="GET" path="/v1/widgets/{id}"}
|
|
142
|
+
|
|
143
|
+
| Parameter | Type | Description |
|
|
144
|
+
| --- | --- | --- |
|
|
145
|
+
| `id` | string | Widget identifier. |
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
The required `method` accepts GET, HEAD, POST, PUT, PATCH, DELETE, OPTIONS, TRACE, or CONNECT, normalizing lowercase/mixed case to uppercase. The required `path` starts with a single `/` and contains no whitespace or control characters; braces and query punctuation remain literal text. Labels, extra attributes, and inline/container forms are errors. The engine emits a `div.mdd-endpoint` containing `strong.mdd-endpoint-method.mdd-method-get` (or the corresponding lowercase method) and `code.mdd-endpoint-path`. Normalized Markdown contains `**GET**` followed by the code-formatted path, so the signature stays readable and searchable. Endpoint paths are not rewritten with the documentation base path. This is documentation only: no HTTP requests run. MDD and its themes provide presentation.
|
|
149
|
+
|
|
132
150
|
Title precedence is frontmatter title, first H1, then readable filename. Navigation uses `navTitle` when supplied. Explicit `order` sorts first; remaining siblings sort deterministically by label and path.
|
|
133
151
|
|
|
134
152
|
Local `.md` links, extensionless routes, reference links, and images resolve from their source document. A leading slash addresses the documentation root. `basePath` prefixes public links, including `/docs/` and `/repository/docs/`. When a file-style URL matches both an existing supported asset and a page route, the asset wins: `chart.png` selects the image, while `/chart.png/` explicitly selects the page. If no regular asset exists, dotted page routes still resolve. External links are preserved without network requests. Headings have `mdd-`-prefixed GitHub-style slugs; author links such as `#installation` are rewritten to `#mdd-installation`. Duplicate headings receive `-1`, `-2`, and subsequent suffixes.
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
const methods = new Set([
|
|
2
|
+
"GET",
|
|
3
|
+
"HEAD",
|
|
4
|
+
"POST",
|
|
5
|
+
"PUT",
|
|
6
|
+
"PATCH",
|
|
7
|
+
"DELETE",
|
|
8
|
+
"OPTIONS",
|
|
9
|
+
"TRACE",
|
|
10
|
+
"CONNECT",
|
|
11
|
+
]);
|
|
12
|
+
/** Endpoint signatures are documentation text, never links or executable requests. */
|
|
13
|
+
export function endpointParagraph(node, source, report) {
|
|
14
|
+
if (node.type !== "leafDirective") {
|
|
15
|
+
report('Use the leaf form ::endpoint{method="GET" path="/example"}.');
|
|
16
|
+
return;
|
|
17
|
+
}
|
|
18
|
+
const offset = node.position?.start.offset;
|
|
19
|
+
if (node.children.length ||
|
|
20
|
+
(offset !== undefined && source.startsWith("::endpoint[", offset))) {
|
|
21
|
+
report("The endpoint component does not support a label.");
|
|
22
|
+
return;
|
|
23
|
+
}
|
|
24
|
+
const attributes = node.attributes ?? {};
|
|
25
|
+
if (Object.keys(attributes).some((key) => key !== "method" && key !== "path")) {
|
|
26
|
+
report("The endpoint component supports only method and path attributes.");
|
|
27
|
+
return;
|
|
28
|
+
}
|
|
29
|
+
const method = (attributes.method ?? "").toUpperCase();
|
|
30
|
+
if (!methods.has(method)) {
|
|
31
|
+
report("The endpoint method must be GET, HEAD, POST, PUT, PATCH, DELETE, OPTIONS, TRACE, or CONNECT.");
|
|
32
|
+
return;
|
|
33
|
+
}
|
|
34
|
+
const path = attributes.path ?? "";
|
|
35
|
+
if (!path.startsWith("/") ||
|
|
36
|
+
path.startsWith("//") ||
|
|
37
|
+
// biome-ignore lint/suspicious/noControlCharactersInRegex: Endpoint text cannot contain invisible controls or whitespace.
|
|
38
|
+
/[\s\u0000-\u001f\u007f-\u009f]/u.test(path)) {
|
|
39
|
+
report("The endpoint path must start with a single / and contain no whitespace or control characters.");
|
|
40
|
+
return;
|
|
41
|
+
}
|
|
42
|
+
return {
|
|
43
|
+
type: "paragraph",
|
|
44
|
+
position: node.position,
|
|
45
|
+
// Tight lists unwrap ordinary paragraphs; retain the signature's block hook.
|
|
46
|
+
data: { hName: "div", hProperties: { className: ["mdd-endpoint"] } },
|
|
47
|
+
children: [
|
|
48
|
+
{
|
|
49
|
+
type: "strong",
|
|
50
|
+
data: {
|
|
51
|
+
hProperties: {
|
|
52
|
+
className: [
|
|
53
|
+
"mdd-endpoint-method",
|
|
54
|
+
`mdd-method-${method.toLowerCase()}`,
|
|
55
|
+
],
|
|
56
|
+
},
|
|
57
|
+
},
|
|
58
|
+
children: [{ type: "text", value: method }],
|
|
59
|
+
},
|
|
60
|
+
{ type: "text", value: " " },
|
|
61
|
+
{
|
|
62
|
+
type: "inlineCode",
|
|
63
|
+
value: path,
|
|
64
|
+
data: { hProperties: { className: ["mdd-endpoint-path"] } },
|
|
65
|
+
},
|
|
66
|
+
],
|
|
67
|
+
};
|
|
68
|
+
}
|
package/dist/index.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { CompileOptions, CompileResult } from "./types.js";
|
|
2
|
-
export { type SearchIndex, type SearchOptions, type SearchPage, type SearchResult, type SearchSection, search, validateSearchIndex, } from "./search.js";
|
|
2
|
+
export { type SearchIndex, type SearchMatch, type SearchOptions, type SearchPage, type SearchResult, type SearchSection, search, validateSearchIndex, } from "./search.js";
|
|
3
3
|
export { createSearchIndex } from "./search-index.js";
|
|
4
4
|
export type { Asset, CompileOptions, CompileResult, Diagnostic, Footer, Heading, Metadata, NavigationItem, Page, Site, SocialLink, Theme, } from "./types.js";
|
|
5
5
|
/** Compile a local checkout; never fetch repositories, execute author code, or emit files. */
|
package/dist/markdown.js
CHANGED
|
@@ -13,6 +13,7 @@ import { unified } from "unified";
|
|
|
13
13
|
import { SKIP, visit } from "unist-util-visit";
|
|
14
14
|
import { parseDocument as parseYaml } from "yaml";
|
|
15
15
|
import { transformAlerts } from "./alerts.js";
|
|
16
|
+
import { endpointParagraph } from "./endpoints.js";
|
|
16
17
|
const parser = unified()
|
|
17
18
|
.use(remarkParse)
|
|
18
19
|
.use(remarkGfm)
|
|
@@ -29,10 +30,20 @@ const htmlRenderer = unified()
|
|
|
29
30
|
...defaultSchema.attributes,
|
|
30
31
|
aside: [["className", /^mdd-/]],
|
|
31
32
|
details: [["className", "mdd-details"]],
|
|
33
|
+
div: [["className", "mdd-endpoint"]],
|
|
32
34
|
p: [
|
|
33
35
|
...(defaultSchema.attributes?.p ?? []),
|
|
34
36
|
["className", "mdd-component-label"],
|
|
35
37
|
],
|
|
38
|
+
strong: [
|
|
39
|
+
...(defaultSchema.attributes?.strong ?? []),
|
|
40
|
+
[
|
|
41
|
+
"className",
|
|
42
|
+
"mdd-endpoint-method",
|
|
43
|
+
/^mdd-method-(get|head|post|put|patch|delete|options|trace|connect)$/,
|
|
44
|
+
],
|
|
45
|
+
],
|
|
46
|
+
code: [["className", /^language-./, "mdd-endpoint-path"]],
|
|
36
47
|
summary: [
|
|
37
48
|
...(defaultSchema.attributes?.summary ?? []),
|
|
38
49
|
["className", "mdd-component-label"],
|
|
@@ -206,8 +217,16 @@ export function parseDocument(source, file, diagnostics) {
|
|
|
206
217
|
report("REMOVED_COMPONENT", `The ${node.name} directive was removed; use > [!${node.name.toUpperCase()}] followed by quoted body lines. Keep any optional custom title as bold text in the alert body.`, node);
|
|
207
218
|
return;
|
|
208
219
|
}
|
|
220
|
+
if (node.name === "endpoint") {
|
|
221
|
+
const paragraph = endpointParagraph(node, source, (message) => report("INVALID_COMPONENT", message, node));
|
|
222
|
+
if (paragraph && parent && index !== undefined) {
|
|
223
|
+
parent.children[index] = paragraph;
|
|
224
|
+
return [SKIP, index];
|
|
225
|
+
}
|
|
226
|
+
return;
|
|
227
|
+
}
|
|
209
228
|
if (node.type !== "containerDirective" || node.name !== "details") {
|
|
210
|
-
report("UNKNOWN_COMPONENT", `Unsupported component ${node.name}; use a GitHub alert or
|
|
229
|
+
report("UNKNOWN_COMPONENT", `Unsupported component ${node.name}; use a GitHub alert, details container, or endpoint leaf.`, node);
|
|
211
230
|
return;
|
|
212
231
|
}
|
|
213
232
|
if (Object.keys(node.attributes ?? {}).length) {
|
package/dist/search-index.js
CHANGED
|
@@ -7,8 +7,9 @@ import { transformAlerts } from "./alerts.js";
|
|
|
7
7
|
import { validateSearchIndex, } from "./search.js";
|
|
8
8
|
const parser = unified().use(remarkParse).use(remarkGfm);
|
|
9
9
|
const blockTypes = new Set(["paragraph", "code", "tableCell", "listItem"]);
|
|
10
|
-
function indexPage(page) {
|
|
10
|
+
function indexPage(page, breadcrumbs) {
|
|
11
11
|
const sections = [{ title: "", url: page.url, text: "" }];
|
|
12
|
+
const ancestors = [];
|
|
12
13
|
let headingIndex = 0;
|
|
13
14
|
const chunks = [[]];
|
|
14
15
|
const tree = parser.parse(page.markdown);
|
|
@@ -23,11 +24,20 @@ function indexPage(page) {
|
|
|
23
24
|
heading.depth !== node.depth) {
|
|
24
25
|
throw new TypeError("Search indexing requires matching compiled Markdown and headings.");
|
|
25
26
|
}
|
|
27
|
+
while (ancestors.length &&
|
|
28
|
+
(ancestors.at(-1)?.depth ?? 0) >= heading.depth)
|
|
29
|
+
ancestors.pop();
|
|
26
30
|
sections.push({
|
|
27
31
|
title: heading.text,
|
|
28
32
|
url: `${page.url}#${encodeURIComponent(heading.id)}`,
|
|
29
33
|
text: "",
|
|
34
|
+
breadcrumbs: ancestors
|
|
35
|
+
.filter((ancestor) => !(ancestor.depth === 1 &&
|
|
36
|
+
ancestor.text.normalize("NFKC").toLowerCase() ===
|
|
37
|
+
page.title.normalize("NFKC").toLowerCase()))
|
|
38
|
+
.map((ancestor) => ancestor.text),
|
|
30
39
|
});
|
|
40
|
+
ancestors.push(heading);
|
|
31
41
|
chunks.push([]);
|
|
32
42
|
return SKIP;
|
|
33
43
|
}
|
|
@@ -55,12 +65,33 @@ function indexPage(page) {
|
|
|
55
65
|
url: page.url,
|
|
56
66
|
title: page.title,
|
|
57
67
|
description: page.description ?? "",
|
|
68
|
+
breadcrumbs,
|
|
58
69
|
sections,
|
|
59
70
|
};
|
|
60
71
|
}
|
|
61
72
|
/** Build only from successfully compiled pages; no source paths or HTML are indexed. */
|
|
62
73
|
export function createSearchIndex(site) {
|
|
63
|
-
const
|
|
74
|
+
const navigation = new Map();
|
|
75
|
+
const pending = site.navigation.map((item) => ({
|
|
76
|
+
item,
|
|
77
|
+
ancestors: [],
|
|
78
|
+
}));
|
|
79
|
+
while (pending.length) {
|
|
80
|
+
const entry = pending.pop();
|
|
81
|
+
if (!entry)
|
|
82
|
+
break;
|
|
83
|
+
const { item, ancestors } = entry;
|
|
84
|
+
if (ancestors.length > 64)
|
|
85
|
+
throw new TypeError("Search navigation exceeds 64 ancestor labels.");
|
|
86
|
+
if (item.url)
|
|
87
|
+
navigation.set(item.url, ancestors);
|
|
88
|
+
for (const child of item.children ?? [])
|
|
89
|
+
pending.push({ item: child, ancestors: [...ancestors, item.title] });
|
|
90
|
+
}
|
|
91
|
+
const index = {
|
|
92
|
+
version: 1,
|
|
93
|
+
pages: site.pages.map((page) => indexPage(page, navigation.get(page.url) ?? [])),
|
|
94
|
+
};
|
|
64
95
|
index.pages.sort((a, b) => (a.url < b.url ? -1 : a.url > b.url ? 1 : 0));
|
|
65
96
|
validateSearchIndex(index);
|
|
66
97
|
return index;
|
package/dist/search.d.ts
CHANGED
|
@@ -7,6 +7,8 @@ export interface SearchPage {
|
|
|
7
7
|
url: string;
|
|
8
8
|
title: string;
|
|
9
9
|
description: string;
|
|
10
|
+
/** Navigation ancestor labels, excluding this page's own label. */
|
|
11
|
+
breadcrumbs?: string[];
|
|
10
12
|
sections: SearchSection[];
|
|
11
13
|
}
|
|
12
14
|
export interface SearchSection {
|
|
@@ -14,20 +16,37 @@ export interface SearchSection {
|
|
|
14
16
|
title: string;
|
|
15
17
|
url: string;
|
|
16
18
|
text: string;
|
|
19
|
+
/** Ancestor headings, excluding this heading and a duplicate page-title H1. */
|
|
20
|
+
breadcrumbs?: string[];
|
|
17
21
|
}
|
|
18
22
|
export interface SearchOptions {
|
|
19
23
|
/** Maximum results, from 0 to 100. Defaults to 10. */
|
|
20
24
|
limit?: number;
|
|
25
|
+
/** Defaults to one result per page; sections returns independent heading hits. */
|
|
26
|
+
mode?: "pages" | "sections";
|
|
27
|
+
/** Allow one single-edit query correction per result. Requires sections mode. */
|
|
28
|
+
fuzzy?: boolean;
|
|
21
29
|
}
|
|
30
|
+
/** UTF-16 offsets into the original displayed field; end is exclusive. */
|
|
31
|
+
export type SearchMatch = [start: number, end: number];
|
|
22
32
|
export interface SearchResult {
|
|
33
|
+
kind: "page" | "section";
|
|
23
34
|
title: string;
|
|
24
35
|
url: string;
|
|
36
|
+
pageUrl: string;
|
|
37
|
+
breadcrumbs: string[];
|
|
25
38
|
/** Present when the destination is a compiled heading. */
|
|
26
39
|
section?: string;
|
|
27
40
|
excerpt: string;
|
|
28
41
|
score: number;
|
|
42
|
+
matches: {
|
|
43
|
+
title: SearchMatch[];
|
|
44
|
+
section: SearchMatch[];
|
|
45
|
+
excerpt: SearchMatch[];
|
|
46
|
+
breadcrumbs: SearchMatch[][];
|
|
47
|
+
};
|
|
29
48
|
}
|
|
30
49
|
/** Validate deserialized data before any result can become a link. */
|
|
31
50
|
export declare function validateSearchIndex(value: unknown): asserts value is SearchIndex;
|
|
32
|
-
/**
|
|
51
|
+
/** Query without filesystem, network, or DOM access; page mode preserves legacy ranking. */
|
|
33
52
|
export declare function search(index: SearchIndex, query: string, options?: SearchOptions): SearchResult[];
|
package/dist/search.js
CHANGED
|
@@ -1,6 +1,17 @@
|
|
|
1
1
|
function record(value) {
|
|
2
2
|
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
3
3
|
}
|
|
4
|
+
function breadcrumbs(value, maximum) {
|
|
5
|
+
if (value === undefined)
|
|
6
|
+
return true;
|
|
7
|
+
if (!Array.isArray(value) || value.length > maximum)
|
|
8
|
+
return false;
|
|
9
|
+
// Iterate holes too: a sparse array does not contain only string labels.
|
|
10
|
+
for (const label of value)
|
|
11
|
+
if (typeof label !== "string")
|
|
12
|
+
return false;
|
|
13
|
+
return true;
|
|
14
|
+
}
|
|
4
15
|
function pageUrl(value) {
|
|
5
16
|
if (typeof value !== "string" ||
|
|
6
17
|
!value.startsWith("/") ||
|
|
@@ -48,6 +59,7 @@ export function validateSearchIndex(value) {
|
|
|
48
59
|
urls.has(page.url) ||
|
|
49
60
|
typeof page.title !== "string" ||
|
|
50
61
|
typeof page.description !== "string" ||
|
|
62
|
+
!breadcrumbs(page.breadcrumbs, 64) ||
|
|
51
63
|
!Array.isArray(page.sections))
|
|
52
64
|
throw new TypeError("Invalid search index page.");
|
|
53
65
|
urls.add(page.url);
|
|
@@ -57,7 +69,8 @@ export function validateSearchIndex(value) {
|
|
|
57
69
|
!sectionUrl(section.url, page.url) ||
|
|
58
70
|
sections.has(section.url) ||
|
|
59
71
|
typeof section.title !== "string" ||
|
|
60
|
-
typeof section.text !== "string"
|
|
72
|
+
typeof section.text !== "string" ||
|
|
73
|
+
!breadcrumbs(section.breadcrumbs, 6))
|
|
61
74
|
throw new TypeError("Invalid search index section.");
|
|
62
75
|
sections.add(section.url);
|
|
63
76
|
}
|
|
@@ -70,29 +83,152 @@ function normalize(value) {
|
|
|
70
83
|
.replace(/[^\p{L}\p{N}\p{M}_]+/gu, " ")
|
|
71
84
|
.trim();
|
|
72
85
|
}
|
|
73
|
-
function
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
: characters.join("");
|
|
86
|
+
function contains(value, pattern) {
|
|
87
|
+
return pattern.whole
|
|
88
|
+
? ` ${value} `.includes(` ${pattern.text} `)
|
|
89
|
+
: value.includes(pattern.text);
|
|
78
90
|
}
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
const
|
|
83
|
-
|
|
84
|
-
|
|
91
|
+
function mergeRanges(ranges) {
|
|
92
|
+
ranges.sort((a, b) => a[0] - b[0] || a[1] - b[1]);
|
|
93
|
+
const merged = [];
|
|
94
|
+
for (const range of ranges) {
|
|
95
|
+
const previous = merged.at(-1);
|
|
96
|
+
if (previous && range[0] <= previous[1])
|
|
97
|
+
previous[1] = Math.max(previous[1], range[1]);
|
|
98
|
+
else
|
|
99
|
+
merged.push([...range]);
|
|
85
100
|
}
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
const
|
|
91
|
-
const
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
101
|
+
return merged;
|
|
102
|
+
}
|
|
103
|
+
/** Map normalized matches to complete original graphemes, including NFKC expansions. */
|
|
104
|
+
function matchRanges(value, patterns, firstOnly = false) {
|
|
105
|
+
const normalized = value.normalize("NFKC").toLowerCase();
|
|
106
|
+
const found = [];
|
|
107
|
+
for (const pattern of patterns) {
|
|
108
|
+
let offset = 0;
|
|
109
|
+
while (offset < normalized.length) {
|
|
110
|
+
const start = normalized.indexOf(pattern.text, offset);
|
|
111
|
+
if (start < 0)
|
|
112
|
+
break;
|
|
113
|
+
const end = start + pattern.text.length;
|
|
114
|
+
// One neighboring code point needs at most two UTF-16 units. Avoid scanning
|
|
115
|
+
// a growing prefix for each occurrence, or any boundary work for substrings.
|
|
116
|
+
if (!pattern.whole ||
|
|
117
|
+
(!/[\p{L}\p{N}\p{M}_]$/u.test(normalized.slice(Math.max(0, start - 2), start)) &&
|
|
118
|
+
!/^[\p{L}\p{N}\p{M}_]/u.test(normalized.slice(end, end + 2)))) {
|
|
119
|
+
found.push([start, end]);
|
|
120
|
+
if (firstOnly)
|
|
121
|
+
break;
|
|
122
|
+
}
|
|
123
|
+
offset = start + 1;
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
let ranges = mergeRanges(found);
|
|
127
|
+
if (!ranges.length)
|
|
95
128
|
return [];
|
|
129
|
+
if (firstOnly)
|
|
130
|
+
ranges = ranges.slice(0, 1);
|
|
131
|
+
const groups = [];
|
|
132
|
+
for (const { segment, index } of new Intl.Segmenter(undefined, {
|
|
133
|
+
granularity: "grapheme",
|
|
134
|
+
}).segment(value)) {
|
|
135
|
+
let group = {
|
|
136
|
+
text: segment.normalize("NFKC"),
|
|
137
|
+
start: index,
|
|
138
|
+
end: index + segment.length,
|
|
139
|
+
};
|
|
140
|
+
// NFKC can combine adjacent graphemes, e.g. compatibility Hangul ㄱㅏ → 가.
|
|
141
|
+
while (groups.length) {
|
|
142
|
+
const previous = groups.at(-1);
|
|
143
|
+
if (!previous)
|
|
144
|
+
break;
|
|
145
|
+
const joined = previous.text + group.text;
|
|
146
|
+
const combined = joined.normalize("NFKC");
|
|
147
|
+
if (combined === joined)
|
|
148
|
+
break;
|
|
149
|
+
groups.pop();
|
|
150
|
+
group = { text: combined, start: previous.start, end: group.end };
|
|
151
|
+
}
|
|
152
|
+
groups.push(group);
|
|
153
|
+
}
|
|
154
|
+
// Whole-field lowercase retains contextual letters such as final sigma.
|
|
155
|
+
// Per-group lowercase supplies only lengths, including expansions like İ.
|
|
156
|
+
const mapped = [];
|
|
157
|
+
let offset = 0;
|
|
158
|
+
let rangeIndex = 0;
|
|
159
|
+
let originalStart = 0;
|
|
160
|
+
for (const group of groups) {
|
|
161
|
+
const next = offset + group.text.toLowerCase().length;
|
|
162
|
+
let range = ranges[rangeIndex];
|
|
163
|
+
while (range && range[0] < next) {
|
|
164
|
+
if (range[0] >= offset)
|
|
165
|
+
originalStart = group.start;
|
|
166
|
+
if (range[1] > next)
|
|
167
|
+
break;
|
|
168
|
+
mapped.push([originalStart, group.end]);
|
|
169
|
+
range = ranges[++rangeIndex];
|
|
170
|
+
}
|
|
171
|
+
if (!range)
|
|
172
|
+
break;
|
|
173
|
+
offset = next;
|
|
174
|
+
}
|
|
175
|
+
return mergeRanges(mapped);
|
|
176
|
+
}
|
|
177
|
+
function excerpt(value, patterns) {
|
|
178
|
+
const text = value.replace(/\s+/gu, " ").trim();
|
|
179
|
+
const characters = Array.from(text);
|
|
180
|
+
if (characters.length <= 160)
|
|
181
|
+
return text;
|
|
182
|
+
const match = matchRanges(text, patterns, true)[0];
|
|
183
|
+
const matchStart = match ? Array.from(text.slice(0, match[0])).length : 0;
|
|
184
|
+
const matchLength = match
|
|
185
|
+
? Array.from(text.slice(match[0], match[1])).length
|
|
186
|
+
: 0;
|
|
187
|
+
// Reserve room for both ellipses and the matched term before adding context.
|
|
188
|
+
const context = Math.min(40, Math.max(0, 158 - matchLength));
|
|
189
|
+
const start = Math.min(Math.max(0, matchStart - context), characters.length - 159);
|
|
190
|
+
let end = Math.min(characters.length, start + 160 - (start > 0 ? 1 : 0));
|
|
191
|
+
if (end < characters.length)
|
|
192
|
+
end--;
|
|
193
|
+
const startOffset = characters.slice(0, start).join("").length;
|
|
194
|
+
const endOffset = characters.slice(0, end).join("").length;
|
|
195
|
+
// Move clipped edges inward to complete graphemes without exceeding the cap.
|
|
196
|
+
let safeStart = text.length;
|
|
197
|
+
let safeEnd = 0;
|
|
198
|
+
for (const { segment, index } of new Intl.Segmenter(undefined, {
|
|
199
|
+
granularity: "grapheme",
|
|
200
|
+
}).segment(text)) {
|
|
201
|
+
if (index >= startOffset && safeStart === text.length)
|
|
202
|
+
safeStart = index;
|
|
203
|
+
if (index + segment.length <= endOffset)
|
|
204
|
+
safeEnd = index + segment.length;
|
|
205
|
+
if (index >= endOffset)
|
|
206
|
+
break;
|
|
207
|
+
}
|
|
208
|
+
return `${safeStart > 0 ? "…" : ""}${text.slice(safeStart, safeEnd)}${safeEnd < text.length ? "…" : ""}`;
|
|
209
|
+
}
|
|
210
|
+
function candidateUrl(candidate) {
|
|
211
|
+
return candidate.destination?.url ?? candidate.page.url;
|
|
212
|
+
}
|
|
213
|
+
function compareCandidates(a, b) {
|
|
214
|
+
const aUrl = candidateUrl(a);
|
|
215
|
+
const bUrl = candidateUrl(b);
|
|
216
|
+
return (b.rank - a.rank ||
|
|
217
|
+
b.score - a.score ||
|
|
218
|
+
(aUrl < bUrl ? -1 : aUrl > bUrl ? 1 : 0));
|
|
219
|
+
}
|
|
220
|
+
function sourceText(sources, patterns, fallback) {
|
|
221
|
+
return (sources.find((text) => {
|
|
222
|
+
const normalized = normalize(text);
|
|
223
|
+
return patterns.some((pattern) => contains(normalized, pattern));
|
|
224
|
+
}) ??
|
|
225
|
+
sources.find((text) => text) ??
|
|
226
|
+
fallback);
|
|
227
|
+
}
|
|
228
|
+
/** Preserve page-mode weighting, cross-section coverage, and destination selection. */
|
|
229
|
+
function pageCandidates(index, terms) {
|
|
230
|
+
const phrase = terms.join(" ");
|
|
231
|
+
const patterns = terms.map((text) => ({ text }));
|
|
96
232
|
const results = [];
|
|
97
233
|
for (const page of index.pages) {
|
|
98
234
|
const title = normalize(page.title);
|
|
@@ -131,23 +267,219 @@ export function search(index, query, options = {}) {
|
|
|
131
267
|
// A stable sort preserves document order when multiple sections tie.
|
|
132
268
|
matches.sort((a, b) => b.score - a.score);
|
|
133
269
|
const best = matches[0]?.section;
|
|
134
|
-
const destination = metadataMatch ? undefined : best;
|
|
135
270
|
results.push({
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
page.description
|
|
141
|
-
page.sections.
|
|
142
|
-
|
|
271
|
+
page,
|
|
272
|
+
destination: metadataMatch ? undefined : best,
|
|
273
|
+
source: sourceText([
|
|
274
|
+
best?.text ?? "",
|
|
275
|
+
page.description,
|
|
276
|
+
...page.sections.map((section) => section.text),
|
|
277
|
+
], patterns, page.title),
|
|
143
278
|
score: weights.reduce((sum, weight) => sum + weight, 0) +
|
|
144
279
|
(title === phrase
|
|
145
280
|
? 24
|
|
146
281
|
: sections.some((section) => section.title === phrase)
|
|
147
282
|
? 12
|
|
148
283
|
: 0),
|
|
284
|
+
rank: 0,
|
|
285
|
+
patterns,
|
|
149
286
|
});
|
|
150
287
|
}
|
|
151
|
-
results
|
|
152
|
-
|
|
288
|
+
return results;
|
|
289
|
+
}
|
|
290
|
+
/** Single insertion, deletion, substitution, or adjacent transposition, on code points. */
|
|
291
|
+
function singleEdit(query, token) {
|
|
292
|
+
if (token.length > 66)
|
|
293
|
+
return false;
|
|
294
|
+
const text = Array.from(token);
|
|
295
|
+
if (Math.abs(query.length - text.length) > 1)
|
|
296
|
+
return false;
|
|
297
|
+
let start = 0;
|
|
298
|
+
while (start < query.length && query[start] === text[start])
|
|
299
|
+
start++;
|
|
300
|
+
if (start === Math.min(query.length, text.length))
|
|
301
|
+
return true;
|
|
302
|
+
if (query.length === text.length) {
|
|
303
|
+
if (query.slice(start + 1).join("") === text.slice(start + 1).join(""))
|
|
304
|
+
return true;
|
|
305
|
+
return (query[start] === text[start + 1] &&
|
|
306
|
+
query[start + 1] === text[start] &&
|
|
307
|
+
query.slice(start + 2).join("") === text.slice(start + 2).join(""));
|
|
308
|
+
}
|
|
309
|
+
return query.length > text.length
|
|
310
|
+
? query.slice(start + 1).join("") === text.slice(start).join("")
|
|
311
|
+
: query.slice(start).join("") === text.slice(start + 1).join("");
|
|
312
|
+
}
|
|
313
|
+
function fieldScore(fields, pattern) {
|
|
314
|
+
return Math.max(0, ...fields
|
|
315
|
+
.filter((field) => contains(field.text, pattern))
|
|
316
|
+
.map((field) => field.weight));
|
|
317
|
+
}
|
|
318
|
+
function matchFields(fields, terms, fuzzy, required = []) {
|
|
319
|
+
const patterns = terms.map((text) => ({ text }));
|
|
320
|
+
const weights = patterns.map((pattern) => fieldScore(fields, pattern));
|
|
321
|
+
const missing = weights.flatMap((weight, index) => (weight ? [] : [index]));
|
|
322
|
+
if (!missing.length)
|
|
323
|
+
return {
|
|
324
|
+
patterns,
|
|
325
|
+
score: weights.reduce((a, b) => a + b, 0),
|
|
326
|
+
corrected: false,
|
|
327
|
+
};
|
|
328
|
+
const missingIndex = missing[0];
|
|
329
|
+
if (!fuzzy || missing.length !== 1 || missingIndex === undefined)
|
|
330
|
+
return;
|
|
331
|
+
const query = Array.from(terms[missingIndex] ?? "");
|
|
332
|
+
if (query.length < 4 || query.length > 32)
|
|
333
|
+
return;
|
|
334
|
+
const needsOwnCorrection = required.length > 0 &&
|
|
335
|
+
!patterns.some((pattern) => fieldScore(required, pattern));
|
|
336
|
+
let correction;
|
|
337
|
+
let correctionWeight = 0;
|
|
338
|
+
for (const field of fields) {
|
|
339
|
+
if (field.weight < correctionWeight)
|
|
340
|
+
continue;
|
|
341
|
+
for (const token of field.text.split(" ")) {
|
|
342
|
+
if (singleEdit(query, token) &&
|
|
343
|
+
(!needsOwnCorrection ||
|
|
344
|
+
fieldScore(required, { text: token, whole: true }) > 0) &&
|
|
345
|
+
(field.weight > correctionWeight ||
|
|
346
|
+
correction === undefined ||
|
|
347
|
+
token < correction)) {
|
|
348
|
+
correction = token;
|
|
349
|
+
correctionWeight = field.weight;
|
|
350
|
+
}
|
|
351
|
+
}
|
|
352
|
+
}
|
|
353
|
+
if (!correction)
|
|
354
|
+
return;
|
|
355
|
+
patterns[missingIndex] = { text: correction, whole: true };
|
|
356
|
+
weights[missingIndex] = correctionWeight;
|
|
357
|
+
return {
|
|
358
|
+
patterns,
|
|
359
|
+
score: weights.reduce((a, b) => a + b, 0),
|
|
360
|
+
corrected: true,
|
|
361
|
+
};
|
|
362
|
+
}
|
|
363
|
+
/** Score only page metadata plus a single coherent section, never sibling bodies. */
|
|
364
|
+
function sectionCandidates(index, terms, fuzzy) {
|
|
365
|
+
const phrase = terms.join(" ");
|
|
366
|
+
const results = [];
|
|
367
|
+
for (const page of index.pages) {
|
|
368
|
+
const title = normalize(page.title);
|
|
369
|
+
const metadata = [
|
|
370
|
+
{ text: title, weight: 8 },
|
|
371
|
+
{ text: normalize(page.description), weight: 2 },
|
|
372
|
+
];
|
|
373
|
+
const pageMatch = matchFields(metadata, terms, fuzzy);
|
|
374
|
+
let pageCandidate;
|
|
375
|
+
if (pageMatch) {
|
|
376
|
+
const exact = title === phrase;
|
|
377
|
+
pageCandidate = {
|
|
378
|
+
page,
|
|
379
|
+
source: sourceText([page.description, ...page.sections.map((section) => section.text)], pageMatch.patterns, page.title),
|
|
380
|
+
score: pageMatch.score + (exact ? 24 : 0),
|
|
381
|
+
rank: pageMatch.corrected ? 0 : exact ? 2 : 1,
|
|
382
|
+
patterns: pageMatch.patterns,
|
|
383
|
+
};
|
|
384
|
+
}
|
|
385
|
+
for (const section of page.sections) {
|
|
386
|
+
const heading = normalize(section.title);
|
|
387
|
+
const body = normalize(section.text);
|
|
388
|
+
const ownFields = [
|
|
389
|
+
{ text: heading, weight: 4 },
|
|
390
|
+
{ text: body, weight: 1 },
|
|
391
|
+
];
|
|
392
|
+
const match = matchFields([...metadata, ...ownFields], terms, fuzzy, ownFields);
|
|
393
|
+
if (!match?.patterns.some((pattern) => fieldScore(ownFields, pattern)))
|
|
394
|
+
continue;
|
|
395
|
+
// Repeated page headings without their own matching body add no useful destination.
|
|
396
|
+
if (pageMatch &&
|
|
397
|
+
heading === title &&
|
|
398
|
+
!match.patterns.some((pattern) => contains(body, pattern)))
|
|
399
|
+
continue;
|
|
400
|
+
const exact = heading === phrase;
|
|
401
|
+
const candidate = {
|
|
402
|
+
page,
|
|
403
|
+
destination: section.url === page.url ? undefined : section,
|
|
404
|
+
source: sourceText([section.text, page.description], match.patterns, section.title || page.title),
|
|
405
|
+
score: match.score + (exact ? 12 : 0),
|
|
406
|
+
rank: match.corrected ? 0 : exact ? 2 : 1,
|
|
407
|
+
patterns: match.patterns,
|
|
408
|
+
};
|
|
409
|
+
if (candidate.destination)
|
|
410
|
+
results.push(candidate);
|
|
411
|
+
else if (!pageCandidate ||
|
|
412
|
+
compareCandidates(candidate, pageCandidate) < 0)
|
|
413
|
+
pageCandidate = candidate;
|
|
414
|
+
}
|
|
415
|
+
if (pageCandidate)
|
|
416
|
+
results.push(pageCandidate);
|
|
417
|
+
}
|
|
418
|
+
return results;
|
|
419
|
+
}
|
|
420
|
+
// Greater than the largest raw score: 32 terms × weight 8 + title bonus 24.
|
|
421
|
+
const sectionTierWeight = 281;
|
|
422
|
+
function displayResult(candidate) {
|
|
423
|
+
const { page, destination, patterns } = candidate;
|
|
424
|
+
const isSection = destination !== undefined && destination.url !== page.url;
|
|
425
|
+
const trail = isSection
|
|
426
|
+
? [
|
|
427
|
+
...(page.breadcrumbs ?? []),
|
|
428
|
+
page.title,
|
|
429
|
+
...(destination.breadcrumbs ?? []),
|
|
430
|
+
]
|
|
431
|
+
: [...(page.breadcrumbs ?? [])];
|
|
432
|
+
const snippet = excerpt(candidate.source, patterns);
|
|
433
|
+
return {
|
|
434
|
+
kind: isSection ? "section" : "page",
|
|
435
|
+
title: page.title,
|
|
436
|
+
pageUrl: page.url,
|
|
437
|
+
url: candidateUrl(candidate),
|
|
438
|
+
...(destination?.title ? { section: destination.title } : {}),
|
|
439
|
+
breadcrumbs: trail,
|
|
440
|
+
excerpt: snippet,
|
|
441
|
+
score: candidate.rank * sectionTierWeight + candidate.score,
|
|
442
|
+
matches: {
|
|
443
|
+
title: matchRanges(page.title, patterns),
|
|
444
|
+
section: matchRanges(destination?.title ?? "", patterns),
|
|
445
|
+
excerpt: matchRanges(snippet, patterns),
|
|
446
|
+
breadcrumbs: trail.map((label) => matchRanges(label, patterns)),
|
|
447
|
+
},
|
|
448
|
+
};
|
|
449
|
+
}
|
|
450
|
+
/** Query without filesystem, network, or DOM access; page mode preserves legacy ranking. */
|
|
451
|
+
export function search(index, query, options = {}) {
|
|
452
|
+
validateSearchIndex(index);
|
|
453
|
+
if (options === null || typeof options !== "object" || Array.isArray(options))
|
|
454
|
+
throw new TypeError("Search options must be an object.");
|
|
455
|
+
const limit = options.limit ?? 10;
|
|
456
|
+
const mode = options.mode === undefined ? "pages" : options.mode;
|
|
457
|
+
const fuzzy = options.fuzzy === undefined ? false : options.fuzzy;
|
|
458
|
+
if (!Number.isInteger(limit) || limit < 0 || limit > 100)
|
|
459
|
+
throw new RangeError("Search limit must be an integer from 0 to 100.");
|
|
460
|
+
if (mode !== "pages" && mode !== "sections")
|
|
461
|
+
throw new RangeError("Search mode must be pages or sections.");
|
|
462
|
+
if (typeof fuzzy !== "boolean")
|
|
463
|
+
throw new TypeError("Search fuzzy option must be a boolean.");
|
|
464
|
+
if (fuzzy && mode !== "sections")
|
|
465
|
+
throw new RangeError("Fuzzy search requires sections mode.");
|
|
466
|
+
if (typeof query !== "string")
|
|
467
|
+
throw new TypeError("Search query must be a string.");
|
|
468
|
+
if (query.length > 512)
|
|
469
|
+
throw new RangeError("Search query must not exceed 512 characters.");
|
|
470
|
+
const terms = [...new Set(normalize(query).split(" ").filter(Boolean))];
|
|
471
|
+
if (terms.length > 32)
|
|
472
|
+
throw new RangeError("Search query must not exceed 32 unique terms.");
|
|
473
|
+
if (!terms.length || !limit)
|
|
474
|
+
return [];
|
|
475
|
+
const results = mode === "sections"
|
|
476
|
+
? sectionCandidates(index, terms, false)
|
|
477
|
+
: pageCandidates(index, terms);
|
|
478
|
+
if (fuzzy && results.length < limit) {
|
|
479
|
+
const literalUrls = new Set(results.map(candidateUrl));
|
|
480
|
+
results.push(...sectionCandidates(index, terms, true).filter((candidate) => candidate.rank === 0 && !literalUrls.has(candidateUrl(candidate))));
|
|
481
|
+
}
|
|
482
|
+
results.sort(compareCandidates);
|
|
483
|
+
// Original-text mapping and excerpt windows are computed only for final hits.
|
|
484
|
+
return results.slice(0, limit).map(displayResult);
|
|
153
485
|
}
|