@wgtechlabs/mdd-engine 0.1.0-pr.04f8fe2 → 0.1.0-pr.0cb765a
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +0 -4
- package/dist/index.d.ts +1 -1
- package/dist/search-index.js +2 -33
- package/dist/search.d.ts +1 -20
- package/dist/search.js +62 -309
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -85,10 +85,6 @@ import { search } from '@wgtechlabs/mdd-engine/search';
|
|
|
85
85
|
const index = createSearchIndex(result.site);
|
|
86
86
|
const restored = JSON.parse(JSON.stringify(index));
|
|
87
87
|
console.log(search(restored, 'installation', { limit: 10 }));
|
|
88
|
-
// Multiple heading hits, breadcrumbs, and ranges for UI highlights.
|
|
89
|
-
console.log(search(restored, 'instalation', {
|
|
90
|
-
mode: 'sections', fuzzy: true, limit: 8,
|
|
91
|
-
}));
|
|
92
88
|
// [{ title, url, section?, excerpt, score }]
|
|
93
89
|
```
|
|
94
90
|
|
package/dist/index.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { CompileOptions, CompileResult } from "./types.js";
|
|
2
|
-
export { type SearchIndex, type
|
|
2
|
+
export { type SearchIndex, type SearchOptions, type SearchPage, type SearchResult, type SearchSection, search, validateSearchIndex, } from "./search.js";
|
|
3
3
|
export { createSearchIndex } from "./search-index.js";
|
|
4
4
|
export type { Asset, CompileOptions, CompileResult, Diagnostic, Footer, Heading, Metadata, NavigationItem, Page, Site, SocialLink, Theme, } from "./types.js";
|
|
5
5
|
/** Compile a local checkout; never fetch repositories, execute author code, or emit files. */
|
package/dist/search-index.js
CHANGED
|
@@ -7,9 +7,8 @@ import { transformAlerts } from "./alerts.js";
|
|
|
7
7
|
import { validateSearchIndex, } from "./search.js";
|
|
8
8
|
const parser = unified().use(remarkParse).use(remarkGfm);
|
|
9
9
|
const blockTypes = new Set(["paragraph", "code", "tableCell", "listItem"]);
|
|
10
|
-
function indexPage(page
|
|
10
|
+
function indexPage(page) {
|
|
11
11
|
const sections = [{ title: "", url: page.url, text: "" }];
|
|
12
|
-
const ancestors = [];
|
|
13
12
|
let headingIndex = 0;
|
|
14
13
|
const chunks = [[]];
|
|
15
14
|
const tree = parser.parse(page.markdown);
|
|
@@ -24,20 +23,11 @@ function indexPage(page, breadcrumbs) {
|
|
|
24
23
|
heading.depth !== node.depth) {
|
|
25
24
|
throw new TypeError("Search indexing requires matching compiled Markdown and headings.");
|
|
26
25
|
}
|
|
27
|
-
while (ancestors.length &&
|
|
28
|
-
(ancestors.at(-1)?.depth ?? 0) >= heading.depth)
|
|
29
|
-
ancestors.pop();
|
|
30
26
|
sections.push({
|
|
31
27
|
title: heading.text,
|
|
32
28
|
url: `${page.url}#${encodeURIComponent(heading.id)}`,
|
|
33
29
|
text: "",
|
|
34
|
-
breadcrumbs: ancestors
|
|
35
|
-
.filter((ancestor) => !(ancestor.depth === 1 &&
|
|
36
|
-
ancestor.text.normalize("NFKC").toLowerCase() ===
|
|
37
|
-
page.title.normalize("NFKC").toLowerCase()))
|
|
38
|
-
.map((ancestor) => ancestor.text),
|
|
39
30
|
});
|
|
40
|
-
ancestors.push(heading);
|
|
41
31
|
chunks.push([]);
|
|
42
32
|
return SKIP;
|
|
43
33
|
}
|
|
@@ -65,33 +55,12 @@ function indexPage(page, breadcrumbs) {
|
|
|
65
55
|
url: page.url,
|
|
66
56
|
title: page.title,
|
|
67
57
|
description: page.description ?? "",
|
|
68
|
-
breadcrumbs,
|
|
69
58
|
sections,
|
|
70
59
|
};
|
|
71
60
|
}
|
|
72
61
|
/** Build only from successfully compiled pages; no source paths or HTML are indexed. */
|
|
73
62
|
export function createSearchIndex(site) {
|
|
74
|
-
const
|
|
75
|
-
const pending = site.navigation.map((item) => ({
|
|
76
|
-
item,
|
|
77
|
-
ancestors: [],
|
|
78
|
-
}));
|
|
79
|
-
while (pending.length) {
|
|
80
|
-
const entry = pending.pop();
|
|
81
|
-
if (!entry)
|
|
82
|
-
break;
|
|
83
|
-
const { item, ancestors } = entry;
|
|
84
|
-
if (ancestors.length > 64)
|
|
85
|
-
throw new TypeError("Search navigation exceeds 64 ancestor labels.");
|
|
86
|
-
if (item.url)
|
|
87
|
-
navigation.set(item.url, ancestors);
|
|
88
|
-
for (const child of item.children ?? [])
|
|
89
|
-
pending.push({ item: child, ancestors: [...ancestors, item.title] });
|
|
90
|
-
}
|
|
91
|
-
const index = {
|
|
92
|
-
version: 1,
|
|
93
|
-
pages: site.pages.map((page) => indexPage(page, navigation.get(page.url) ?? [])),
|
|
94
|
-
};
|
|
63
|
+
const index = { version: 1, pages: site.pages.map(indexPage) };
|
|
95
64
|
index.pages.sort((a, b) => (a.url < b.url ? -1 : a.url > b.url ? 1 : 0));
|
|
96
65
|
validateSearchIndex(index);
|
|
97
66
|
return index;
|
package/dist/search.d.ts
CHANGED
|
@@ -7,8 +7,6 @@ export interface SearchPage {
|
|
|
7
7
|
url: string;
|
|
8
8
|
title: string;
|
|
9
9
|
description: string;
|
|
10
|
-
/** Navigation ancestor labels, excluding this page's own label. */
|
|
11
|
-
breadcrumbs?: string[];
|
|
12
10
|
sections: SearchSection[];
|
|
13
11
|
}
|
|
14
12
|
export interface SearchSection {
|
|
@@ -16,37 +14,20 @@ export interface SearchSection {
|
|
|
16
14
|
title: string;
|
|
17
15
|
url: string;
|
|
18
16
|
text: string;
|
|
19
|
-
/** Ancestor headings, excluding this heading and a duplicate page-title H1. */
|
|
20
|
-
breadcrumbs?: string[];
|
|
21
17
|
}
|
|
22
18
|
export interface SearchOptions {
|
|
23
19
|
/** Maximum results, from 0 to 100. Defaults to 10. */
|
|
24
20
|
limit?: number;
|
|
25
|
-
/** Defaults to one result per page; sections returns independent heading hits. */
|
|
26
|
-
mode?: "pages" | "sections";
|
|
27
|
-
/** Allow one single-edit query correction per result. Requires sections mode. */
|
|
28
|
-
fuzzy?: boolean;
|
|
29
21
|
}
|
|
30
|
-
/** UTF-16 offsets into the original displayed field; end is exclusive. */
|
|
31
|
-
export type SearchMatch = [start: number, end: number];
|
|
32
22
|
export interface SearchResult {
|
|
33
|
-
kind: "page" | "section";
|
|
34
23
|
title: string;
|
|
35
24
|
url: string;
|
|
36
|
-
pageUrl: string;
|
|
37
|
-
breadcrumbs: string[];
|
|
38
25
|
/** Present when the destination is a compiled heading. */
|
|
39
26
|
section?: string;
|
|
40
27
|
excerpt: string;
|
|
41
28
|
score: number;
|
|
42
|
-
matches: {
|
|
43
|
-
title: SearchMatch[];
|
|
44
|
-
section: SearchMatch[];
|
|
45
|
-
excerpt: SearchMatch[];
|
|
46
|
-
breadcrumbs: SearchMatch[][];
|
|
47
|
-
};
|
|
48
29
|
}
|
|
49
30
|
/** Validate deserialized data before any result can become a link. */
|
|
50
31
|
export declare function validateSearchIndex(value: unknown): asserts value is SearchIndex;
|
|
51
|
-
/**
|
|
32
|
+
/** Return at most one hit per page, without filesystem, network, or DOM access. */
|
|
52
33
|
export declare function search(index: SearchIndex, query: string, options?: SearchOptions): SearchResult[];
|
package/dist/search.js
CHANGED
|
@@ -1,12 +1,6 @@
|
|
|
1
1
|
function record(value) {
|
|
2
2
|
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
3
3
|
}
|
|
4
|
-
function breadcrumbs(value, maximum) {
|
|
5
|
-
return (value === undefined ||
|
|
6
|
-
(Array.isArray(value) &&
|
|
7
|
-
value.length <= maximum &&
|
|
8
|
-
value.every((label) => typeof label === "string")));
|
|
9
|
-
}
|
|
10
4
|
function pageUrl(value) {
|
|
11
5
|
if (typeof value !== "string" ||
|
|
12
6
|
!value.startsWith("/") ||
|
|
@@ -54,7 +48,6 @@ export function validateSearchIndex(value) {
|
|
|
54
48
|
urls.has(page.url) ||
|
|
55
49
|
typeof page.title !== "string" ||
|
|
56
50
|
typeof page.description !== "string" ||
|
|
57
|
-
!breadcrumbs(page.breadcrumbs, 64) ||
|
|
58
51
|
!Array.isArray(page.sections))
|
|
59
52
|
throw new TypeError("Invalid search index page.");
|
|
60
53
|
urls.add(page.url);
|
|
@@ -64,8 +57,7 @@ export function validateSearchIndex(value) {
|
|
|
64
57
|
!sectionUrl(section.url, page.url) ||
|
|
65
58
|
sections.has(section.url) ||
|
|
66
59
|
typeof section.title !== "string" ||
|
|
67
|
-
typeof section.text !== "string"
|
|
68
|
-
!breadcrumbs(section.breadcrumbs, 6))
|
|
60
|
+
typeof section.text !== "string")
|
|
69
61
|
throw new TypeError("Invalid search index section.");
|
|
70
62
|
sections.add(section.url);
|
|
71
63
|
}
|
|
@@ -78,50 +70,23 @@ function normalize(value) {
|
|
|
78
70
|
.replace(/[^\p{L}\p{N}\p{M}_]+/gu, " ")
|
|
79
71
|
.trim();
|
|
80
72
|
}
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
? ` ${value} `.includes(` ${pattern.text} `)
|
|
84
|
-
: value.includes(pattern.text);
|
|
85
|
-
}
|
|
86
|
-
function mergeRanges(ranges) {
|
|
87
|
-
ranges.sort((a, b) => a[0] - b[0] || a[1] - b[1]);
|
|
88
|
-
const merged = [];
|
|
89
|
-
for (const range of ranges) {
|
|
90
|
-
const previous = merged.at(-1);
|
|
91
|
-
if (previous && range[0] <= previous[1])
|
|
92
|
-
previous[1] = Math.max(previous[1], range[1]);
|
|
93
|
-
else
|
|
94
|
-
merged.push([...range]);
|
|
95
|
-
}
|
|
96
|
-
return merged;
|
|
97
|
-
}
|
|
98
|
-
/** Map normalized matches to complete original graphemes, including NFKC expansions. */
|
|
99
|
-
function matchRanges(value, patterns, firstOnly = false) {
|
|
73
|
+
/** Map a normalized match back to the original text, including NFKC expansions. */
|
|
74
|
+
function matchRange(value, terms) {
|
|
100
75
|
const normalized = value.normalize("NFKC").toLowerCase();
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
const after = /^[\p{L}\p{N}\p{M}_]/u.test(normalized.slice(end));
|
|
112
|
-
if (!pattern.whole || (!before && !after)) {
|
|
113
|
-
found.push([start, end]);
|
|
114
|
-
if (firstOnly)
|
|
115
|
-
break;
|
|
116
|
-
}
|
|
117
|
-
offset = start + 1;
|
|
76
|
+
let start = -1;
|
|
77
|
+
let end = -1;
|
|
78
|
+
for (const term of terms) {
|
|
79
|
+
const found = normalized.indexOf(term);
|
|
80
|
+
if (found >= 0 &&
|
|
81
|
+
(start < 0 ||
|
|
82
|
+
found < start ||
|
|
83
|
+
(found === start && found + term.length > end))) {
|
|
84
|
+
start = found;
|
|
85
|
+
end = found + term.length;
|
|
118
86
|
}
|
|
119
87
|
}
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
return [];
|
|
123
|
-
if (firstOnly)
|
|
124
|
-
ranges = ranges.slice(0, 1);
|
|
88
|
+
if (start < 0)
|
|
89
|
+
return;
|
|
125
90
|
const groups = [];
|
|
126
91
|
for (const { segment, index } of new Intl.Segmenter(undefined, {
|
|
127
92
|
granularity: "grapheme",
|
|
@@ -145,35 +110,25 @@ function matchRanges(value, patterns, firstOnly = false) {
|
|
|
145
110
|
}
|
|
146
111
|
groups.push(group);
|
|
147
112
|
}
|
|
148
|
-
//
|
|
149
|
-
// Per-group lowercase
|
|
150
|
-
const mapped = [];
|
|
113
|
+
// Match against whole-field lowercase for contextual letters such as sigma.
|
|
114
|
+
// Per-group lowercase is used only for lengths (including expansions like İ).
|
|
151
115
|
let offset = 0;
|
|
152
|
-
let rangeIndex = 0;
|
|
153
116
|
let originalStart = 0;
|
|
154
117
|
for (const group of groups) {
|
|
155
118
|
const next = offset + group.text.toLowerCase().length;
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
if (range[1] > next)
|
|
161
|
-
break;
|
|
162
|
-
mapped.push([originalStart, group.end]);
|
|
163
|
-
range = ranges[++rangeIndex];
|
|
164
|
-
}
|
|
165
|
-
if (!range)
|
|
166
|
-
break;
|
|
119
|
+
if (offset <= start && start < next)
|
|
120
|
+
originalStart = group.start;
|
|
121
|
+
if (offset < end && end <= next)
|
|
122
|
+
return [originalStart, group.end];
|
|
167
123
|
offset = next;
|
|
168
124
|
}
|
|
169
|
-
return mergeRanges(mapped);
|
|
170
125
|
}
|
|
171
|
-
function excerpt(value,
|
|
126
|
+
function excerpt(value, terms) {
|
|
172
127
|
const text = value.replace(/\s+/gu, " ").trim();
|
|
173
128
|
const characters = Array.from(text);
|
|
174
129
|
if (characters.length <= 160)
|
|
175
130
|
return text;
|
|
176
|
-
const match =
|
|
131
|
+
const match = matchRange(text, terms);
|
|
177
132
|
const matchStart = match ? Array.from(text.slice(0, match[0])).length : 0;
|
|
178
133
|
const matchLength = match
|
|
179
134
|
? Array.from(text.slice(match[0], match[1])).length
|
|
@@ -184,45 +139,25 @@ function excerpt(value, patterns) {
|
|
|
184
139
|
let end = Math.min(characters.length, start + 160 - (start > 0 ? 1 : 0));
|
|
185
140
|
if (end < characters.length)
|
|
186
141
|
end--;
|
|
187
|
-
|
|
188
|
-
const endOffset = characters.slice(0, end).join("").length;
|
|
189
|
-
// Move clipped edges inward to complete graphemes without exceeding the cap.
|
|
190
|
-
let safeStart = text.length;
|
|
191
|
-
let safeEnd = 0;
|
|
192
|
-
for (const { segment, index } of new Intl.Segmenter(undefined, {
|
|
193
|
-
granularity: "grapheme",
|
|
194
|
-
}).segment(text)) {
|
|
195
|
-
if (index >= startOffset && safeStart === text.length)
|
|
196
|
-
safeStart = index;
|
|
197
|
-
if (index + segment.length <= endOffset)
|
|
198
|
-
safeEnd = index + segment.length;
|
|
199
|
-
if (index >= endOffset)
|
|
200
|
-
break;
|
|
201
|
-
}
|
|
202
|
-
return `${safeStart > 0 ? "…" : ""}${text.slice(safeStart, safeEnd)}${safeEnd < text.length ? "…" : ""}`;
|
|
203
|
-
}
|
|
204
|
-
function candidateUrl(candidate) {
|
|
205
|
-
return candidate.destination?.url ?? candidate.page.url;
|
|
206
|
-
}
|
|
207
|
-
function compareCandidates(a, b) {
|
|
208
|
-
const aUrl = candidateUrl(a);
|
|
209
|
-
const bUrl = candidateUrl(b);
|
|
210
|
-
return (b.rank - a.rank ||
|
|
211
|
-
b.score - a.score ||
|
|
212
|
-
(aUrl < bUrl ? -1 : aUrl > bUrl ? 1 : 0));
|
|
142
|
+
return `${start > 0 ? "…" : ""}${characters.slice(start, end).join("")}${end < characters.length ? "…" : ""}`;
|
|
213
143
|
}
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
144
|
+
/** Return at most one hit per page, without filesystem, network, or DOM access. */
|
|
145
|
+
export function search(index, query, options = {}) {
|
|
146
|
+
validateSearchIndex(index);
|
|
147
|
+
const limit = options.limit ?? 10;
|
|
148
|
+
if (!Number.isInteger(limit) || limit < 0 || limit > 100) {
|
|
149
|
+
throw new RangeError("Search limit must be an integer from 0 to 100.");
|
|
150
|
+
}
|
|
151
|
+
if (typeof query !== "string")
|
|
152
|
+
throw new TypeError("Search query must be a string.");
|
|
153
|
+
if (query.length > 512)
|
|
154
|
+
throw new RangeError("Search query must not exceed 512 characters.");
|
|
155
|
+
const terms = [...new Set(normalize(query).split(" ").filter(Boolean))];
|
|
224
156
|
const phrase = terms.join(" ");
|
|
225
|
-
|
|
157
|
+
if (terms.length > 32)
|
|
158
|
+
throw new RangeError("Search query must not exceed 32 unique terms.");
|
|
159
|
+
if (!terms.length || !limit)
|
|
160
|
+
return [];
|
|
226
161
|
const results = [];
|
|
227
162
|
for (const page of index.pages) {
|
|
228
163
|
const title = normalize(page.title);
|
|
@@ -261,217 +196,35 @@ function pageCandidates(index, terms) {
|
|
|
261
196
|
// A stable sort preserves document order when multiple sections tie.
|
|
262
197
|
matches.sort((a, b) => b.score - a.score);
|
|
263
198
|
const best = matches[0]?.section;
|
|
199
|
+
const destination = metadataMatch ? undefined : best;
|
|
200
|
+
const sources = [
|
|
201
|
+
best?.text ?? "",
|
|
202
|
+
page.description,
|
|
203
|
+
...page.sections.map((section) => section.text),
|
|
204
|
+
];
|
|
205
|
+
const source = sources.find((text) => {
|
|
206
|
+
const normalized = normalize(text);
|
|
207
|
+
return terms.some((term) => normalized.includes(term));
|
|
208
|
+
}) ??
|
|
209
|
+
sources.find((text) => text) ??
|
|
210
|
+
page.title;
|
|
264
211
|
results.push({
|
|
265
|
-
page,
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
...page.sections.map((section) => section.text),
|
|
271
|
-
], patterns, page.title),
|
|
212
|
+
title: page.title,
|
|
213
|
+
url: destination?.url ?? page.url,
|
|
214
|
+
...(destination?.title ? { section: destination.title } : {}),
|
|
215
|
+
// Keep source text until ranking so only returned hits need a snippet.
|
|
216
|
+
excerpt: source,
|
|
272
217
|
score: weights.reduce((sum, weight) => sum + weight, 0) +
|
|
273
218
|
(title === phrase
|
|
274
219
|
? 24
|
|
275
220
|
: sections.some((section) => section.title === phrase)
|
|
276
221
|
? 12
|
|
277
222
|
: 0),
|
|
278
|
-
rank: 0,
|
|
279
|
-
patterns,
|
|
280
223
|
});
|
|
281
224
|
}
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
return false;
|
|
288
|
-
const text = Array.from(token);
|
|
289
|
-
if (Math.abs(query.length - text.length) > 1)
|
|
290
|
-
return false;
|
|
291
|
-
let start = 0;
|
|
292
|
-
while (start < query.length && query[start] === text[start])
|
|
293
|
-
start++;
|
|
294
|
-
if (start === Math.min(query.length, text.length))
|
|
295
|
-
return true;
|
|
296
|
-
if (query.length === text.length) {
|
|
297
|
-
if (query.slice(start + 1).join("") === text.slice(start + 1).join(""))
|
|
298
|
-
return true;
|
|
299
|
-
return (query[start] === text[start + 1] &&
|
|
300
|
-
query[start + 1] === text[start] &&
|
|
301
|
-
query.slice(start + 2).join("") === text.slice(start + 2).join(""));
|
|
302
|
-
}
|
|
303
|
-
return query.length > text.length
|
|
304
|
-
? query.slice(start + 1).join("") === text.slice(start).join("")
|
|
305
|
-
: query.slice(start).join("") === text.slice(start + 1).join("");
|
|
306
|
-
}
|
|
307
|
-
function fieldScore(fields, pattern) {
|
|
308
|
-
return Math.max(0, ...fields
|
|
309
|
-
.filter((field) => contains(field.text, pattern))
|
|
310
|
-
.map((field) => field.weight));
|
|
311
|
-
}
|
|
312
|
-
function matchFields(fields, terms, fuzzy, required = []) {
|
|
313
|
-
const patterns = terms.map((text) => ({ text }));
|
|
314
|
-
const weights = patterns.map((pattern) => fieldScore(fields, pattern));
|
|
315
|
-
const missing = weights.flatMap((weight, index) => (weight ? [] : [index]));
|
|
316
|
-
if (!missing.length)
|
|
317
|
-
return {
|
|
318
|
-
patterns,
|
|
319
|
-
score: weights.reduce((a, b) => a + b, 0),
|
|
320
|
-
corrected: false,
|
|
321
|
-
};
|
|
322
|
-
const missingIndex = missing[0];
|
|
323
|
-
if (!fuzzy || missing.length !== 1 || missingIndex === undefined)
|
|
324
|
-
return;
|
|
325
|
-
const query = Array.from(terms[missingIndex] ?? "");
|
|
326
|
-
if (query.length < 4 || query.length > 32)
|
|
327
|
-
return;
|
|
328
|
-
const needsOwnCorrection = required.length > 0 &&
|
|
329
|
-
!patterns.some((pattern) => fieldScore(required, pattern));
|
|
330
|
-
let correction;
|
|
331
|
-
let correctionWeight = 0;
|
|
332
|
-
for (const field of fields) {
|
|
333
|
-
if (field.weight < correctionWeight)
|
|
334
|
-
continue;
|
|
335
|
-
for (const token of field.text.split(" ")) {
|
|
336
|
-
if (singleEdit(query, token) &&
|
|
337
|
-
(!needsOwnCorrection ||
|
|
338
|
-
fieldScore(required, { text: token, whole: true }) > 0) &&
|
|
339
|
-
(field.weight > correctionWeight ||
|
|
340
|
-
correction === undefined ||
|
|
341
|
-
token < correction)) {
|
|
342
|
-
correction = token;
|
|
343
|
-
correctionWeight = field.weight;
|
|
344
|
-
}
|
|
345
|
-
}
|
|
346
|
-
}
|
|
347
|
-
if (!correction)
|
|
348
|
-
return;
|
|
349
|
-
patterns[missingIndex] = { text: correction, whole: true };
|
|
350
|
-
weights[missingIndex] = correctionWeight;
|
|
351
|
-
return {
|
|
352
|
-
patterns,
|
|
353
|
-
score: weights.reduce((a, b) => a + b, 0),
|
|
354
|
-
corrected: true,
|
|
355
|
-
};
|
|
356
|
-
}
|
|
357
|
-
/** Score only page metadata plus a single coherent section, never sibling bodies. */
|
|
358
|
-
function sectionCandidates(index, terms, fuzzy) {
|
|
359
|
-
const phrase = terms.join(" ");
|
|
360
|
-
const results = [];
|
|
361
|
-
for (const page of index.pages) {
|
|
362
|
-
const title = normalize(page.title);
|
|
363
|
-
const metadata = [
|
|
364
|
-
{ text: title, weight: 8 },
|
|
365
|
-
{ text: normalize(page.description), weight: 2 },
|
|
366
|
-
];
|
|
367
|
-
const pageMatch = matchFields(metadata, terms, fuzzy);
|
|
368
|
-
let pageCandidate;
|
|
369
|
-
if (pageMatch) {
|
|
370
|
-
const exact = title === phrase;
|
|
371
|
-
pageCandidate = {
|
|
372
|
-
page,
|
|
373
|
-
source: sourceText([page.description, ...page.sections.map((section) => section.text)], pageMatch.patterns, page.title),
|
|
374
|
-
score: pageMatch.score + (exact ? 24 : 0),
|
|
375
|
-
rank: pageMatch.corrected ? 0 : exact ? 2 : 1,
|
|
376
|
-
patterns: pageMatch.patterns,
|
|
377
|
-
};
|
|
378
|
-
}
|
|
379
|
-
for (const section of page.sections) {
|
|
380
|
-
const heading = normalize(section.title);
|
|
381
|
-
const body = normalize(section.text);
|
|
382
|
-
const ownFields = [
|
|
383
|
-
{ text: heading, weight: 4 },
|
|
384
|
-
{ text: body, weight: 1 },
|
|
385
|
-
];
|
|
386
|
-
const match = matchFields([...metadata, ...ownFields], terms, fuzzy, ownFields);
|
|
387
|
-
if (!match?.patterns.some((pattern) => fieldScore(ownFields, pattern)))
|
|
388
|
-
continue;
|
|
389
|
-
// Repeated page headings without their own matching body add no useful destination.
|
|
390
|
-
if (pageMatch &&
|
|
391
|
-
heading === title &&
|
|
392
|
-
!match.patterns.some((pattern) => contains(body, pattern)))
|
|
393
|
-
continue;
|
|
394
|
-
const exact = heading === phrase;
|
|
395
|
-
const candidate = {
|
|
396
|
-
page,
|
|
397
|
-
destination: section.url === page.url ? undefined : section,
|
|
398
|
-
source: sourceText([section.text, page.description], match.patterns, section.title || page.title),
|
|
399
|
-
score: match.score + (exact ? 12 : 0),
|
|
400
|
-
rank: match.corrected ? 0 : exact ? 2 : 1,
|
|
401
|
-
patterns: match.patterns,
|
|
402
|
-
};
|
|
403
|
-
if (candidate.destination)
|
|
404
|
-
results.push(candidate);
|
|
405
|
-
else if (!pageCandidate ||
|
|
406
|
-
compareCandidates(candidate, pageCandidate) < 0)
|
|
407
|
-
pageCandidate = candidate;
|
|
408
|
-
}
|
|
409
|
-
if (pageCandidate)
|
|
410
|
-
results.push(pageCandidate);
|
|
411
|
-
}
|
|
412
|
-
return results;
|
|
413
|
-
}
|
|
414
|
-
function displayResult(candidate) {
|
|
415
|
-
const { page, destination, patterns } = candidate;
|
|
416
|
-
const isSection = destination !== undefined && destination.url !== page.url;
|
|
417
|
-
const trail = isSection
|
|
418
|
-
? [
|
|
419
|
-
...(page.breadcrumbs ?? []),
|
|
420
|
-
page.title,
|
|
421
|
-
...(destination.breadcrumbs ?? []),
|
|
422
|
-
]
|
|
423
|
-
: [...(page.breadcrumbs ?? [])];
|
|
424
|
-
const snippet = excerpt(candidate.source, patterns);
|
|
425
|
-
return {
|
|
426
|
-
kind: isSection ? "section" : "page",
|
|
427
|
-
title: page.title,
|
|
428
|
-
pageUrl: page.url,
|
|
429
|
-
url: candidateUrl(candidate),
|
|
430
|
-
...(destination?.title ? { section: destination.title } : {}),
|
|
431
|
-
breadcrumbs: trail,
|
|
432
|
-
excerpt: snippet,
|
|
433
|
-
score: candidate.score,
|
|
434
|
-
matches: {
|
|
435
|
-
title: matchRanges(page.title, patterns),
|
|
436
|
-
section: matchRanges(destination?.title ?? "", patterns),
|
|
437
|
-
excerpt: matchRanges(snippet, patterns),
|
|
438
|
-
breadcrumbs: trail.map((label) => matchRanges(label, patterns)),
|
|
439
|
-
},
|
|
440
|
-
};
|
|
441
|
-
}
|
|
442
|
-
/** Query without filesystem, network, or DOM access; page mode preserves legacy ranking. */
|
|
443
|
-
export function search(index, query, options = {}) {
|
|
444
|
-
validateSearchIndex(index);
|
|
445
|
-
if (options === null || typeof options !== "object" || Array.isArray(options))
|
|
446
|
-
throw new TypeError("Search options must be an object.");
|
|
447
|
-
const limit = options.limit ?? 10;
|
|
448
|
-
const mode = options.mode === undefined ? "pages" : options.mode;
|
|
449
|
-
const fuzzy = options.fuzzy === undefined ? false : options.fuzzy;
|
|
450
|
-
if (!Number.isInteger(limit) || limit < 0 || limit > 100)
|
|
451
|
-
throw new RangeError("Search limit must be an integer from 0 to 100.");
|
|
452
|
-
if (mode !== "pages" && mode !== "sections")
|
|
453
|
-
throw new RangeError("Search mode must be pages or sections.");
|
|
454
|
-
if (typeof fuzzy !== "boolean")
|
|
455
|
-
throw new TypeError("Search fuzzy option must be a boolean.");
|
|
456
|
-
if (fuzzy && mode !== "sections")
|
|
457
|
-
throw new RangeError("Fuzzy search requires sections mode.");
|
|
458
|
-
if (typeof query !== "string")
|
|
459
|
-
throw new TypeError("Search query must be a string.");
|
|
460
|
-
if (query.length > 512)
|
|
461
|
-
throw new RangeError("Search query must not exceed 512 characters.");
|
|
462
|
-
const terms = [...new Set(normalize(query).split(" ").filter(Boolean))];
|
|
463
|
-
if (terms.length > 32)
|
|
464
|
-
throw new RangeError("Search query must not exceed 32 unique terms.");
|
|
465
|
-
if (!terms.length || !limit)
|
|
466
|
-
return [];
|
|
467
|
-
const results = mode === "sections"
|
|
468
|
-
? sectionCandidates(index, terms, false)
|
|
469
|
-
: pageCandidates(index, terms);
|
|
470
|
-
if (fuzzy && results.length < limit) {
|
|
471
|
-
const literalUrls = new Set(results.map(candidateUrl));
|
|
472
|
-
results.push(...sectionCandidates(index, terms, true).filter((candidate) => candidate.rank === 0 && !literalUrls.has(candidateUrl(candidate))));
|
|
473
|
-
}
|
|
474
|
-
results.sort(compareCandidates);
|
|
475
|
-
// Original-text mapping and excerpt windows are computed only for final hits.
|
|
476
|
-
return results.slice(0, limit).map(displayResult);
|
|
225
|
+
results.sort((a, b) => b.score - a.score || (a.url < b.url ? -1 : a.url > b.url ? 1 : 0));
|
|
226
|
+
return results.slice(0, limit).map((result) => ({
|
|
227
|
+
...result,
|
|
228
|
+
excerpt: excerpt(result.excerpt, terms),
|
|
229
|
+
}));
|
|
477
230
|
}
|