@wgtechlabs/mdd-engine 0.1.0-pr.0cb765a → 0.1.0-pr.8fa2dc9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -0
- package/dist/index.d.ts +1 -1
- package/dist/search-index.js +33 -2
- package/dist/search.d.ts +20 -1
- package/dist/search.js +317 -62
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -85,6 +85,10 @@ import { search } from '@wgtechlabs/mdd-engine/search';
|
|
|
85
85
|
const index = createSearchIndex(result.site);
|
|
86
86
|
const restored = JSON.parse(JSON.stringify(index));
|
|
87
87
|
console.log(search(restored, 'installation', { limit: 10 }));
|
|
88
|
+
// Multiple heading hits, breadcrumbs, and ranges for UI highlights.
|
|
89
|
+
console.log(search(restored, 'instalation', {
|
|
90
|
+
mode: 'sections', fuzzy: true, limit: 8,
|
|
91
|
+
}));
|
|
88
92
|
// [{ title, url, section?, excerpt, score }]
|
|
89
93
|
```
|
|
90
94
|
|
package/dist/index.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { CompileOptions, CompileResult } from "./types.js";
|
|
2
|
-
export { type SearchIndex, type SearchOptions, type SearchPage, type SearchResult, type SearchSection, search, validateSearchIndex, } from "./search.js";
|
|
2
|
+
export { type SearchIndex, type SearchMatch, type SearchOptions, type SearchPage, type SearchResult, type SearchSection, search, validateSearchIndex, } from "./search.js";
|
|
3
3
|
export { createSearchIndex } from "./search-index.js";
|
|
4
4
|
export type { Asset, CompileOptions, CompileResult, Diagnostic, Footer, Heading, Metadata, NavigationItem, Page, Site, SocialLink, Theme, } from "./types.js";
|
|
5
5
|
/** Compile a local checkout; never fetch repositories, execute author code, or emit files. */
|
package/dist/search-index.js
CHANGED
|
@@ -7,8 +7,9 @@ import { transformAlerts } from "./alerts.js";
|
|
|
7
7
|
import { validateSearchIndex, } from "./search.js";
|
|
8
8
|
const parser = unified().use(remarkParse).use(remarkGfm);
|
|
9
9
|
const blockTypes = new Set(["paragraph", "code", "tableCell", "listItem"]);
|
|
10
|
-
function indexPage(page) {
|
|
10
|
+
function indexPage(page, breadcrumbs) {
|
|
11
11
|
const sections = [{ title: "", url: page.url, text: "" }];
|
|
12
|
+
const ancestors = [];
|
|
12
13
|
let headingIndex = 0;
|
|
13
14
|
const chunks = [[]];
|
|
14
15
|
const tree = parser.parse(page.markdown);
|
|
@@ -23,11 +24,20 @@ function indexPage(page) {
|
|
|
23
24
|
heading.depth !== node.depth) {
|
|
24
25
|
throw new TypeError("Search indexing requires matching compiled Markdown and headings.");
|
|
25
26
|
}
|
|
27
|
+
while (ancestors.length &&
|
|
28
|
+
(ancestors.at(-1)?.depth ?? 0) >= heading.depth)
|
|
29
|
+
ancestors.pop();
|
|
26
30
|
sections.push({
|
|
27
31
|
title: heading.text,
|
|
28
32
|
url: `${page.url}#${encodeURIComponent(heading.id)}`,
|
|
29
33
|
text: "",
|
|
34
|
+
breadcrumbs: ancestors
|
|
35
|
+
.filter((ancestor) => !(ancestor.depth === 1 &&
|
|
36
|
+
ancestor.text.normalize("NFKC").toLowerCase() ===
|
|
37
|
+
page.title.normalize("NFKC").toLowerCase()))
|
|
38
|
+
.map((ancestor) => ancestor.text),
|
|
30
39
|
});
|
|
40
|
+
ancestors.push(heading);
|
|
31
41
|
chunks.push([]);
|
|
32
42
|
return SKIP;
|
|
33
43
|
}
|
|
@@ -55,12 +65,33 @@ function indexPage(page) {
|
|
|
55
65
|
url: page.url,
|
|
56
66
|
title: page.title,
|
|
57
67
|
description: page.description ?? "",
|
|
68
|
+
breadcrumbs,
|
|
58
69
|
sections,
|
|
59
70
|
};
|
|
60
71
|
}
|
|
61
72
|
/** Build only from successfully compiled pages; no source paths or HTML are indexed. */
|
|
62
73
|
export function createSearchIndex(site) {
|
|
63
|
-
const
|
|
74
|
+
const navigation = new Map();
|
|
75
|
+
const pending = site.navigation.map((item) => ({
|
|
76
|
+
item,
|
|
77
|
+
ancestors: [],
|
|
78
|
+
}));
|
|
79
|
+
while (pending.length) {
|
|
80
|
+
const entry = pending.pop();
|
|
81
|
+
if (!entry)
|
|
82
|
+
break;
|
|
83
|
+
const { item, ancestors } = entry;
|
|
84
|
+
if (ancestors.length > 64)
|
|
85
|
+
throw new TypeError("Search navigation exceeds 64 ancestor labels.");
|
|
86
|
+
if (item.url)
|
|
87
|
+
navigation.set(item.url, ancestors);
|
|
88
|
+
for (const child of item.children ?? [])
|
|
89
|
+
pending.push({ item: child, ancestors: [...ancestors, item.title] });
|
|
90
|
+
}
|
|
91
|
+
const index = {
|
|
92
|
+
version: 1,
|
|
93
|
+
pages: site.pages.map((page) => indexPage(page, navigation.get(page.url) ?? [])),
|
|
94
|
+
};
|
|
64
95
|
index.pages.sort((a, b) => (a.url < b.url ? -1 : a.url > b.url ? 1 : 0));
|
|
65
96
|
validateSearchIndex(index);
|
|
66
97
|
return index;
|
package/dist/search.d.ts
CHANGED
|
@@ -7,6 +7,8 @@ export interface SearchPage {
|
|
|
7
7
|
url: string;
|
|
8
8
|
title: string;
|
|
9
9
|
description: string;
|
|
10
|
+
/** Navigation ancestor labels, excluding this page's own label. */
|
|
11
|
+
breadcrumbs?: string[];
|
|
10
12
|
sections: SearchSection[];
|
|
11
13
|
}
|
|
12
14
|
export interface SearchSection {
|
|
@@ -14,20 +16,37 @@ export interface SearchSection {
|
|
|
14
16
|
title: string;
|
|
15
17
|
url: string;
|
|
16
18
|
text: string;
|
|
19
|
+
/** Ancestor headings, excluding this heading and a duplicate page-title H1. */
|
|
20
|
+
breadcrumbs?: string[];
|
|
17
21
|
}
|
|
18
22
|
export interface SearchOptions {
|
|
19
23
|
/** Maximum results, from 0 to 100. Defaults to 10. */
|
|
20
24
|
limit?: number;
|
|
25
|
+
/** Defaults to one result per page; sections returns independent heading hits. */
|
|
26
|
+
mode?: "pages" | "sections";
|
|
27
|
+
/** Allow one single-edit query correction per result. Requires sections mode. */
|
|
28
|
+
fuzzy?: boolean;
|
|
21
29
|
}
|
|
30
|
+
/** UTF-16 offsets into the original displayed field; end is exclusive. */
|
|
31
|
+
export type SearchMatch = [start: number, end: number];
|
|
22
32
|
export interface SearchResult {
|
|
33
|
+
kind: "page" | "section";
|
|
23
34
|
title: string;
|
|
24
35
|
url: string;
|
|
36
|
+
pageUrl: string;
|
|
37
|
+
breadcrumbs: string[];
|
|
25
38
|
/** Present when the destination is a compiled heading. */
|
|
26
39
|
section?: string;
|
|
27
40
|
excerpt: string;
|
|
28
41
|
score: number;
|
|
42
|
+
matches: {
|
|
43
|
+
title: SearchMatch[];
|
|
44
|
+
section: SearchMatch[];
|
|
45
|
+
excerpt: SearchMatch[];
|
|
46
|
+
breadcrumbs: SearchMatch[][];
|
|
47
|
+
};
|
|
29
48
|
}
|
|
30
49
|
/** Validate deserialized data before any result can become a link. */
|
|
31
50
|
export declare function validateSearchIndex(value: unknown): asserts value is SearchIndex;
|
|
32
|
-
/**
|
|
51
|
+
/** Query without filesystem, network, or DOM access; page mode preserves legacy ranking. */
|
|
33
52
|
export declare function search(index: SearchIndex, query: string, options?: SearchOptions): SearchResult[];
|
package/dist/search.js
CHANGED
|
@@ -1,6 +1,17 @@
|
|
|
1
1
|
function record(value) {
|
|
2
2
|
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
3
3
|
}
|
|
4
|
+
function breadcrumbs(value, maximum) {
|
|
5
|
+
if (value === undefined)
|
|
6
|
+
return true;
|
|
7
|
+
if (!Array.isArray(value) || value.length > maximum)
|
|
8
|
+
return false;
|
|
9
|
+
// Iterate holes too: a sparse array does not contain only string labels.
|
|
10
|
+
for (const label of value)
|
|
11
|
+
if (typeof label !== "string")
|
|
12
|
+
return false;
|
|
13
|
+
return true;
|
|
14
|
+
}
|
|
4
15
|
function pageUrl(value) {
|
|
5
16
|
if (typeof value !== "string" ||
|
|
6
17
|
!value.startsWith("/") ||
|
|
@@ -48,6 +59,7 @@ export function validateSearchIndex(value) {
|
|
|
48
59
|
urls.has(page.url) ||
|
|
49
60
|
typeof page.title !== "string" ||
|
|
50
61
|
typeof page.description !== "string" ||
|
|
62
|
+
!breadcrumbs(page.breadcrumbs, 64) ||
|
|
51
63
|
!Array.isArray(page.sections))
|
|
52
64
|
throw new TypeError("Invalid search index page.");
|
|
53
65
|
urls.add(page.url);
|
|
@@ -57,7 +69,8 @@ export function validateSearchIndex(value) {
|
|
|
57
69
|
!sectionUrl(section.url, page.url) ||
|
|
58
70
|
sections.has(section.url) ||
|
|
59
71
|
typeof section.title !== "string" ||
|
|
60
|
-
typeof section.text !== "string"
|
|
72
|
+
typeof section.text !== "string" ||
|
|
73
|
+
!breadcrumbs(section.breadcrumbs, 6))
|
|
61
74
|
throw new TypeError("Invalid search index section.");
|
|
62
75
|
sections.add(section.url);
|
|
63
76
|
}
|
|
@@ -70,23 +83,51 @@ function normalize(value) {
|
|
|
70
83
|
.replace(/[^\p{L}\p{N}\p{M}_]+/gu, " ")
|
|
71
84
|
.trim();
|
|
72
85
|
}
|
|
73
|
-
|
|
74
|
-
|
|
86
|
+
function contains(value, pattern) {
|
|
87
|
+
return pattern.whole
|
|
88
|
+
? ` ${value} `.includes(` ${pattern.text} `)
|
|
89
|
+
: value.includes(pattern.text);
|
|
90
|
+
}
|
|
91
|
+
function mergeRanges(ranges) {
|
|
92
|
+
ranges.sort((a, b) => a[0] - b[0] || a[1] - b[1]);
|
|
93
|
+
const merged = [];
|
|
94
|
+
for (const range of ranges) {
|
|
95
|
+
const previous = merged.at(-1);
|
|
96
|
+
if (previous && range[0] <= previous[1])
|
|
97
|
+
previous[1] = Math.max(previous[1], range[1]);
|
|
98
|
+
else
|
|
99
|
+
merged.push([...range]);
|
|
100
|
+
}
|
|
101
|
+
return merged;
|
|
102
|
+
}
|
|
103
|
+
/** Map normalized matches to complete original graphemes, including NFKC expansions. */
|
|
104
|
+
function matchRanges(value, patterns, firstOnly = false) {
|
|
75
105
|
const normalized = value.normalize("NFKC").toLowerCase();
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
(start < 0
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
106
|
+
const found = [];
|
|
107
|
+
for (const pattern of patterns) {
|
|
108
|
+
let offset = 0;
|
|
109
|
+
while (offset < normalized.length) {
|
|
110
|
+
const start = normalized.indexOf(pattern.text, offset);
|
|
111
|
+
if (start < 0)
|
|
112
|
+
break;
|
|
113
|
+
const end = start + pattern.text.length;
|
|
114
|
+
// One neighboring code point needs at most two UTF-16 units. Avoid scanning
|
|
115
|
+
// a growing prefix for each occurrence, or any boundary work for substrings.
|
|
116
|
+
if (!pattern.whole ||
|
|
117
|
+
(!/[\p{L}\p{N}\p{M}_]$/u.test(normalized.slice(Math.max(0, start - 2), start)) &&
|
|
118
|
+
!/^[\p{L}\p{N}\p{M}_]/u.test(normalized.slice(end, end + 2)))) {
|
|
119
|
+
found.push([start, end]);
|
|
120
|
+
if (firstOnly)
|
|
121
|
+
break;
|
|
122
|
+
}
|
|
123
|
+
offset = start + 1;
|
|
86
124
|
}
|
|
87
125
|
}
|
|
88
|
-
|
|
89
|
-
|
|
126
|
+
let ranges = mergeRanges(found);
|
|
127
|
+
if (!ranges.length)
|
|
128
|
+
return [];
|
|
129
|
+
if (firstOnly)
|
|
130
|
+
ranges = ranges.slice(0, 1);
|
|
90
131
|
const groups = [];
|
|
91
132
|
for (const { segment, index } of new Intl.Segmenter(undefined, {
|
|
92
133
|
granularity: "grapheme",
|
|
@@ -110,25 +151,35 @@ function matchRange(value, terms) {
|
|
|
110
151
|
}
|
|
111
152
|
groups.push(group);
|
|
112
153
|
}
|
|
113
|
-
//
|
|
114
|
-
// Per-group lowercase
|
|
154
|
+
// Whole-field lowercase retains contextual letters such as final sigma.
|
|
155
|
+
// Per-group lowercase supplies only lengths, including expansions like İ.
|
|
156
|
+
const mapped = [];
|
|
115
157
|
let offset = 0;
|
|
158
|
+
let rangeIndex = 0;
|
|
116
159
|
let originalStart = 0;
|
|
117
160
|
for (const group of groups) {
|
|
118
161
|
const next = offset + group.text.toLowerCase().length;
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
162
|
+
let range = ranges[rangeIndex];
|
|
163
|
+
while (range && range[0] < next) {
|
|
164
|
+
if (range[0] >= offset)
|
|
165
|
+
originalStart = group.start;
|
|
166
|
+
if (range[1] > next)
|
|
167
|
+
break;
|
|
168
|
+
mapped.push([originalStart, group.end]);
|
|
169
|
+
range = ranges[++rangeIndex];
|
|
170
|
+
}
|
|
171
|
+
if (!range)
|
|
172
|
+
break;
|
|
123
173
|
offset = next;
|
|
124
174
|
}
|
|
175
|
+
return mergeRanges(mapped);
|
|
125
176
|
}
|
|
126
|
-
function excerpt(value,
|
|
177
|
+
function excerpt(value, patterns) {
|
|
127
178
|
const text = value.replace(/\s+/gu, " ").trim();
|
|
128
179
|
const characters = Array.from(text);
|
|
129
180
|
if (characters.length <= 160)
|
|
130
181
|
return text;
|
|
131
|
-
const match =
|
|
182
|
+
const match = matchRanges(text, patterns, true)[0];
|
|
132
183
|
const matchStart = match ? Array.from(text.slice(0, match[0])).length : 0;
|
|
133
184
|
const matchLength = match
|
|
134
185
|
? Array.from(text.slice(match[0], match[1])).length
|
|
@@ -139,25 +190,45 @@ function excerpt(value, terms) {
|
|
|
139
190
|
let end = Math.min(characters.length, start + 160 - (start > 0 ? 1 : 0));
|
|
140
191
|
if (end < characters.length)
|
|
141
192
|
end--;
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
const
|
|
148
|
-
|
|
149
|
-
|
|
193
|
+
const startOffset = characters.slice(0, start).join("").length;
|
|
194
|
+
const endOffset = characters.slice(0, end).join("").length;
|
|
195
|
+
// Move clipped edges inward to complete graphemes without exceeding the cap.
|
|
196
|
+
let safeStart = text.length;
|
|
197
|
+
let safeEnd = 0;
|
|
198
|
+
for (const { segment, index } of new Intl.Segmenter(undefined, {
|
|
199
|
+
granularity: "grapheme",
|
|
200
|
+
}).segment(text)) {
|
|
201
|
+
if (index >= startOffset && safeStart === text.length)
|
|
202
|
+
safeStart = index;
|
|
203
|
+
if (index + segment.length <= endOffset)
|
|
204
|
+
safeEnd = index + segment.length;
|
|
205
|
+
if (index >= endOffset)
|
|
206
|
+
break;
|
|
150
207
|
}
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
208
|
+
return `${safeStart > 0 ? "…" : ""}${text.slice(safeStart, safeEnd)}${safeEnd < text.length ? "…" : ""}`;
|
|
209
|
+
}
|
|
210
|
+
function candidateUrl(candidate) {
|
|
211
|
+
return candidate.destination?.url ?? candidate.page.url;
|
|
212
|
+
}
|
|
213
|
+
function compareCandidates(a, b) {
|
|
214
|
+
const aUrl = candidateUrl(a);
|
|
215
|
+
const bUrl = candidateUrl(b);
|
|
216
|
+
return (b.rank - a.rank ||
|
|
217
|
+
b.score - a.score ||
|
|
218
|
+
(aUrl < bUrl ? -1 : aUrl > bUrl ? 1 : 0));
|
|
219
|
+
}
|
|
220
|
+
function sourceText(sources, patterns, fallback) {
|
|
221
|
+
return (sources.find((text) => {
|
|
222
|
+
const normalized = normalize(text);
|
|
223
|
+
return patterns.some((pattern) => contains(normalized, pattern));
|
|
224
|
+
}) ??
|
|
225
|
+
sources.find((text) => text) ??
|
|
226
|
+
fallback);
|
|
227
|
+
}
|
|
228
|
+
/** Preserve page-mode weighting, cross-section coverage, and destination selection. */
|
|
229
|
+
function pageCandidates(index, terms) {
|
|
156
230
|
const phrase = terms.join(" ");
|
|
157
|
-
|
|
158
|
-
throw new RangeError("Search query must not exceed 32 unique terms.");
|
|
159
|
-
if (!terms.length || !limit)
|
|
160
|
-
return [];
|
|
231
|
+
const patterns = terms.map((text) => ({ text }));
|
|
161
232
|
const results = [];
|
|
162
233
|
for (const page of index.pages) {
|
|
163
234
|
const title = normalize(page.title);
|
|
@@ -196,35 +267,219 @@ export function search(index, query, options = {}) {
|
|
|
196
267
|
// A stable sort preserves document order when multiple sections tie.
|
|
197
268
|
matches.sort((a, b) => b.score - a.score);
|
|
198
269
|
const best = matches[0]?.section;
|
|
199
|
-
const destination = metadataMatch ? undefined : best;
|
|
200
|
-
const sources = [
|
|
201
|
-
best?.text ?? "",
|
|
202
|
-
page.description,
|
|
203
|
-
...page.sections.map((section) => section.text),
|
|
204
|
-
];
|
|
205
|
-
const source = sources.find((text) => {
|
|
206
|
-
const normalized = normalize(text);
|
|
207
|
-
return terms.some((term) => normalized.includes(term));
|
|
208
|
-
}) ??
|
|
209
|
-
sources.find((text) => text) ??
|
|
210
|
-
page.title;
|
|
211
270
|
results.push({
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
271
|
+
page,
|
|
272
|
+
destination: metadataMatch ? undefined : best,
|
|
273
|
+
source: sourceText([
|
|
274
|
+
best?.text ?? "",
|
|
275
|
+
page.description,
|
|
276
|
+
...page.sections.map((section) => section.text),
|
|
277
|
+
], patterns, page.title),
|
|
217
278
|
score: weights.reduce((sum, weight) => sum + weight, 0) +
|
|
218
279
|
(title === phrase
|
|
219
280
|
? 24
|
|
220
281
|
: sections.some((section) => section.title === phrase)
|
|
221
282
|
? 12
|
|
222
283
|
: 0),
|
|
284
|
+
rank: 0,
|
|
285
|
+
patterns,
|
|
223
286
|
});
|
|
224
287
|
}
|
|
225
|
-
results
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
288
|
+
return results;
|
|
289
|
+
}
|
|
290
|
+
/** Single insertion, deletion, substitution, or adjacent transposition, on code points. */
|
|
291
|
+
function singleEdit(query, token) {
|
|
292
|
+
if (token.length > 66)
|
|
293
|
+
return false;
|
|
294
|
+
const text = Array.from(token);
|
|
295
|
+
if (Math.abs(query.length - text.length) > 1)
|
|
296
|
+
return false;
|
|
297
|
+
let start = 0;
|
|
298
|
+
while (start < query.length && query[start] === text[start])
|
|
299
|
+
start++;
|
|
300
|
+
if (start === Math.min(query.length, text.length))
|
|
301
|
+
return true;
|
|
302
|
+
if (query.length === text.length) {
|
|
303
|
+
if (query.slice(start + 1).join("") === text.slice(start + 1).join(""))
|
|
304
|
+
return true;
|
|
305
|
+
return (query[start] === text[start + 1] &&
|
|
306
|
+
query[start + 1] === text[start] &&
|
|
307
|
+
query.slice(start + 2).join("") === text.slice(start + 2).join(""));
|
|
308
|
+
}
|
|
309
|
+
return query.length > text.length
|
|
310
|
+
? query.slice(start + 1).join("") === text.slice(start).join("")
|
|
311
|
+
: query.slice(start).join("") === text.slice(start + 1).join("");
|
|
312
|
+
}
|
|
313
|
+
function fieldScore(fields, pattern) {
|
|
314
|
+
return Math.max(0, ...fields
|
|
315
|
+
.filter((field) => contains(field.text, pattern))
|
|
316
|
+
.map((field) => field.weight));
|
|
317
|
+
}
|
|
318
|
+
function matchFields(fields, terms, fuzzy, required = []) {
|
|
319
|
+
const patterns = terms.map((text) => ({ text }));
|
|
320
|
+
const weights = patterns.map((pattern) => fieldScore(fields, pattern));
|
|
321
|
+
const missing = weights.flatMap((weight, index) => (weight ? [] : [index]));
|
|
322
|
+
if (!missing.length)
|
|
323
|
+
return {
|
|
324
|
+
patterns,
|
|
325
|
+
score: weights.reduce((a, b) => a + b, 0),
|
|
326
|
+
corrected: false,
|
|
327
|
+
};
|
|
328
|
+
const missingIndex = missing[0];
|
|
329
|
+
if (!fuzzy || missing.length !== 1 || missingIndex === undefined)
|
|
330
|
+
return;
|
|
331
|
+
const query = Array.from(terms[missingIndex] ?? "");
|
|
332
|
+
if (query.length < 4 || query.length > 32)
|
|
333
|
+
return;
|
|
334
|
+
const needsOwnCorrection = required.length > 0 &&
|
|
335
|
+
!patterns.some((pattern) => fieldScore(required, pattern));
|
|
336
|
+
let correction;
|
|
337
|
+
let correctionWeight = 0;
|
|
338
|
+
for (const field of fields) {
|
|
339
|
+
if (field.weight < correctionWeight)
|
|
340
|
+
continue;
|
|
341
|
+
for (const token of field.text.split(" ")) {
|
|
342
|
+
if (singleEdit(query, token) &&
|
|
343
|
+
(!needsOwnCorrection ||
|
|
344
|
+
fieldScore(required, { text: token, whole: true }) > 0) &&
|
|
345
|
+
(field.weight > correctionWeight ||
|
|
346
|
+
correction === undefined ||
|
|
347
|
+
token < correction)) {
|
|
348
|
+
correction = token;
|
|
349
|
+
correctionWeight = field.weight;
|
|
350
|
+
}
|
|
351
|
+
}
|
|
352
|
+
}
|
|
353
|
+
if (!correction)
|
|
354
|
+
return;
|
|
355
|
+
patterns[missingIndex] = { text: correction, whole: true };
|
|
356
|
+
weights[missingIndex] = correctionWeight;
|
|
357
|
+
return {
|
|
358
|
+
patterns,
|
|
359
|
+
score: weights.reduce((a, b) => a + b, 0),
|
|
360
|
+
corrected: true,
|
|
361
|
+
};
|
|
362
|
+
}
|
|
363
|
+
/** Score only page metadata plus a single coherent section, never sibling bodies. */
|
|
364
|
+
function sectionCandidates(index, terms, fuzzy) {
|
|
365
|
+
const phrase = terms.join(" ");
|
|
366
|
+
const results = [];
|
|
367
|
+
for (const page of index.pages) {
|
|
368
|
+
const title = normalize(page.title);
|
|
369
|
+
const metadata = [
|
|
370
|
+
{ text: title, weight: 8 },
|
|
371
|
+
{ text: normalize(page.description), weight: 2 },
|
|
372
|
+
];
|
|
373
|
+
const pageMatch = matchFields(metadata, terms, fuzzy);
|
|
374
|
+
let pageCandidate;
|
|
375
|
+
if (pageMatch) {
|
|
376
|
+
const exact = title === phrase;
|
|
377
|
+
pageCandidate = {
|
|
378
|
+
page,
|
|
379
|
+
source: sourceText([page.description, ...page.sections.map((section) => section.text)], pageMatch.patterns, page.title),
|
|
380
|
+
score: pageMatch.score + (exact ? 24 : 0),
|
|
381
|
+
rank: pageMatch.corrected ? 0 : exact ? 2 : 1,
|
|
382
|
+
patterns: pageMatch.patterns,
|
|
383
|
+
};
|
|
384
|
+
}
|
|
385
|
+
for (const section of page.sections) {
|
|
386
|
+
const heading = normalize(section.title);
|
|
387
|
+
const body = normalize(section.text);
|
|
388
|
+
const ownFields = [
|
|
389
|
+
{ text: heading, weight: 4 },
|
|
390
|
+
{ text: body, weight: 1 },
|
|
391
|
+
];
|
|
392
|
+
const match = matchFields([...metadata, ...ownFields], terms, fuzzy, ownFields);
|
|
393
|
+
if (!match?.patterns.some((pattern) => fieldScore(ownFields, pattern)))
|
|
394
|
+
continue;
|
|
395
|
+
// Repeated page headings without their own matching body add no useful destination.
|
|
396
|
+
if (pageMatch &&
|
|
397
|
+
heading === title &&
|
|
398
|
+
!match.patterns.some((pattern) => contains(body, pattern)))
|
|
399
|
+
continue;
|
|
400
|
+
const exact = heading === phrase;
|
|
401
|
+
const candidate = {
|
|
402
|
+
page,
|
|
403
|
+
destination: section.url === page.url ? undefined : section,
|
|
404
|
+
source: sourceText([section.text, page.description], match.patterns, section.title || page.title),
|
|
405
|
+
score: match.score + (exact ? 12 : 0),
|
|
406
|
+
rank: match.corrected ? 0 : exact ? 2 : 1,
|
|
407
|
+
patterns: match.patterns,
|
|
408
|
+
};
|
|
409
|
+
if (candidate.destination)
|
|
410
|
+
results.push(candidate);
|
|
411
|
+
else if (!pageCandidate ||
|
|
412
|
+
compareCandidates(candidate, pageCandidate) < 0)
|
|
413
|
+
pageCandidate = candidate;
|
|
414
|
+
}
|
|
415
|
+
if (pageCandidate)
|
|
416
|
+
results.push(pageCandidate);
|
|
417
|
+
}
|
|
418
|
+
return results;
|
|
419
|
+
}
|
|
420
|
+
// Greater than the largest raw score: 32 terms × weight 8 + title bonus 24.
|
|
421
|
+
const sectionTierWeight = 281;
|
|
422
|
+
function displayResult(candidate) {
|
|
423
|
+
const { page, destination, patterns } = candidate;
|
|
424
|
+
const isSection = destination !== undefined && destination.url !== page.url;
|
|
425
|
+
const trail = isSection
|
|
426
|
+
? [
|
|
427
|
+
...(page.breadcrumbs ?? []),
|
|
428
|
+
page.title,
|
|
429
|
+
...(destination.breadcrumbs ?? []),
|
|
430
|
+
]
|
|
431
|
+
: [...(page.breadcrumbs ?? [])];
|
|
432
|
+
const snippet = excerpt(candidate.source, patterns);
|
|
433
|
+
return {
|
|
434
|
+
kind: isSection ? "section" : "page",
|
|
435
|
+
title: page.title,
|
|
436
|
+
pageUrl: page.url,
|
|
437
|
+
url: candidateUrl(candidate),
|
|
438
|
+
...(destination?.title ? { section: destination.title } : {}),
|
|
439
|
+
breadcrumbs: trail,
|
|
440
|
+
excerpt: snippet,
|
|
441
|
+
score: candidate.rank * sectionTierWeight + candidate.score,
|
|
442
|
+
matches: {
|
|
443
|
+
title: matchRanges(page.title, patterns),
|
|
444
|
+
section: matchRanges(destination?.title ?? "", patterns),
|
|
445
|
+
excerpt: matchRanges(snippet, patterns),
|
|
446
|
+
breadcrumbs: trail.map((label) => matchRanges(label, patterns)),
|
|
447
|
+
},
|
|
448
|
+
};
|
|
449
|
+
}
|
|
450
|
+
/** Query without filesystem, network, or DOM access; page mode preserves legacy ranking. */
|
|
451
|
+
export function search(index, query, options = {}) {
|
|
452
|
+
validateSearchIndex(index);
|
|
453
|
+
if (options === null || typeof options !== "object" || Array.isArray(options))
|
|
454
|
+
throw new TypeError("Search options must be an object.");
|
|
455
|
+
const limit = options.limit ?? 10;
|
|
456
|
+
const mode = options.mode === undefined ? "pages" : options.mode;
|
|
457
|
+
const fuzzy = options.fuzzy === undefined ? false : options.fuzzy;
|
|
458
|
+
if (!Number.isInteger(limit) || limit < 0 || limit > 100)
|
|
459
|
+
throw new RangeError("Search limit must be an integer from 0 to 100.");
|
|
460
|
+
if (mode !== "pages" && mode !== "sections")
|
|
461
|
+
throw new RangeError("Search mode must be pages or sections.");
|
|
462
|
+
if (typeof fuzzy !== "boolean")
|
|
463
|
+
throw new TypeError("Search fuzzy option must be a boolean.");
|
|
464
|
+
if (fuzzy && mode !== "sections")
|
|
465
|
+
throw new RangeError("Fuzzy search requires sections mode.");
|
|
466
|
+
if (typeof query !== "string")
|
|
467
|
+
throw new TypeError("Search query must be a string.");
|
|
468
|
+
if (query.length > 512)
|
|
469
|
+
throw new RangeError("Search query must not exceed 512 characters.");
|
|
470
|
+
const terms = [...new Set(normalize(query).split(" ").filter(Boolean))];
|
|
471
|
+
if (terms.length > 32)
|
|
472
|
+
throw new RangeError("Search query must not exceed 32 unique terms.");
|
|
473
|
+
if (!terms.length || !limit)
|
|
474
|
+
return [];
|
|
475
|
+
const results = mode === "sections"
|
|
476
|
+
? sectionCandidates(index, terms, false)
|
|
477
|
+
: pageCandidates(index, terms);
|
|
478
|
+
if (fuzzy && results.length < limit) {
|
|
479
|
+
const literalUrls = new Set(results.map(candidateUrl));
|
|
480
|
+
results.push(...sectionCandidates(index, terms, true).filter((candidate) => candidate.rank === 0 && !literalUrls.has(candidateUrl(candidate))));
|
|
481
|
+
}
|
|
482
|
+
results.sort(compareCandidates);
|
|
483
|
+
// Original-text mapping and excerpt windows are computed only for final hits.
|
|
484
|
+
return results.slice(0, limit).map(displayResult);
|
|
230
485
|
}
|