@wgtechlabs/mdd-engine 0.1.0-pr.08f3b80 → 0.1.0-pr.0cb765a
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/search.js +87 -10
- package/package.json +1 -1
package/dist/search.js
CHANGED
|
@@ -70,11 +70,76 @@ function normalize(value) {
|
|
|
70
70
|
.replace(/[^\p{L}\p{N}\p{M}_]+/gu, " ")
|
|
71
71
|
.trim();
|
|
72
72
|
}
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
73
|
+
/** Map a normalized match back to the original text, including NFKC expansions. */
|
|
74
|
+
function matchRange(value, terms) {
|
|
75
|
+
const normalized = value.normalize("NFKC").toLowerCase();
|
|
76
|
+
let start = -1;
|
|
77
|
+
let end = -1;
|
|
78
|
+
for (const term of terms) {
|
|
79
|
+
const found = normalized.indexOf(term);
|
|
80
|
+
if (found >= 0 &&
|
|
81
|
+
(start < 0 ||
|
|
82
|
+
found < start ||
|
|
83
|
+
(found === start && found + term.length > end))) {
|
|
84
|
+
start = found;
|
|
85
|
+
end = found + term.length;
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
if (start < 0)
|
|
89
|
+
return;
|
|
90
|
+
const groups = [];
|
|
91
|
+
for (const { segment, index } of new Intl.Segmenter(undefined, {
|
|
92
|
+
granularity: "grapheme",
|
|
93
|
+
}).segment(value)) {
|
|
94
|
+
let group = {
|
|
95
|
+
text: segment.normalize("NFKC"),
|
|
96
|
+
start: index,
|
|
97
|
+
end: index + segment.length,
|
|
98
|
+
};
|
|
99
|
+
// NFKC can combine adjacent graphemes, e.g. compatibility Hangul ㄱㅏ → 가.
|
|
100
|
+
while (groups.length) {
|
|
101
|
+
const previous = groups.at(-1);
|
|
102
|
+
if (!previous)
|
|
103
|
+
break;
|
|
104
|
+
const joined = previous.text + group.text;
|
|
105
|
+
const combined = joined.normalize("NFKC");
|
|
106
|
+
if (combined === joined)
|
|
107
|
+
break;
|
|
108
|
+
groups.pop();
|
|
109
|
+
group = { text: combined, start: previous.start, end: group.end };
|
|
110
|
+
}
|
|
111
|
+
groups.push(group);
|
|
112
|
+
}
|
|
113
|
+
// Match against whole-field lowercase for contextual letters such as sigma.
|
|
114
|
+
// Per-group lowercase is used only for lengths (including expansions like İ).
|
|
115
|
+
let offset = 0;
|
|
116
|
+
let originalStart = 0;
|
|
117
|
+
for (const group of groups) {
|
|
118
|
+
const next = offset + group.text.toLowerCase().length;
|
|
119
|
+
if (offset <= start && start < next)
|
|
120
|
+
originalStart = group.start;
|
|
121
|
+
if (offset < end && end <= next)
|
|
122
|
+
return [originalStart, group.end];
|
|
123
|
+
offset = next;
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
function excerpt(value, terms) {
|
|
127
|
+
const text = value.replace(/\s+/gu, " ").trim();
|
|
128
|
+
const characters = Array.from(text);
|
|
129
|
+
if (characters.length <= 160)
|
|
130
|
+
return text;
|
|
131
|
+
const match = matchRange(text, terms);
|
|
132
|
+
const matchStart = match ? Array.from(text.slice(0, match[0])).length : 0;
|
|
133
|
+
const matchLength = match
|
|
134
|
+
? Array.from(text.slice(match[0], match[1])).length
|
|
135
|
+
: 0;
|
|
136
|
+
// Reserve room for both ellipses and the matched term before adding context.
|
|
137
|
+
const context = Math.min(40, Math.max(0, 158 - matchLength));
|
|
138
|
+
const start = Math.min(Math.max(0, matchStart - context), characters.length - 159);
|
|
139
|
+
let end = Math.min(characters.length, start + 160 - (start > 0 ? 1 : 0));
|
|
140
|
+
if (end < characters.length)
|
|
141
|
+
end--;
|
|
142
|
+
return `${start > 0 ? "…" : ""}${characters.slice(start, end).join("")}${end < characters.length ? "…" : ""}`;
|
|
78
143
|
}
|
|
79
144
|
/** Return at most one hit per page, without filesystem, network, or DOM access. */
|
|
80
145
|
export function search(index, query, options = {}) {
|
|
@@ -132,14 +197,23 @@ export function search(index, query, options = {}) {
|
|
|
132
197
|
matches.sort((a, b) => b.score - a.score);
|
|
133
198
|
const best = matches[0]?.section;
|
|
134
199
|
const destination = metadataMatch ? undefined : best;
|
|
200
|
+
const sources = [
|
|
201
|
+
best?.text ?? "",
|
|
202
|
+
page.description,
|
|
203
|
+
...page.sections.map((section) => section.text),
|
|
204
|
+
];
|
|
205
|
+
const source = sources.find((text) => {
|
|
206
|
+
const normalized = normalize(text);
|
|
207
|
+
return terms.some((term) => normalized.includes(term));
|
|
208
|
+
}) ??
|
|
209
|
+
sources.find((text) => text) ??
|
|
210
|
+
page.title;
|
|
135
211
|
results.push({
|
|
136
212
|
title: page.title,
|
|
137
213
|
url: destination?.url ?? page.url,
|
|
138
214
|
...(destination?.title ? { section: destination.title } : {}),
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
page.sections.find((section) => section.text)?.text ||
|
|
142
|
-
page.title),
|
|
215
|
+
// Keep source text until ranking so only returned hits need a snippet.
|
|
216
|
+
excerpt: source,
|
|
143
217
|
score: weights.reduce((sum, weight) => sum + weight, 0) +
|
|
144
218
|
(title === phrase
|
|
145
219
|
? 24
|
|
@@ -149,5 +223,8 @@ export function search(index, query, options = {}) {
|
|
|
149
223
|
});
|
|
150
224
|
}
|
|
151
225
|
results.sort((a, b) => b.score - a.score || (a.url < b.url ? -1 : a.url > b.url ? 1 : 0));
|
|
152
|
-
return results.slice(0, limit)
|
|
226
|
+
return results.slice(0, limit).map((result) => ({
|
|
227
|
+
...result,
|
|
228
|
+
excerpt: excerpt(result.excerpt, terms),
|
|
229
|
+
}));
|
|
153
230
|
}
|