@lacspace/nepali-match 1.0.0 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -6
- package/dist/index.cjs +66 -21
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +47 -11
- package/dist/index.d.ts +47 -11
- package/dist/index.js +66 -22
- package/dist/index.js.map +1 -1
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -36,17 +36,20 @@ normaliseNe("काठमाडौँ") === normaliseNe("काठमाडौ
|
|
|
36
36
|
6. Devanagari digits → ASCII.
|
|
37
37
|
7. Whitespace collapses.
|
|
38
38
|
|
|
39
|
-
`loose: true` also folds श/ष → स, व → ब and ण → न. Latin text is left as it is.
|
|
39
|
+
`loose: true` also folds श/ष → स, व → ब and ण → न. `hyphenAsSpace: true` treats hyphens as spaces, so "Solu Khumbu" matches "Solu-Khumbu" and "Bardaghat-Susta". Latin text is left as it is.
|
|
40
40
|
|
|
41
41
|
**Nepali terms** match as whole words:
|
|
42
42
|
- Nothing may stand before the word except a space or punctuation.
|
|
43
43
|
- After the word, a chain of up to three `POSTPOSITIONS` may follow: मा, को, का, की, ले, लाई, बाट, सँग, देखि, सम्म, तिर, भित्र, हरू, मै, बाटै, नै … Anything else after it means it's a different word.
|
|
44
|
-
- Turn
|
|
44
|
+
- Turn the built-in list off with `postpositions: false`. Add your own with `extraSuffixes`; with `postpositions: false` they become the only suffixes allowed.
|
|
45
|
+
- "Like" (जस्तो) is deliberately not in the list: "काठमाडौंजस्तो ठूलो सहर" is a comparison, not a location.
|
|
46
|
+
- With `joinNepali: true`, any space in a multi-word Nepali term is optional, so "बर्दघाट सुस्ता पूर्व" also matches "बर्दघाट सुस्तापूर्व".
|
|
45
47
|
|
|
46
48
|
**English terms** match as whole words, ignoring case. Exceptions:
|
|
47
49
|
- Under the default `caseSensitive: "auto"` rule, an all-caps term of 2–5 letters (SEE, NEB, PSC) must match its case exactly.
|
|
48
50
|
- Set `caseSensitive` per term or per matcher to change this.
|
|
49
|
-
- An
|
|
51
|
+
- An exact-case term longer than 5 letters also matches in ALL CAPS, so "KATHMANDU" in a headline counts. Turn this off with `allCaps: false`.
|
|
52
|
+
- Short acronyms can't tell a shouted headline apart: "COME AND SEE" still matches SEE. Check the case of the surrounding text if your input has shouty headlines.
|
|
50
53
|
|
|
51
54
|
**Overlaps:** the longest match wins, so "Nawalparasi West" beats "Nawalparasi". Pass `overlaps: true` to keep both.
|
|
52
55
|
|
|
@@ -54,20 +57,27 @@ Every match carries `index`/`end` offsets into the original text (after NFC), so
|
|
|
54
57
|
|
|
55
58
|
## API
|
|
56
59
|
|
|
57
|
-
- **`createMatcher(terms, options?)`** returns `{ find(text), test(text, id?), ids(text), size }`. Compile it once and reuse it.
|
|
60
|
+
- **`createMatcher(terms, options?)`** returns `{ find(text), test(text, id?), ids(text), prepare(text), size }`. Compile it once and reuse it.
|
|
61
|
+
- **`prepare(text, options?)`** / **`matcher.prepare(text)`** normalise text once. Every `find`, `test`, `ids`, `findTerms`, `contains` and `near` accepts the result in place of a string, so a rules loop over one article normalises it once, not once per rule. A prepared text made with different options is re-prepared automatically.
|
|
58
62
|
- A term is either a string, or `{ id?, en?, ne?, aliases?, caseSensitive? }`. `en` and `ne` each take one spelling or an array.
|
|
59
63
|
- **`findTerms(text, terms, options?)`** and **`contains(text, terms, options?)`** are one-off shortcuts.
|
|
60
64
|
- **`near(text, a, b, maxChars | options, sameSentence?)`** returns the closest `{ a, b, gap }` pair of matches within `maxChars` (default 60), in either order, or null.
|
|
65
|
+
- `a` and `b` can be terms, a compiled `Matcher`, or a `Match[]` you already found in the same text.
|
|
61
66
|
- By default both must sit in the same sentence. A sentence ends at । ॥ ? ! or a newline, or at "." before a space.
|
|
62
67
|
- Dotted abbreviations such as ने.क.पा. and U.S. never end a sentence.
|
|
63
68
|
- **`sentenceSpans(text)`** returns the sentence boundaries `near` uses.
|
|
64
69
|
- **`splitSuffix(word)`**: for example, `"जिल्लाहरूमा"` → `{ stem: "जिल्ला", suffixes: ["हरु", "मा"] }`.
|
|
65
70
|
- **`districtTerms()`** returns all 77 districts.
|
|
66
71
|
- **Names and ids:** names come from `@lacspace/nepali-utils`. Ids are slugs such as `"nawalparasi-east"`, and `province` is a number.
|
|
67
|
-
- **Spellings:** each district carries the spellings seen in real copy
|
|
68
|
-
- **Case:** English district names are exact-case, so "dang it" is not Dang.
|
|
72
|
+
- **Spellings:** each district carries the spellings seen in real copy, plus every alias in `@lacspace/nepali-utils`: Kavre/काभ्रे, Rukum East/Rukum Purba/रुकुम पूर्व, Parasi/परासी, Bardaghat Susta West, Kapilbastu, मोरंग/मोरङ…
|
|
73
|
+
- **Case:** English district names are exact-case, so "dang it" is not Dang. Names over 5 letters also match in ALL CAPS.
|
|
74
|
+
- **Known edge:** a place that shares a district's name still matches it. For example, Nawalpur in Sarlahi matches the Nawalpur alias of Nawalparasi East.
|
|
69
75
|
- **`ambiguous: true`:** marks Parbat, because पर्वत also means "mountain". Confirm it with `near()` or with district/area context before you tag a story.
|
|
70
76
|
|
|
77
|
+
## Speed
|
|
78
|
+
|
|
79
|
+
A ~1,500-character article checked against all 77 districts (258 spellings) takes about 0.16 ms once prepared, and about 0.3 ms from a raw string.
|
|
80
|
+
|
|
71
81
|
## Limits
|
|
72
82
|
|
|
73
83
|
- The matcher is rule-based and does no stemming beyond postpositions. Verb forms and compounds won't match, which is the point.
|
package/dist/index.cjs
CHANGED
|
@@ -93,8 +93,9 @@ var ALIASES = {
|
|
|
93
93
|
Kathmandu: { en: ["Katmandu"], ne: ["\u0915\u093E\u0920\u092E\u093E\u0923\u094D\u0921\u094C", "\u0915\u093E\u0920\u092E\u093E\u0928\u094D\u0921\u0941"] },
|
|
94
94
|
Kavrepalanchok: { en: ["Kavre", "Kabhre", "Kabhrepalanchok", "Kavrepalanchowk"], ne: ["\u0915\u093E\u092D\u094D\u0930\u0947", "\u0915\u093E\u092D\u094D\u0930\u0947\u092A\u0932\u093E\u0928\u094D\u091A\u094B\u0915"] },
|
|
95
95
|
Sindhupalchok: { en: ["Sindhupalchowk"] },
|
|
96
|
-
|
|
97
|
-
"Nawalparasi
|
|
96
|
+
Solukhumbu: { en: ["Solu Khumbu"], ne: ["\u0938\u094B\u0932\u0941 \u0916\u0941\u092E\u094D\u092C\u0941"] },
|
|
97
|
+
"Nawalparasi East": { en: ["Nawalparasi (East)", "East Nawalparasi", "Nawalpur", "Bardaghat Susta East"], ne: ["\u0928\u0935\u0932\u092A\u0930\u093E\u0938\u0940 \u092A\u0942\u0930\u094D\u0935", "\u092A\u0942\u0930\u094D\u0935\u0940 \u0928\u0935\u0932\u092A\u0930\u093E\u0938\u0940", "\u0928\u0935\u0932\u092A\u0941\u0930", "\u092C\u0930\u094D\u0926\u0918\u093E\u091F \u0938\u0941\u0938\u094D\u0924\u093E \u092A\u0942\u0930\u094D\u0935"] },
|
|
98
|
+
"Nawalparasi West": { en: ["Nawalparasi (West)", "West Nawalparasi", "Bardaghat Susta West"], ne: ["\u0928\u0935\u0932\u092A\u0930\u093E\u0938\u0940 \u092A\u0936\u094D\u091A\u093F\u092E", "\u092A\u0936\u094D\u091A\u093F\u092E\u0940 \u0928\u0935\u0932\u092A\u0930\u093E\u0938\u0940", "\u092C\u0930\u094D\u0926\u0918\u093E\u091F \u0938\u0941\u0938\u094D\u0924\u093E \u092A\u0936\u094D\u091A\u093F\u092E"] },
|
|
98
99
|
"Eastern Rukum": { en: ["Rukum East", "East Rukum", "Rukum (East)"], ne: ["\u0930\u0941\u0915\u0941\u092E \u092A\u0942\u0930\u094D\u0935"] },
|
|
99
100
|
"Western Rukum": { en: ["Rukum West", "West Rukum", "Rukum (West)"], ne: ["\u0930\u0941\u0915\u0941\u092E \u092A\u0936\u094D\u091A\u093F\u092E"] },
|
|
100
101
|
Tanahun: { en: ["Tanahu"], ne: ["\u0924\u0928\u0939\u0941"] },
|
|
@@ -119,11 +120,13 @@ var slug = (s) => s.toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-|-$/g,
|
|
|
119
120
|
function districtTerms() {
|
|
120
121
|
return DISTRICTS.map((d) => {
|
|
121
122
|
const extra = ALIASES[d.name] ?? {};
|
|
122
|
-
const
|
|
123
|
+
const own = d.aliases ?? [];
|
|
124
|
+
const isNe = (x) => /[\u0900-\u097f]/.test(x);
|
|
125
|
+
const ne = /* @__PURE__ */ new Set([d.nameNp, ...extra.ne ?? [], ...own.filter(isNe)]);
|
|
123
126
|
for (const n of [...ne]) if (n.endsWith("\u0919")) ne.add(n.slice(0, -1) + "\u0902\u0917");
|
|
124
127
|
return {
|
|
125
128
|
id: slug(d.name),
|
|
126
|
-
en: [d.name, ...extra.en ?? []],
|
|
129
|
+
en: [.../* @__PURE__ */ new Set([d.name, ...extra.en ?? [], ...own.filter((x) => !isNe(x))])],
|
|
127
130
|
ne: [...ne],
|
|
128
131
|
province: d.province,
|
|
129
132
|
...AMBIGUOUS.has(d.name) ? { ambiguous: true } : {},
|
|
@@ -133,7 +136,7 @@ function districtTerms() {
|
|
|
133
136
|
}
|
|
134
137
|
|
|
135
138
|
// src/index.ts
|
|
136
|
-
var VERSION = "1.
|
|
139
|
+
var VERSION = "1.1.0";
|
|
137
140
|
var NUKTA = 2364;
|
|
138
141
|
var HALANT = 2381;
|
|
139
142
|
var NASALS = /* @__PURE__ */ new Set([2329, 2334, 2339, 2344, 2350]);
|
|
@@ -160,6 +163,8 @@ var LOOSE = {
|
|
|
160
163
|
2339: 2344
|
|
161
164
|
// ण → न
|
|
162
165
|
};
|
|
166
|
+
var isSpace = (c) => c === 32 || c >= 9 && c <= 13 || c === 160 || c === 5760 || c >= 8192 && c <= 8202 || c === 8232 || c === 8233 || c === 8239 || c === 8287 || c === 12288;
|
|
167
|
+
var isHyphen = (c) => c === 45 || c === 8208 || c === 8209;
|
|
163
168
|
var isConsonant = (c) => c >= 2325 && c <= 2361 || c >= 2392 && c <= 2399;
|
|
164
169
|
function mapNormalise(input, opts = {}) {
|
|
165
170
|
const digits = opts.digits !== false;
|
|
@@ -179,7 +184,7 @@ function mapNormalise(input, opts = {}) {
|
|
|
179
184
|
for (let i = 0; i < s.length; i++) {
|
|
180
185
|
let c = s.charCodeAt(i);
|
|
181
186
|
if (DROP.has(c)) continue;
|
|
182
|
-
if (
|
|
187
|
+
if (isSpace(c) || opts.hyphenAsSpace && isHyphen(c)) {
|
|
183
188
|
if (!lastSpace) push(" ", i);
|
|
184
189
|
lastSpace = true;
|
|
185
190
|
continue;
|
|
@@ -253,7 +258,12 @@ var POSTPOSITIONS = [
|
|
|
253
258
|
"\u0935\u093E\u0938\u0940",
|
|
254
259
|
"\u092C\u093E\u0938\u0940",
|
|
255
260
|
"\u0938\u094D\u0925\u093F\u0924",
|
|
256
|
-
"\u0928\u093F\u0935\u093E\u0938\u0940"
|
|
261
|
+
"\u0928\u093F\u0935\u093E\u0938\u0940",
|
|
262
|
+
"\u092E\u093E\u0930\u094D\u092B\u0924",
|
|
263
|
+
"\u092C\u093E\u0930\u0947",
|
|
264
|
+
"\u0905\u0928\u094D\u0924\u0930\u094D\u0917\u0924",
|
|
265
|
+
"\u092E\u093E\u0924\u094D\u0930",
|
|
266
|
+
"\u091C\u0938\u094D\u0924\u093E"
|
|
257
267
|
];
|
|
258
268
|
var isNeLetter = (c) => c >= 2304 && c <= 2403 || c >= 2417 && c <= 2431;
|
|
259
269
|
var isWordChar = (ch) => !!ch && /[\p{L}\p{M}\p{N}]/u.test(ch);
|
|
@@ -263,9 +273,18 @@ function autoCase(term) {
|
|
|
263
273
|
const letters = term.replace(/[^\p{L}]/gu, "");
|
|
264
274
|
return letters.length >= 2 && letters.length <= 5 && letters === letters.toUpperCase() && letters !== letters.toLowerCase();
|
|
265
275
|
}
|
|
276
|
+
var optsKey = (o) => `${o.loose ? 1 : 0}${o.digits === false ? 0 : 1}${o.hyphenAsSpace ? 1 : 0}`;
|
|
277
|
+
function prepare(text, opts = {}) {
|
|
278
|
+
const nfc = text.normalize("NFC");
|
|
279
|
+
const { text: n, map } = mapNormalise(nfc, opts);
|
|
280
|
+
return { kind: "prepared", original: nfc, text: n, map, key: optsKey(opts) };
|
|
281
|
+
}
|
|
282
|
+
var isPrepared = (x) => typeof x === "object" && x !== null && x.kind === "prepared";
|
|
266
283
|
function createMatcher(terms, opts = {}) {
|
|
267
|
-
const nopts = { loose: opts.loose, digits: opts.digits };
|
|
268
|
-
const
|
|
284
|
+
const nopts = { loose: opts.loose, digits: opts.digits, hyphenAsSpace: opts.hyphenAsSpace };
|
|
285
|
+
const key = optsKey(nopts);
|
|
286
|
+
const allCaps = opts.allCaps !== false;
|
|
287
|
+
const suffixes = [...new Set([...opts.postpositions === false ? [] : POSTPOSITIONS, ...opts.extraSuffixes ?? []].map((s) => normaliseNe(s, nopts)))].filter(Boolean).sort((a, b) => b.length - a.length);
|
|
269
288
|
const variants = [];
|
|
270
289
|
for (const t of terms) {
|
|
271
290
|
const obj = typeof t === "string" ? { id: t, aliases: [t] } : t;
|
|
@@ -274,12 +293,26 @@ function createMatcher(terms, opts = {}) {
|
|
|
274
293
|
const id = obj.id ?? spellings[0] ?? "";
|
|
275
294
|
for (const sp of new Set(spellings)) {
|
|
276
295
|
if (hasDevanagari(sp)) {
|
|
277
|
-
|
|
296
|
+
const k = normaliseNe(sp, nopts);
|
|
297
|
+
variants.push({ id, term: sp, lang: "ne", key: k });
|
|
298
|
+
if (opts.joinNepali && k.includes(" ")) {
|
|
299
|
+
const parts = k.split(" ");
|
|
300
|
+
const gaps = Math.min(parts.length - 1, 4);
|
|
301
|
+
for (let mask = 1; mask < 1 << gaps; mask++) {
|
|
302
|
+
let joined = parts[0];
|
|
303
|
+
for (let g = 1; g < parts.length; g++) joined += (g <= gaps && mask & 1 << g - 1 ? "" : " ") + parts[g];
|
|
304
|
+
variants.push({ id, term: sp, lang: "ne", key: joined });
|
|
305
|
+
}
|
|
306
|
+
}
|
|
278
307
|
} else {
|
|
279
308
|
const rule = obj.caseSensitive ?? opts.caseSensitive ?? "auto";
|
|
280
309
|
const exact = rule === "auto" ? autoCase(sp) : rule;
|
|
281
|
-
const
|
|
282
|
-
|
|
310
|
+
const words = normaliseNe(sp, nopts).split(" ").filter(Boolean);
|
|
311
|
+
const body = words.map(escapeRe).join(" ");
|
|
312
|
+
const letters = sp.replace(/[^\p{L}]/gu, "");
|
|
313
|
+
const shout = exact && allCaps && letters.length > 5 && sp !== sp.toUpperCase();
|
|
314
|
+
const alt = shout ? `|${words.map((w) => escapeRe(w.toUpperCase())).join(" ")}` : "";
|
|
315
|
+
variants.push({ id, term: sp, lang: "en", key: sp, re: new RegExp(`(?:${body}${alt})`, exact ? "gu" : "giu") });
|
|
283
316
|
}
|
|
284
317
|
}
|
|
285
318
|
}
|
|
@@ -294,10 +327,11 @@ function createMatcher(terms, opts = {}) {
|
|
|
294
327
|
}
|
|
295
328
|
return -1;
|
|
296
329
|
};
|
|
297
|
-
const
|
|
298
|
-
|
|
299
|
-
const
|
|
300
|
-
const nfc =
|
|
330
|
+
const prep = (text) => prepare(text, nopts);
|
|
331
|
+
const find = (input) => {
|
|
332
|
+
const p = isPrepared(input) ? input.key === key ? input : prep(input.original) : prep(input);
|
|
333
|
+
const { text: n, map, original: nfc } = p;
|
|
334
|
+
const toOrig = (i) => i >= map.length ? nfc.length : map[i];
|
|
301
335
|
const found = [];
|
|
302
336
|
for (const v of variants) {
|
|
303
337
|
if (v.lang === "ne") {
|
|
@@ -345,9 +379,11 @@ function createMatcher(terms, opts = {}) {
|
|
|
345
379
|
find,
|
|
346
380
|
test: (text, id) => find(text).some((m) => id === void 0 || m.id === id),
|
|
347
381
|
ids: (text) => [...new Set(find(text).map((m) => m.id))],
|
|
382
|
+
prepare: prep,
|
|
348
383
|
get size() {
|
|
349
384
|
return variants.length;
|
|
350
|
-
}
|
|
385
|
+
},
|
|
386
|
+
kind: "matcher"
|
|
351
387
|
};
|
|
352
388
|
}
|
|
353
389
|
function findTerms(text, terms, opts) {
|
|
@@ -370,6 +406,8 @@ function splitSuffix(word) {
|
|
|
370
406
|
}
|
|
371
407
|
return { stem: nfc.slice(0, end >= n.length ? nfc.length : map[end]), suffixes: out };
|
|
372
408
|
}
|
|
409
|
+
var isMatcher = (x) => typeof x === "object" && x !== null && x.kind === "matcher";
|
|
410
|
+
var isMatchList = (x) => Array.isArray(x) && (x.length === 0 || typeof x[0] === "object" && x[0] !== null && typeof x[0].index === "number" && typeof x[0].end === "number");
|
|
373
411
|
var dottedAbbr = (s, i) => {
|
|
374
412
|
let j = i - 1;
|
|
375
413
|
while (j >= 0 && !/\s/.test(s[j])) j--;
|
|
@@ -395,10 +433,16 @@ function near(text, a, b, maxCharsOrOpts = {}, sameSentence) {
|
|
|
395
433
|
if (sameSentence !== void 0) opts.sameSentence = sameSentence;
|
|
396
434
|
const max = opts.maxChars ?? 60;
|
|
397
435
|
const same = opts.sameSentence !== false;
|
|
398
|
-
const
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
436
|
+
const resolve = (src) => {
|
|
437
|
+
if (isMatcher(src)) return src.find(text);
|
|
438
|
+
if (isMatchList(src)) return src;
|
|
439
|
+
return findTerms(text, src, opts);
|
|
440
|
+
};
|
|
441
|
+
const am = resolve(a);
|
|
442
|
+
if (!am.length) return null;
|
|
443
|
+
const bm = resolve(b);
|
|
444
|
+
if (!bm.length) return null;
|
|
445
|
+
const spans = same ? sentenceSpans(isPrepared(text) ? text.original : text) : [];
|
|
402
446
|
const sentenceOf = (i) => spans.findIndex(([s, e]) => i >= s && i < e);
|
|
403
447
|
let best = null;
|
|
404
448
|
for (const x of am) {
|
|
@@ -422,6 +466,7 @@ exports.findTerms = findTerms;
|
|
|
422
466
|
exports.near = near;
|
|
423
467
|
exports.normaliseNe = normaliseNe;
|
|
424
468
|
exports.normalizeNe = normalizeNe;
|
|
469
|
+
exports.prepare = prepare;
|
|
425
470
|
exports.sentenceSpans = sentenceSpans;
|
|
426
471
|
exports.splitSuffix = splitSuffix;
|
|
427
472
|
//# sourceMappingURL=index.cjs.map
|