@lacspace/nepali-match 1.0.0 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -36,17 +36,20 @@ normaliseNe("काठमाडौँ") === normaliseNe("काठमाडौ
36
36
  6. Devanagari digits → ASCII.
37
37
  7. Whitespace collapses.
38
38
 
39
- `loose: true` also folds श/ष → स, व → ब and ण → न. Latin text is left as it is.
39
+ `loose: true` also folds श/ष → स, व → ब and ण → न. `hyphenAsSpace: true` treats hyphens as spaces, so "Solu Khumbu" matches "Solu-Khumbu" and "Bardaghat-Susta". Latin text is left as it is.
40
40
 
41
41
  **Nepali terms** match as whole words:
42
42
  - Nothing may stand before the word except a space or punctuation.
43
43
  - After the word, a chain of up to three `POSTPOSITIONS` may follow: मा, को, का, की, ले, लाई, बाट, सँग, देखि, सम्म, तिर, भित्र, हरू, मै, बाटै, नै … Anything else after it means it's a different word.
44
- - Turn this off with `postpositions: false`. Add your own with `extraSuffixes`.
44
+ - Turn the built-in list off with `postpositions: false`. Add your own with `extraSuffixes`; with `postpositions: false` they become the only suffixes allowed.
45
+ - "Like" (जस्तो) is deliberately not in the list: "काठमाडौंजस्तो ठूलो सहर" is a comparison, not a location.
46
+ - With `joinNepali: true`, any space in a multi-word Nepali term is optional, so "बर्दघाट सुस्ता पूर्व" also matches "बर्दघाट सुस्तापूर्व".
45
47
 
46
48
  **English terms** match as whole words, ignoring case. Exceptions:
47
49
  - Under the default `caseSensitive: "auto"` rule, an all-caps term of 2–5 letters (SEE, NEB, PSC) must match its case exactly.
48
50
  - Set `caseSensitive` per term or per matcher to change this.
49
- - An all-caps headline ("COME AND SEE") still matches SEE. Check the case of the surrounding text if your input has shouty headlines.
51
+ - An exact-case term longer than 5 letters also matches in ALL CAPS, so "KATHMANDU" in a headline counts. Turn this off with `allCaps: false`.
52
+ - Short acronyms can't tell a shouted headline apart: "COME AND SEE" still matches SEE. Check the case of the surrounding text if your input has shouty headlines.
50
53
 
51
54
  **Overlaps:** the longest match wins, so "Nawalparasi West" beats "Nawalparasi". Pass `overlaps: true` to keep both.
52
55
 
@@ -54,20 +57,27 @@ Every match carries `index`/`end` offsets into the original text (after NFC), so
54
57
 
55
58
  ## API
56
59
 
57
- - **`createMatcher(terms, options?)`** returns `{ find(text), test(text, id?), ids(text), size }`. Compile it once and reuse it.
60
+ - **`createMatcher(terms, options?)`** returns `{ find(text), test(text, id?), ids(text), prepare(text), size }`. Compile it once and reuse it.
61
+ - **`prepare(text, options?)`** / **`matcher.prepare(text)`** normalise text once. Every `find`, `test`, `ids`, `findTerms`, `contains` and `near` accepts the result in place of a string, so a rules loop over one article normalises it once, not once per rule. A prepared text made with different options is re-prepared automatically.
58
62
  - A term is either a string, or `{ id?, en?, ne?, aliases?, caseSensitive? }`. `en` and `ne` each take one spelling or an array.
59
63
  - **`findTerms(text, terms, options?)`** and **`contains(text, terms, options?)`** are one-off shortcuts.
60
64
  - **`near(text, a, b, maxChars | options, sameSentence?)`** returns the closest `{ a, b, gap }` pair of matches within `maxChars` (default 60), in either order, or null.
65
+ - `a` and `b` can be terms, a compiled `Matcher`, or a `Match[]` you already found in the same text.
61
66
  - By default both must sit in the same sentence. A sentence ends at । ॥ ? ! or a newline, or at "." before a space.
62
67
  - Dotted abbreviations such as ने.क.पा. and U.S. never end a sentence.
63
68
  - **`sentenceSpans(text)`** returns the sentence boundaries `near` uses.
64
69
  - **`splitSuffix(word)`**: for example, `"जिल्लाहरूमा"` → `{ stem: "जिल्ला", suffixes: ["हरु", "मा"] }`.
65
70
  - **`districtTerms()`** returns all 77 districts.
66
71
  - **Names and ids:** names come from `@lacspace/nepali-utils`. Ids are slugs such as `"nawalparasi-east"`, and `province` is a number.
67
- - **Spellings:** each district carries the spellings seen in real copy: Kavre/काभ्रे, Rukum East/रुकुम पूर्व, Kapilbastu, मोरंग/मोरङ…
68
- - **Case:** English district names are exact-case, so "dang it" is not Dang.
72
+ - **Spellings:** each district carries the spellings seen in real copy, plus every alias in `@lacspace/nepali-utils`: Kavre/काभ्रे, Rukum East/Rukum Purba/रुकुम पूर्व, Parasi/परासी, Bardaghat Susta West, Kapilbastu, मोरंग/मोरङ…
73
+ - **Case:** English district names are exact-case, so "dang it" is not Dang. Names over 5 letters also match in ALL CAPS.
74
+ - **Known edge:** a place that shares a district's name still matches it. For example, Nawalpur in Sarlahi matches the Nawalpur alias of Nawalparasi East.
69
75
  - **`ambiguous: true`:** marks Parbat, because पर्वत also means "mountain". Confirm it with `near()` or with district/area context before you tag a story.
70
76
 
77
+ ## Speed
78
+
79
+ A ~1,500-character article checked against all 77 districts (258 spellings) takes about 0.16 ms once prepared, and about 0.3 ms from a raw string.
80
+
71
81
  ## Limits
72
82
 
73
83
  - The matcher is rule-based and does no stemming beyond postpositions. Verb forms and compounds won't match, which is the point.
package/dist/index.cjs CHANGED
@@ -93,8 +93,9 @@ var ALIASES = {
93
93
  Kathmandu: { en: ["Katmandu"], ne: ["\u0915\u093E\u0920\u092E\u093E\u0923\u094D\u0921\u094C", "\u0915\u093E\u0920\u092E\u093E\u0928\u094D\u0921\u0941"] },
94
94
  Kavrepalanchok: { en: ["Kavre", "Kabhre", "Kabhrepalanchok", "Kavrepalanchowk"], ne: ["\u0915\u093E\u092D\u094D\u0930\u0947", "\u0915\u093E\u092D\u094D\u0930\u0947\u092A\u0932\u093E\u0928\u094D\u091A\u094B\u0915"] },
95
95
  Sindhupalchok: { en: ["Sindhupalchowk"] },
96
- "Nawalparasi East": { en: ["Nawalparasi (East)", "East Nawalparasi", "Nawalpur"], ne: ["\u0928\u0935\u0932\u092A\u0930\u093E\u0938\u0940 \u092A\u0942\u0930\u094D\u0935", "\u092A\u0942\u0930\u094D\u0935\u0940 \u0928\u0935\u0932\u092A\u0930\u093E\u0938\u0940", "\u0928\u0935\u0932\u092A\u0941\u0930"] },
97
- "Nawalparasi West": { en: ["Nawalparasi (West)", "West Nawalparasi"], ne: ["\u0928\u0935\u0932\u092A\u0930\u093E\u0938\u0940 \u092A\u0936\u094D\u091A\u093F\u092E", "\u092A\u0936\u094D\u091A\u093F\u092E\u0940 \u0928\u0935\u0932\u092A\u0930\u093E\u0938\u0940"] },
96
+ Solukhumbu: { en: ["Solu Khumbu"], ne: ["\u0938\u094B\u0932\u0941 \u0916\u0941\u092E\u094D\u092C\u0941"] },
97
+ "Nawalparasi East": { en: ["Nawalparasi (East)", "East Nawalparasi", "Nawalpur", "Bardaghat Susta East"], ne: ["\u0928\u0935\u0932\u092A\u0930\u093E\u0938\u0940 \u092A\u0942\u0930\u094D\u0935", "\u092A\u0942\u0930\u094D\u0935\u0940 \u0928\u0935\u0932\u092A\u0930\u093E\u0938\u0940", "\u0928\u0935\u0932\u092A\u0941\u0930", "\u092C\u0930\u094D\u0926\u0918\u093E\u091F \u0938\u0941\u0938\u094D\u0924\u093E \u092A\u0942\u0930\u094D\u0935"] },
98
+ "Nawalparasi West": { en: ["Nawalparasi (West)", "West Nawalparasi", "Bardaghat Susta West"], ne: ["\u0928\u0935\u0932\u092A\u0930\u093E\u0938\u0940 \u092A\u0936\u094D\u091A\u093F\u092E", "\u092A\u0936\u094D\u091A\u093F\u092E\u0940 \u0928\u0935\u0932\u092A\u0930\u093E\u0938\u0940", "\u092C\u0930\u094D\u0926\u0918\u093E\u091F \u0938\u0941\u0938\u094D\u0924\u093E \u092A\u0936\u094D\u091A\u093F\u092E"] },
98
99
  "Eastern Rukum": { en: ["Rukum East", "East Rukum", "Rukum (East)"], ne: ["\u0930\u0941\u0915\u0941\u092E \u092A\u0942\u0930\u094D\u0935"] },
99
100
  "Western Rukum": { en: ["Rukum West", "West Rukum", "Rukum (West)"], ne: ["\u0930\u0941\u0915\u0941\u092E \u092A\u0936\u094D\u091A\u093F\u092E"] },
100
101
  Tanahun: { en: ["Tanahu"], ne: ["\u0924\u0928\u0939\u0941"] },
@@ -119,11 +120,13 @@ var slug = (s) => s.toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-|-$/g,
119
120
  function districtTerms() {
120
121
  return DISTRICTS.map((d) => {
121
122
  const extra = ALIASES[d.name] ?? {};
122
- const ne = /* @__PURE__ */ new Set([d.nameNp, ...extra.ne ?? []]);
123
+ const own = d.aliases ?? [];
124
+ const isNe = (x) => /[\u0900-\u097f]/.test(x);
125
+ const ne = /* @__PURE__ */ new Set([d.nameNp, ...extra.ne ?? [], ...own.filter(isNe)]);
123
126
  for (const n of [...ne]) if (n.endsWith("\u0919")) ne.add(n.slice(0, -1) + "\u0902\u0917");
124
127
  return {
125
128
  id: slug(d.name),
126
- en: [d.name, ...extra.en ?? []],
129
+ en: [.../* @__PURE__ */ new Set([d.name, ...extra.en ?? [], ...own.filter((x) => !isNe(x))])],
127
130
  ne: [...ne],
128
131
  province: d.province,
129
132
  ...AMBIGUOUS.has(d.name) ? { ambiguous: true } : {},
@@ -133,7 +136,7 @@ function districtTerms() {
133
136
  }
134
137
 
135
138
  // src/index.ts
136
- var VERSION = "1.0.0";
139
+ var VERSION = "1.1.0";
137
140
  var NUKTA = 2364;
138
141
  var HALANT = 2381;
139
142
  var NASALS = /* @__PURE__ */ new Set([2329, 2334, 2339, 2344, 2350]);
@@ -160,6 +163,8 @@ var LOOSE = {
160
163
  2339: 2344
161
164
  // ण → न
162
165
  };
166
+ var isSpace = (c) => c === 32 || c >= 9 && c <= 13 || c === 160 || c === 5760 || c >= 8192 && c <= 8202 || c === 8232 || c === 8233 || c === 8239 || c === 8287 || c === 12288;
167
+ var isHyphen = (c) => c === 45 || c === 8208 || c === 8209;
163
168
  var isConsonant = (c) => c >= 2325 && c <= 2361 || c >= 2392 && c <= 2399;
164
169
  function mapNormalise(input, opts = {}) {
165
170
  const digits = opts.digits !== false;
@@ -179,7 +184,7 @@ function mapNormalise(input, opts = {}) {
179
184
  for (let i = 0; i < s.length; i++) {
180
185
  let c = s.charCodeAt(i);
181
186
  if (DROP.has(c)) continue;
182
- if (/\s/.test(s[i])) {
187
+ if (isSpace(c) || opts.hyphenAsSpace && isHyphen(c)) {
183
188
  if (!lastSpace) push(" ", i);
184
189
  lastSpace = true;
185
190
  continue;
@@ -253,7 +258,12 @@ var POSTPOSITIONS = [
253
258
  "\u0935\u093E\u0938\u0940",
254
259
  "\u092C\u093E\u0938\u0940",
255
260
  "\u0938\u094D\u0925\u093F\u0924",
256
- "\u0928\u093F\u0935\u093E\u0938\u0940"
261
+ "\u0928\u093F\u0935\u093E\u0938\u0940",
262
+ "\u092E\u093E\u0930\u094D\u092B\u0924",
263
+ "\u092C\u093E\u0930\u0947",
264
+ "\u0905\u0928\u094D\u0924\u0930\u094D\u0917\u0924",
265
+ "\u092E\u093E\u0924\u094D\u0930",
266
+ "\u091C\u0938\u094D\u0924\u093E"
257
267
  ];
258
268
  var isNeLetter = (c) => c >= 2304 && c <= 2403 || c >= 2417 && c <= 2431;
259
269
  var isWordChar = (ch) => !!ch && /[\p{L}\p{M}\p{N}]/u.test(ch);
@@ -263,9 +273,18 @@ function autoCase(term) {
263
273
  const letters = term.replace(/[^\p{L}]/gu, "");
264
274
  return letters.length >= 2 && letters.length <= 5 && letters === letters.toUpperCase() && letters !== letters.toLowerCase();
265
275
  }
276
+ var optsKey = (o) => `${o.loose ? 1 : 0}${o.digits === false ? 0 : 1}${o.hyphenAsSpace ? 1 : 0}`;
277
+ function prepare(text, opts = {}) {
278
+ const nfc = text.normalize("NFC");
279
+ const { text: n, map } = mapNormalise(nfc, opts);
280
+ return { kind: "prepared", original: nfc, text: n, map, key: optsKey(opts) };
281
+ }
282
+ var isPrepared = (x) => typeof x === "object" && x !== null && x.kind === "prepared";
266
283
  function createMatcher(terms, opts = {}) {
267
- const nopts = { loose: opts.loose, digits: opts.digits };
268
- const suffixes = opts.postpositions === false ? [] : [...new Set([...POSTPOSITIONS, ...opts.extraSuffixes ?? []].map((s) => normaliseNe(s, nopts)))].sort((a, b) => b.length - a.length);
284
+ const nopts = { loose: opts.loose, digits: opts.digits, hyphenAsSpace: opts.hyphenAsSpace };
285
+ const key = optsKey(nopts);
286
+ const allCaps = opts.allCaps !== false;
287
+ const suffixes = [...new Set([...opts.postpositions === false ? [] : POSTPOSITIONS, ...opts.extraSuffixes ?? []].map((s) => normaliseNe(s, nopts)))].filter(Boolean).sort((a, b) => b.length - a.length);
269
288
  const variants = [];
270
289
  for (const t of terms) {
271
290
  const obj = typeof t === "string" ? { id: t, aliases: [t] } : t;
@@ -274,12 +293,26 @@ function createMatcher(terms, opts = {}) {
274
293
  const id = obj.id ?? spellings[0] ?? "";
275
294
  for (const sp of new Set(spellings)) {
276
295
  if (hasDevanagari(sp)) {
277
- variants.push({ id, term: sp, lang: "ne", key: normaliseNe(sp, nopts) });
296
+ const k = normaliseNe(sp, nopts);
297
+ variants.push({ id, term: sp, lang: "ne", key: k });
298
+ if (opts.joinNepali && k.includes(" ")) {
299
+ const parts = k.split(" ");
300
+ const gaps = Math.min(parts.length - 1, 4);
301
+ for (let mask = 1; mask < 1 << gaps; mask++) {
302
+ let joined = parts[0];
303
+ for (let g = 1; g < parts.length; g++) joined += (g <= gaps && mask & 1 << g - 1 ? "" : " ") + parts[g];
304
+ variants.push({ id, term: sp, lang: "ne", key: joined });
305
+ }
306
+ }
278
307
  } else {
279
308
  const rule = obj.caseSensitive ?? opts.caseSensitive ?? "auto";
280
309
  const exact = rule === "auto" ? autoCase(sp) : rule;
281
- const body = sp.split(/\s+/).map(escapeRe).join("\\s+");
282
- variants.push({ id, term: sp, lang: "en", key: sp, re: new RegExp(body, exact ? "gu" : "giu") });
310
+ const words = normaliseNe(sp, nopts).split(" ").filter(Boolean);
311
+ const body = words.map(escapeRe).join(" ");
312
+ const letters = sp.replace(/[^\p{L}]/gu, "");
313
+ const shout = exact && allCaps && letters.length > 5 && sp !== sp.toUpperCase();
314
+ const alt = shout ? `|${words.map((w) => escapeRe(w.toUpperCase())).join(" ")}` : "";
315
+ variants.push({ id, term: sp, lang: "en", key: sp, re: new RegExp(`(?:${body}${alt})`, exact ? "gu" : "giu") });
283
316
  }
284
317
  }
285
318
  }
@@ -294,10 +327,11 @@ function createMatcher(terms, opts = {}) {
294
327
  }
295
328
  return -1;
296
329
  };
297
- const find = (text) => {
298
- const { text: n, map } = mapNormalise(text, nopts);
299
- const toOrig = (i) => i >= map.length ? text.normalize("NFC").length : map[i];
300
- const nfc = text.normalize("NFC");
330
+ const prep = (text) => prepare(text, nopts);
331
+ const find = (input) => {
332
+ const p = isPrepared(input) ? input.key === key ? input : prep(input.original) : prep(input);
333
+ const { text: n, map, original: nfc } = p;
334
+ const toOrig = (i) => i >= map.length ? nfc.length : map[i];
301
335
  const found = [];
302
336
  for (const v of variants) {
303
337
  if (v.lang === "ne") {
@@ -345,9 +379,11 @@ function createMatcher(terms, opts = {}) {
345
379
  find,
346
380
  test: (text, id) => find(text).some((m) => id === void 0 || m.id === id),
347
381
  ids: (text) => [...new Set(find(text).map((m) => m.id))],
382
+ prepare: prep,
348
383
  get size() {
349
384
  return variants.length;
350
- }
385
+ },
386
+ kind: "matcher"
351
387
  };
352
388
  }
353
389
  function findTerms(text, terms, opts) {
@@ -370,6 +406,8 @@ function splitSuffix(word) {
370
406
  }
371
407
  return { stem: nfc.slice(0, end >= n.length ? nfc.length : map[end]), suffixes: out };
372
408
  }
409
+ var isMatcher = (x) => typeof x === "object" && x !== null && x.kind === "matcher";
410
+ var isMatchList = (x) => Array.isArray(x) && (x.length === 0 || typeof x[0] === "object" && x[0] !== null && typeof x[0].index === "number" && typeof x[0].end === "number");
373
411
  var dottedAbbr = (s, i) => {
374
412
  let j = i - 1;
375
413
  while (j >= 0 && !/\s/.test(s[j])) j--;
@@ -395,10 +433,16 @@ function near(text, a, b, maxCharsOrOpts = {}, sameSentence) {
395
433
  if (sameSentence !== void 0) opts.sameSentence = sameSentence;
396
434
  const max = opts.maxChars ?? 60;
397
435
  const same = opts.sameSentence !== false;
398
- const am = findTerms(text, a, opts);
399
- const bm = findTerms(text, b, opts);
400
- if (!am.length || !bm.length) return null;
401
- const spans = same ? sentenceSpans(text) : [];
436
+ const resolve = (src) => {
437
+ if (isMatcher(src)) return src.find(text);
438
+ if (isMatchList(src)) return src;
439
+ return findTerms(text, src, opts);
440
+ };
441
+ const am = resolve(a);
442
+ if (!am.length) return null;
443
+ const bm = resolve(b);
444
+ if (!bm.length) return null;
445
+ const spans = same ? sentenceSpans(isPrepared(text) ? text.original : text) : [];
402
446
  const sentenceOf = (i) => spans.findIndex(([s, e]) => i >= s && i < e);
403
447
  let best = null;
404
448
  for (const x of am) {
@@ -422,6 +466,7 @@ exports.findTerms = findTerms;
422
466
  exports.near = near;
423
467
  exports.normaliseNe = normaliseNe;
424
468
  exports.normalizeNe = normalizeNe;
469
+ exports.prepare = prepare;
425
470
  exports.sentenceSpans = sentenceSpans;
426
471
  exports.splitSuffix = splitSuffix;
427
472
  //# sourceMappingURL=index.cjs.map