@lacspace/nepali-match 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +51 -0
- package/README.md +78 -0
- package/dist/index.cjs +428 -0
- package/dist/index.cjs.map +1 -0
- package/dist/index.d.cts +122 -0
- package/dist/index.d.ts +122 -0
- package/dist/index.js +416 -0
- package/dist/index.js.map +1 -0
- package/package.json +66 -0
package/LICENSE
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
Lacspace Free Licence
|
|
2
|
+
Version 1.0, August 2026
|
|
3
|
+
|
|
4
|
+
Copyright (c) 2026 Lacspace
|
|
5
|
+
|
|
6
|
+
PREAMBLE
|
|
7
|
+
|
|
8
|
+
This software is published by Lacspace under the Lacspace Free Licence — a free,
|
|
9
|
+
permissive licence that lets you use this software for any purpose, including in
|
|
10
|
+
commercial products and services, at no cost. It grants the same freedoms as
|
|
11
|
+
common permissive open-source licences; the only condition is that this notice
|
|
12
|
+
travels with the software. The canonical, always-current text of this licence is
|
|
13
|
+
maintained at https://lacspace.com/licenses/lacspace-free-1.0
|
|
14
|
+
|
|
15
|
+
GRANT OF RIGHTS
|
|
16
|
+
|
|
17
|
+
Permission is hereby granted, free of charge, to any person or organisation
|
|
18
|
+
obtaining a copy of this software and its associated documentation and data files
|
|
19
|
+
(the "Software"), to deal in the Software without restriction, including without
|
|
20
|
+
limitation the rights to use, copy, modify, merge, publish, distribute,
|
|
21
|
+
sublicense, and/or sell copies of the Software, and to permit persons to whom the
|
|
22
|
+
Software is furnished to do so, subject to the conditions below. These rights are
|
|
23
|
+
granted for any purpose, personal or commercial, and are perpetual, worldwide,
|
|
24
|
+
non-exclusive, and royalty-free.
|
|
25
|
+
|
|
26
|
+
CONDITIONS
|
|
27
|
+
|
|
28
|
+
The above copyright notice, this permission notice, and the name of this licence
|
|
29
|
+
("Lacspace Free Licence") shall be included in all copies or substantial portions
|
|
30
|
+
of the Software.
|
|
31
|
+
|
|
32
|
+
TRADEMARKS
|
|
33
|
+
|
|
34
|
+
This licence does not grant permission to use the trade names, trademarks, service
|
|
35
|
+
marks, logos, or product names of Lacspace, except as required to reproduce the
|
|
36
|
+
notice above or to describe the origin of the Software in a truthful manner.
|
|
37
|
+
|
|
38
|
+
DISCLAIMER OF WARRANTY AND LIMITATION OF LIABILITY
|
|
39
|
+
|
|
40
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
41
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
|
42
|
+
FOR A PARTICULAR PURPOSE, AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
|
43
|
+
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES, OR OTHER LIABILITY, WHETHER IN
|
|
44
|
+
AN ACTION OF CONTRACT, TORT, OR OTHERWISE, ARISING FROM, OUT OF, OR IN CONNECTION
|
|
45
|
+
WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
|
46
|
+
|
|
47
|
+
---
|
|
48
|
+
|
|
49
|
+
The Lacspace Free Licence is a source-available, permissive licence and is not (as
|
|
50
|
+
of this version) an OSI-approved licence. In substance it grants the same freedoms
|
|
51
|
+
as the MIT Licence. Learn more at https://lacspace.com/licenses
|
package/README.md
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
# @lacspace/nepali-match
|
|
2
|
+
|
|
3
|
+
Find names and keywords in Nepali and English text the way a Nepali reader would:
|
|
4
|
+
- **Spelling variants match:** काठमाडौँ, काठमाडौं and काठमाण्डौ, or चन्द्र and चंद्र.
|
|
5
|
+
- **Words with a postposition match:** झापाको is Jhapa and चितवनमा is Chitwan.
|
|
6
|
+
- **Different words don't match:** पर्वतारोही (climber) is not Parbat.
|
|
7
|
+
- **Short English acronyms are exact-case:** "SEE results" counts, "Come and see" doesn't.
|
|
8
|
+
|
|
9
|
+
It is pure JS with no dependencies, and it is safe for React Native.
|
|
10
|
+
|
|
11
|
+
```ts
|
|
12
|
+
import { createMatcher, districtTerms, near, normaliseNe } from "@lacspace/nepali-match";
|
|
13
|
+
|
|
14
|
+
const areas = createMatcher(districtTerms());
|
|
15
|
+
areas.ids("चितवनमा बाढी, झापाको मेचीनगरमा पहिरो"); // ["chitwan", "jhapa"]
|
|
16
|
+
areas.test("दुई पर्वतारोही बेपत्ता"); // false
|
|
17
|
+
areas.find("काठमाण्डौबाटै आएका")[0];
|
|
18
|
+
// { id: "kathmandu", term: "काठमाण्डौ", lang: "ne", index: 0, end: 13, text: "काठमाण्डौबाटै", suffix: "बाटै" }
|
|
19
|
+
|
|
20
|
+
const exam = ["परीक्षा", "नतिजा", "विज्ञापन", "exam", "result"];
|
|
21
|
+
near("लोकसेवा आयोगले निजामती विधेयकमा राय दियो", "लोकसेवा", exam, 60); // null: not exam news
|
|
22
|
+
near("लोकसेवा आयोगको खरिदार परीक्षाको नतिजा", "लोकसेवा", exam, 60); // { a, b, gap }
|
|
23
|
+
|
|
24
|
+
createMatcher([{ id: "see", en: "SEE", ne: "एसईई" }]).test("Come and see"); // false
|
|
25
|
+
normaliseNe("काठमाडौँ") === normaliseNe("काठमाडौं"); // true
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
## How matching works
|
|
29
|
+
|
|
30
|
+
**`normaliseNe(text, { loose?, digits? })`** applies these steps in order:
|
|
31
|
+
1. NFC.
|
|
32
|
+
2. Chandrabindu → anusvara.
|
|
33
|
+
3. A half nasal before a consonant (ङ्, ञ्, ण्, न्, म्) → anusvara.
|
|
34
|
+
4. ई/ी → इ/ि and ऊ/ू → उ/ु.
|
|
35
|
+
5. Nukta, ZWJ, ZWNJ and soft hyphens are removed.
|
|
36
|
+
6. Devanagari digits → ASCII.
|
|
37
|
+
7. Whitespace collapses.
|
|
38
|
+
|
|
39
|
+
`loose: true` also folds श/ष → स, व → ब and ण → न. Latin text is left as it is.
|
|
40
|
+
|
|
41
|
+
**Nepali terms** match as whole words:
|
|
42
|
+
- Nothing may stand before the word except a space or punctuation.
|
|
43
|
+
- After the word, a chain of up to three `POSTPOSITIONS` may follow: मा, को, का, की, ले, लाई, बाट, सँग, देखि, सम्म, तिर, भित्र, हरू, मै, बाटै, नै … Anything else after it means it's a different word.
|
|
44
|
+
- Turn this off with `postpositions: false`. Add your own with `extraSuffixes`.
|
|
45
|
+
|
|
46
|
+
**English terms** match as whole words, ignoring case. Exceptions:
|
|
47
|
+
- Under the default `caseSensitive: "auto"` rule, an all-caps term of 2–5 letters (SEE, NEB, PSC) must match its case exactly.
|
|
48
|
+
- Set `caseSensitive` per term or per matcher to change this.
|
|
49
|
+
- An all-caps headline ("COME AND SEE") still matches SEE. Check the case of the surrounding text if your input has shouty headlines.
|
|
50
|
+
|
|
51
|
+
**Overlaps:** the longest match wins, so "Nawalparasi West" beats "Nawalparasi". Pass `overlaps: true` to keep both.
|
|
52
|
+
|
|
53
|
+
Every match carries `index`/`end` offsets into the original text (after NFC), so you can highlight or cut it.
|
|
54
|
+
|
|
55
|
+
## API
|
|
56
|
+
|
|
57
|
+
- **`createMatcher(terms, options?)`** returns `{ find(text), test(text, id?), ids(text), size }`. Compile it once and reuse it.
|
|
58
|
+
- A term is either a string, or `{ id?, en?, ne?, aliases?, caseSensitive? }`. `en` and `ne` each take one spelling or an array.
|
|
59
|
+
- **`findTerms(text, terms, options?)`** and **`contains(text, terms, options?)`** are one-off shortcuts.
|
|
60
|
+
- **`near(text, a, b, maxChars | options, sameSentence?)`** returns the closest `{ a, b, gap }` pair of matches within `maxChars` (default 60), in either order, or null.
|
|
61
|
+
- By default both must sit in the same sentence. A sentence ends at । ॥ ? ! or a newline, or at "." before a space.
|
|
62
|
+
- Dotted abbreviations such as ने.क.पा. and U.S. never end a sentence.
|
|
63
|
+
- **`sentenceSpans(text)`** returns the sentence boundaries `near` uses.
|
|
64
|
+
- **`splitSuffix(word)`**: for example, `"जिल्लाहरूमा"` → `{ stem: "जिल्ला", suffixes: ["हरु", "मा"] }`.
|
|
65
|
+
- **`districtTerms()`** returns all 77 districts.
|
|
66
|
+
- **Names and ids:** names come from `@lacspace/nepali-utils`. Ids are slugs such as `"nawalparasi-east"`, and `province` is a number.
|
|
67
|
+
- **Spellings:** each district carries the spellings seen in real copy: Kavre/काभ्रे, Rukum East/रुकुम पूर्व, Kapilbastu, मोरंग/मोरङ…
|
|
68
|
+
- **Case:** English district names are exact-case, so "dang it" is not Dang.
|
|
69
|
+
- **`ambiguous: true`:** marks Parbat, because पर्वत also means "mountain". Confirm it with `near()` or with district/area context before you tag a story.
|
|
70
|
+
|
|
71
|
+
## Limits
|
|
72
|
+
|
|
73
|
+
- The matcher is rule-based and does no stemming beyond postpositions. Verb forms and compounds won't match, which is the point.
|
|
74
|
+
- Spellings that differ in more than the folded letters (for example गोर्खा and गोरखा) need an alias.
|
|
75
|
+
|
|
76
|
+
---
|
|
77
|
+
|
|
78
|
+
Built by [Lacspace](https://lacspace.com). Licensed under the Lacspace Free Licence v1.0 (see LICENSE).
|
package/dist/index.cjs
ADDED
|
@@ -0,0 +1,428 @@
|
|
|
1
|
+
'use strict';
|
|
2
|
+
|
|
3
|
+
// ../nepali-utils/dist/index.js
|
|
4
|
+
var DISTRICTS = [
|
|
5
|
+
// Koshi (1)
|
|
6
|
+
{ name: "Bhojpur", nameNp: "\u092D\u094B\u091C\u092A\u0941\u0930", province: 1 },
|
|
7
|
+
{ name: "Dhankuta", nameNp: "\u0927\u0928\u0915\u0941\u091F\u093E", province: 1 },
|
|
8
|
+
{ name: "Ilam", nameNp: "\u0907\u0932\u093E\u092E", province: 1 },
|
|
9
|
+
{ name: "Jhapa", nameNp: "\u091D\u093E\u092A\u093E", province: 1 },
|
|
10
|
+
{ name: "Khotang", nameNp: "\u0916\u094B\u091F\u093E\u0919", province: 1 },
|
|
11
|
+
{ name: "Morang", nameNp: "\u092E\u094B\u0930\u0919", province: 1 },
|
|
12
|
+
{ name: "Okhaldhunga", nameNp: "\u0913\u0916\u0932\u0922\u0941\u0902\u0917\u093E", province: 1 },
|
|
13
|
+
{ name: "Panchthar", nameNp: "\u092A\u093E\u0901\u091A\u0925\u0930", province: 1 },
|
|
14
|
+
{ name: "Sankhuwasabha", nameNp: "\u0938\u0902\u0916\u0941\u0935\u093E\u0938\u092D\u093E", province: 1 },
|
|
15
|
+
{ name: "Solukhumbu", nameNp: "\u0938\u094B\u0932\u0941\u0916\u0941\u092E\u094D\u092C\u0941", province: 1 },
|
|
16
|
+
{ name: "Sunsari", nameNp: "\u0938\u0941\u0928\u0938\u0930\u0940", province: 1 },
|
|
17
|
+
{ name: "Taplejung", nameNp: "\u0924\u093E\u092A\u094D\u0932\u0947\u091C\u0941\u0919", province: 1 },
|
|
18
|
+
{ name: "Terhathum", nameNp: "\u0924\u0947\u0939\u094D\u0930\u0925\u0941\u092E", province: 1 },
|
|
19
|
+
{ name: "Udayapur", nameNp: "\u0909\u0926\u092F\u092A\u0941\u0930", province: 1 },
|
|
20
|
+
// Madhesh (2)
|
|
21
|
+
{ name: "Bara", nameNp: "\u092C\u093E\u0930\u093E", province: 2 },
|
|
22
|
+
{ name: "Dhanusha", nameNp: "\u0927\u0928\u0941\u0937\u093E", province: 2 },
|
|
23
|
+
{ name: "Mahottari", nameNp: "\u092E\u0939\u094B\u0924\u094D\u0924\u0930\u0940", province: 2 },
|
|
24
|
+
{ name: "Parsa", nameNp: "\u092A\u0930\u094D\u0938\u093E", province: 2 },
|
|
25
|
+
{ name: "Rautahat", nameNp: "\u0930\u094C\u0924\u0939\u091F", province: 2 },
|
|
26
|
+
{ name: "Saptari", nameNp: "\u0938\u092A\u094D\u0924\u0930\u0940", province: 2 },
|
|
27
|
+
{ name: "Sarlahi", nameNp: "\u0938\u0930\u094D\u0932\u093E\u0939\u0940", province: 2 },
|
|
28
|
+
{ name: "Siraha", nameNp: "\u0938\u093F\u0930\u0939\u093E", province: 2 },
|
|
29
|
+
// Bagmati (3)
|
|
30
|
+
{ name: "Bhaktapur", nameNp: "\u092D\u0915\u094D\u0924\u092A\u0941\u0930", province: 3 },
|
|
31
|
+
{ name: "Chitwan", nameNp: "\u091A\u093F\u0924\u0935\u0928", province: 3 },
|
|
32
|
+
{ name: "Dhading", nameNp: "\u0927\u093E\u0926\u093F\u0919", province: 3 },
|
|
33
|
+
{ name: "Dolakha", nameNp: "\u0926\u094B\u0932\u0916\u093E", province: 3 },
|
|
34
|
+
{ name: "Kathmandu", nameNp: "\u0915\u093E\u0920\u092E\u093E\u0921\u094C\u0902", province: 3 },
|
|
35
|
+
{ name: "Kavrepalanchok", nameNp: "\u0915\u093E\u092D\u094D\u0930\u0947\u092A\u0932\u093E\u091E\u094D\u091A\u094B\u0915", province: 3 },
|
|
36
|
+
{ name: "Lalitpur", nameNp: "\u0932\u0932\u093F\u0924\u092A\u0941\u0930", province: 3 },
|
|
37
|
+
{ name: "Makwanpur", nameNp: "\u092E\u0915\u0935\u093E\u0928\u092A\u0941\u0930", province: 3 },
|
|
38
|
+
{ name: "Nuwakot", nameNp: "\u0928\u0941\u0935\u093E\u0915\u094B\u091F", province: 3 },
|
|
39
|
+
{ name: "Ramechhap", nameNp: "\u0930\u093E\u092E\u0947\u091B\u093E\u092A", province: 3 },
|
|
40
|
+
{ name: "Rasuwa", nameNp: "\u0930\u0938\u0941\u0935\u093E", province: 3 },
|
|
41
|
+
{ name: "Sindhuli", nameNp: "\u0938\u093F\u0928\u094D\u0927\u0941\u0932\u0940", province: 3 },
|
|
42
|
+
{ name: "Sindhupalchok", nameNp: "\u0938\u093F\u0928\u094D\u0927\u0941\u092A\u093E\u0932\u094D\u091A\u094B\u0915", province: 3 },
|
|
43
|
+
// Gandaki (4)
|
|
44
|
+
{ name: "Baglung", nameNp: "\u092C\u093E\u0917\u0932\u0941\u0919", province: 4 },
|
|
45
|
+
{ name: "Gorkha", nameNp: "\u0917\u094B\u0930\u0916\u093E", province: 4 },
|
|
46
|
+
{ name: "Kaski", nameNp: "\u0915\u093E\u0938\u094D\u0915\u0940", province: 4 },
|
|
47
|
+
{ name: "Lamjung", nameNp: "\u0932\u092E\u091C\u0941\u0919", province: 4 },
|
|
48
|
+
{ name: "Manang", nameNp: "\u092E\u0928\u093E\u0919", province: 4 },
|
|
49
|
+
{ name: "Mustang", nameNp: "\u092E\u0941\u0938\u094D\u0924\u093E\u0919", province: 4 },
|
|
50
|
+
{ name: "Myagdi", nameNp: "\u092E\u094D\u092F\u093E\u0917\u094D\u0926\u0940", province: 4 },
|
|
51
|
+
{ name: "Nawalparasi East", nameNp: "\u0928\u0935\u0932\u092A\u0930\u093E\u0938\u0940 (\u092C\u0930\u094D\u0926\u0918\u093E\u091F \u0938\u0941\u0938\u094D\u0924\u093E \u092A\u0942\u0930\u094D\u0935)", province: 4, aliases: ["Nawalpur", "\u0928\u0935\u0932\u092A\u0941\u0930", "Nawalparasi (East of Bardaghat Susta)", "Nawalparasi (Bardaghat Susta East)", "Nawalparasi Purba", "\u0928\u0935\u0932\u092A\u0930\u093E\u0938\u0940 \u092A\u0942\u0930\u094D\u0935"] },
|
|
52
|
+
{ name: "Parbat", nameNp: "\u092A\u0930\u094D\u0935\u0924", province: 4 },
|
|
53
|
+
{ name: "Syangja", nameNp: "\u0938\u094D\u092F\u093E\u0919\u094D\u091C\u093E", province: 4 },
|
|
54
|
+
{ name: "Tanahun", nameNp: "\u0924\u0928\u0939\u0941\u0901", province: 4 },
|
|
55
|
+
// Lumbini (5)
|
|
56
|
+
{ name: "Arghakhanchi", nameNp: "\u0905\u0930\u094D\u0918\u093E\u0916\u093E\u0901\u091A\u0940", province: 5 },
|
|
57
|
+
{ name: "Banke", nameNp: "\u092C\u093E\u0901\u0915\u0947", province: 5 },
|
|
58
|
+
{ name: "Bardiya", nameNp: "\u092C\u0930\u094D\u0926\u093F\u092F\u093E", province: 5 },
|
|
59
|
+
{ name: "Dang", nameNp: "\u0926\u093E\u0919", province: 5 },
|
|
60
|
+
{ name: "Eastern Rukum", nameNp: "\u092A\u0942\u0930\u094D\u0935\u0940 \u0930\u0941\u0915\u0941\u092E", province: 5, aliases: ["Rukum East", "Rukum Purba", "\u0930\u0941\u0915\u0941\u092E \u092A\u0942\u0930\u094D\u0935"] },
|
|
61
|
+
{ name: "Gulmi", nameNp: "\u0917\u0941\u0932\u094D\u092E\u0940", province: 5 },
|
|
62
|
+
{ name: "Kapilvastu", nameNp: "\u0915\u092A\u093F\u0932\u0935\u0938\u094D\u0924\u0941", province: 5 },
|
|
63
|
+
{ name: "Nawalparasi West", nameNp: "\u0928\u0935\u0932\u092A\u0930\u093E\u0938\u0940 (\u092C\u0930\u094D\u0926\u0918\u093E\u091F \u0938\u0941\u0938\u094D\u0924\u093E \u092A\u0936\u094D\u091A\u093F\u092E)", province: 5, aliases: ["Parasi", "\u092A\u0930\u093E\u0938\u0940", "Nawalparasi (West of Bardaghat Susta)", "Nawalparasi (Bardaghat Susta West)", "Nawalparasi Paschim", "\u0928\u0935\u0932\u092A\u0930\u093E\u0938\u0940 \u092A\u0936\u094D\u091A\u093F\u092E"] },
|
|
64
|
+
{ name: "Palpa", nameNp: "\u092A\u093E\u0932\u094D\u092A\u093E", province: 5 },
|
|
65
|
+
{ name: "Pyuthan", nameNp: "\u092A\u094D\u092F\u0941\u0920\u093E\u0928", province: 5 },
|
|
66
|
+
{ name: "Rolpa", nameNp: "\u0930\u094B\u0932\u094D\u092A\u093E", province: 5 },
|
|
67
|
+
{ name: "Rupandehi", nameNp: "\u0930\u0942\u092A\u0928\u094D\u0926\u0947\u0939\u0940", province: 5 },
|
|
68
|
+
// Karnali (6)
|
|
69
|
+
{ name: "Dailekh", nameNp: "\u0926\u0948\u0932\u0947\u0916", province: 6 },
|
|
70
|
+
{ name: "Dolpa", nameNp: "\u0921\u094B\u0932\u094D\u092A\u093E", province: 6 },
|
|
71
|
+
{ name: "Humla", nameNp: "\u0939\u0941\u092E\u094D\u0932\u093E", province: 6 },
|
|
72
|
+
{ name: "Jajarkot", nameNp: "\u091C\u093E\u091C\u0930\u0915\u094B\u091F", province: 6 },
|
|
73
|
+
{ name: "Jumla", nameNp: "\u091C\u0941\u092E\u094D\u0932\u093E", province: 6 },
|
|
74
|
+
{ name: "Kalikot", nameNp: "\u0915\u093E\u0932\u093F\u0915\u094B\u091F", province: 6 },
|
|
75
|
+
{ name: "Mugu", nameNp: "\u092E\u0941\u0917\u0941", province: 6 },
|
|
76
|
+
{ name: "Salyan", nameNp: "\u0938\u0932\u094D\u092F\u093E\u0928", province: 6 },
|
|
77
|
+
{ name: "Surkhet", nameNp: "\u0938\u0941\u0930\u094D\u0916\u0947\u0924", province: 6 },
|
|
78
|
+
{ name: "Western Rukum", nameNp: "\u092A\u0936\u094D\u091A\u093F\u092E\u0940 \u0930\u0941\u0915\u0941\u092E", province: 6, aliases: ["Rukum West", "Rukum Paschim", "\u0930\u0941\u0915\u0941\u092E \u092A\u0936\u094D\u091A\u093F\u092E"] },
|
|
79
|
+
// Sudurpashchim (7)
|
|
80
|
+
{ name: "Achham", nameNp: "\u0905\u091B\u093E\u092E", province: 7 },
|
|
81
|
+
{ name: "Baitadi", nameNp: "\u092C\u0948\u0924\u0921\u0940", province: 7 },
|
|
82
|
+
{ name: "Bajhang", nameNp: "\u092C\u091D\u093E\u0919", province: 7 },
|
|
83
|
+
{ name: "Bajura", nameNp: "\u092C\u093E\u091C\u0941\u0930\u093E", province: 7 },
|
|
84
|
+
{ name: "Dadeldhura", nameNp: "\u0921\u0921\u0947\u0932\u094D\u0927\u0941\u0930\u093E", province: 7 },
|
|
85
|
+
{ name: "Darchula", nameNp: "\u0926\u093E\u0930\u094D\u091A\u0941\u0932\u093E", province: 7 },
|
|
86
|
+
{ name: "Doti", nameNp: "\u0921\u094B\u091F\u0940", province: 7 },
|
|
87
|
+
{ name: "Kailali", nameNp: "\u0915\u0948\u0932\u093E\u0932\u0940", province: 7 },
|
|
88
|
+
{ name: "Kanchanpur", nameNp: "\u0915\u091E\u094D\u091A\u0928\u092A\u0941\u0930", province: 7 }
|
|
89
|
+
];
|
|
90
|
+
|
|
91
|
+
// src/districts.ts
|
|
92
|
+
var ALIASES = {
|
|
93
|
+
Kathmandu: { en: ["Katmandu"], ne: ["\u0915\u093E\u0920\u092E\u093E\u0923\u094D\u0921\u094C", "\u0915\u093E\u0920\u092E\u093E\u0928\u094D\u0921\u0941"] },
|
|
94
|
+
Kavrepalanchok: { en: ["Kavre", "Kabhre", "Kabhrepalanchok", "Kavrepalanchowk"], ne: ["\u0915\u093E\u092D\u094D\u0930\u0947", "\u0915\u093E\u092D\u094D\u0930\u0947\u092A\u0932\u093E\u0928\u094D\u091A\u094B\u0915"] },
|
|
95
|
+
Sindhupalchok: { en: ["Sindhupalchowk"] },
|
|
96
|
+
"Nawalparasi East": { en: ["Nawalparasi (East)", "East Nawalparasi", "Nawalpur"], ne: ["\u0928\u0935\u0932\u092A\u0930\u093E\u0938\u0940 \u092A\u0942\u0930\u094D\u0935", "\u092A\u0942\u0930\u094D\u0935\u0940 \u0928\u0935\u0932\u092A\u0930\u093E\u0938\u0940", "\u0928\u0935\u0932\u092A\u0941\u0930"] },
|
|
97
|
+
"Nawalparasi West": { en: ["Nawalparasi (West)", "West Nawalparasi"], ne: ["\u0928\u0935\u0932\u092A\u0930\u093E\u0938\u0940 \u092A\u0936\u094D\u091A\u093F\u092E", "\u092A\u0936\u094D\u091A\u093F\u092E\u0940 \u0928\u0935\u0932\u092A\u0930\u093E\u0938\u0940"] },
|
|
98
|
+
"Eastern Rukum": { en: ["Rukum East", "East Rukum", "Rukum (East)"], ne: ["\u0930\u0941\u0915\u0941\u092E \u092A\u0942\u0930\u094D\u0935"] },
|
|
99
|
+
"Western Rukum": { en: ["Rukum West", "West Rukum", "Rukum (West)"], ne: ["\u0930\u0941\u0915\u0941\u092E \u092A\u0936\u094D\u091A\u093F\u092E"] },
|
|
100
|
+
Tanahun: { en: ["Tanahu"], ne: ["\u0924\u0928\u0939\u0941"] },
|
|
101
|
+
Dhanusha: { en: ["Dhanusa"] },
|
|
102
|
+
Kapilvastu: { en: ["Kapilbastu"], ne: ["\u0915\u092A\u093F\u0932\u092C\u0938\u094D\u0924\u0941"] },
|
|
103
|
+
Makwanpur: { en: ["Makawanpur"] },
|
|
104
|
+
Sankhuwasabha: { en: ["Sankhuwasava"] },
|
|
105
|
+
Terhathum: { en: ["Tehrathum"], ne: ["\u0924\u0947\u0939\u0925\u0941\u092E"] },
|
|
106
|
+
Okhaldhunga: { ne: ["\u0913\u0916\u0932\u0922\u0941\u0919\u094D\u0917\u093E"] },
|
|
107
|
+
Syangja: { en: ["Syanja"], ne: ["\u0938\u094D\u092F\u093E\u0919\u091C\u093E"] },
|
|
108
|
+
Achham: { en: ["Accham"] },
|
|
109
|
+
Udayapur: { en: ["Udaypur"] },
|
|
110
|
+
Bardiya: { en: ["Bardia"] },
|
|
111
|
+
Ilam: { en: ["Illam"] },
|
|
112
|
+
Baglung: { ne: ["\u092C\u093E\u0917\u094D\u0932\u0941\u0919"] },
|
|
113
|
+
Rupandehi: { ne: ["\u0930\u0941\u092A\u0928\u094D\u0926\u0947\u0939\u0940"] },
|
|
114
|
+
Dadeldhura: { ne: ["\u0921\u0901\u0921\u0947\u0932\u0927\u0941\u0930\u093E"] },
|
|
115
|
+
Mahottari: { ne: ["\u092E\u0939\u094B\u0924\u0930\u0940"] }
|
|
116
|
+
};
|
|
117
|
+
var AMBIGUOUS = /* @__PURE__ */ new Set(["Parbat"]);
|
|
118
|
+
var slug = (s) => s.toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-|-$/g, "");
|
|
119
|
+
function districtTerms() {
|
|
120
|
+
return DISTRICTS.map((d) => {
|
|
121
|
+
const extra = ALIASES[d.name] ?? {};
|
|
122
|
+
const ne = /* @__PURE__ */ new Set([d.nameNp, ...extra.ne ?? []]);
|
|
123
|
+
for (const n of [...ne]) if (n.endsWith("\u0919")) ne.add(n.slice(0, -1) + "\u0902\u0917");
|
|
124
|
+
return {
|
|
125
|
+
id: slug(d.name),
|
|
126
|
+
en: [d.name, ...extra.en ?? []],
|
|
127
|
+
ne: [...ne],
|
|
128
|
+
province: d.province,
|
|
129
|
+
...AMBIGUOUS.has(d.name) ? { ambiguous: true } : {},
|
|
130
|
+
caseSensitive: true
|
|
131
|
+
};
|
|
132
|
+
});
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
// src/index.ts
|
|
136
|
+
var VERSION = "1.0.0";
|
|
137
|
+
var NUKTA = 2364;
|
|
138
|
+
var HALANT = 2381;
|
|
139
|
+
var NASALS = /* @__PURE__ */ new Set([2329, 2334, 2339, 2344, 2350]);
|
|
140
|
+
var DROP = /* @__PURE__ */ new Set([NUKTA, 8204, 8205, 173, 65279]);
|
|
141
|
+
var FOLD = {
|
|
142
|
+
2305: 2306,
|
|
143
|
+
// ँ → ं
|
|
144
|
+
2312: 2311,
|
|
145
|
+
// ई → इ
|
|
146
|
+
2314: 2313,
|
|
147
|
+
// ऊ → उ
|
|
148
|
+
2368: 2367,
|
|
149
|
+
// ी → ि
|
|
150
|
+
2370: 2369
|
|
151
|
+
// ू → ु
|
|
152
|
+
};
|
|
153
|
+
var LOOSE = {
|
|
154
|
+
2358: 2360,
|
|
155
|
+
// श → स
|
|
156
|
+
2359: 2360,
|
|
157
|
+
// ष → स
|
|
158
|
+
2357: 2348,
|
|
159
|
+
// व → ब
|
|
160
|
+
2339: 2344
|
|
161
|
+
// ण → न
|
|
162
|
+
};
|
|
163
|
+
var isConsonant = (c) => c >= 2325 && c <= 2361 || c >= 2392 && c <= 2399;
|
|
164
|
+
function mapNormalise(input, opts = {}) {
|
|
165
|
+
const digits = opts.digits !== false;
|
|
166
|
+
const s = input.normalize("NFC");
|
|
167
|
+
const out = [];
|
|
168
|
+
const map = [];
|
|
169
|
+
const push = (ch, i) => {
|
|
170
|
+
out.push(ch);
|
|
171
|
+
map.push(i);
|
|
172
|
+
};
|
|
173
|
+
const next = (i) => {
|
|
174
|
+
let j = i + 1;
|
|
175
|
+
while (j < s.length && DROP.has(s.charCodeAt(j))) j++;
|
|
176
|
+
return j;
|
|
177
|
+
};
|
|
178
|
+
let lastSpace = false;
|
|
179
|
+
for (let i = 0; i < s.length; i++) {
|
|
180
|
+
let c = s.charCodeAt(i);
|
|
181
|
+
if (DROP.has(c)) continue;
|
|
182
|
+
if (/\s/.test(s[i])) {
|
|
183
|
+
if (!lastSpace) push(" ", i);
|
|
184
|
+
lastSpace = true;
|
|
185
|
+
continue;
|
|
186
|
+
}
|
|
187
|
+
lastSpace = false;
|
|
188
|
+
if (NASALS.has(c)) {
|
|
189
|
+
const h = next(i);
|
|
190
|
+
if (s.charCodeAt(h) === HALANT) {
|
|
191
|
+
const k = next(h);
|
|
192
|
+
if (k < s.length && isConsonant(s.charCodeAt(k))) {
|
|
193
|
+
push("\u0902", i);
|
|
194
|
+
i = h;
|
|
195
|
+
continue;
|
|
196
|
+
}
|
|
197
|
+
}
|
|
198
|
+
}
|
|
199
|
+
if (FOLD[c] !== void 0) c = FOLD[c];
|
|
200
|
+
if (opts.loose && LOOSE[c] !== void 0) c = LOOSE[c];
|
|
201
|
+
if (digits && c >= 2406 && c <= 2415) c = 48 + (c - 2406);
|
|
202
|
+
if (c >= 55296 && c <= 56319 && i + 1 < s.length) {
|
|
203
|
+
push(s[i], i);
|
|
204
|
+
push(s[i + 1], i + 1);
|
|
205
|
+
i++;
|
|
206
|
+
continue;
|
|
207
|
+
}
|
|
208
|
+
push(String.fromCharCode(c), i);
|
|
209
|
+
}
|
|
210
|
+
return { text: out.join(""), map };
|
|
211
|
+
}
|
|
212
|
+
function normaliseNe(text, opts = {}) {
|
|
213
|
+
return mapNormalise(text, opts).text;
|
|
214
|
+
}
|
|
215
|
+
var normalizeNe = normaliseNe;
|
|
216
|
+
var POSTPOSITIONS = [
|
|
217
|
+
"\u092E\u093E",
|
|
218
|
+
"\u092E\u0948",
|
|
219
|
+
"\u0915\u094B",
|
|
220
|
+
"\u0915\u093E",
|
|
221
|
+
"\u0915\u0940",
|
|
222
|
+
"\u0915\u0947",
|
|
223
|
+
"\u0915\u0948",
|
|
224
|
+
"\u0932\u0947",
|
|
225
|
+
"\u0932\u093E\u0908",
|
|
226
|
+
"\u092C\u093E\u091F",
|
|
227
|
+
"\u092C\u093E\u091F\u0948",
|
|
228
|
+
"\u0938\u0901\u0917",
|
|
229
|
+
"\u0938\u0902\u0917",
|
|
230
|
+
"\u0938\u093F\u0924",
|
|
231
|
+
"\u0926\u0947\u0916\u093F",
|
|
232
|
+
"\u0926\u0947\u0916\u093F\u0928\u0948",
|
|
233
|
+
"\u0938\u092E\u094D\u092E",
|
|
234
|
+
"\u0938\u092E\u094D\u092E\u0948",
|
|
235
|
+
"\u0924\u093F\u0930",
|
|
236
|
+
"\u0924\u0930\u094D\u092B",
|
|
237
|
+
"\u092D\u093F\u0924\u094D\u0930",
|
|
238
|
+
"\u092C\u093E\u0939\u093F\u0930",
|
|
239
|
+
"\u092E\u093E\u0925\u093F",
|
|
240
|
+
"\u092E\u0941\u0928\u093F",
|
|
241
|
+
"\u0928\u091C\u093F\u0915",
|
|
242
|
+
"\u092A\u091B\u093F",
|
|
243
|
+
"\u0905\u0918\u093F",
|
|
244
|
+
"\u0905\u0917\u093E\u0921\u093F",
|
|
245
|
+
"\u092A\u093E\u0930\u093F",
|
|
246
|
+
"\u0935\u093E\u0930\u093F",
|
|
247
|
+
"\u0926\u094D\u0935\u093E\u0930\u093E",
|
|
248
|
+
"\u0939\u0930\u0942",
|
|
249
|
+
"\u0939\u0930\u0941",
|
|
250
|
+
"\u0928\u0948",
|
|
251
|
+
"\u092D\u0930",
|
|
252
|
+
"\u092D\u0930\u093F",
|
|
253
|
+
"\u0935\u093E\u0938\u0940",
|
|
254
|
+
"\u092C\u093E\u0938\u0940",
|
|
255
|
+
"\u0938\u094D\u0925\u093F\u0924",
|
|
256
|
+
"\u0928\u093F\u0935\u093E\u0938\u0940"
|
|
257
|
+
];
|
|
258
|
+
var isNeLetter = (c) => c >= 2304 && c <= 2403 || c >= 2417 && c <= 2431;
|
|
259
|
+
var isWordChar = (ch) => !!ch && /[\p{L}\p{M}\p{N}]/u.test(ch);
|
|
260
|
+
var hasDevanagari = (s) => /[ऀ-ॿ]/.test(s);
|
|
261
|
+
var escapeRe = (s) => s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
262
|
+
function autoCase(term) {
|
|
263
|
+
const letters = term.replace(/[^\p{L}]/gu, "");
|
|
264
|
+
return letters.length >= 2 && letters.length <= 5 && letters === letters.toUpperCase() && letters !== letters.toLowerCase();
|
|
265
|
+
}
|
|
266
|
+
function createMatcher(terms, opts = {}) {
|
|
267
|
+
const nopts = { loose: opts.loose, digits: opts.digits };
|
|
268
|
+
const suffixes = opts.postpositions === false ? [] : [...new Set([...POSTPOSITIONS, ...opts.extraSuffixes ?? []].map((s) => normaliseNe(s, nopts)))].sort((a, b) => b.length - a.length);
|
|
269
|
+
const variants = [];
|
|
270
|
+
for (const t of terms) {
|
|
271
|
+
const obj = typeof t === "string" ? { id: t, aliases: [t] } : t;
|
|
272
|
+
const list = (v) => v === void 0 ? [] : typeof v === "string" ? [v] : [...v];
|
|
273
|
+
const spellings = [...list(obj.en), ...list(obj.ne), ...obj.aliases ?? []].map((s) => s.trim()).filter(Boolean);
|
|
274
|
+
const id = obj.id ?? spellings[0] ?? "";
|
|
275
|
+
for (const sp of new Set(spellings)) {
|
|
276
|
+
if (hasDevanagari(sp)) {
|
|
277
|
+
variants.push({ id, term: sp, lang: "ne", key: normaliseNe(sp, nopts) });
|
|
278
|
+
} else {
|
|
279
|
+
const rule = obj.caseSensitive ?? opts.caseSensitive ?? "auto";
|
|
280
|
+
const exact = rule === "auto" ? autoCase(sp) : rule;
|
|
281
|
+
const body = sp.split(/\s+/).map(escapeRe).join("\\s+");
|
|
282
|
+
variants.push({ id, term: sp, lang: "en", key: sp, re: new RegExp(body, exact ? "gu" : "giu") });
|
|
283
|
+
}
|
|
284
|
+
}
|
|
285
|
+
}
|
|
286
|
+
const suffixEnd = (s, pos, depth) => {
|
|
287
|
+
if (pos >= s.length || !isNeLetter(s.charCodeAt(pos))) return pos;
|
|
288
|
+
if (depth >= 3) return -1;
|
|
289
|
+
for (const suf of suffixes) {
|
|
290
|
+
if (s.startsWith(suf, pos)) {
|
|
291
|
+
const e = suffixEnd(s, pos + suf.length, depth + 1);
|
|
292
|
+
if (e >= 0) return e;
|
|
293
|
+
}
|
|
294
|
+
}
|
|
295
|
+
return -1;
|
|
296
|
+
};
|
|
297
|
+
const find = (text) => {
|
|
298
|
+
const { text: n, map } = mapNormalise(text, nopts);
|
|
299
|
+
const toOrig = (i) => i >= map.length ? text.normalize("NFC").length : map[i];
|
|
300
|
+
const nfc = text.normalize("NFC");
|
|
301
|
+
const found = [];
|
|
302
|
+
for (const v of variants) {
|
|
303
|
+
if (v.lang === "ne") {
|
|
304
|
+
if (!v.key) continue;
|
|
305
|
+
let at = n.indexOf(v.key);
|
|
306
|
+
while (at >= 0) {
|
|
307
|
+
const before = n[at - 1];
|
|
308
|
+
if (!isWordChar(before)) {
|
|
309
|
+
const wordEnd = at + v.key.length;
|
|
310
|
+
const end = suffixEnd(n, wordEnd, 0);
|
|
311
|
+
const after = n[end];
|
|
312
|
+
if (end >= 0 && !isWordChar(after)) {
|
|
313
|
+
const s = toOrig(at);
|
|
314
|
+
const e = end >= n.length ? nfc.length : toOrig(end);
|
|
315
|
+
found.push({ id: v.id, term: v.term, lang: "ne", index: s, end: e, text: nfc.slice(s, e), ...end > wordEnd ? { suffix: n.slice(wordEnd, end) } : {} });
|
|
316
|
+
}
|
|
317
|
+
}
|
|
318
|
+
at = n.indexOf(v.key, at + 1);
|
|
319
|
+
}
|
|
320
|
+
} else {
|
|
321
|
+
v.re.lastIndex = 0;
|
|
322
|
+
for (const m of n.matchAll(v.re)) {
|
|
323
|
+
const at = m.index;
|
|
324
|
+
const end = at + m[0].length;
|
|
325
|
+
if (isWordChar(n[at - 1]) || isWordChar(n[end])) continue;
|
|
326
|
+
const s = toOrig(at);
|
|
327
|
+
const e = end >= n.length ? nfc.length : toOrig(end);
|
|
328
|
+
found.push({ id: v.id, term: v.term, lang: "en", index: s, end: e, text: nfc.slice(s, e) });
|
|
329
|
+
}
|
|
330
|
+
}
|
|
331
|
+
}
|
|
332
|
+
found.sort((a, b) => a.index - b.index || b.end - a.end);
|
|
333
|
+
if (opts.overlaps) return found;
|
|
334
|
+
const kept = [];
|
|
335
|
+
let reach = -1;
|
|
336
|
+
for (const m of found) {
|
|
337
|
+
if (m.index >= reach) {
|
|
338
|
+
kept.push(m);
|
|
339
|
+
reach = m.end;
|
|
340
|
+
}
|
|
341
|
+
}
|
|
342
|
+
return kept;
|
|
343
|
+
};
|
|
344
|
+
return {
|
|
345
|
+
find,
|
|
346
|
+
test: (text, id) => find(text).some((m) => id === void 0 || m.id === id),
|
|
347
|
+
ids: (text) => [...new Set(find(text).map((m) => m.id))],
|
|
348
|
+
get size() {
|
|
349
|
+
return variants.length;
|
|
350
|
+
}
|
|
351
|
+
};
|
|
352
|
+
}
|
|
353
|
+
function findTerms(text, terms, opts) {
|
|
354
|
+
return createMatcher(Array.isArray(terms) ? terms : [terms], opts).find(text);
|
|
355
|
+
}
|
|
356
|
+
function contains(text, term, opts) {
|
|
357
|
+
return findTerms(text, term, opts).length > 0;
|
|
358
|
+
}
|
|
359
|
+
function splitSuffix(word) {
|
|
360
|
+
const { text: n, map } = mapNormalise(word.trim());
|
|
361
|
+
const sufs = [...new Set(POSTPOSITIONS.map((s) => normaliseNe(s)))].sort((a, b) => b.length - a.length);
|
|
362
|
+
const nfc = word.trim().normalize("NFC");
|
|
363
|
+
let end = n.length;
|
|
364
|
+
const out = [];
|
|
365
|
+
while (out.length < 3) {
|
|
366
|
+
const hit = sufs.find((s) => end - s.length >= 2 && n.slice(end - s.length, end) === s);
|
|
367
|
+
if (!hit) break;
|
|
368
|
+
out.unshift(hit);
|
|
369
|
+
end -= hit.length;
|
|
370
|
+
}
|
|
371
|
+
return { stem: nfc.slice(0, end >= n.length ? nfc.length : map[end]), suffixes: out };
|
|
372
|
+
}
|
|
373
|
+
var dottedAbbr = (s, i) => {
|
|
374
|
+
let j = i - 1;
|
|
375
|
+
while (j >= 0 && !/\s/.test(s[j])) j--;
|
|
376
|
+
return s.slice(j + 1, i).includes(".");
|
|
377
|
+
};
|
|
378
|
+
function sentenceSpans(text) {
|
|
379
|
+
const s = text.normalize("NFC");
|
|
380
|
+
const spans = [];
|
|
381
|
+
let start = 0;
|
|
382
|
+
for (let i = 0; i < s.length; i++) {
|
|
383
|
+
const ch = s[i];
|
|
384
|
+
const end = ch === "\u0964" || ch === "\u0965" || ch === "?" || ch === "!" || ch === "\n" || ch === "." && (i + 1 >= s.length || /\s/.test(s[i + 1])) && !dottedAbbr(s, i);
|
|
385
|
+
if (end) {
|
|
386
|
+
spans.push([start, i + 1]);
|
|
387
|
+
start = i + 1;
|
|
388
|
+
}
|
|
389
|
+
}
|
|
390
|
+
if (start < s.length) spans.push([start, s.length]);
|
|
391
|
+
return spans;
|
|
392
|
+
}
|
|
393
|
+
function near(text, a, b, maxCharsOrOpts = {}, sameSentence) {
|
|
394
|
+
const opts = typeof maxCharsOrOpts === "number" ? { maxChars: maxCharsOrOpts } : { ...maxCharsOrOpts };
|
|
395
|
+
if (sameSentence !== void 0) opts.sameSentence = sameSentence;
|
|
396
|
+
const max = opts.maxChars ?? 60;
|
|
397
|
+
const same = opts.sameSentence !== false;
|
|
398
|
+
const am = findTerms(text, a, opts);
|
|
399
|
+
const bm = findTerms(text, b, opts);
|
|
400
|
+
if (!am.length || !bm.length) return null;
|
|
401
|
+
const spans = same ? sentenceSpans(text) : [];
|
|
402
|
+
const sentenceOf = (i) => spans.findIndex(([s, e]) => i >= s && i < e);
|
|
403
|
+
let best = null;
|
|
404
|
+
for (const x of am) {
|
|
405
|
+
for (const y of bm) {
|
|
406
|
+
if (x.index < y.end && y.index < x.end) continue;
|
|
407
|
+
const gap = x.end <= y.index ? y.index - x.end : x.index - y.end;
|
|
408
|
+
if (gap > max) continue;
|
|
409
|
+
if (same && sentenceOf(x.index) !== sentenceOf(y.index)) continue;
|
|
410
|
+
if (!best || gap < best.gap) best = { a: x, b: y, gap };
|
|
411
|
+
}
|
|
412
|
+
}
|
|
413
|
+
return best;
|
|
414
|
+
}
|
|
415
|
+
|
|
416
|
+
exports.POSTPOSITIONS = POSTPOSITIONS;
|
|
417
|
+
exports.VERSION = VERSION;
|
|
418
|
+
exports.contains = contains;
|
|
419
|
+
exports.createMatcher = createMatcher;
|
|
420
|
+
exports.districtTerms = districtTerms;
|
|
421
|
+
exports.findTerms = findTerms;
|
|
422
|
+
exports.near = near;
|
|
423
|
+
exports.normaliseNe = normaliseNe;
|
|
424
|
+
exports.normalizeNe = normalizeNe;
|
|
425
|
+
exports.sentenceSpans = sentenceSpans;
|
|
426
|
+
exports.splitSuffix = splitSuffix;
|
|
427
|
+
//# sourceMappingURL=index.cjs.map
|
|
428
|
+
//# sourceMappingURL=index.cjs.map
|