@lacspace/keyphrase 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +51 -0
- package/README.md +33 -0
- package/dist/index.cjs +166 -0
- package/dist/index.cjs.map +1 -0
- package/dist/index.d.cts +53 -0
- package/dist/index.d.ts +53 -0
- package/dist/index.js +161 -0
- package/dist/index.js.map +1 -0
- package/package.json +26 -0
package/LICENSE
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
Lacspace Free Licence
|
|
2
|
+
Version 1.0, August 2026
|
|
3
|
+
|
|
4
|
+
Copyright (c) 2026 Lacspace
|
|
5
|
+
|
|
6
|
+
PREAMBLE
|
|
7
|
+
|
|
8
|
+
This software is published by Lacspace under the Lacspace Free Licence — a free,
|
|
9
|
+
permissive licence that lets you use this software for any purpose, including in
|
|
10
|
+
commercial products and services, at no cost. It grants the same freedoms as
|
|
11
|
+
common permissive open-source licences; the only condition is that this notice
|
|
12
|
+
travels with the software. The canonical, always-current text of this licence is
|
|
13
|
+
maintained at https://lacspace.com/licenses/lacspace-free-1.0
|
|
14
|
+
|
|
15
|
+
GRANT OF RIGHTS
|
|
16
|
+
|
|
17
|
+
Permission is hereby granted, free of charge, to any person or organisation
|
|
18
|
+
obtaining a copy of this software and its associated documentation and data files
|
|
19
|
+
(the "Software"), to deal in the Software without restriction, including without
|
|
20
|
+
limitation the rights to use, copy, modify, merge, publish, distribute,
|
|
21
|
+
sublicense, and/or sell copies of the Software, and to permit persons to whom the
|
|
22
|
+
Software is furnished to do so, subject to the conditions below. These rights are
|
|
23
|
+
granted for any purpose, personal or commercial, and are perpetual, worldwide,
|
|
24
|
+
non-exclusive, and royalty-free.
|
|
25
|
+
|
|
26
|
+
CONDITIONS
|
|
27
|
+
|
|
28
|
+
The above copyright notice, this permission notice, and the name of this licence
|
|
29
|
+
("Lacspace Free Licence") shall be included in all copies or substantial portions
|
|
30
|
+
of the Software.
|
|
31
|
+
|
|
32
|
+
TRADEMARKS
|
|
33
|
+
|
|
34
|
+
This licence does not grant permission to use the trade names, trademarks, service
|
|
35
|
+
marks, logos, or product names of Lacspace, except as required to reproduce the
|
|
36
|
+
notice above or to describe the origin of the Software in a truthful manner.
|
|
37
|
+
|
|
38
|
+
DISCLAIMER OF WARRANTY AND LIMITATION OF LIABILITY
|
|
39
|
+
|
|
40
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
41
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
|
42
|
+
FOR A PARTICULAR PURPOSE, AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
|
43
|
+
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES, OR OTHER LIABILITY, WHETHER IN
|
|
44
|
+
AN ACTION OF CONTRACT, TORT, OR OTHERWISE, ARISING FROM, OUT OF, OR IN CONNECTION
|
|
45
|
+
WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
|
46
|
+
|
|
47
|
+
---
|
|
48
|
+
|
|
49
|
+
The Lacspace Free Licence is a source-available, permissive licence and is not (as
|
|
50
|
+
of this version) an OSI-approved licence. In substance it grants the same freedoms
|
|
51
|
+
as the MIT Licence. Learn more at https://lacspace.com/licenses
|
package/README.md
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
# @lacspace/keyphrase
|
|
2
|
+
|
|
3
|
+
**Stop asking the model to tag things.** Extract keyphrases, tags, hashtags, named entities and category votes from text with a zero-dependency RAKE + TF-IDF engine — built-in English and Nepali stopwords, Devanagari-aware, deterministic. Do the ~15–20% of an LLM's output that is really just extraction, for free.
|
|
4
|
+
|
|
5
|
+
```bash
|
|
6
|
+
npm i @lacspace/keyphrase
|
|
7
|
+
```
|
|
8
|
+
|
|
9
|
+
```ts
|
|
10
|
+
import { keyphrase } from "@lacspace/keyphrase";
|
|
11
|
+
|
|
12
|
+
const r = keyphrase(articleText, {
|
|
13
|
+
gazetteer: ["Nepal Rastra Bank", "नेपाल राष्ट्र बैंक"],
|
|
14
|
+
categories: { economy: ["rate", "inflation", "bank"], sports: ["match", "goal"] },
|
|
15
|
+
});
|
|
16
|
+
|
|
17
|
+
r.tags; // ["policy interest rate", "central bank", ...]
|
|
18
|
+
r.hashtags; // ["#PolicyInterestRate", "#CentralBank", ...]
|
|
19
|
+
r.entities; // [{ text: "Nepal Rastra Bank", count: 2 }, ...]
|
|
20
|
+
r.categories; // [{ category: "economy", score: 4 }]
|
|
21
|
+
r.language; // "en" | "ne" (auto-detected)
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
- **RAKE keyphrases** — candidate phrases split at stopwords/punctuation, scored by word degree/frequency; `topK`, `maxWords`, dedupe, gazetteer boost.
|
|
25
|
+
- **Hashtags** — CamelCase for Latin, Devanagari kept whole (matras preserved), punctuation stripped, 2–30 chars.
|
|
26
|
+
- **Entities** — Latin Title-Case runs plus every gazetteer term (including Devanagari, which has no case).
|
|
27
|
+
- **Category votes** — pass `{ category: [terms] }` and get a ranked vote by term hits.
|
|
28
|
+
- **Bilingual** — ships `ENGLISH_STOPWORDS` and `NEPALI_STOPWORDS`; auto-detects language, or force it; add your own with `extraStopwords` or replace with `stopwords`.
|
|
29
|
+
|
|
30
|
+
Deterministic and isomorphic. You bring domain stopwords/gazetteers; it brings the engine. Exports `toHashtag` and the two stopword lists too.
|
|
31
|
+
|
|
32
|
+
## Licence
|
|
33
|
+
[Lacspace Free Licence v1.0](https://developer.lacspace.com/licenses/lacspace-free-1.0) — free for personal and commercial use.
|
package/dist/index.cjs
ADDED
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
'use strict';
|
|
2
|
+
|
|
3
|
+
// src/stopwords.ts
|
|
4
|
+
var ENGLISH_STOPWORDS = "a an and are as at be but by for from has have he her his i in is it its of on or that the their them they this to was were will with would you your we our us she him not no do does did done can could should may might must shall about after all also any because been before being between both during each few more most other over own same so some such than then there these those through under until up very what when where which who whom why how had having into out off down again further once here said says say according reported reports report told tuesday monday wednesday thursday friday saturday sunday am pm mr mrs ms dr new one two three per amid across".split(/\s+/);
|
|
5
|
+
var NEPALI_STOPWORDS = "\u0930 \u0915\u094B \u0915\u093E \u0915\u0940 \u092E\u093E \u0932\u0947 \u0939\u094B \u091B \u0925\u093F\u092F\u094B \u0925\u093F\u090F \u0939\u0941\u0928 \u0939\u0941\u0928\u094D\u091B \u092D\u0928\u0947 \u092D\u0928\u0940 \u092D\u0928\u094D\u0928\u0947 \u092A\u0928\u093F \u092F\u094B \u0924\u094D\u092F\u094B \u0924\u093F \u0924\u094D\u092F\u0938 \u092F\u0938 \u0917\u0930\u0947\u0915\u094B \u092D\u090F\u0915\u094B \u0932\u093E\u0917\u093F \u0924\u0925\u093E \u090F\u0935\u0902 \u0905\u0928\u0940 \u091C\u0938\u094D\u0924\u094B \u091C\u0938\u094D\u0924\u0948 \u0939\u094B\u0938\u094D \u091B\u0928\u094D \u0917\u0930\u094D\u0928 \u0917\u0930\u094D\u091B \u0917\u0930\u094D\u092F\u094B \u092D\u092F\u094B \u0905\u092C \u0938\u092C\u0948 \u0915\u0947\u0939\u0940 \u0906\u092B\u094D\u0928\u094B \u0909\u0928\u0940 \u0909\u0938\u0932\u0947 \u0909\u0928\u0932\u0947 \u092E\u093E\u0925\u093F \u0924\u0932 \u092D\u0928\u094D\u0926\u093E \u0938\u092E\u094D\u092C\u0928\u094D\u0927\u0940 \u092C\u093E\u091F \u0926\u0947\u0916\u093F \u0938\u092E\u094D\u092E".split(/\s+/).filter(Boolean);
|
|
6
|
+
|
|
7
|
+
// src/index.ts
|
|
8
|
+
var WORD_RE = /[\p{L}\p{N}][\p{L}\p{M}\p{N}]*/gu;
|
|
9
|
+
var DEVANAGARI_RE = /[ऀ-ॿ]/;
|
|
10
|
+
function detectLanguage(text) {
|
|
11
|
+
const deva = (text.match(/[ऀ-ॿ]/g) || []).length;
|
|
12
|
+
const latin = (text.match(/[A-Za-z]/g) || []).length;
|
|
13
|
+
return deva > latin ? "ne" : "en";
|
|
14
|
+
}
|
|
15
|
+
function keyphrase(text, options = {}) {
|
|
16
|
+
const language = options.language && options.language !== "auto" ? options.language : detectLanguage(text || "");
|
|
17
|
+
const topK = options.topK ?? 10;
|
|
18
|
+
const maxWords = options.maxWords ?? 4;
|
|
19
|
+
const minLen = options.minWordLength ?? 2;
|
|
20
|
+
const base = options.stopwords ?? (language === "ne" ? NEPALI_STOPWORDS : ENGLISH_STOPWORDS);
|
|
21
|
+
const stop = new Set([...base, ...options.extraStopwords ?? []].map((s) => s.toLowerCase()));
|
|
22
|
+
const gaz = (options.gazetteer ?? []).filter(Boolean);
|
|
23
|
+
const empty = { phrases: [], tags: [], hashtags: [], entities: [], categories: [], language };
|
|
24
|
+
if (!text || !text.trim()) return empty;
|
|
25
|
+
const candidates = [];
|
|
26
|
+
let current = [];
|
|
27
|
+
const tokenStream = tokenizeWithGaps(text);
|
|
28
|
+
for (const tok of tokenStream) {
|
|
29
|
+
if (tok.isBreak) {
|
|
30
|
+
if (current.length) candidates.push(current);
|
|
31
|
+
current = [];
|
|
32
|
+
continue;
|
|
33
|
+
}
|
|
34
|
+
const lw = tok.w.toLowerCase();
|
|
35
|
+
if (stop.has(lw) || tok.w.length < minLen || /^\d+$/.test(tok.w)) {
|
|
36
|
+
if (current.length) candidates.push(current);
|
|
37
|
+
current = [];
|
|
38
|
+
} else {
|
|
39
|
+
current.push(tok.w);
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
if (current.length) candidates.push(current);
|
|
43
|
+
const freq = /* @__PURE__ */ new Map();
|
|
44
|
+
const degree = /* @__PURE__ */ new Map();
|
|
45
|
+
for (const phrase of candidates) {
|
|
46
|
+
const deg = phrase.length - 1;
|
|
47
|
+
for (const w of phrase) {
|
|
48
|
+
const k = w.toLowerCase();
|
|
49
|
+
freq.set(k, (freq.get(k) ?? 0) + 1);
|
|
50
|
+
degree.set(k, (degree.get(k) ?? 0) + deg + 1);
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
const wordScore = (w) => {
|
|
54
|
+
const k = w.toLowerCase();
|
|
55
|
+
const f = freq.get(k) ?? 1;
|
|
56
|
+
return (degree.get(k) ?? f) / f;
|
|
57
|
+
};
|
|
58
|
+
const gazLower = new Set(gaz.map((g) => g.toLowerCase()));
|
|
59
|
+
const phraseScores = /* @__PURE__ */ new Map();
|
|
60
|
+
let order = 0;
|
|
61
|
+
for (const phrase of candidates) {
|
|
62
|
+
if (phrase.length === 0 || phrase.length > maxWords) continue;
|
|
63
|
+
const text2 = phrase.join(" ");
|
|
64
|
+
const key = text2.toLowerCase();
|
|
65
|
+
let score = phrase.reduce((s, w) => s + wordScore(w), 0);
|
|
66
|
+
if (gazLower.has(key) || gaz.some((g) => text2.includes(g))) score *= 1.5;
|
|
67
|
+
const prev = phraseScores.get(key);
|
|
68
|
+
if (prev) prev.score = Math.max(prev.score, score);
|
|
69
|
+
else phraseScores.set(key, { phrase: text2, score, order: order++ });
|
|
70
|
+
}
|
|
71
|
+
const phrases = [...phraseScores.values()].sort((a, b) => b.score - a.score || a.order - b.order).slice(0, topK).map((p) => ({ phrase: p.phrase, score: round(p.score) }));
|
|
72
|
+
const tags = phrases.map((p) => p.phrase);
|
|
73
|
+
const hashtags = dedupe(tags.map(toHashtag).filter((h) => h.length > 1));
|
|
74
|
+
const entities = extractEntities(text, gaz);
|
|
75
|
+
const categories = [];
|
|
76
|
+
if (options.categories) {
|
|
77
|
+
const lower = text.toLowerCase();
|
|
78
|
+
for (const [cat, terms] of Object.entries(options.categories)) {
|
|
79
|
+
let score = 0;
|
|
80
|
+
for (const term of terms) {
|
|
81
|
+
if (!term) continue;
|
|
82
|
+
if (DEVANAGARI_RE.test(term)) {
|
|
83
|
+
score += countOccurrences(text, term);
|
|
84
|
+
} else {
|
|
85
|
+
const re = new RegExp(`(?:^|[^a-z0-9])${escapeRe(term.toLowerCase())}(?:[^a-z0-9]|$)`, "g");
|
|
86
|
+
score += (lower.match(re) || []).length;
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
if (score > 0) categories.push({ category: cat, score });
|
|
90
|
+
}
|
|
91
|
+
categories.sort((a, b) => b.score - a.score || a.category.localeCompare(b.category));
|
|
92
|
+
}
|
|
93
|
+
return { phrases, tags, hashtags, entities, categories, language };
|
|
94
|
+
}
|
|
95
|
+
function tokenizeWithGaps(text) {
|
|
96
|
+
const out = [];
|
|
97
|
+
let last = 0;
|
|
98
|
+
let m;
|
|
99
|
+
WORD_RE.lastIndex = 0;
|
|
100
|
+
while ((m = WORD_RE.exec(text)) !== null) {
|
|
101
|
+
if (m.index > last) {
|
|
102
|
+
const gap = text.slice(last, m.index);
|
|
103
|
+
if (/[.!?;:,।॥\n\-–—/()\[\]"'“”]/.test(gap)) out.push({ isBreak: true });
|
|
104
|
+
}
|
|
105
|
+
out.push({ w: m[0], isBreak: false });
|
|
106
|
+
last = m.index + m[0].length;
|
|
107
|
+
}
|
|
108
|
+
return out;
|
|
109
|
+
}
|
|
110
|
+
function toHashtag(input) {
|
|
111
|
+
const cleaned = input.replace(/^#/, "");
|
|
112
|
+
const parts = cleaned.split(/[\s\-_/]+/).filter(Boolean);
|
|
113
|
+
const joined = parts.map((p) => /^[a-z]/.test(p) ? p.charAt(0).toUpperCase() + p.slice(1) : p).join("");
|
|
114
|
+
const kept = [...joined].filter((ch) => /[\p{L}\p{M}\p{N}]/u.test(ch)).join("");
|
|
115
|
+
return kept.length >= 2 && kept.length <= 30 ? "#" + kept : "";
|
|
116
|
+
}
|
|
117
|
+
function extractEntities(text, gaz) {
|
|
118
|
+
const counts = /* @__PURE__ */ new Map();
|
|
119
|
+
const re = /\b([A-Z][a-zA-Z]+(?:\s+[A-Z][a-zA-Z]+){0,3})\b/g;
|
|
120
|
+
let m;
|
|
121
|
+
while ((m = re.exec(text)) !== null) {
|
|
122
|
+
const e = m[1];
|
|
123
|
+
counts.set(e, (counts.get(e) ?? 0) + 1);
|
|
124
|
+
}
|
|
125
|
+
for (const g of gaz) {
|
|
126
|
+
if (!g) continue;
|
|
127
|
+
const c = countOccurrences(text, g);
|
|
128
|
+
if (c > 0) counts.set(g, Math.max(counts.get(g) ?? 0, c));
|
|
129
|
+
}
|
|
130
|
+
return [...counts.entries()].map(([t, c]) => ({ text: t, count: c })).sort((a, b) => b.count - a.count || a.text.localeCompare(b.text)).filter((e) => e.text.includes(" ") || gaz.includes(e.text) || e.count > 1 || /[ऀ-ॿ]/.test(e.text));
|
|
131
|
+
}
|
|
132
|
+
function countOccurrences(hay, needle) {
|
|
133
|
+
if (!needle) return 0;
|
|
134
|
+
let n = 0;
|
|
135
|
+
let i = hay.indexOf(needle);
|
|
136
|
+
while (i !== -1) {
|
|
137
|
+
n++;
|
|
138
|
+
i = hay.indexOf(needle, i + needle.length);
|
|
139
|
+
}
|
|
140
|
+
return n;
|
|
141
|
+
}
|
|
142
|
+
function dedupe(arr) {
|
|
143
|
+
const seen = /* @__PURE__ */ new Set();
|
|
144
|
+
const out = [];
|
|
145
|
+
for (const x of arr) {
|
|
146
|
+
const k = x.toLowerCase();
|
|
147
|
+
if (!seen.has(k)) {
|
|
148
|
+
seen.add(k);
|
|
149
|
+
out.push(x);
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
return out;
|
|
153
|
+
}
|
|
154
|
+
function escapeRe(s) {
|
|
155
|
+
return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
156
|
+
}
|
|
157
|
+
function round(n) {
|
|
158
|
+
return Math.round(n * 1e3) / 1e3;
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
exports.ENGLISH_STOPWORDS = ENGLISH_STOPWORDS;
|
|
162
|
+
exports.NEPALI_STOPWORDS = NEPALI_STOPWORDS;
|
|
163
|
+
exports.keyphrase = keyphrase;
|
|
164
|
+
exports.toHashtag = toHashtag;
|
|
165
|
+
//# sourceMappingURL=index.cjs.map
|
|
166
|
+
//# sourceMappingURL=index.cjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"sources":["../src/stopwords.ts","../src/index.ts"],"names":[],"mappings":";;;AACO,IAAM,iBAAA,GACX,wpBAAA,CAQA,KAAA,CAAM,KAAK;AAGN,IAAM,mBACX,4hCAAA,CAUA,KAAA,CAAM,KAAK,CAAA,CAAE,OAAO,OAAO;;;AC0B7B,IAAM,OAAA,GAAU,oCAAA;AAChB,IAAM,aAAA,GAAgB,OAAA;AAEtB,SAAS,eAAe,IAAA,EAA2B;AACjD,EAAA,MAAM,QAAQ,IAAA,CAAK,KAAA,CAAM,QAAQ,CAAA,IAAK,EAAC,EAAG,MAAA;AAC1C,EAAA,MAAM,SAAS,IAAA,CAAK,KAAA,CAAM,WAAW,CAAA,IAAK,EAAC,EAAG,MAAA;AAC9C,EAAA,OAAO,IAAA,GAAO,QAAQ,IAAA,GAAO,IAAA;AAC/B;AAWO,SAAS,SAAA,CAAU,IAAA,EAAc,OAAA,GAA4B,EAAC,EAAoB;AACvF,EAAA,MAAM,QAAA,GAAW,OAAA,CAAQ,QAAA,IAAY,OAAA,CAAQ,QAAA,KAAa,SAAS,OAAA,CAAQ,QAAA,GAAW,cAAA,CAAe,IAAA,IAAQ,EAAE,CAAA;AAC/G,EAAA,MAAM,IAAA,GAAO,QAAQ,IAAA,IAAQ,EAAA;AAC7B,EAAA,MAAM,QAAA,GAAW,QAAQ,QAAA,IAAY,CAAA;AACrC,EAAA,MAAM,MAAA,GAAS,QAAQ,aAAA,IAAiB,CAAA;AACxC,EAAA,MAAM,IAAA,GAAO,OAAA,CAAQ,SAAA,KAAc,QAAA,KAAa,OAAO,gBAAA,GAAmB,iBAAA,CAAA;AAC1E,EAAA,MAAM,OAAO,IAAI,GAAA,CAAI,CAAC,GAAG,IAAA,EAAM,GAAI,OAAA,CAAQ,cAAA,IAAkB,EAAG,EAAE,GAAA,CAAI,CAAC,MAAM,CAAA,CAAE,WAAA,EAAa,CAAC,CAAA;AAC7F,EAAA,MAAM,OAAO,OAAA,CAAQ,SAAA,IAAa,EAAC,EAAG,OAAO,OAAO,CAAA;AAEpD,EAAA,MAAM,QAAyB,EAAE,OAAA,EAAS,EAAC,EAAG,MAAM,EAAC,EAAG,QAAA,EAAU,IAAI,QAAA,EAAU,IAAI,UAAA,EAAY,IAAI,QAAA,EAAS;AAC7G,EAAA,IAAI,CAAC,IAAA,IAAQ,CAAC,IAAA,CAAK,IAAA,IAAQ,OAAO,KAAA;AAGlC,EAAA,MAAM,aAAyB,EAAC;AAChC,EAAA,IAAI,UAAoB,EAAC;AAEzB,EAAA,MAAM,WAAA,GAAc,iBAAiB,IAAI,CAAA;AACzC,EAAA,KAAA,MAAW,OAAO,WAAA,EAAa;AAC7B,IAAA,IAAI,IAAI,OAAA,EAAS;AACf,MAAA,IAAI,OAAA,CAAQ,MAAA,EAAQ,UAAA,CAAW,IAAA,CAAK,OAAO,CAAA;AAC3C,MAAA,OAAA,GAAU,EAAC;AACX,MAAA;AAAA,IACF;AACA,IAAA,MAAM,EAAA,GAAK,GAAA,CAAI,CAAA,CAAG,WAAA,EAAY;AAC9B,IAAA,IAAI,IAAA,CAAK,GAAA,CAAI,EAAE,CAAA,IAAK,GAAA,CAAI,CAAA,CAAG,MAAA,GAAS,MAAA,IAAU,OAAA,CAAQ,IAAA,CAAK,GAAA,CAAI,CAAE,CAAA,EAAG;AAClE,MAAA,IAAI,OAAA,CAAQ,MAAA,EAAQ,UAAA,CAAW,IAAA,CAAK,OAAO,CAAA;AAC3C,MAAA,OAAA,GAAU,EAAC;AAAA,IACb,CAAA,MAAO;AACL,MAAA,OAAA,CAAQ,IAAA,CAAK,IAAI,CAAE,CAAA;AAAA,IACrB;AAAA,EACF;AACA,EAAA,IAAI,OAAA,CAAQ,MAAA,EAAQ,UAAA,CAAW,IAAA,CAAK,OAAO,CAAA;AAG3C,EAAA,MAAM,IAAA,uBAAW,GAAA,EAAoB;AACrC,EAAA,MAAM,MAAA,uBAAa,GAAA,EAAoB;AACvC,EAAA,KAAA,MAAW,UAAU,UAAA,EAAY;AAC/B,IAAA,MAAM,GAAA,GAAM,OAAO,MAAA,GAAS,CAAA;AAC5B,IAAA,KAAA,MAAW,KAAK,MAAA,EAAQ;AACtB,MAAA,MAAM,CAAA,GAAI,EAAE,WAAA,EAAY;AACxB,MAAA,IAAA,CAAK,IAAI,CAAA,EAAA,CAAI,IAAA,CAAK,IAAI,CAAC,CAAA,IAAK,KAAK,CAAC,CAAA;AAClC,MAAA,MAAA,CAAO,GAAA,CAAI,IAAI,MAAA,CAAO,GAAA,CAAI,CAAC,CAAA,IAAK,CAAA,IAAK,MAAM,CAAC,CAAA;AAAA,IAC9C;AAAA,EACF;AACA,EAAA,MAAM,SAAA,GAAY,CAAC,CAAA,KAAc;AAC/B,IAAA,MAAM,CAAA,GAAI,EAAE,WAAA,EAAY;AACxB,IAAA,MAAM,CAAA,GAAI,IAAA,CAAK,GAAA,CAAI,CAAC,CAAA,IAAK,CAAA;AACzB,IAAA,OAAA,CAAQ,MAAA,CAAO,GAAA,CAAI,CAAC,CAAA,IAAK,CAAA,IAAK,CAAA;AAAA,EAChC,CAAA;AAGA,EAAA,MAAM,QAAA,GAAW,IAAI,GAAA,CAAI,GAAA,CAAI,GAAA,CAAI,CAAC,CAAA,KAAM,CAAA,CAAE,WAAA,EAAa,CAAC,CAAA;AACxD,EAAA,MAAM,YAAA,uBAAmB,GAAA,EAA8D;AACvF,EAAA,IAAI,KAAA,GAAQ,CAAA;AACZ,EAAA,KAAA,MAAW,UAAU,UAAA,EAAY;AAC/B,IAAA,IAAI,MAAA,CAAO,MAAA,KAAW,CAAA,IAAK,MAAA,CAAO,SAAS,QAAA,EAAU;AACrD,IAAA,MAAM,KAAA,GAAQ,MAAA,CAAO,IAAA,CAAK,GAAG,CAAA;AAC7B,IAAA,MAAM,GAAA,GAAM,MAAM,WAAA,EAAY;AAC9B,IAAA,IAAI,KAAA,GAAQ,MAAA,CAAO,MAAA,CAAO,CAAC,CAAA,EAAG,MAAM,CAAA,GAAI,SAAA,CAAU,CAAC,CAAA,EAAG,CAAC,CAAA;AACvD,IAAA,IAAI,QAAA,CAAS,GAAA,CAAI,GAAG,CAAA,IAAK,GAAA,CAAI,IAAA,CAAK,CAAC,CAAA,KAAM,KAAA,CAAM,QAAA,CAAS,CAAC,CAAC,GAAG,KAAA,IAAS,GAAA;AACtE,IAAA,MAAM,IAAA,GAAO,YAAA,CAAa,GAAA,CAAI,GAAG,CAAA;AACjC,IAAA,IAAI,MAAM,IAAA,CAAK,KAAA,GAAQ,KAAK,GAAA,CAAI,IAAA,CAAK,OAAO,KAAK,CAAA;AAAA,SAC5C,YAAA,CAAa,IAAI,GAAA,EAAK,EAAE,QAAQ,KAAA,EAAO,KAAA,EAAO,KAAA,EAAO,KAAA,EAAA,EAAS,CAAA;AAAA,EACrE;AACA,EAAA,MAAM,UAAU,CAAC,GAAG,YAAA,CAAa,MAAA,EAAQ,CAAA,CACtC,IAAA,CAAK,CAAC,CAAA,EAAG,MAAM,CAAA,CAAE,KAAA,GAAQ,CAAA,CAAE,KAAA,IAAS,EAAE,KAAA,GAAQ,CAAA,CAAE,KAAK,CAAA,CACrD,MAAM,CAAA,EAAG,IAAI,CAAA,CACb,GAAA,CAAI,CAAC,CAAA,MAAO,EAAE,MAAA,EAAQ,CAAA,CAAE,QAAQ,KAAA,EAAO,KAAA,CAAM,CAAA,CAAE,KAAK,GAAE,CAAE,CAAA;AAG3D,EAAA,MAAM,OAAO,OAAA,CAAQ,GAAA,CAAI,CAAC,CAAA,KAAM,EAAE,MAAM,CAAA;AACxC,EAAA,MAAM,QAAA,GAAW,MAAA,CAAO,IAAA,CAAK,GAAA,CAAI,SAAS,CAAA,CAAE,MAAA,CAAO,CAAC,CAAA,KAAM,CAAA,CAAE,MAAA,GAAS,CAAC,CAAC,CAAA;AAGvE,EAAA,MAAM,QAAA,GAAW,eAAA,CAAgB,IAAA,EAAM,GAAG,CAAA;AAG1C,EAAA,MAAM,aAA6B,EAAC;AACpC,EAAA,IAAI,QAAQ,UAAA,EAAY;AACtB,IAAA,MAAM,KAAA,GAAQ,KAAK,WAAA,EAAY;AAC/B,IAAA,KAAA,MAAW,CAAC,KAAK,KAAK,CAAA,IAAK,OAAO,OAAA,CAAQ,OAAA,CAAQ,UAAU,CAAA,EAAG;AAC7D,MAAA,IAAI,KAAA,GAAQ,CAAA;AACZ,MAAA,KAAA,MAAW,QAAQ,KAAA,EAAO;AACxB,QAAA,IAAI,CAAC,IAAA,EAAM;AACX,QAAA,IAAI,aAAA,CAAc,IAAA,CAAK,IAAI,CAAA,EAAG;AAC5B,UAAA,KAAA,IAAS,gBAAA,CAAiB,MAAM,IAAI,CAAA;AAAA,QACtC,CAAA,MAAO;AACL,UAAA,MAAM,EAAA,GAAK,IAAI,MAAA,CAAO,CAAA,eAAA,EAAkB,QAAA,CAAS,KAAK,WAAA,EAAa,CAAC,CAAA,eAAA,CAAA,EAAmB,GAAG,CAAA;AAC1F,UAAA,KAAA,IAAA,CAAU,KAAA,CAAM,KAAA,CAAM,EAAE,CAAA,IAAK,EAAC,EAAG,MAAA;AAAA,QACnC;AAAA,MACF;AACA,MAAA,IAAI,KAAA,GAAQ,GAAG,UAAA,CAAW,IAAA,CAAK,EAAE,QAAA,EAAU,GAAA,EAAK,OAAO,CAAA;AAAA,IACzD;AACA,IAAA,UAAA,CAAW,IAAA,CAAK,CAAC,CAAA,EAAG,CAAA,KAAM,CAAA,CAAE,KAAA,GAAQ,CAAA,CAAE,KAAA,IAAS,CAAA,CAAE,QAAA,CAAS,aAAA,CAAc,CAAA,CAAE,QAAQ,CAAC,CAAA;AAAA,EACrF;AAEA,EAAA,OAAO,EAAE,OAAA,EAAS,IAAA,EAAM,QAAA,EAAU,QAAA,EAAU,YAAY,QAAA,EAAS;AACnE;AAMA,SAAS,iBAAiB,IAAA,EAA2B;AACnD,EAAA,MAAM,MAAmB,EAAC;AAC1B,EAAA,IAAI,IAAA,GAAO,CAAA;AACX,EAAA,IAAI,CAAA;AACJ,EAAA,OAAA,CAAQ,SAAA,GAAY,CAAA;AACpB,EAAA,OAAA,CAAQ,CAAA,GAAI,OAAA,CAAQ,IAAA,CAAK,IAAI,OAAO,IAAA,EAAM;AACxC,IAAA,IAAI,CAAA,CAAE,QAAQ,IAAA,EAAM;AAElB,MAAA,MAAM,GAAA,GAAM,IAAA,CAAK,KAAA,CAAM,IAAA,EAAM,EAAE,KAAK,CAAA;AACpC,MAAA,IAAI,6BAAA,CAA8B,KAAK,GAAG,CAAA,MAAO,IAAA,CAAK,EAAE,OAAA,EAAS,IAAA,EAAM,CAAA;AAAA,IACzE;AACA,IAAA,GAAA,CAAI,IAAA,CAAK,EAAE,CAAA,EAAG,CAAA,CAAE,CAAC,CAAA,EAAG,OAAA,EAAS,OAAO,CAAA;AACpC,IAAA,IAAA,GAAO,CAAA,CAAE,KAAA,GAAQ,CAAA,CAAE,CAAC,CAAA,CAAE,MAAA;AAAA,EACxB;AACA,EAAA,OAAO,GAAA;AACT;AAGO,SAAS,UAAU,KAAA,EAAuB;AAC/C,EAAA,MAAM,OAAA,GAAU,KAAA,CAAM,OAAA,CAAQ,IAAA,EAAM,EAAE,CAAA;AACtC,EAAA,MAAM,QAAQ,OAAA,CAAQ,KAAA,CAAM,WAAW,CAAA,CAAE,OAAO,OAAO,CAAA;AACvD,EAAA,MAAM,MAAA,GAAS,MACZ,GAAA,CAAI,CAAC,MAAO,QAAA,CAAS,IAAA,CAAK,CAAC,CAAA,GAAI,CAAA,CAAE,MAAA,CAAO,CAAC,CAAA,CAAE,WAAA,KAAgB,CAAA,CAAE,KAAA,CAAM,CAAC,CAAA,GAAI,CAAE,CAAA,CAC1E,IAAA,CAAK,EAAE,CAAA;AACV,EAAA,MAAM,IAAA,GAAO,CAAC,GAAG,MAAM,EAAE,MAAA,CAAO,CAAC,EAAA,KAAO,oBAAA,CAAqB,IAAA,CAAK,EAAE,CAAC,CAAA,CAAE,KAAK,EAAE,CAAA;AAC9E,EAAA,OAAO,KAAK,MAAA,IAAU,CAAA,IAAK,KAAK,MAAA,IAAU,EAAA,GAAK,MAAM,IAAA,GAAO,EAAA;AAC9D;AAEA,SAAS,eAAA,CAAgB,MAAc,GAAA,EAA4B;AACjE,EAAA,MAAM,MAAA,uBAAa,GAAA,EAAoB;AAEvC,EAAA,MAAM,EAAA,GAAK,iDAAA;AACX,EAAA,IAAI,CAAA;AACJ,EAAA,OAAA,CAAQ,CAAA,GAAI,EAAA,CAAG,IAAA,CAAK,IAAI,OAAO,IAAA,EAAM;AACnC,IAAA,MAAM,CAAA,GAAI,EAAE,CAAC,CAAA;AAEb,IAAA,MAAA,CAAO,IAAI,CAAA,EAAA,CAAI,MAAA,CAAO,IAAI,CAAC,CAAA,IAAK,KAAK,CAAC,CAAA;AAAA,EACxC;AAEA,EAAA,KAAA,MAAW,KAAK,GAAA,EAAK;AACnB,IAAA,IAAI,CAAC,CAAA,EAAG;AACR,IAAA,MAAM,CAAA,GAAI,gBAAA,CAAiB,IAAA,EAAM,CAAC,CAAA;AAClC,IAAA,IAAI,CAAA,GAAI,CAAA,EAAG,MAAA,CAAO,GAAA,CAAI,CAAA,EAAG,IAAA,CAAK,GAAA,CAAI,MAAA,CAAO,GAAA,CAAI,CAAC,CAAA,IAAK,CAAA,EAAG,CAAC,CAAC,CAAA;AAAA,EAC1D;AACA,EAAA,OAAO,CAAC,GAAG,MAAA,CAAO,OAAA,EAAS,EACxB,GAAA,CAAI,CAAC,CAAC,CAAA,EAAG,CAAC,CAAA,MAAO,EAAE,IAAA,EAAM,CAAA,EAAG,KAAA,EAAO,CAAA,EAAE,CAAE,CAAA,CACvC,IAAA,CAAK,CAAC,CAAA,EAAG,CAAA,KAAM,CAAA,CAAE,KAAA,GAAQ,CAAA,CAAE,KAAA,IAAS,EAAE,IAAA,CAAK,aAAA,CAAc,CAAA,CAAE,IAAI,CAAC,CAAA,CAChE,OAAO,CAAC,CAAA,KAAM,CAAA,CAAE,IAAA,CAAK,QAAA,CAAS,GAAG,KAAK,GAAA,CAAI,QAAA,CAAS,CAAA,CAAE,IAAI,CAAA,IAAK,CAAA,CAAE,KAAA,GAAQ,CAAA,IAAK,OAAA,CAAQ,IAAA,CAAK,CAAA,CAAE,IAAI,CAAC,CAAA;AACtG;AAEA,SAAS,gBAAA,CAAiB,KAAa,MAAA,EAAwB;AAC7D,EAAA,IAAI,CAAC,QAAQ,OAAO,CAAA;AACpB,EAAA,IAAI,CAAA,GAAI,CAAA;AACR,EAAA,IAAI,CAAA,GAAI,GAAA,CAAI,OAAA,CAAQ,MAAM,CAAA;AAC1B,EAAA,OAAO,MAAM,EAAA,EAAI;AACf,IAAA,CAAA,EAAA;AACA,IAAA,CAAA,GAAI,GAAA,CAAI,OAAA,CAAQ,MAAA,EAAQ,CAAA,GAAI,OAAO,MAAM,CAAA;AAAA,EAC3C;AACA,EAAA,OAAO,CAAA;AACT;AACA,SAAS,OAAO,GAAA,EAAyB;AACvC,EAAA,MAAM,IAAA,uBAAW,GAAA,EAAY;AAC7B,EAAA,MAAM,MAAgB,EAAC;AACvB,EAAA,KAAA,MAAW,KAAK,GAAA,EAAK;AACnB,IAAA,MAAM,CAAA,GAAI,EAAE,WAAA,EAAY;AACxB,IAAA,IAAI,CAAC,IAAA,CAAK,GAAA,CAAI,CAAC,CAAA,EAAG;AAChB,MAAA,IAAA,CAAK,IAAI,CAAC,CAAA;AACV,MAAA,GAAA,CAAI,KAAK,CAAC,CAAA;AAAA,IACZ;AAAA,EACF;AACA,EAAA,OAAO,GAAA;AACT;AACA,SAAS,SAAS,CAAA,EAAmB;AACnC,EAAA,OAAO,CAAA,CAAE,OAAA,CAAQ,qBAAA,EAAuB,MAAM,CAAA;AAChD;AACA,SAAS,MAAM,CAAA,EAAmB;AAChC,EAAA,OAAO,IAAA,CAAK,KAAA,CAAM,CAAA,GAAI,GAAI,CAAA,GAAI,GAAA;AAChC","file":"index.cjs","sourcesContent":["// Compact, high-frequency stopword lists. Extend via options.extraStopwords.\nexport const ENGLISH_STOPWORDS: string[] = (\n \"a an and are as at be but by for from has have he her his i in is it its of on or \" +\n \"that the their them they this to was were will with would you your we our us she him \" +\n \"not no do does did done can could should may might must shall about after all also \" +\n \"any because been before being between both during each few more most other over own \" +\n \"same so some such than then there these those through under until up very what when \" +\n \"where which who whom why how had having into out off down again further once here \" +\n \"said says say according reported reports report told tuesday monday wednesday thursday \" +\n \"friday saturday sunday am pm mr mrs ms dr new one two three per amid across\"\n).split(/\\s+/);\n\n// Common Nepali (Devanagari) function words and news filler.\nexport const NEPALI_STOPWORDS: string[] = (\n \"र को का की मा ले हो \" +\n \"छ थियो थिए हुन हुन्छ \" +\n \"भने भनी भन्ने पनि यो \" +\n \"त्यो ति त्यस यस गरेको \" +\n \"भएको लागि तथा एवं अनी \" +\n \"जस्तो जस्तै होस् छन् \" +\n \"गर्न गर्छ गर्यो भयो \" +\n \"अब सबै केही आफ्नो उनी \" +\n \"उसले उनले माथि तल भन्दा \" +\n \"सम्बन्धी बाट देखि सम्म\"\n).split(/\\s+/).filter(Boolean);\n","import { ENGLISH_STOPWORDS, NEPALI_STOPWORDS } from \"./stopwords.js\";\n\nexport { ENGLISH_STOPWORDS, NEPALI_STOPWORDS } from \"./stopwords.js\";\n\nexport interface KeyphraseOptions {\n /** \"en\", \"ne\", or \"auto\" (default) — picks built-in stopwords. */\n language?: \"en\" | \"ne\" | \"auto\";\n /** How many keyphrases/tags to return. Default 10. */\n topK?: number;\n /** Longest phrase, in words. Default 4. */\n maxWords?: number;\n /** Replace the built-in stopwords entirely. */\n stopwords?: string[];\n /** Add to the built-in stopwords. */\n extraStopwords?: string[];\n /** Known entities (people/places), en and ne forms — always surfaced and boosted. */\n gazetteer?: string[];\n /** Category lexicons: { category: [terms] }. Returns a vote per category by term hits. */\n categories?: Record<string, string[]>;\n /** Minimum characters for a candidate word. Default 2. */\n minWordLength?: number;\n}\n\nexport interface ScoredPhrase {\n phrase: string;\n score: number;\n}\nexport interface EntityHit {\n text: string;\n count: number;\n}\nexport interface CategoryVote {\n category: string;\n score: number;\n}\nexport interface KeyphraseResult {\n /** Top keyphrases by RAKE degree/frequency score. */\n phrases: ScoredPhrase[];\n /** Flat, de-duplicated tag strings (top phrases, normalized). */\n tags: string[];\n /** Hashtags built from the top tags (CamelCase for Latin, Devanagari kept whole). */\n hashtags: string[];\n /** Candidate named entities: Latin Title-Case runs + gazetteer hits. */\n entities: EntityHit[];\n /** Category votes (only when `categories` is provided), highest first. */\n categories: CategoryVote[];\n /** Detected/So-used language. */\n language: \"en\" | \"ne\";\n}\n\nconst WORD_RE = /[\\p{L}\\p{N}][\\p{L}\\p{M}\\p{N}]*/gu;\nconst DEVANAGARI_RE = /[ऀ-ॿ]/;\n\nfunction detectLanguage(text: string): \"en\" | \"ne\" {\n const deva = (text.match(/[ऀ-ॿ]/g) || []).length;\n const latin = (text.match(/[A-Za-z]/g) || []).length;\n return deva > latin ? \"ne\" : \"en\";\n}\n\nfunction words(text: string): { w: string; start: number }[] {\n const out: { w: string; start: number }[] = [];\n let m: RegExpExecArray | null;\n WORD_RE.lastIndex = 0;\n while ((m = WORD_RE.exec(text)) !== null) out.push({ w: m[0], start: m.index });\n return out;\n}\n\n/** Extract keyphrases, tags, hashtags, entities and category votes — no LLM. */\nexport function keyphrase(text: string, options: KeyphraseOptions = {}): KeyphraseResult {\n const language = options.language && options.language !== \"auto\" ? options.language : detectLanguage(text || \"\");\n const topK = options.topK ?? 10;\n const maxWords = options.maxWords ?? 4;\n const minLen = options.minWordLength ?? 2;\n const base = options.stopwords ?? (language === \"ne\" ? NEPALI_STOPWORDS : ENGLISH_STOPWORDS);\n const stop = new Set([...base, ...(options.extraStopwords ?? [])].map((s) => s.toLowerCase()));\n const gaz = (options.gazetteer ?? []).filter(Boolean);\n\n const empty: KeyphraseResult = { phrases: [], tags: [], hashtags: [], entities: [], categories: [], language };\n if (!text || !text.trim()) return empty;\n\n // 1. RAKE: break into candidate phrases at stopwords and non-word chars.\n const candidates: string[][] = [];\n let current: string[] = [];\n // Walk the raw text so punctuation also breaks phrases.\n const tokenStream = tokenizeWithGaps(text);\n for (const tok of tokenStream) {\n if (tok.isBreak) {\n if (current.length) candidates.push(current);\n current = [];\n continue;\n }\n const lw = tok.w!.toLowerCase();\n if (stop.has(lw) || tok.w!.length < minLen || /^\\d+$/.test(tok.w!)) {\n if (current.length) candidates.push(current);\n current = [];\n } else {\n current.push(tok.w!);\n }\n }\n if (current.length) candidates.push(current);\n\n // 2. Word scores: degree / frequency (classic RAKE).\n const freq = new Map<string, number>();\n const degree = new Map<string, number>();\n for (const phrase of candidates) {\n const deg = phrase.length - 1;\n for (const w of phrase) {\n const k = w.toLowerCase();\n freq.set(k, (freq.get(k) ?? 0) + 1);\n degree.set(k, (degree.get(k) ?? 0) + deg + 1);\n }\n }\n const wordScore = (w: string) => {\n const k = w.toLowerCase();\n const f = freq.get(k) ?? 1;\n return (degree.get(k) ?? f) / f;\n };\n\n // 3. Phrase scores; keep ≤ maxWords; dedupe by lowercase text; boost gazetteer.\n const gazLower = new Set(gaz.map((g) => g.toLowerCase()));\n const phraseScores = new Map<string, { phrase: string; score: number; order: number }>();\n let order = 0;\n for (const phrase of candidates) {\n if (phrase.length === 0 || phrase.length > maxWords) continue;\n const text2 = phrase.join(\" \");\n const key = text2.toLowerCase();\n let score = phrase.reduce((s, w) => s + wordScore(w), 0);\n if (gazLower.has(key) || gaz.some((g) => text2.includes(g))) score *= 1.5;\n const prev = phraseScores.get(key);\n if (prev) prev.score = Math.max(prev.score, score);\n else phraseScores.set(key, { phrase: text2, score, order: order++ });\n }\n const phrases = [...phraseScores.values()]\n .sort((a, b) => b.score - a.score || a.order - b.order)\n .slice(0, topK)\n .map((p) => ({ phrase: p.phrase, score: round(p.score) }));\n\n // 4. Tags + hashtags.\n const tags = phrases.map((p) => p.phrase);\n const hashtags = dedupe(tags.map(toHashtag).filter((h) => h.length > 1));\n\n // 5. Entities: Latin Title-Case runs + gazetteer hits.\n const entities = extractEntities(text, gaz);\n\n // 6. Category votes.\n const categories: CategoryVote[] = [];\n if (options.categories) {\n const lower = text.toLowerCase();\n for (const [cat, terms] of Object.entries(options.categories)) {\n let score = 0;\n for (const term of terms) {\n if (!term) continue;\n if (DEVANAGARI_RE.test(term)) {\n score += countOccurrences(text, term);\n } else {\n const re = new RegExp(`(?:^|[^a-z0-9])${escapeRe(term.toLowerCase())}(?:[^a-z0-9]|$)`, \"g\");\n score += (lower.match(re) || []).length;\n }\n }\n if (score > 0) categories.push({ category: cat, score });\n }\n categories.sort((a, b) => b.score - a.score || a.category.localeCompare(b.category));\n }\n\n return { phrases, tags, hashtags, entities, categories, language };\n}\n\ninterface StreamTok {\n w?: string;\n isBreak: boolean;\n}\nfunction tokenizeWithGaps(text: string): StreamTok[] {\n const out: StreamTok[] = [];\n let last = 0;\n let m: RegExpExecArray | null;\n WORD_RE.lastIndex = 0;\n while ((m = WORD_RE.exec(text)) !== null) {\n if (m.index > last) {\n // any non-word gap that contains sentence punctuation is a hard break\n const gap = text.slice(last, m.index);\n if (/[.!?;:,।॥\\n\\-–—/()\\[\\]\"'“”]/.test(gap)) out.push({ isBreak: true });\n }\n out.push({ w: m[0], isBreak: false });\n last = m.index + m[0].length;\n }\n return out;\n}\n\n/** Build a hashtag: strip #, CamelCase Latin words, keep Devanagari, keep only letters/marks/numbers. */\nexport function toHashtag(input: string): string {\n const cleaned = input.replace(/^#/, \"\");\n const parts = cleaned.split(/[\\s\\-_/]+/).filter(Boolean);\n const joined = parts\n .map((p) => (/^[a-z]/.test(p) ? p.charAt(0).toUpperCase() + p.slice(1) : p))\n .join(\"\");\n const kept = [...joined].filter((ch) => /[\\p{L}\\p{M}\\p{N}]/u.test(ch)).join(\"\");\n return kept.length >= 2 && kept.length <= 30 ? \"#\" + kept : \"\";\n}\n\nfunction extractEntities(text: string, gaz: string[]): EntityHit[] {\n const counts = new Map<string, number>();\n // Latin Title-Case runs of 1–4 words.\n const re = /\\b([A-Z][a-zA-Z]+(?:\\s+[A-Z][a-zA-Z]+){0,3})\\b/g;\n let m: RegExpExecArray | null;\n while ((m = re.exec(text)) !== null) {\n const e = m[1]!;\n // skip a lone word that starts a sentence and is common (heuristic: keep multiword or gazetteer)\n counts.set(e, (counts.get(e) ?? 0) + 1);\n }\n // Gazetteer (incl. Devanagari) always counted.\n for (const g of gaz) {\n if (!g) continue;\n const c = countOccurrences(text, g);\n if (c > 0) counts.set(g, Math.max(counts.get(g) ?? 0, c));\n }\n return [...counts.entries()]\n .map(([t, c]) => ({ text: t, count: c }))\n .sort((a, b) => b.count - a.count || a.text.localeCompare(b.text))\n .filter((e) => e.text.includes(\" \") || gaz.includes(e.text) || e.count > 1 || /[ऀ-ॿ]/.test(e.text));\n}\n\nfunction countOccurrences(hay: string, needle: string): number {\n if (!needle) return 0;\n let n = 0;\n let i = hay.indexOf(needle);\n while (i !== -1) {\n n++;\n i = hay.indexOf(needle, i + needle.length);\n }\n return n;\n}\nfunction dedupe(arr: string[]): string[] {\n const seen = new Set<string>();\n const out: string[] = [];\n for (const x of arr) {\n const k = x.toLowerCase();\n if (!seen.has(k)) {\n seen.add(k);\n out.push(x);\n }\n }\n return out;\n}\nfunction escapeRe(s: string): string {\n return s.replace(/[.*+?^${}()|[\\]\\\\]/g, \"\\\\$&\");\n}\nfunction round(n: number): number {\n return Math.round(n * 1000) / 1000;\n}\n"]}
|
package/dist/index.d.cts
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
declare const ENGLISH_STOPWORDS: string[];
|
|
2
|
+
declare const NEPALI_STOPWORDS: string[];
|
|
3
|
+
|
|
4
|
+
interface KeyphraseOptions {
|
|
5
|
+
/** "en", "ne", or "auto" (default) — picks built-in stopwords. */
|
|
6
|
+
language?: "en" | "ne" | "auto";
|
|
7
|
+
/** How many keyphrases/tags to return. Default 10. */
|
|
8
|
+
topK?: number;
|
|
9
|
+
/** Longest phrase, in words. Default 4. */
|
|
10
|
+
maxWords?: number;
|
|
11
|
+
/** Replace the built-in stopwords entirely. */
|
|
12
|
+
stopwords?: string[];
|
|
13
|
+
/** Add to the built-in stopwords. */
|
|
14
|
+
extraStopwords?: string[];
|
|
15
|
+
/** Known entities (people/places), en and ne forms — always surfaced and boosted. */
|
|
16
|
+
gazetteer?: string[];
|
|
17
|
+
/** Category lexicons: { category: [terms] }. Returns a vote per category by term hits. */
|
|
18
|
+
categories?: Record<string, string[]>;
|
|
19
|
+
/** Minimum characters for a candidate word. Default 2. */
|
|
20
|
+
minWordLength?: number;
|
|
21
|
+
}
|
|
22
|
+
interface ScoredPhrase {
|
|
23
|
+
phrase: string;
|
|
24
|
+
score: number;
|
|
25
|
+
}
|
|
26
|
+
interface EntityHit {
|
|
27
|
+
text: string;
|
|
28
|
+
count: number;
|
|
29
|
+
}
|
|
30
|
+
interface CategoryVote {
|
|
31
|
+
category: string;
|
|
32
|
+
score: number;
|
|
33
|
+
}
|
|
34
|
+
interface KeyphraseResult {
|
|
35
|
+
/** Top keyphrases by RAKE degree/frequency score. */
|
|
36
|
+
phrases: ScoredPhrase[];
|
|
37
|
+
/** Flat, de-duplicated tag strings (top phrases, normalized). */
|
|
38
|
+
tags: string[];
|
|
39
|
+
/** Hashtags built from the top tags (CamelCase for Latin, Devanagari kept whole). */
|
|
40
|
+
hashtags: string[];
|
|
41
|
+
/** Candidate named entities: Latin Title-Case runs + gazetteer hits. */
|
|
42
|
+
entities: EntityHit[];
|
|
43
|
+
/** Category votes (only when `categories` is provided), highest first. */
|
|
44
|
+
categories: CategoryVote[];
|
|
45
|
+
/** Detected/So-used language. */
|
|
46
|
+
language: "en" | "ne";
|
|
47
|
+
}
|
|
48
|
+
/** Extract keyphrases, tags, hashtags, entities and category votes — no LLM. */
|
|
49
|
+
declare function keyphrase(text: string, options?: KeyphraseOptions): KeyphraseResult;
|
|
50
|
+
/** Build a hashtag: strip #, CamelCase Latin words, keep Devanagari, keep only letters/marks/numbers. */
|
|
51
|
+
declare function toHashtag(input: string): string;
|
|
52
|
+
|
|
53
|
+
export { type CategoryVote, ENGLISH_STOPWORDS, type EntityHit, type KeyphraseOptions, type KeyphraseResult, NEPALI_STOPWORDS, type ScoredPhrase, keyphrase, toHashtag };
|
package/dist/index.d.ts
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
declare const ENGLISH_STOPWORDS: string[];
|
|
2
|
+
declare const NEPALI_STOPWORDS: string[];
|
|
3
|
+
|
|
4
|
+
interface KeyphraseOptions {
|
|
5
|
+
/** "en", "ne", or "auto" (default) — picks built-in stopwords. */
|
|
6
|
+
language?: "en" | "ne" | "auto";
|
|
7
|
+
/** How many keyphrases/tags to return. Default 10. */
|
|
8
|
+
topK?: number;
|
|
9
|
+
/** Longest phrase, in words. Default 4. */
|
|
10
|
+
maxWords?: number;
|
|
11
|
+
/** Replace the built-in stopwords entirely. */
|
|
12
|
+
stopwords?: string[];
|
|
13
|
+
/** Add to the built-in stopwords. */
|
|
14
|
+
extraStopwords?: string[];
|
|
15
|
+
/** Known entities (people/places), en and ne forms — always surfaced and boosted. */
|
|
16
|
+
gazetteer?: string[];
|
|
17
|
+
/** Category lexicons: { category: [terms] }. Returns a vote per category by term hits. */
|
|
18
|
+
categories?: Record<string, string[]>;
|
|
19
|
+
/** Minimum characters for a candidate word. Default 2. */
|
|
20
|
+
minWordLength?: number;
|
|
21
|
+
}
|
|
22
|
+
interface ScoredPhrase {
|
|
23
|
+
phrase: string;
|
|
24
|
+
score: number;
|
|
25
|
+
}
|
|
26
|
+
interface EntityHit {
|
|
27
|
+
text: string;
|
|
28
|
+
count: number;
|
|
29
|
+
}
|
|
30
|
+
interface CategoryVote {
|
|
31
|
+
category: string;
|
|
32
|
+
score: number;
|
|
33
|
+
}
|
|
34
|
+
interface KeyphraseResult {
|
|
35
|
+
/** Top keyphrases by RAKE degree/frequency score. */
|
|
36
|
+
phrases: ScoredPhrase[];
|
|
37
|
+
/** Flat, de-duplicated tag strings (top phrases, normalized). */
|
|
38
|
+
tags: string[];
|
|
39
|
+
/** Hashtags built from the top tags (CamelCase for Latin, Devanagari kept whole). */
|
|
40
|
+
hashtags: string[];
|
|
41
|
+
/** Candidate named entities: Latin Title-Case runs + gazetteer hits. */
|
|
42
|
+
entities: EntityHit[];
|
|
43
|
+
/** Category votes (only when `categories` is provided), highest first. */
|
|
44
|
+
categories: CategoryVote[];
|
|
45
|
+
/** Detected/So-used language. */
|
|
46
|
+
language: "en" | "ne";
|
|
47
|
+
}
|
|
48
|
+
/** Extract keyphrases, tags, hashtags, entities and category votes — no LLM. */
|
|
49
|
+
declare function keyphrase(text: string, options?: KeyphraseOptions): KeyphraseResult;
|
|
50
|
+
/** Build a hashtag: strip #, CamelCase Latin words, keep Devanagari, keep only letters/marks/numbers. */
|
|
51
|
+
declare function toHashtag(input: string): string;
|
|
52
|
+
|
|
53
|
+
export { type CategoryVote, ENGLISH_STOPWORDS, type EntityHit, type KeyphraseOptions, type KeyphraseResult, NEPALI_STOPWORDS, type ScoredPhrase, keyphrase, toHashtag };
|
package/dist/index.js
ADDED
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
// src/stopwords.ts
|
|
2
|
+
var ENGLISH_STOPWORDS = "a an and are as at be but by for from has have he her his i in is it its of on or that the their them they this to was were will with would you your we our us she him not no do does did done can could should may might must shall about after all also any because been before being between both during each few more most other over own same so some such than then there these those through under until up very what when where which who whom why how had having into out off down again further once here said says say according reported reports report told tuesday monday wednesday thursday friday saturday sunday am pm mr mrs ms dr new one two three per amid across".split(/\s+/);
|
|
3
|
+
var NEPALI_STOPWORDS = "\u0930 \u0915\u094B \u0915\u093E \u0915\u0940 \u092E\u093E \u0932\u0947 \u0939\u094B \u091B \u0925\u093F\u092F\u094B \u0925\u093F\u090F \u0939\u0941\u0928 \u0939\u0941\u0928\u094D\u091B \u092D\u0928\u0947 \u092D\u0928\u0940 \u092D\u0928\u094D\u0928\u0947 \u092A\u0928\u093F \u092F\u094B \u0924\u094D\u092F\u094B \u0924\u093F \u0924\u094D\u092F\u0938 \u092F\u0938 \u0917\u0930\u0947\u0915\u094B \u092D\u090F\u0915\u094B \u0932\u093E\u0917\u093F \u0924\u0925\u093E \u090F\u0935\u0902 \u0905\u0928\u0940 \u091C\u0938\u094D\u0924\u094B \u091C\u0938\u094D\u0924\u0948 \u0939\u094B\u0938\u094D \u091B\u0928\u094D \u0917\u0930\u094D\u0928 \u0917\u0930\u094D\u091B \u0917\u0930\u094D\u092F\u094B \u092D\u092F\u094B \u0905\u092C \u0938\u092C\u0948 \u0915\u0947\u0939\u0940 \u0906\u092B\u094D\u0928\u094B \u0909\u0928\u0940 \u0909\u0938\u0932\u0947 \u0909\u0928\u0932\u0947 \u092E\u093E\u0925\u093F \u0924\u0932 \u092D\u0928\u094D\u0926\u093E \u0938\u092E\u094D\u092C\u0928\u094D\u0927\u0940 \u092C\u093E\u091F \u0926\u0947\u0916\u093F \u0938\u092E\u094D\u092E".split(/\s+/).filter(Boolean);
|
|
4
|
+
|
|
5
|
+
// src/index.ts
|
|
6
|
+
var WORD_RE = /[\p{L}\p{N}][\p{L}\p{M}\p{N}]*/gu;
|
|
7
|
+
var DEVANAGARI_RE = /[ऀ-ॿ]/;
|
|
8
|
+
function detectLanguage(text) {
|
|
9
|
+
const deva = (text.match(/[ऀ-ॿ]/g) || []).length;
|
|
10
|
+
const latin = (text.match(/[A-Za-z]/g) || []).length;
|
|
11
|
+
return deva > latin ? "ne" : "en";
|
|
12
|
+
}
|
|
13
|
+
function keyphrase(text, options = {}) {
|
|
14
|
+
const language = options.language && options.language !== "auto" ? options.language : detectLanguage(text || "");
|
|
15
|
+
const topK = options.topK ?? 10;
|
|
16
|
+
const maxWords = options.maxWords ?? 4;
|
|
17
|
+
const minLen = options.minWordLength ?? 2;
|
|
18
|
+
const base = options.stopwords ?? (language === "ne" ? NEPALI_STOPWORDS : ENGLISH_STOPWORDS);
|
|
19
|
+
const stop = new Set([...base, ...options.extraStopwords ?? []].map((s) => s.toLowerCase()));
|
|
20
|
+
const gaz = (options.gazetteer ?? []).filter(Boolean);
|
|
21
|
+
const empty = { phrases: [], tags: [], hashtags: [], entities: [], categories: [], language };
|
|
22
|
+
if (!text || !text.trim()) return empty;
|
|
23
|
+
const candidates = [];
|
|
24
|
+
let current = [];
|
|
25
|
+
const tokenStream = tokenizeWithGaps(text);
|
|
26
|
+
for (const tok of tokenStream) {
|
|
27
|
+
if (tok.isBreak) {
|
|
28
|
+
if (current.length) candidates.push(current);
|
|
29
|
+
current = [];
|
|
30
|
+
continue;
|
|
31
|
+
}
|
|
32
|
+
const lw = tok.w.toLowerCase();
|
|
33
|
+
if (stop.has(lw) || tok.w.length < minLen || /^\d+$/.test(tok.w)) {
|
|
34
|
+
if (current.length) candidates.push(current);
|
|
35
|
+
current = [];
|
|
36
|
+
} else {
|
|
37
|
+
current.push(tok.w);
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
if (current.length) candidates.push(current);
|
|
41
|
+
const freq = /* @__PURE__ */ new Map();
|
|
42
|
+
const degree = /* @__PURE__ */ new Map();
|
|
43
|
+
for (const phrase of candidates) {
|
|
44
|
+
const deg = phrase.length - 1;
|
|
45
|
+
for (const w of phrase) {
|
|
46
|
+
const k = w.toLowerCase();
|
|
47
|
+
freq.set(k, (freq.get(k) ?? 0) + 1);
|
|
48
|
+
degree.set(k, (degree.get(k) ?? 0) + deg + 1);
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
const wordScore = (w) => {
|
|
52
|
+
const k = w.toLowerCase();
|
|
53
|
+
const f = freq.get(k) ?? 1;
|
|
54
|
+
return (degree.get(k) ?? f) / f;
|
|
55
|
+
};
|
|
56
|
+
const gazLower = new Set(gaz.map((g) => g.toLowerCase()));
|
|
57
|
+
const phraseScores = /* @__PURE__ */ new Map();
|
|
58
|
+
let order = 0;
|
|
59
|
+
for (const phrase of candidates) {
|
|
60
|
+
if (phrase.length === 0 || phrase.length > maxWords) continue;
|
|
61
|
+
const text2 = phrase.join(" ");
|
|
62
|
+
const key = text2.toLowerCase();
|
|
63
|
+
let score = phrase.reduce((s, w) => s + wordScore(w), 0);
|
|
64
|
+
if (gazLower.has(key) || gaz.some((g) => text2.includes(g))) score *= 1.5;
|
|
65
|
+
const prev = phraseScores.get(key);
|
|
66
|
+
if (prev) prev.score = Math.max(prev.score, score);
|
|
67
|
+
else phraseScores.set(key, { phrase: text2, score, order: order++ });
|
|
68
|
+
}
|
|
69
|
+
const phrases = [...phraseScores.values()].sort((a, b) => b.score - a.score || a.order - b.order).slice(0, topK).map((p) => ({ phrase: p.phrase, score: round(p.score) }));
|
|
70
|
+
const tags = phrases.map((p) => p.phrase);
|
|
71
|
+
const hashtags = dedupe(tags.map(toHashtag).filter((h) => h.length > 1));
|
|
72
|
+
const entities = extractEntities(text, gaz);
|
|
73
|
+
const categories = [];
|
|
74
|
+
if (options.categories) {
|
|
75
|
+
const lower = text.toLowerCase();
|
|
76
|
+
for (const [cat, terms] of Object.entries(options.categories)) {
|
|
77
|
+
let score = 0;
|
|
78
|
+
for (const term of terms) {
|
|
79
|
+
if (!term) continue;
|
|
80
|
+
if (DEVANAGARI_RE.test(term)) {
|
|
81
|
+
score += countOccurrences(text, term);
|
|
82
|
+
} else {
|
|
83
|
+
const re = new RegExp(`(?:^|[^a-z0-9])${escapeRe(term.toLowerCase())}(?:[^a-z0-9]|$)`, "g");
|
|
84
|
+
score += (lower.match(re) || []).length;
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
if (score > 0) categories.push({ category: cat, score });
|
|
88
|
+
}
|
|
89
|
+
categories.sort((a, b) => b.score - a.score || a.category.localeCompare(b.category));
|
|
90
|
+
}
|
|
91
|
+
return { phrases, tags, hashtags, entities, categories, language };
|
|
92
|
+
}
|
|
93
|
+
function tokenizeWithGaps(text) {
|
|
94
|
+
const out = [];
|
|
95
|
+
let last = 0;
|
|
96
|
+
let m;
|
|
97
|
+
WORD_RE.lastIndex = 0;
|
|
98
|
+
while ((m = WORD_RE.exec(text)) !== null) {
|
|
99
|
+
if (m.index > last) {
|
|
100
|
+
const gap = text.slice(last, m.index);
|
|
101
|
+
if (/[.!?;:,।॥\n\-–—/()\[\]"'“”]/.test(gap)) out.push({ isBreak: true });
|
|
102
|
+
}
|
|
103
|
+
out.push({ w: m[0], isBreak: false });
|
|
104
|
+
last = m.index + m[0].length;
|
|
105
|
+
}
|
|
106
|
+
return out;
|
|
107
|
+
}
|
|
108
|
+
function toHashtag(input) {
|
|
109
|
+
const cleaned = input.replace(/^#/, "");
|
|
110
|
+
const parts = cleaned.split(/[\s\-_/]+/).filter(Boolean);
|
|
111
|
+
const joined = parts.map((p) => /^[a-z]/.test(p) ? p.charAt(0).toUpperCase() + p.slice(1) : p).join("");
|
|
112
|
+
const kept = [...joined].filter((ch) => /[\p{L}\p{M}\p{N}]/u.test(ch)).join("");
|
|
113
|
+
return kept.length >= 2 && kept.length <= 30 ? "#" + kept : "";
|
|
114
|
+
}
|
|
115
|
+
function extractEntities(text, gaz) {
|
|
116
|
+
const counts = /* @__PURE__ */ new Map();
|
|
117
|
+
const re = /\b([A-Z][a-zA-Z]+(?:\s+[A-Z][a-zA-Z]+){0,3})\b/g;
|
|
118
|
+
let m;
|
|
119
|
+
while ((m = re.exec(text)) !== null) {
|
|
120
|
+
const e = m[1];
|
|
121
|
+
counts.set(e, (counts.get(e) ?? 0) + 1);
|
|
122
|
+
}
|
|
123
|
+
for (const g of gaz) {
|
|
124
|
+
if (!g) continue;
|
|
125
|
+
const c = countOccurrences(text, g);
|
|
126
|
+
if (c > 0) counts.set(g, Math.max(counts.get(g) ?? 0, c));
|
|
127
|
+
}
|
|
128
|
+
return [...counts.entries()].map(([t, c]) => ({ text: t, count: c })).sort((a, b) => b.count - a.count || a.text.localeCompare(b.text)).filter((e) => e.text.includes(" ") || gaz.includes(e.text) || e.count > 1 || /[ऀ-ॿ]/.test(e.text));
|
|
129
|
+
}
|
|
130
|
+
function countOccurrences(hay, needle) {
|
|
131
|
+
if (!needle) return 0;
|
|
132
|
+
let n = 0;
|
|
133
|
+
let i = hay.indexOf(needle);
|
|
134
|
+
while (i !== -1) {
|
|
135
|
+
n++;
|
|
136
|
+
i = hay.indexOf(needle, i + needle.length);
|
|
137
|
+
}
|
|
138
|
+
return n;
|
|
139
|
+
}
|
|
140
|
+
function dedupe(arr) {
|
|
141
|
+
const seen = /* @__PURE__ */ new Set();
|
|
142
|
+
const out = [];
|
|
143
|
+
for (const x of arr) {
|
|
144
|
+
const k = x.toLowerCase();
|
|
145
|
+
if (!seen.has(k)) {
|
|
146
|
+
seen.add(k);
|
|
147
|
+
out.push(x);
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
return out;
|
|
151
|
+
}
|
|
152
|
+
function escapeRe(s) {
|
|
153
|
+
return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
154
|
+
}
|
|
155
|
+
function round(n) {
|
|
156
|
+
return Math.round(n * 1e3) / 1e3;
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
export { ENGLISH_STOPWORDS, NEPALI_STOPWORDS, keyphrase, toHashtag };
|
|
160
|
+
//# sourceMappingURL=index.js.map
|
|
161
|
+
//# sourceMappingURL=index.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"sources":["../src/stopwords.ts","../src/index.ts"],"names":[],"mappings":";AACO,IAAM,iBAAA,GACX,wpBAAA,CAQA,KAAA,CAAM,KAAK;AAGN,IAAM,mBACX,4hCAAA,CAUA,KAAA,CAAM,KAAK,CAAA,CAAE,OAAO,OAAO;;;AC0B7B,IAAM,OAAA,GAAU,oCAAA;AAChB,IAAM,aAAA,GAAgB,OAAA;AAEtB,SAAS,eAAe,IAAA,EAA2B;AACjD,EAAA,MAAM,QAAQ,IAAA,CAAK,KAAA,CAAM,QAAQ,CAAA,IAAK,EAAC,EAAG,MAAA;AAC1C,EAAA,MAAM,SAAS,IAAA,CAAK,KAAA,CAAM,WAAW,CAAA,IAAK,EAAC,EAAG,MAAA;AAC9C,EAAA,OAAO,IAAA,GAAO,QAAQ,IAAA,GAAO,IAAA;AAC/B;AAWO,SAAS,SAAA,CAAU,IAAA,EAAc,OAAA,GAA4B,EAAC,EAAoB;AACvF,EAAA,MAAM,QAAA,GAAW,OAAA,CAAQ,QAAA,IAAY,OAAA,CAAQ,QAAA,KAAa,SAAS,OAAA,CAAQ,QAAA,GAAW,cAAA,CAAe,IAAA,IAAQ,EAAE,CAAA;AAC/G,EAAA,MAAM,IAAA,GAAO,QAAQ,IAAA,IAAQ,EAAA;AAC7B,EAAA,MAAM,QAAA,GAAW,QAAQ,QAAA,IAAY,CAAA;AACrC,EAAA,MAAM,MAAA,GAAS,QAAQ,aAAA,IAAiB,CAAA;AACxC,EAAA,MAAM,IAAA,GAAO,OAAA,CAAQ,SAAA,KAAc,QAAA,KAAa,OAAO,gBAAA,GAAmB,iBAAA,CAAA;AAC1E,EAAA,MAAM,OAAO,IAAI,GAAA,CAAI,CAAC,GAAG,IAAA,EAAM,GAAI,OAAA,CAAQ,cAAA,IAAkB,EAAG,EAAE,GAAA,CAAI,CAAC,MAAM,CAAA,CAAE,WAAA,EAAa,CAAC,CAAA;AAC7F,EAAA,MAAM,OAAO,OAAA,CAAQ,SAAA,IAAa,EAAC,EAAG,OAAO,OAAO,CAAA;AAEpD,EAAA,MAAM,QAAyB,EAAE,OAAA,EAAS,EAAC,EAAG,MAAM,EAAC,EAAG,QAAA,EAAU,IAAI,QAAA,EAAU,IAAI,UAAA,EAAY,IAAI,QAAA,EAAS;AAC7G,EAAA,IAAI,CAAC,IAAA,IAAQ,CAAC,IAAA,CAAK,IAAA,IAAQ,OAAO,KAAA;AAGlC,EAAA,MAAM,aAAyB,EAAC;AAChC,EAAA,IAAI,UAAoB,EAAC;AAEzB,EAAA,MAAM,WAAA,GAAc,iBAAiB,IAAI,CAAA;AACzC,EAAA,KAAA,MAAW,OAAO,WAAA,EAAa;AAC7B,IAAA,IAAI,IAAI,OAAA,EAAS;AACf,MAAA,IAAI,OAAA,CAAQ,MAAA,EAAQ,UAAA,CAAW,IAAA,CAAK,OAAO,CAAA;AAC3C,MAAA,OAAA,GAAU,EAAC;AACX,MAAA;AAAA,IACF;AACA,IAAA,MAAM,EAAA,GAAK,GAAA,CAAI,CAAA,CAAG,WAAA,EAAY;AAC9B,IAAA,IAAI,IAAA,CAAK,GAAA,CAAI,EAAE,CAAA,IAAK,GAAA,CAAI,CAAA,CAAG,MAAA,GAAS,MAAA,IAAU,OAAA,CAAQ,IAAA,CAAK,GAAA,CAAI,CAAE,CAAA,EAAG;AAClE,MAAA,IAAI,OAAA,CAAQ,MAAA,EAAQ,UAAA,CAAW,IAAA,CAAK,OAAO,CAAA;AAC3C,MAAA,OAAA,GAAU,EAAC;AAAA,IACb,CAAA,MAAO;AACL,MAAA,OAAA,CAAQ,IAAA,CAAK,IAAI,CAAE,CAAA;AAAA,IACrB;AAAA,EACF;AACA,EAAA,IAAI,OAAA,CAAQ,MAAA,EAAQ,UAAA,CAAW,IAAA,CAAK,OAAO,CAAA;AAG3C,EAAA,MAAM,IAAA,uBAAW,GAAA,EAAoB;AACrC,EAAA,MAAM,MAAA,uBAAa,GAAA,EAAoB;AACvC,EAAA,KAAA,MAAW,UAAU,UAAA,EAAY;AAC/B,IAAA,MAAM,GAAA,GAAM,OAAO,MAAA,GAAS,CAAA;AAC5B,IAAA,KAAA,MAAW,KAAK,MAAA,EAAQ;AACtB,MAAA,MAAM,CAAA,GAAI,EAAE,WAAA,EAAY;AACxB,MAAA,IAAA,CAAK,IAAI,CAAA,EAAA,CAAI,IAAA,CAAK,IAAI,CAAC,CAAA,IAAK,KAAK,CAAC,CAAA;AAClC,MAAA,MAAA,CAAO,GAAA,CAAI,IAAI,MAAA,CAAO,GAAA,CAAI,CAAC,CAAA,IAAK,CAAA,IAAK,MAAM,CAAC,CAAA;AAAA,IAC9C;AAAA,EACF;AACA,EAAA,MAAM,SAAA,GAAY,CAAC,CAAA,KAAc;AAC/B,IAAA,MAAM,CAAA,GAAI,EAAE,WAAA,EAAY;AACxB,IAAA,MAAM,CAAA,GAAI,IAAA,CAAK,GAAA,CAAI,CAAC,CAAA,IAAK,CAAA;AACzB,IAAA,OAAA,CAAQ,MAAA,CAAO,GAAA,CAAI,CAAC,CAAA,IAAK,CAAA,IAAK,CAAA;AAAA,EAChC,CAAA;AAGA,EAAA,MAAM,QAAA,GAAW,IAAI,GAAA,CAAI,GAAA,CAAI,GAAA,CAAI,CAAC,CAAA,KAAM,CAAA,CAAE,WAAA,EAAa,CAAC,CAAA;AACxD,EAAA,MAAM,YAAA,uBAAmB,GAAA,EAA8D;AACvF,EAAA,IAAI,KAAA,GAAQ,CAAA;AACZ,EAAA,KAAA,MAAW,UAAU,UAAA,EAAY;AAC/B,IAAA,IAAI,MAAA,CAAO,MAAA,KAAW,CAAA,IAAK,MAAA,CAAO,SAAS,QAAA,EAAU;AACrD,IAAA,MAAM,KAAA,GAAQ,MAAA,CAAO,IAAA,CAAK,GAAG,CAAA;AAC7B,IAAA,MAAM,GAAA,GAAM,MAAM,WAAA,EAAY;AAC9B,IAAA,IAAI,KAAA,GAAQ,MAAA,CAAO,MAAA,CAAO,CAAC,CAAA,EAAG,MAAM,CAAA,GAAI,SAAA,CAAU,CAAC,CAAA,EAAG,CAAC,CAAA;AACvD,IAAA,IAAI,QAAA,CAAS,GAAA,CAAI,GAAG,CAAA,IAAK,GAAA,CAAI,IAAA,CAAK,CAAC,CAAA,KAAM,KAAA,CAAM,QAAA,CAAS,CAAC,CAAC,GAAG,KAAA,IAAS,GAAA;AACtE,IAAA,MAAM,IAAA,GAAO,YAAA,CAAa,GAAA,CAAI,GAAG,CAAA;AACjC,IAAA,IAAI,MAAM,IAAA,CAAK,KAAA,GAAQ,KAAK,GAAA,CAAI,IAAA,CAAK,OAAO,KAAK,CAAA;AAAA,SAC5C,YAAA,CAAa,IAAI,GAAA,EAAK,EAAE,QAAQ,KAAA,EAAO,KAAA,EAAO,KAAA,EAAO,KAAA,EAAA,EAAS,CAAA;AAAA,EACrE;AACA,EAAA,MAAM,UAAU,CAAC,GAAG,YAAA,CAAa,MAAA,EAAQ,CAAA,CACtC,IAAA,CAAK,CAAC,CAAA,EAAG,MAAM,CAAA,CAAE,KAAA,GAAQ,CAAA,CAAE,KAAA,IAAS,EAAE,KAAA,GAAQ,CAAA,CAAE,KAAK,CAAA,CACrD,MAAM,CAAA,EAAG,IAAI,CAAA,CACb,GAAA,CAAI,CAAC,CAAA,MAAO,EAAE,MAAA,EAAQ,CAAA,CAAE,QAAQ,KAAA,EAAO,KAAA,CAAM,CAAA,CAAE,KAAK,GAAE,CAAE,CAAA;AAG3D,EAAA,MAAM,OAAO,OAAA,CAAQ,GAAA,CAAI,CAAC,CAAA,KAAM,EAAE,MAAM,CAAA;AACxC,EAAA,MAAM,QAAA,GAAW,MAAA,CAAO,IAAA,CAAK,GAAA,CAAI,SAAS,CAAA,CAAE,MAAA,CAAO,CAAC,CAAA,KAAM,CAAA,CAAE,MAAA,GAAS,CAAC,CAAC,CAAA;AAGvE,EAAA,MAAM,QAAA,GAAW,eAAA,CAAgB,IAAA,EAAM,GAAG,CAAA;AAG1C,EAAA,MAAM,aAA6B,EAAC;AACpC,EAAA,IAAI,QAAQ,UAAA,EAAY;AACtB,IAAA,MAAM,KAAA,GAAQ,KAAK,WAAA,EAAY;AAC/B,IAAA,KAAA,MAAW,CAAC,KAAK,KAAK,CAAA,IAAK,OAAO,OAAA,CAAQ,OAAA,CAAQ,UAAU,CAAA,EAAG;AAC7D,MAAA,IAAI,KAAA,GAAQ,CAAA;AACZ,MAAA,KAAA,MAAW,QAAQ,KAAA,EAAO;AACxB,QAAA,IAAI,CAAC,IAAA,EAAM;AACX,QAAA,IAAI,aAAA,CAAc,IAAA,CAAK,IAAI,CAAA,EAAG;AAC5B,UAAA,KAAA,IAAS,gBAAA,CAAiB,MAAM,IAAI,CAAA;AAAA,QACtC,CAAA,MAAO;AACL,UAAA,MAAM,EAAA,GAAK,IAAI,MAAA,CAAO,CAAA,eAAA,EAAkB,QAAA,CAAS,KAAK,WAAA,EAAa,CAAC,CAAA,eAAA,CAAA,EAAmB,GAAG,CAAA;AAC1F,UAAA,KAAA,IAAA,CAAU,KAAA,CAAM,KAAA,CAAM,EAAE,CAAA,IAAK,EAAC,EAAG,MAAA;AAAA,QACnC;AAAA,MACF;AACA,MAAA,IAAI,KAAA,GAAQ,GAAG,UAAA,CAAW,IAAA,CAAK,EAAE,QAAA,EAAU,GAAA,EAAK,OAAO,CAAA;AAAA,IACzD;AACA,IAAA,UAAA,CAAW,IAAA,CAAK,CAAC,CAAA,EAAG,CAAA,KAAM,CAAA,CAAE,KAAA,GAAQ,CAAA,CAAE,KAAA,IAAS,CAAA,CAAE,QAAA,CAAS,aAAA,CAAc,CAAA,CAAE,QAAQ,CAAC,CAAA;AAAA,EACrF;AAEA,EAAA,OAAO,EAAE,OAAA,EAAS,IAAA,EAAM,QAAA,EAAU,QAAA,EAAU,YAAY,QAAA,EAAS;AACnE;AAMA,SAAS,iBAAiB,IAAA,EAA2B;AACnD,EAAA,MAAM,MAAmB,EAAC;AAC1B,EAAA,IAAI,IAAA,GAAO,CAAA;AACX,EAAA,IAAI,CAAA;AACJ,EAAA,OAAA,CAAQ,SAAA,GAAY,CAAA;AACpB,EAAA,OAAA,CAAQ,CAAA,GAAI,OAAA,CAAQ,IAAA,CAAK,IAAI,OAAO,IAAA,EAAM;AACxC,IAAA,IAAI,CAAA,CAAE,QAAQ,IAAA,EAAM;AAElB,MAAA,MAAM,GAAA,GAAM,IAAA,CAAK,KAAA,CAAM,IAAA,EAAM,EAAE,KAAK,CAAA;AACpC,MAAA,IAAI,6BAAA,CAA8B,KAAK,GAAG,CAAA,MAAO,IAAA,CAAK,EAAE,OAAA,EAAS,IAAA,EAAM,CAAA;AAAA,IACzE;AACA,IAAA,GAAA,CAAI,IAAA,CAAK,EAAE,CAAA,EAAG,CAAA,CAAE,CAAC,CAAA,EAAG,OAAA,EAAS,OAAO,CAAA;AACpC,IAAA,IAAA,GAAO,CAAA,CAAE,KAAA,GAAQ,CAAA,CAAE,CAAC,CAAA,CAAE,MAAA;AAAA,EACxB;AACA,EAAA,OAAO,GAAA;AACT;AAGO,SAAS,UAAU,KAAA,EAAuB;AAC/C,EAAA,MAAM,OAAA,GAAU,KAAA,CAAM,OAAA,CAAQ,IAAA,EAAM,EAAE,CAAA;AACtC,EAAA,MAAM,QAAQ,OAAA,CAAQ,KAAA,CAAM,WAAW,CAAA,CAAE,OAAO,OAAO,CAAA;AACvD,EAAA,MAAM,MAAA,GAAS,MACZ,GAAA,CAAI,CAAC,MAAO,QAAA,CAAS,IAAA,CAAK,CAAC,CAAA,GAAI,CAAA,CAAE,MAAA,CAAO,CAAC,CAAA,CAAE,WAAA,KAAgB,CAAA,CAAE,KAAA,CAAM,CAAC,CAAA,GAAI,CAAE,CAAA,CAC1E,IAAA,CAAK,EAAE,CAAA;AACV,EAAA,MAAM,IAAA,GAAO,CAAC,GAAG,MAAM,EAAE,MAAA,CAAO,CAAC,EAAA,KAAO,oBAAA,CAAqB,IAAA,CAAK,EAAE,CAAC,CAAA,CAAE,KAAK,EAAE,CAAA;AAC9E,EAAA,OAAO,KAAK,MAAA,IAAU,CAAA,IAAK,KAAK,MAAA,IAAU,EAAA,GAAK,MAAM,IAAA,GAAO,EAAA;AAC9D;AAEA,SAAS,eAAA,CAAgB,MAAc,GAAA,EAA4B;AACjE,EAAA,MAAM,MAAA,uBAAa,GAAA,EAAoB;AAEvC,EAAA,MAAM,EAAA,GAAK,iDAAA;AACX,EAAA,IAAI,CAAA;AACJ,EAAA,OAAA,CAAQ,CAAA,GAAI,EAAA,CAAG,IAAA,CAAK,IAAI,OAAO,IAAA,EAAM;AACnC,IAAA,MAAM,CAAA,GAAI,EAAE,CAAC,CAAA;AAEb,IAAA,MAAA,CAAO,IAAI,CAAA,EAAA,CAAI,MAAA,CAAO,IAAI,CAAC,CAAA,IAAK,KAAK,CAAC,CAAA;AAAA,EACxC;AAEA,EAAA,KAAA,MAAW,KAAK,GAAA,EAAK;AACnB,IAAA,IAAI,CAAC,CAAA,EAAG;AACR,IAAA,MAAM,CAAA,GAAI,gBAAA,CAAiB,IAAA,EAAM,CAAC,CAAA;AAClC,IAAA,IAAI,CAAA,GAAI,CAAA,EAAG,MAAA,CAAO,GAAA,CAAI,CAAA,EAAG,IAAA,CAAK,GAAA,CAAI,MAAA,CAAO,GAAA,CAAI,CAAC,CAAA,IAAK,CAAA,EAAG,CAAC,CAAC,CAAA;AAAA,EAC1D;AACA,EAAA,OAAO,CAAC,GAAG,MAAA,CAAO,OAAA,EAAS,EACxB,GAAA,CAAI,CAAC,CAAC,CAAA,EAAG,CAAC,CAAA,MAAO,EAAE,IAAA,EAAM,CAAA,EAAG,KAAA,EAAO,CAAA,EAAE,CAAE,CAAA,CACvC,IAAA,CAAK,CAAC,CAAA,EAAG,CAAA,KAAM,CAAA,CAAE,KAAA,GAAQ,CAAA,CAAE,KAAA,IAAS,EAAE,IAAA,CAAK,aAAA,CAAc,CAAA,CAAE,IAAI,CAAC,CAAA,CAChE,OAAO,CAAC,CAAA,KAAM,CAAA,CAAE,IAAA,CAAK,QAAA,CAAS,GAAG,KAAK,GAAA,CAAI,QAAA,CAAS,CAAA,CAAE,IAAI,CAAA,IAAK,CAAA,CAAE,KAAA,GAAQ,CAAA,IAAK,OAAA,CAAQ,IAAA,CAAK,CAAA,CAAE,IAAI,CAAC,CAAA;AACtG;AAEA,SAAS,gBAAA,CAAiB,KAAa,MAAA,EAAwB;AAC7D,EAAA,IAAI,CAAC,QAAQ,OAAO,CAAA;AACpB,EAAA,IAAI,CAAA,GAAI,CAAA;AACR,EAAA,IAAI,CAAA,GAAI,GAAA,CAAI,OAAA,CAAQ,MAAM,CAAA;AAC1B,EAAA,OAAO,MAAM,EAAA,EAAI;AACf,IAAA,CAAA,EAAA;AACA,IAAA,CAAA,GAAI,GAAA,CAAI,OAAA,CAAQ,MAAA,EAAQ,CAAA,GAAI,OAAO,MAAM,CAAA;AAAA,EAC3C;AACA,EAAA,OAAO,CAAA;AACT;AACA,SAAS,OAAO,GAAA,EAAyB;AACvC,EAAA,MAAM,IAAA,uBAAW,GAAA,EAAY;AAC7B,EAAA,MAAM,MAAgB,EAAC;AACvB,EAAA,KAAA,MAAW,KAAK,GAAA,EAAK;AACnB,IAAA,MAAM,CAAA,GAAI,EAAE,WAAA,EAAY;AACxB,IAAA,IAAI,CAAC,IAAA,CAAK,GAAA,CAAI,CAAC,CAAA,EAAG;AAChB,MAAA,IAAA,CAAK,IAAI,CAAC,CAAA;AACV,MAAA,GAAA,CAAI,KAAK,CAAC,CAAA;AAAA,IACZ;AAAA,EACF;AACA,EAAA,OAAO,GAAA;AACT;AACA,SAAS,SAAS,CAAA,EAAmB;AACnC,EAAA,OAAO,CAAA,CAAE,OAAA,CAAQ,qBAAA,EAAuB,MAAM,CAAA;AAChD;AACA,SAAS,MAAM,CAAA,EAAmB;AAChC,EAAA,OAAO,IAAA,CAAK,KAAA,CAAM,CAAA,GAAI,GAAI,CAAA,GAAI,GAAA;AAChC","file":"index.js","sourcesContent":["// Compact, high-frequency stopword lists. Extend via options.extraStopwords.\nexport const ENGLISH_STOPWORDS: string[] = (\n \"a an and are as at be but by for from has have he her his i in is it its of on or \" +\n \"that the their them they this to was were will with would you your we our us she him \" +\n \"not no do does did done can could should may might must shall about after all also \" +\n \"any because been before being between both during each few more most other over own \" +\n \"same so some such than then there these those through under until up very what when \" +\n \"where which who whom why how had having into out off down again further once here \" +\n \"said says say according reported reports report told tuesday monday wednesday thursday \" +\n \"friday saturday sunday am pm mr mrs ms dr new one two three per amid across\"\n).split(/\\s+/);\n\n// Common Nepali (Devanagari) function words and news filler.\nexport const NEPALI_STOPWORDS: string[] = (\n \"र को का की मा ले हो \" +\n \"छ थियो थिए हुन हुन्छ \" +\n \"भने भनी भन्ने पनि यो \" +\n \"त्यो ति त्यस यस गरेको \" +\n \"भएको लागि तथा एवं अनी \" +\n \"जस्तो जस्तै होस् छन् \" +\n \"गर्न गर्छ गर्यो भयो \" +\n \"अब सबै केही आफ्नो उनी \" +\n \"उसले उनले माथि तल भन्दा \" +\n \"सम्बन्धी बाट देखि सम्म\"\n).split(/\\s+/).filter(Boolean);\n","import { ENGLISH_STOPWORDS, NEPALI_STOPWORDS } from \"./stopwords.js\";\n\nexport { ENGLISH_STOPWORDS, NEPALI_STOPWORDS } from \"./stopwords.js\";\n\nexport interface KeyphraseOptions {\n /** \"en\", \"ne\", or \"auto\" (default) — picks built-in stopwords. */\n language?: \"en\" | \"ne\" | \"auto\";\n /** How many keyphrases/tags to return. Default 10. */\n topK?: number;\n /** Longest phrase, in words. Default 4. */\n maxWords?: number;\n /** Replace the built-in stopwords entirely. */\n stopwords?: string[];\n /** Add to the built-in stopwords. */\n extraStopwords?: string[];\n /** Known entities (people/places), en and ne forms — always surfaced and boosted. */\n gazetteer?: string[];\n /** Category lexicons: { category: [terms] }. Returns a vote per category by term hits. */\n categories?: Record<string, string[]>;\n /** Minimum characters for a candidate word. Default 2. */\n minWordLength?: number;\n}\n\nexport interface ScoredPhrase {\n phrase: string;\n score: number;\n}\nexport interface EntityHit {\n text: string;\n count: number;\n}\nexport interface CategoryVote {\n category: string;\n score: number;\n}\nexport interface KeyphraseResult {\n /** Top keyphrases by RAKE degree/frequency score. */\n phrases: ScoredPhrase[];\n /** Flat, de-duplicated tag strings (top phrases, normalized). */\n tags: string[];\n /** Hashtags built from the top tags (CamelCase for Latin, Devanagari kept whole). */\n hashtags: string[];\n /** Candidate named entities: Latin Title-Case runs + gazetteer hits. */\n entities: EntityHit[];\n /** Category votes (only when `categories` is provided), highest first. */\n categories: CategoryVote[];\n /** Detected/So-used language. */\n language: \"en\" | \"ne\";\n}\n\nconst WORD_RE = /[\\p{L}\\p{N}][\\p{L}\\p{M}\\p{N}]*/gu;\nconst DEVANAGARI_RE = /[ऀ-ॿ]/;\n\nfunction detectLanguage(text: string): \"en\" | \"ne\" {\n const deva = (text.match(/[ऀ-ॿ]/g) || []).length;\n const latin = (text.match(/[A-Za-z]/g) || []).length;\n return deva > latin ? \"ne\" : \"en\";\n}\n\nfunction words(text: string): { w: string; start: number }[] {\n const out: { w: string; start: number }[] = [];\n let m: RegExpExecArray | null;\n WORD_RE.lastIndex = 0;\n while ((m = WORD_RE.exec(text)) !== null) out.push({ w: m[0], start: m.index });\n return out;\n}\n\n/** Extract keyphrases, tags, hashtags, entities and category votes — no LLM. */\nexport function keyphrase(text: string, options: KeyphraseOptions = {}): KeyphraseResult {\n const language = options.language && options.language !== \"auto\" ? options.language : detectLanguage(text || \"\");\n const topK = options.topK ?? 10;\n const maxWords = options.maxWords ?? 4;\n const minLen = options.minWordLength ?? 2;\n const base = options.stopwords ?? (language === \"ne\" ? NEPALI_STOPWORDS : ENGLISH_STOPWORDS);\n const stop = new Set([...base, ...(options.extraStopwords ?? [])].map((s) => s.toLowerCase()));\n const gaz = (options.gazetteer ?? []).filter(Boolean);\n\n const empty: KeyphraseResult = { phrases: [], tags: [], hashtags: [], entities: [], categories: [], language };\n if (!text || !text.trim()) return empty;\n\n // 1. RAKE: break into candidate phrases at stopwords and non-word chars.\n const candidates: string[][] = [];\n let current: string[] = [];\n // Walk the raw text so punctuation also breaks phrases.\n const tokenStream = tokenizeWithGaps(text);\n for (const tok of tokenStream) {\n if (tok.isBreak) {\n if (current.length) candidates.push(current);\n current = [];\n continue;\n }\n const lw = tok.w!.toLowerCase();\n if (stop.has(lw) || tok.w!.length < minLen || /^\\d+$/.test(tok.w!)) {\n if (current.length) candidates.push(current);\n current = [];\n } else {\n current.push(tok.w!);\n }\n }\n if (current.length) candidates.push(current);\n\n // 2. Word scores: degree / frequency (classic RAKE).\n const freq = new Map<string, number>();\n const degree = new Map<string, number>();\n for (const phrase of candidates) {\n const deg = phrase.length - 1;\n for (const w of phrase) {\n const k = w.toLowerCase();\n freq.set(k, (freq.get(k) ?? 0) + 1);\n degree.set(k, (degree.get(k) ?? 0) + deg + 1);\n }\n }\n const wordScore = (w: string) => {\n const k = w.toLowerCase();\n const f = freq.get(k) ?? 1;\n return (degree.get(k) ?? f) / f;\n };\n\n // 3. Phrase scores; keep ≤ maxWords; dedupe by lowercase text; boost gazetteer.\n const gazLower = new Set(gaz.map((g) => g.toLowerCase()));\n const phraseScores = new Map<string, { phrase: string; score: number; order: number }>();\n let order = 0;\n for (const phrase of candidates) {\n if (phrase.length === 0 || phrase.length > maxWords) continue;\n const text2 = phrase.join(\" \");\n const key = text2.toLowerCase();\n let score = phrase.reduce((s, w) => s + wordScore(w), 0);\n if (gazLower.has(key) || gaz.some((g) => text2.includes(g))) score *= 1.5;\n const prev = phraseScores.get(key);\n if (prev) prev.score = Math.max(prev.score, score);\n else phraseScores.set(key, { phrase: text2, score, order: order++ });\n }\n const phrases = [...phraseScores.values()]\n .sort((a, b) => b.score - a.score || a.order - b.order)\n .slice(0, topK)\n .map((p) => ({ phrase: p.phrase, score: round(p.score) }));\n\n // 4. Tags + hashtags.\n const tags = phrases.map((p) => p.phrase);\n const hashtags = dedupe(tags.map(toHashtag).filter((h) => h.length > 1));\n\n // 5. Entities: Latin Title-Case runs + gazetteer hits.\n const entities = extractEntities(text, gaz);\n\n // 6. Category votes.\n const categories: CategoryVote[] = [];\n if (options.categories) {\n const lower = text.toLowerCase();\n for (const [cat, terms] of Object.entries(options.categories)) {\n let score = 0;\n for (const term of terms) {\n if (!term) continue;\n if (DEVANAGARI_RE.test(term)) {\n score += countOccurrences(text, term);\n } else {\n const re = new RegExp(`(?:^|[^a-z0-9])${escapeRe(term.toLowerCase())}(?:[^a-z0-9]|$)`, \"g\");\n score += (lower.match(re) || []).length;\n }\n }\n if (score > 0) categories.push({ category: cat, score });\n }\n categories.sort((a, b) => b.score - a.score || a.category.localeCompare(b.category));\n }\n\n return { phrases, tags, hashtags, entities, categories, language };\n}\n\ninterface StreamTok {\n w?: string;\n isBreak: boolean;\n}\nfunction tokenizeWithGaps(text: string): StreamTok[] {\n const out: StreamTok[] = [];\n let last = 0;\n let m: RegExpExecArray | null;\n WORD_RE.lastIndex = 0;\n while ((m = WORD_RE.exec(text)) !== null) {\n if (m.index > last) {\n // any non-word gap that contains sentence punctuation is a hard break\n const gap = text.slice(last, m.index);\n if (/[.!?;:,।॥\\n\\-–—/()\\[\\]\"'“”]/.test(gap)) out.push({ isBreak: true });\n }\n out.push({ w: m[0], isBreak: false });\n last = m.index + m[0].length;\n }\n return out;\n}\n\n/** Build a hashtag: strip #, CamelCase Latin words, keep Devanagari, keep only letters/marks/numbers. */\nexport function toHashtag(input: string): string {\n const cleaned = input.replace(/^#/, \"\");\n const parts = cleaned.split(/[\\s\\-_/]+/).filter(Boolean);\n const joined = parts\n .map((p) => (/^[a-z]/.test(p) ? p.charAt(0).toUpperCase() + p.slice(1) : p))\n .join(\"\");\n const kept = [...joined].filter((ch) => /[\\p{L}\\p{M}\\p{N}]/u.test(ch)).join(\"\");\n return kept.length >= 2 && kept.length <= 30 ? \"#\" + kept : \"\";\n}\n\nfunction extractEntities(text: string, gaz: string[]): EntityHit[] {\n const counts = new Map<string, number>();\n // Latin Title-Case runs of 1–4 words.\n const re = /\\b([A-Z][a-zA-Z]+(?:\\s+[A-Z][a-zA-Z]+){0,3})\\b/g;\n let m: RegExpExecArray | null;\n while ((m = re.exec(text)) !== null) {\n const e = m[1]!;\n // skip a lone word that starts a sentence and is common (heuristic: keep multiword or gazetteer)\n counts.set(e, (counts.get(e) ?? 0) + 1);\n }\n // Gazetteer (incl. Devanagari) always counted.\n for (const g of gaz) {\n if (!g) continue;\n const c = countOccurrences(text, g);\n if (c > 0) counts.set(g, Math.max(counts.get(g) ?? 0, c));\n }\n return [...counts.entries()]\n .map(([t, c]) => ({ text: t, count: c }))\n .sort((a, b) => b.count - a.count || a.text.localeCompare(b.text))\n .filter((e) => e.text.includes(\" \") || gaz.includes(e.text) || e.count > 1 || /[ऀ-ॿ]/.test(e.text));\n}\n\nfunction countOccurrences(hay: string, needle: string): number {\n if (!needle) return 0;\n let n = 0;\n let i = hay.indexOf(needle);\n while (i !== -1) {\n n++;\n i = hay.indexOf(needle, i + needle.length);\n }\n return n;\n}\nfunction dedupe(arr: string[]): string[] {\n const seen = new Set<string>();\n const out: string[] = [];\n for (const x of arr) {\n const k = x.toLowerCase();\n if (!seen.has(k)) {\n seen.add(k);\n out.push(x);\n }\n }\n return out;\n}\nfunction escapeRe(s: string): string {\n return s.replace(/[.*+?^${}()|[\\]\\\\]/g, \"\\\\$&\");\n}\nfunction round(n: number): number {\n return Math.round(n * 1000) / 1000;\n}\n"]}
|
package/package.json
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "@lacspace/keyphrase",
|
|
3
|
+
"version": "1.0.0",
|
|
4
|
+
"description": "Extractive keyphrases, tags, hashtags, named entities and category votes — a zero-dependency RAKE + TF-IDF engine with built-in English and Nepali stopwords, Devanagari-aware, so you can stop asking an LLM to generate tags/hashtags/entities. Deterministic, isomorphic, typed.",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"main": "./dist/index.cjs",
|
|
7
|
+
"module": "./dist/index.js",
|
|
8
|
+
"types": "./dist/index.d.ts",
|
|
9
|
+
"exports": {
|
|
10
|
+
".": {
|
|
11
|
+
"import": { "types": "./dist/index.d.ts", "default": "./dist/index.js" },
|
|
12
|
+
"require": { "types": "./dist/index.d.cts", "default": "./dist/index.cjs" }
|
|
13
|
+
}
|
|
14
|
+
},
|
|
15
|
+
"files": ["dist"],
|
|
16
|
+
"sideEffects": false,
|
|
17
|
+
"scripts": { "build": "tsup", "prepublishOnly": "npm run build" },
|
|
18
|
+
"keywords": ["keyphrase", "keyword-extraction", "rake", "tfidf", "tags", "hashtags", "named-entity", "category", "extractive", "nepali", "devanagari", "stopwords", "zero-dependency", "isomorphic", "typescript"],
|
|
19
|
+
"author": "Lacspace <contact@lacspace.com>",
|
|
20
|
+
"license": "SEE LICENSE IN LICENSE",
|
|
21
|
+
"homepage": "https://developer.lacspace.com/packages/keyphrase",
|
|
22
|
+
"repository": { "type": "git", "url": "git+https://github.com/lacspace/npm-packages.git", "directory": "keyphrase" },
|
|
23
|
+
"bugs": { "url": "https://github.com/lacspace/npm-packages/issues" },
|
|
24
|
+
"engines": { "node": ">=18" },
|
|
25
|
+
"publishConfig": { "access": "public" }
|
|
26
|
+
}
|