@lacspace/keyphrase 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/LICENSE ADDED
@@ -0,0 +1,51 @@
1
+ Lacspace Free Licence
2
+ Version 1.0, August 2026
3
+
4
+ Copyright (c) 2026 Lacspace
5
+
6
+ PREAMBLE
7
+
8
+ This software is published by Lacspace under the Lacspace Free Licence — a free,
9
+ permissive licence that lets you use this software for any purpose, including in
10
+ commercial products and services, at no cost. It grants the same freedoms as
11
+ common permissive open-source licences; the only condition is that this notice
12
+ travels with the software. The canonical, always-current text of this licence is
13
+ maintained at https://lacspace.com/licenses/lacspace-free-1.0
14
+
15
+ GRANT OF RIGHTS
16
+
17
+ Permission is hereby granted, free of charge, to any person or organisation
18
+ obtaining a copy of this software and its associated documentation and data files
19
+ (the "Software"), to deal in the Software without restriction, including without
20
+ limitation the rights to use, copy, modify, merge, publish, distribute,
21
+ sublicense, and/or sell copies of the Software, and to permit persons to whom the
22
+ Software is furnished to do so, subject to the conditions below. These rights are
23
+ granted for any purpose, personal or commercial, and are perpetual, worldwide,
24
+ non-exclusive, and royalty-free.
25
+
26
+ CONDITIONS
27
+
28
+ The above copyright notice, this permission notice, and the name of this licence
29
+ ("Lacspace Free Licence") shall be included in all copies or substantial portions
30
+ of the Software.
31
+
32
+ TRADEMARKS
33
+
34
+ This licence does not grant permission to use the trade names, trademarks, service
35
+ marks, logos, or product names of Lacspace, except as required to reproduce the
36
+ notice above or to describe the origin of the Software in a truthful manner.
37
+
38
+ DISCLAIMER OF WARRANTY AND LIMITATION OF LIABILITY
39
+
40
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
41
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
42
+ FOR A PARTICULAR PURPOSE, AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
43
+ COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES, OR OTHER LIABILITY, WHETHER IN
44
+ AN ACTION OF CONTRACT, TORT, OR OTHERWISE, ARISING FROM, OUT OF, OR IN CONNECTION
45
+ WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
46
+
47
+ ---
48
+
49
+ The Lacspace Free Licence is a source-available, permissive licence and is not (as
50
+ of this version) an OSI-approved licence. In substance it grants the same freedoms
51
+ as the MIT Licence. Learn more at https://lacspace.com/licenses
package/README.md ADDED
@@ -0,0 +1,33 @@
1
+ # @lacspace/keyphrase
2
+
3
+ **Stop asking the model to tag things.** Extract keyphrases, tags, hashtags, named entities and category votes from text with a zero-dependency RAKE + TF-IDF engine — built-in English and Nepali stopwords, Devanagari-aware, deterministic. Do the ~15–20% of an LLM's output that is really just extraction, for free.
4
+
5
+ ```bash
6
+ npm i @lacspace/keyphrase
7
+ ```
8
+
9
+ ```ts
10
+ import { keyphrase } from "@lacspace/keyphrase";
11
+
12
+ const r = keyphrase(articleText, {
13
+ gazetteer: ["Nepal Rastra Bank", "नेपाल राष्ट्र बैंक"],
14
+ categories: { economy: ["rate", "inflation", "bank"], sports: ["match", "goal"] },
15
+ });
16
+
17
+ r.tags; // ["policy interest rate", "central bank", ...]
18
+ r.hashtags; // ["#PolicyInterestRate", "#CentralBank", ...]
19
+ r.entities; // [{ text: "Nepal Rastra Bank", count: 2 }, ...]
20
+ r.categories; // [{ category: "economy", score: 4 }]
21
+ r.language; // "en" | "ne" (auto-detected)
22
+ ```
23
+
24
+ - **RAKE keyphrases** — candidate phrases split at stopwords/punctuation, scored by word degree/frequency; `topK`, `maxWords`, dedupe, gazetteer boost.
25
+ - **Hashtags** — CamelCase for Latin, Devanagari kept whole (matras preserved), punctuation stripped, 2–30 chars.
26
+ - **Entities** — Latin Title-Case runs plus every gazetteer term (including Devanagari, which has no case).
27
+ - **Category votes** — pass `{ category: [terms] }` and get a ranked vote by term hits.
28
+ - **Bilingual** — ships `ENGLISH_STOPWORDS` and `NEPALI_STOPWORDS`; auto-detects language, or force it; add your own with `extraStopwords` or replace with `stopwords`.
29
+
30
+ Deterministic and isomorphic. You bring domain stopwords/gazetteers; it brings the engine. Exports `toHashtag` and the two stopword lists too.
31
+
32
+ ## Licence
33
+ [Lacspace Free Licence v1.0](https://developer.lacspace.com/licenses/lacspace-free-1.0) — free for personal and commercial use.
package/dist/index.cjs ADDED
@@ -0,0 +1,166 @@
1
+ 'use strict';
2
+
3
+ // src/stopwords.ts
4
+ var ENGLISH_STOPWORDS = "a an and are as at be but by for from has have he her his i in is it its of on or that the their them they this to was were will with would you your we our us she him not no do does did done can could should may might must shall about after all also any because been before being between both during each few more most other over own same so some such than then there these those through under until up very what when where which who whom why how had having into out off down again further once here said says say according reported reports report told tuesday monday wednesday thursday friday saturday sunday am pm mr mrs ms dr new one two three per amid across".split(/\s+/);
5
+ var NEPALI_STOPWORDS = "\u0930 \u0915\u094B \u0915\u093E \u0915\u0940 \u092E\u093E \u0932\u0947 \u0939\u094B \u091B \u0925\u093F\u092F\u094B \u0925\u093F\u090F \u0939\u0941\u0928 \u0939\u0941\u0928\u094D\u091B \u092D\u0928\u0947 \u092D\u0928\u0940 \u092D\u0928\u094D\u0928\u0947 \u092A\u0928\u093F \u092F\u094B \u0924\u094D\u092F\u094B \u0924\u093F \u0924\u094D\u092F\u0938 \u092F\u0938 \u0917\u0930\u0947\u0915\u094B \u092D\u090F\u0915\u094B \u0932\u093E\u0917\u093F \u0924\u0925\u093E \u090F\u0935\u0902 \u0905\u0928\u0940 \u091C\u0938\u094D\u0924\u094B \u091C\u0938\u094D\u0924\u0948 \u0939\u094B\u0938\u094D \u091B\u0928\u094D \u0917\u0930\u094D\u0928 \u0917\u0930\u094D\u091B \u0917\u0930\u094D\u092F\u094B \u092D\u092F\u094B \u0905\u092C \u0938\u092C\u0948 \u0915\u0947\u0939\u0940 \u0906\u092B\u094D\u0928\u094B \u0909\u0928\u0940 \u0909\u0938\u0932\u0947 \u0909\u0928\u0932\u0947 \u092E\u093E\u0925\u093F \u0924\u0932 \u092D\u0928\u094D\u0926\u093E \u0938\u092E\u094D\u092C\u0928\u094D\u0927\u0940 \u092C\u093E\u091F \u0926\u0947\u0916\u093F \u0938\u092E\u094D\u092E".split(/\s+/).filter(Boolean);
6
+
7
+ // src/index.ts
8
+ var WORD_RE = /[\p{L}\p{N}][\p{L}\p{M}\p{N}‌‍]*/gu;
9
+ var DEVANAGARI_RE = /[ऀ-ॿ]/;
10
+ function detectLanguage(text) {
11
+ const deva = (text.match(/[ऀ-ॿ]/g) || []).length;
12
+ const latin = (text.match(/[A-Za-z]/g) || []).length;
13
+ return deva > latin ? "ne" : "en";
14
+ }
15
+ function keyphrase(text, options = {}) {
16
+ const language = options.language && options.language !== "auto" ? options.language : detectLanguage(text || "");
17
+ const topK = options.topK ?? 10;
18
+ const maxWords = options.maxWords ?? 4;
19
+ const minLen = options.minWordLength ?? 2;
20
+ const base = options.stopwords ?? (language === "ne" ? NEPALI_STOPWORDS : ENGLISH_STOPWORDS);
21
+ const stop = new Set([...base, ...options.extraStopwords ?? []].map((s) => s.toLowerCase()));
22
+ const gaz = (options.gazetteer ?? []).filter(Boolean);
23
+ const empty = { phrases: [], tags: [], hashtags: [], entities: [], categories: [], language };
24
+ if (!text || !text.trim()) return empty;
25
+ const candidates = [];
26
+ let current = [];
27
+ const tokenStream = tokenizeWithGaps(text);
28
+ for (const tok of tokenStream) {
29
+ if (tok.isBreak) {
30
+ if (current.length) candidates.push(current);
31
+ current = [];
32
+ continue;
33
+ }
34
+ const lw = tok.w.toLowerCase();
35
+ if (stop.has(lw) || tok.w.length < minLen || /^\d+$/.test(tok.w)) {
36
+ if (current.length) candidates.push(current);
37
+ current = [];
38
+ } else {
39
+ current.push(tok.w);
40
+ }
41
+ }
42
+ if (current.length) candidates.push(current);
43
+ const freq = /* @__PURE__ */ new Map();
44
+ const degree = /* @__PURE__ */ new Map();
45
+ for (const phrase of candidates) {
46
+ const deg = phrase.length - 1;
47
+ for (const w of phrase) {
48
+ const k = w.toLowerCase();
49
+ freq.set(k, (freq.get(k) ?? 0) + 1);
50
+ degree.set(k, (degree.get(k) ?? 0) + deg + 1);
51
+ }
52
+ }
53
+ const wordScore = (w) => {
54
+ const k = w.toLowerCase();
55
+ const f = freq.get(k) ?? 1;
56
+ return (degree.get(k) ?? f) / f;
57
+ };
58
+ const gazLower = new Set(gaz.map((g) => g.toLowerCase()));
59
+ const phraseScores = /* @__PURE__ */ new Map();
60
+ let order = 0;
61
+ for (const phrase of candidates) {
62
+ if (phrase.length === 0 || phrase.length > maxWords) continue;
63
+ const text2 = phrase.join(" ");
64
+ const key = text2.toLowerCase();
65
+ let score = phrase.reduce((s, w) => s + wordScore(w), 0);
66
+ if (gazLower.has(key) || gaz.some((g) => text2.includes(g))) score *= 1.5;
67
+ const prev = phraseScores.get(key);
68
+ if (prev) prev.score = Math.max(prev.score, score);
69
+ else phraseScores.set(key, { phrase: text2, score, order: order++ });
70
+ }
71
+ const phrases = [...phraseScores.values()].sort((a, b) => b.score - a.score || a.order - b.order).slice(0, topK).map((p) => ({ phrase: p.phrase, score: round(p.score) }));
72
+ const tags = phrases.map((p) => p.phrase);
73
+ const hashtags = dedupe(tags.map(toHashtag).filter((h) => h.length > 1));
74
+ const entities = extractEntities(text, gaz);
75
+ const categories = [];
76
+ if (options.categories) {
77
+ const lower = text.toLowerCase();
78
+ for (const [cat, terms] of Object.entries(options.categories)) {
79
+ let score = 0;
80
+ for (const term of terms) {
81
+ if (!term) continue;
82
+ if (DEVANAGARI_RE.test(term)) {
83
+ score += countOccurrences(text, term);
84
+ } else {
85
+ const re = new RegExp(`(?:^|[^a-z0-9])${escapeRe(term.toLowerCase())}(?:[^a-z0-9]|$)`, "g");
86
+ score += (lower.match(re) || []).length;
87
+ }
88
+ }
89
+ if (score > 0) categories.push({ category: cat, score });
90
+ }
91
+ categories.sort((a, b) => b.score - a.score || a.category.localeCompare(b.category));
92
+ }
93
+ return { phrases, tags, hashtags, entities, categories, language };
94
+ }
95
+ function tokenizeWithGaps(text) {
96
+ const out = [];
97
+ let last = 0;
98
+ let m;
99
+ WORD_RE.lastIndex = 0;
100
+ while ((m = WORD_RE.exec(text)) !== null) {
101
+ if (m.index > last) {
102
+ const gap = text.slice(last, m.index);
103
+ if (/[.!?;:,।॥\n\-–—/()\[\]"'“”]/.test(gap)) out.push({ isBreak: true });
104
+ }
105
+ out.push({ w: m[0], isBreak: false });
106
+ last = m.index + m[0].length;
107
+ }
108
+ return out;
109
+ }
110
+ function toHashtag(input) {
111
+ const cleaned = input.replace(/^#/, "");
112
+ const parts = cleaned.split(/[\s\-_/]+/).filter(Boolean);
113
+ const joined = parts.map((p) => /^[a-z]/.test(p) ? p.charAt(0).toUpperCase() + p.slice(1) : p).join("");
114
+ const kept = [...joined].filter((ch) => /[\p{L}\p{M}\p{N}]/u.test(ch)).join("");
115
+ return kept.length >= 2 && kept.length <= 30 ? "#" + kept : "";
116
+ }
117
+ function extractEntities(text, gaz) {
118
+ const counts = /* @__PURE__ */ new Map();
119
+ const re = /\b([A-Z][a-zA-Z]+(?:\s+[A-Z][a-zA-Z]+){0,3})\b/g;
120
+ let m;
121
+ while ((m = re.exec(text)) !== null) {
122
+ const e = m[1];
123
+ counts.set(e, (counts.get(e) ?? 0) + 1);
124
+ }
125
+ for (const g of gaz) {
126
+ if (!g) continue;
127
+ const c = countOccurrences(text, g);
128
+ if (c > 0) counts.set(g, Math.max(counts.get(g) ?? 0, c));
129
+ }
130
+ return [...counts.entries()].map(([t, c]) => ({ text: t, count: c })).sort((a, b) => b.count - a.count || a.text.localeCompare(b.text)).filter((e) => e.text.includes(" ") || gaz.includes(e.text) || e.count > 1 || /[ऀ-ॿ]/.test(e.text));
131
+ }
132
+ function countOccurrences(hay, needle) {
133
+ if (!needle) return 0;
134
+ let n = 0;
135
+ let i = hay.indexOf(needle);
136
+ while (i !== -1) {
137
+ n++;
138
+ i = hay.indexOf(needle, i + needle.length);
139
+ }
140
+ return n;
141
+ }
142
+ function dedupe(arr) {
143
+ const seen = /* @__PURE__ */ new Set();
144
+ const out = [];
145
+ for (const x of arr) {
146
+ const k = x.toLowerCase();
147
+ if (!seen.has(k)) {
148
+ seen.add(k);
149
+ out.push(x);
150
+ }
151
+ }
152
+ return out;
153
+ }
154
+ function escapeRe(s) {
155
+ return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
156
+ }
157
+ function round(n) {
158
+ return Math.round(n * 1e3) / 1e3;
159
+ }
160
+
161
+ exports.ENGLISH_STOPWORDS = ENGLISH_STOPWORDS;
162
+ exports.NEPALI_STOPWORDS = NEPALI_STOPWORDS;
163
+ exports.keyphrase = keyphrase;
164
+ exports.toHashtag = toHashtag;
165
+ //# sourceMappingURL=index.cjs.map
166
+ //# sourceMappingURL=index.cjs.map
@@ -0,0 +1 @@
1
+ {"version":3,"sources":["../src/stopwords.ts","../src/index.ts"],"names":[],"mappings":";;;AACO,IAAM,iBAAA,GACX,wpBAAA,CAQA,KAAA,CAAM,KAAK;AAGN,IAAM,mBACX,4hCAAA,CAUA,KAAA,CAAM,KAAK,CAAA,CAAE,OAAO,OAAO;;;AC0B7B,IAAM,OAAA,GAAU,oCAAA;AAChB,IAAM,aAAA,GAAgB,OAAA;AAEtB,SAAS,eAAe,IAAA,EAA2B;AACjD,EAAA,MAAM,QAAQ,IAAA,CAAK,KAAA,CAAM,QAAQ,CAAA,IAAK,EAAC,EAAG,MAAA;AAC1C,EAAA,MAAM,SAAS,IAAA,CAAK,KAAA,CAAM,WAAW,CAAA,IAAK,EAAC,EAAG,MAAA;AAC9C,EAAA,OAAO,IAAA,GAAO,QAAQ,IAAA,GAAO,IAAA;AAC/B;AAWO,SAAS,SAAA,CAAU,IAAA,EAAc,OAAA,GAA4B,EAAC,EAAoB;AACvF,EAAA,MAAM,QAAA,GAAW,OAAA,CAAQ,QAAA,IAAY,OAAA,CAAQ,QAAA,KAAa,SAAS,OAAA,CAAQ,QAAA,GAAW,cAAA,CAAe,IAAA,IAAQ,EAAE,CAAA;AAC/G,EAAA,MAAM,IAAA,GAAO,QAAQ,IAAA,IAAQ,EAAA;AAC7B,EAAA,MAAM,QAAA,GAAW,QAAQ,QAAA,IAAY,CAAA;AACrC,EAAA,MAAM,MAAA,GAAS,QAAQ,aAAA,IAAiB,CAAA;AACxC,EAAA,MAAM,IAAA,GAAO,OAAA,CAAQ,SAAA,KAAc,QAAA,KAAa,OAAO,gBAAA,GAAmB,iBAAA,CAAA;AAC1E,EAAA,MAAM,OAAO,IAAI,GAAA,CAAI,CAAC,GAAG,IAAA,EAAM,GAAI,OAAA,CAAQ,cAAA,IAAkB,EAAG,EAAE,GAAA,CAAI,CAAC,MAAM,CAAA,CAAE,WAAA,EAAa,CAAC,CAAA;AAC7F,EAAA,MAAM,OAAO,OAAA,CAAQ,SAAA,IAAa,EAAC,EAAG,OAAO,OAAO,CAAA;AAEpD,EAAA,MAAM,QAAyB,EAAE,OAAA,EAAS,EAAC,EAAG,MAAM,EAAC,EAAG,QAAA,EAAU,IAAI,QAAA,EAAU,IAAI,UAAA,EAAY,IAAI,QAAA,EAAS;AAC7G,EAAA,IAAI,CAAC,IAAA,IAAQ,CAAC,IAAA,CAAK,IAAA,IAAQ,OAAO,KAAA;AAGlC,EAAA,MAAM,aAAyB,EAAC;AAChC,EAAA,IAAI,UAAoB,EAAC;AAEzB,EAAA,MAAM,WAAA,GAAc,iBAAiB,IAAI,CAAA;AACzC,EAAA,KAAA,MAAW,OAAO,WAAA,EAAa;AAC7B,IAAA,IAAI,IAAI,OAAA,EAAS;AACf,MAAA,IAAI,OAAA,CAAQ,MAAA,EAAQ,UAAA,CAAW,IAAA,CAAK,OAAO,CAAA;AAC3C,MAAA,OAAA,GAAU,EAAC;AACX,MAAA;AAAA,IACF;AACA,IAAA,MAAM,EAAA,GAAK,GAAA,CAAI,CAAA,CAAG,WAAA,EAAY;AAC9B,IAAA,IAAI,IAAA,CAAK,GAAA,CAAI,EAAE,CAAA,IAAK,GAAA,CAAI,CAAA,CAAG,MAAA,GAAS,MAAA,IAAU,OAAA,CAAQ,IAAA,CAAK,GAAA,CAAI,CAAE,CAAA,EAAG;AAClE,MAAA,IAAI,OAAA,CAAQ,MAAA,EAAQ,UAAA,CAAW,IAAA,CAAK,OAAO,CAAA;AAC3C,MAAA,OAAA,GAAU,EAAC;AAAA,IACb,CAAA,MAAO;AACL,MAAA,OAAA,CAAQ,IAAA,CAAK,IAAI,CAAE,CAAA;AAAA,IACrB;AAAA,EACF;AACA,EAAA,IAAI,OAAA,CAAQ,MAAA,EAAQ,UAAA,CAAW,IAAA,CAAK,OAAO,CAAA;AAG3C,EAAA,MAAM,IAAA,uBAAW,GAAA,EAAoB;AACrC,EAAA,MAAM,MAAA,uBAAa,GAAA,EAAoB;AACvC,EAAA,KAAA,MAAW,UAAU,UAAA,EAAY;AAC/B,IAAA,MAAM,GAAA,GAAM,OAAO,MAAA,GAAS,CAAA;AAC5B,IAAA,KAAA,MAAW,KAAK,MAAA,EAAQ;AACtB,MAAA,MAAM,CAAA,GAAI,EAAE,WAAA,EAAY;AACxB,MAAA,IAAA,CAAK,IAAI,CAAA,EAAA,CAAI,IAAA,CAAK,IAAI,CAAC,CAAA,IAAK,KAAK,CAAC,CAAA;AAClC,MAAA,MAAA,CAAO,GAAA,CAAI,IAAI,MAAA,CAAO,GAAA,CAAI,CAAC,CAAA,IAAK,CAAA,IAAK,MAAM,CAAC,CAAA;AAAA,IAC9C;AAAA,EACF;AACA,EAAA,MAAM,SAAA,GAAY,CAAC,CAAA,KAAc;AAC/B,IAAA,MAAM,CAAA,GAAI,EAAE,WAAA,EAAY;AACxB,IAAA,MAAM,CAAA,GAAI,IAAA,CAAK,GAAA,CAAI,CAAC,CAAA,IAAK,CAAA;AACzB,IAAA,OAAA,CAAQ,MAAA,CAAO,GAAA,CAAI,CAAC,CAAA,IAAK,CAAA,IAAK,CAAA;AAAA,EAChC,CAAA;AAGA,EAAA,MAAM,QAAA,GAAW,IAAI,GAAA,CAAI,GAAA,CAAI,GAAA,CAAI,CAAC,CAAA,KAAM,CAAA,CAAE,WAAA,EAAa,CAAC,CAAA;AACxD,EAAA,MAAM,YAAA,uBAAmB,GAAA,EAA8D;AACvF,EAAA,IAAI,KAAA,GAAQ,CAAA;AACZ,EAAA,KAAA,MAAW,UAAU,UAAA,EAAY;AAC/B,IAAA,IAAI,MAAA,CAAO,MAAA,KAAW,CAAA,IAAK,MAAA,CAAO,SAAS,QAAA,EAAU;AACrD,IAAA,MAAM,KAAA,GAAQ,MAAA,CAAO,IAAA,CAAK,GAAG,CAAA;AAC7B,IAAA,MAAM,GAAA,GAAM,MAAM,WAAA,EAAY;AAC9B,IAAA,IAAI,KAAA,GAAQ,MAAA,CAAO,MAAA,CAAO,CAAC,CAAA,EAAG,MAAM,CAAA,GAAI,SAAA,CAAU,CAAC,CAAA,EAAG,CAAC,CAAA;AACvD,IAAA,IAAI,QAAA,CAAS,GAAA,CAAI,GAAG,CAAA,IAAK,GAAA,CAAI,IAAA,CAAK,CAAC,CAAA,KAAM,KAAA,CAAM,QAAA,CAAS,CAAC,CAAC,GAAG,KAAA,IAAS,GAAA;AACtE,IAAA,MAAM,IAAA,GAAO,YAAA,CAAa,GAAA,CAAI,GAAG,CAAA;AACjC,IAAA,IAAI,MAAM,IAAA,CAAK,KAAA,GAAQ,KAAK,GAAA,CAAI,IAAA,CAAK,OAAO,KAAK,CAAA;AAAA,SAC5C,YAAA,CAAa,IAAI,GAAA,EAAK,EAAE,QAAQ,KAAA,EAAO,KAAA,EAAO,KAAA,EAAO,KAAA,EAAA,EAAS,CAAA;AAAA,EACrE;AACA,EAAA,MAAM,UAAU,CAAC,GAAG,YAAA,CAAa,MAAA,EAAQ,CAAA,CACtC,IAAA,CAAK,CAAC,CAAA,EAAG,MAAM,CAAA,CAAE,KAAA,GAAQ,CAAA,CAAE,KAAA,IAAS,EAAE,KAAA,GAAQ,CAAA,CAAE,KAAK,CAAA,CACrD,MAAM,CAAA,EAAG,IAAI,CAAA,CACb,GAAA,CAAI,CAAC,CAAA,MAAO,EAAE,MAAA,EAAQ,CAAA,CAAE,QAAQ,KAAA,EAAO,KAAA,CAAM,CAAA,CAAE,KAAK,GAAE,CAAE,CAAA;AAG3D,EAAA,MAAM,OAAO,OAAA,CAAQ,GAAA,CAAI,CAAC,CAAA,KAAM,EAAE,MAAM,CAAA;AACxC,EAAA,MAAM,QAAA,GAAW,MAAA,CAAO,IAAA,CAAK,GAAA,CAAI,SAAS,CAAA,CAAE,MAAA,CAAO,CAAC,CAAA,KAAM,CAAA,CAAE,MAAA,GAAS,CAAC,CAAC,CAAA;AAGvE,EAAA,MAAM,QAAA,GAAW,eAAA,CAAgB,IAAA,EAAM,GAAG,CAAA;AAG1C,EAAA,MAAM,aAA6B,EAAC;AACpC,EAAA,IAAI,QAAQ,UAAA,EAAY;AACtB,IAAA,MAAM,KAAA,GAAQ,KAAK,WAAA,EAAY;AAC/B,IAAA,KAAA,MAAW,CAAC,KAAK,KAAK,CAAA,IAAK,OAAO,OAAA,CAAQ,OAAA,CAAQ,UAAU,CAAA,EAAG;AAC7D,MAAA,IAAI,KAAA,GAAQ,CAAA;AACZ,MAAA,KAAA,MAAW,QAAQ,KAAA,EAAO;AACxB,QAAA,IAAI,CAAC,IAAA,EAAM;AACX,QAAA,IAAI,aAAA,CAAc,IAAA,CAAK,IAAI,CAAA,EAAG;AAC5B,UAAA,KAAA,IAAS,gBAAA,CAAiB,MAAM,IAAI,CAAA;AAAA,QACtC,CAAA,MAAO;AACL,UAAA,MAAM,EAAA,GAAK,IAAI,MAAA,CAAO,CAAA,eAAA,EAAkB,QAAA,CAAS,KAAK,WAAA,EAAa,CAAC,CAAA,eAAA,CAAA,EAAmB,GAAG,CAAA;AAC1F,UAAA,KAAA,IAAA,CAAU,KAAA,CAAM,KAAA,CAAM,EAAE,CAAA,IAAK,EAAC,EAAG,MAAA;AAAA,QACnC;AAAA,MACF;AACA,MAAA,IAAI,KAAA,GAAQ,GAAG,UAAA,CAAW,IAAA,CAAK,EAAE,QAAA,EAAU,GAAA,EAAK,OAAO,CAAA;AAAA,IACzD;AACA,IAAA,UAAA,CAAW,IAAA,CAAK,CAAC,CAAA,EAAG,CAAA,KAAM,CAAA,CAAE,KAAA,GAAQ,CAAA,CAAE,KAAA,IAAS,CAAA,CAAE,QAAA,CAAS,aAAA,CAAc,CAAA,CAAE,QAAQ,CAAC,CAAA;AAAA,EACrF;AAEA,EAAA,OAAO,EAAE,OAAA,EAAS,IAAA,EAAM,QAAA,EAAU,QAAA,EAAU,YAAY,QAAA,EAAS;AACnE;AAMA,SAAS,iBAAiB,IAAA,EAA2B;AACnD,EAAA,MAAM,MAAmB,EAAC;AAC1B,EAAA,IAAI,IAAA,GAAO,CAAA;AACX,EAAA,IAAI,CAAA;AACJ,EAAA,OAAA,CAAQ,SAAA,GAAY,CAAA;AACpB,EAAA,OAAA,CAAQ,CAAA,GAAI,OAAA,CAAQ,IAAA,CAAK,IAAI,OAAO,IAAA,EAAM;AACxC,IAAA,IAAI,CAAA,CAAE,QAAQ,IAAA,EAAM;AAElB,MAAA,MAAM,GAAA,GAAM,IAAA,CAAK,KAAA,CAAM,IAAA,EAAM,EAAE,KAAK,CAAA;AACpC,MAAA,IAAI,6BAAA,CAA8B,KAAK,GAAG,CAAA,MAAO,IAAA,CAAK,EAAE,OAAA,EAAS,IAAA,EAAM,CAAA;AAAA,IACzE;AACA,IAAA,GAAA,CAAI,IAAA,CAAK,EAAE,CAAA,EAAG,CAAA,CAAE,CAAC,CAAA,EAAG,OAAA,EAAS,OAAO,CAAA;AACpC,IAAA,IAAA,GAAO,CAAA,CAAE,KAAA,GAAQ,CAAA,CAAE,CAAC,CAAA,CAAE,MAAA;AAAA,EACxB;AACA,EAAA,OAAO,GAAA;AACT;AAGO,SAAS,UAAU,KAAA,EAAuB;AAC/C,EAAA,MAAM,OAAA,GAAU,KAAA,CAAM,OAAA,CAAQ,IAAA,EAAM,EAAE,CAAA;AACtC,EAAA,MAAM,QAAQ,OAAA,CAAQ,KAAA,CAAM,WAAW,CAAA,CAAE,OAAO,OAAO,CAAA;AACvD,EAAA,MAAM,MAAA,GAAS,MACZ,GAAA,CAAI,CAAC,MAAO,QAAA,CAAS,IAAA,CAAK,CAAC,CAAA,GAAI,CAAA,CAAE,MAAA,CAAO,CAAC,CAAA,CAAE,WAAA,KAAgB,CAAA,CAAE,KAAA,CAAM,CAAC,CAAA,GAAI,CAAE,CAAA,CAC1E,IAAA,CAAK,EAAE,CAAA;AACV,EAAA,MAAM,IAAA,GAAO,CAAC,GAAG,MAAM,EAAE,MAAA,CAAO,CAAC,EAAA,KAAO,oBAAA,CAAqB,IAAA,CAAK,EAAE,CAAC,CAAA,CAAE,KAAK,EAAE,CAAA;AAC9E,EAAA,OAAO,KAAK,MAAA,IAAU,CAAA,IAAK,KAAK,MAAA,IAAU,EAAA,GAAK,MAAM,IAAA,GAAO,EAAA;AAC9D;AAEA,SAAS,eAAA,CAAgB,MAAc,GAAA,EAA4B;AACjE,EAAA,MAAM,MAAA,uBAAa,GAAA,EAAoB;AAEvC,EAAA,MAAM,EAAA,GAAK,iDAAA;AACX,EAAA,IAAI,CAAA;AACJ,EAAA,OAAA,CAAQ,CAAA,GAAI,EAAA,CAAG,IAAA,CAAK,IAAI,OAAO,IAAA,EAAM;AACnC,IAAA,MAAM,CAAA,GAAI,EAAE,CAAC,CAAA;AAEb,IAAA,MAAA,CAAO,IAAI,CAAA,EAAA,CAAI,MAAA,CAAO,IAAI,CAAC,CAAA,IAAK,KAAK,CAAC,CAAA;AAAA,EACxC;AAEA,EAAA,KAAA,MAAW,KAAK,GAAA,EAAK;AACnB,IAAA,IAAI,CAAC,CAAA,EAAG;AACR,IAAA,MAAM,CAAA,GAAI,gBAAA,CAAiB,IAAA,EAAM,CAAC,CAAA;AAClC,IAAA,IAAI,CAAA,GAAI,CAAA,EAAG,MAAA,CAAO,GAAA,CAAI,CAAA,EAAG,IAAA,CAAK,GAAA,CAAI,MAAA,CAAO,GAAA,CAAI,CAAC,CAAA,IAAK,CAAA,EAAG,CAAC,CAAC,CAAA;AAAA,EAC1D;AACA,EAAA,OAAO,CAAC,GAAG,MAAA,CAAO,OAAA,EAAS,EACxB,GAAA,CAAI,CAAC,CAAC,CAAA,EAAG,CAAC,CAAA,MAAO,EAAE,IAAA,EAAM,CAAA,EAAG,KAAA,EAAO,CAAA,EAAE,CAAE,CAAA,CACvC,IAAA,CAAK,CAAC,CAAA,EAAG,CAAA,KAAM,CAAA,CAAE,KAAA,GAAQ,CAAA,CAAE,KAAA,IAAS,EAAE,IAAA,CAAK,aAAA,CAAc,CAAA,CAAE,IAAI,CAAC,CAAA,CAChE,OAAO,CAAC,CAAA,KAAM,CAAA,CAAE,IAAA,CAAK,QAAA,CAAS,GAAG,KAAK,GAAA,CAAI,QAAA,CAAS,CAAA,CAAE,IAAI,CAAA,IAAK,CAAA,CAAE,KAAA,GAAQ,CAAA,IAAK,OAAA,CAAQ,IAAA,CAAK,CAAA,CAAE,IAAI,CAAC,CAAA;AACtG;AAEA,SAAS,gBAAA,CAAiB,KAAa,MAAA,EAAwB;AAC7D,EAAA,IAAI,CAAC,QAAQ,OAAO,CAAA;AACpB,EAAA,IAAI,CAAA,GAAI,CAAA;AACR,EAAA,IAAI,CAAA,GAAI,GAAA,CAAI,OAAA,CAAQ,MAAM,CAAA;AAC1B,EAAA,OAAO,MAAM,EAAA,EAAI;AACf,IAAA,CAAA,EAAA;AACA,IAAA,CAAA,GAAI,GAAA,CAAI,OAAA,CAAQ,MAAA,EAAQ,CAAA,GAAI,OAAO,MAAM,CAAA;AAAA,EAC3C;AACA,EAAA,OAAO,CAAA;AACT;AACA,SAAS,OAAO,GAAA,EAAyB;AACvC,EAAA,MAAM,IAAA,uBAAW,GAAA,EAAY;AAC7B,EAAA,MAAM,MAAgB,EAAC;AACvB,EAAA,KAAA,MAAW,KAAK,GAAA,EAAK;AACnB,IAAA,MAAM,CAAA,GAAI,EAAE,WAAA,EAAY;AACxB,IAAA,IAAI,CAAC,IAAA,CAAK,GAAA,CAAI,CAAC,CAAA,EAAG;AAChB,MAAA,IAAA,CAAK,IAAI,CAAC,CAAA;AACV,MAAA,GAAA,CAAI,KAAK,CAAC,CAAA;AAAA,IACZ;AAAA,EACF;AACA,EAAA,OAAO,GAAA;AACT;AACA,SAAS,SAAS,CAAA,EAAmB;AACnC,EAAA,OAAO,CAAA,CAAE,OAAA,CAAQ,qBAAA,EAAuB,MAAM,CAAA;AAChD;AACA,SAAS,MAAM,CAAA,EAAmB;AAChC,EAAA,OAAO,IAAA,CAAK,KAAA,CAAM,CAAA,GAAI,GAAI,CAAA,GAAI,GAAA;AAChC","file":"index.cjs","sourcesContent":["// Compact, high-frequency stopword lists. Extend via options.extraStopwords.\nexport const ENGLISH_STOPWORDS: string[] = (\n \"a an and are as at be but by for from has have he her his i in is it its of on or \" +\n \"that the their them they this to was were will with would you your we our us she him \" +\n \"not no do does did done can could should may might must shall about after all also \" +\n \"any because been before being between both during each few more most other over own \" +\n \"same so some such than then there these those through under until up very what when \" +\n \"where which who whom why how had having into out off down again further once here \" +\n \"said says say according reported reports report told tuesday monday wednesday thursday \" +\n \"friday saturday sunday am pm mr mrs ms dr new one two three per amid across\"\n).split(/\\s+/);\n\n// Common Nepali (Devanagari) function words and news filler.\nexport const NEPALI_STOPWORDS: string[] = (\n \"र को का की मा ले हो \" +\n \"छ थियो थिए हुन हुन्छ \" +\n \"भने भनी भन्ने पनि यो \" +\n \"त्यो ति त्यस यस गरेको \" +\n \"भएको लागि तथा एवं अनी \" +\n \"जस्तो जस्तै होस् छन् \" +\n \"गर्न गर्छ गर्यो भयो \" +\n \"अब सबै केही आफ्नो उनी \" +\n \"उसले उनले माथि तल भन्दा \" +\n \"सम्बन्धी बाट देखि सम्म\"\n).split(/\\s+/).filter(Boolean);\n","import { ENGLISH_STOPWORDS, NEPALI_STOPWORDS } from \"./stopwords.js\";\n\nexport { ENGLISH_STOPWORDS, NEPALI_STOPWORDS } from \"./stopwords.js\";\n\nexport interface KeyphraseOptions {\n /** \"en\", \"ne\", or \"auto\" (default) — picks built-in stopwords. */\n language?: \"en\" | \"ne\" | \"auto\";\n /** How many keyphrases/tags to return. Default 10. */\n topK?: number;\n /** Longest phrase, in words. Default 4. */\n maxWords?: number;\n /** Replace the built-in stopwords entirely. */\n stopwords?: string[];\n /** Add to the built-in stopwords. */\n extraStopwords?: string[];\n /** Known entities (people/places), en and ne forms — always surfaced and boosted. */\n gazetteer?: string[];\n /** Category lexicons: { category: [terms] }. Returns a vote per category by term hits. */\n categories?: Record<string, string[]>;\n /** Minimum characters for a candidate word. Default 2. */\n minWordLength?: number;\n}\n\nexport interface ScoredPhrase {\n phrase: string;\n score: number;\n}\nexport interface EntityHit {\n text: string;\n count: number;\n}\nexport interface CategoryVote {\n category: string;\n score: number;\n}\nexport interface KeyphraseResult {\n /** Top keyphrases by RAKE degree/frequency score. */\n phrases: ScoredPhrase[];\n /** Flat, de-duplicated tag strings (top phrases, normalized). */\n tags: string[];\n /** Hashtags built from the top tags (CamelCase for Latin, Devanagari kept whole). */\n hashtags: string[];\n /** Candidate named entities: Latin Title-Case runs + gazetteer hits. */\n entities: EntityHit[];\n /** Category votes (only when `categories` is provided), highest first. */\n categories: CategoryVote[];\n /** Detected/So-used language. */\n language: \"en\" | \"ne\";\n}\n\nconst WORD_RE = /[\\p{L}\\p{N}][\\p{L}\\p{M}\\p{N}‌‍]*/gu;\nconst DEVANAGARI_RE = /[ऀ-ॿ]/;\n\nfunction detectLanguage(text: string): \"en\" | \"ne\" {\n const deva = (text.match(/[ऀ-ॿ]/g) || []).length;\n const latin = (text.match(/[A-Za-z]/g) || []).length;\n return deva > latin ? \"ne\" : \"en\";\n}\n\nfunction words(text: string): { w: string; start: number }[] {\n const out: { w: string; start: number }[] = [];\n let m: RegExpExecArray | null;\n WORD_RE.lastIndex = 0;\n while ((m = WORD_RE.exec(text)) !== null) out.push({ w: m[0], start: m.index });\n return out;\n}\n\n/** Extract keyphrases, tags, hashtags, entities and category votes — no LLM. */\nexport function keyphrase(text: string, options: KeyphraseOptions = {}): KeyphraseResult {\n const language = options.language && options.language !== \"auto\" ? options.language : detectLanguage(text || \"\");\n const topK = options.topK ?? 10;\n const maxWords = options.maxWords ?? 4;\n const minLen = options.minWordLength ?? 2;\n const base = options.stopwords ?? (language === \"ne\" ? NEPALI_STOPWORDS : ENGLISH_STOPWORDS);\n const stop = new Set([...base, ...(options.extraStopwords ?? [])].map((s) => s.toLowerCase()));\n const gaz = (options.gazetteer ?? []).filter(Boolean);\n\n const empty: KeyphraseResult = { phrases: [], tags: [], hashtags: [], entities: [], categories: [], language };\n if (!text || !text.trim()) return empty;\n\n // 1. RAKE: break into candidate phrases at stopwords and non-word chars.\n const candidates: string[][] = [];\n let current: string[] = [];\n // Walk the raw text so punctuation also breaks phrases.\n const tokenStream = tokenizeWithGaps(text);\n for (const tok of tokenStream) {\n if (tok.isBreak) {\n if (current.length) candidates.push(current);\n current = [];\n continue;\n }\n const lw = tok.w!.toLowerCase();\n if (stop.has(lw) || tok.w!.length < minLen || /^\\d+$/.test(tok.w!)) {\n if (current.length) candidates.push(current);\n current = [];\n } else {\n current.push(tok.w!);\n }\n }\n if (current.length) candidates.push(current);\n\n // 2. Word scores: degree / frequency (classic RAKE).\n const freq = new Map<string, number>();\n const degree = new Map<string, number>();\n for (const phrase of candidates) {\n const deg = phrase.length - 1;\n for (const w of phrase) {\n const k = w.toLowerCase();\n freq.set(k, (freq.get(k) ?? 0) + 1);\n degree.set(k, (degree.get(k) ?? 0) + deg + 1);\n }\n }\n const wordScore = (w: string) => {\n const k = w.toLowerCase();\n const f = freq.get(k) ?? 1;\n return (degree.get(k) ?? f) / f;\n };\n\n // 3. Phrase scores; keep ≤ maxWords; dedupe by lowercase text; boost gazetteer.\n const gazLower = new Set(gaz.map((g) => g.toLowerCase()));\n const phraseScores = new Map<string, { phrase: string; score: number; order: number }>();\n let order = 0;\n for (const phrase of candidates) {\n if (phrase.length === 0 || phrase.length > maxWords) continue;\n const text2 = phrase.join(\" \");\n const key = text2.toLowerCase();\n let score = phrase.reduce((s, w) => s + wordScore(w), 0);\n if (gazLower.has(key) || gaz.some((g) => text2.includes(g))) score *= 1.5;\n const prev = phraseScores.get(key);\n if (prev) prev.score = Math.max(prev.score, score);\n else phraseScores.set(key, { phrase: text2, score, order: order++ });\n }\n const phrases = [...phraseScores.values()]\n .sort((a, b) => b.score - a.score || a.order - b.order)\n .slice(0, topK)\n .map((p) => ({ phrase: p.phrase, score: round(p.score) }));\n\n // 4. Tags + hashtags.\n const tags = phrases.map((p) => p.phrase);\n const hashtags = dedupe(tags.map(toHashtag).filter((h) => h.length > 1));\n\n // 5. Entities: Latin Title-Case runs + gazetteer hits.\n const entities = extractEntities(text, gaz);\n\n // 6. Category votes.\n const categories: CategoryVote[] = [];\n if (options.categories) {\n const lower = text.toLowerCase();\n for (const [cat, terms] of Object.entries(options.categories)) {\n let score = 0;\n for (const term of terms) {\n if (!term) continue;\n if (DEVANAGARI_RE.test(term)) {\n score += countOccurrences(text, term);\n } else {\n const re = new RegExp(`(?:^|[^a-z0-9])${escapeRe(term.toLowerCase())}(?:[^a-z0-9]|$)`, \"g\");\n score += (lower.match(re) || []).length;\n }\n }\n if (score > 0) categories.push({ category: cat, score });\n }\n categories.sort((a, b) => b.score - a.score || a.category.localeCompare(b.category));\n }\n\n return { phrases, tags, hashtags, entities, categories, language };\n}\n\ninterface StreamTok {\n w?: string;\n isBreak: boolean;\n}\nfunction tokenizeWithGaps(text: string): StreamTok[] {\n const out: StreamTok[] = [];\n let last = 0;\n let m: RegExpExecArray | null;\n WORD_RE.lastIndex = 0;\n while ((m = WORD_RE.exec(text)) !== null) {\n if (m.index > last) {\n // any non-word gap that contains sentence punctuation is a hard break\n const gap = text.slice(last, m.index);\n if (/[.!?;:,।॥\\n\\-–—/()\\[\\]\"'“”]/.test(gap)) out.push({ isBreak: true });\n }\n out.push({ w: m[0], isBreak: false });\n last = m.index + m[0].length;\n }\n return out;\n}\n\n/** Build a hashtag: strip #, CamelCase Latin words, keep Devanagari, keep only letters/marks/numbers. */\nexport function toHashtag(input: string): string {\n const cleaned = input.replace(/^#/, \"\");\n const parts = cleaned.split(/[\\s\\-_/]+/).filter(Boolean);\n const joined = parts\n .map((p) => (/^[a-z]/.test(p) ? p.charAt(0).toUpperCase() + p.slice(1) : p))\n .join(\"\");\n const kept = [...joined].filter((ch) => /[\\p{L}\\p{M}\\p{N}]/u.test(ch)).join(\"\");\n return kept.length >= 2 && kept.length <= 30 ? \"#\" + kept : \"\";\n}\n\nfunction extractEntities(text: string, gaz: string[]): EntityHit[] {\n const counts = new Map<string, number>();\n // Latin Title-Case runs of 1–4 words.\n const re = /\\b([A-Z][a-zA-Z]+(?:\\s+[A-Z][a-zA-Z]+){0,3})\\b/g;\n let m: RegExpExecArray | null;\n while ((m = re.exec(text)) !== null) {\n const e = m[1]!;\n // skip a lone word that starts a sentence and is common (heuristic: keep multiword or gazetteer)\n counts.set(e, (counts.get(e) ?? 0) + 1);\n }\n // Gazetteer (incl. Devanagari) always counted.\n for (const g of gaz) {\n if (!g) continue;\n const c = countOccurrences(text, g);\n if (c > 0) counts.set(g, Math.max(counts.get(g) ?? 0, c));\n }\n return [...counts.entries()]\n .map(([t, c]) => ({ text: t, count: c }))\n .sort((a, b) => b.count - a.count || a.text.localeCompare(b.text))\n .filter((e) => e.text.includes(\" \") || gaz.includes(e.text) || e.count > 1 || /[ऀ-ॿ]/.test(e.text));\n}\n\nfunction countOccurrences(hay: string, needle: string): number {\n if (!needle) return 0;\n let n = 0;\n let i = hay.indexOf(needle);\n while (i !== -1) {\n n++;\n i = hay.indexOf(needle, i + needle.length);\n }\n return n;\n}\nfunction dedupe(arr: string[]): string[] {\n const seen = new Set<string>();\n const out: string[] = [];\n for (const x of arr) {\n const k = x.toLowerCase();\n if (!seen.has(k)) {\n seen.add(k);\n out.push(x);\n }\n }\n return out;\n}\nfunction escapeRe(s: string): string {\n return s.replace(/[.*+?^${}()|[\\]\\\\]/g, \"\\\\$&\");\n}\nfunction round(n: number): number {\n return Math.round(n * 1000) / 1000;\n}\n"]}
@@ -0,0 +1,53 @@
1
+ declare const ENGLISH_STOPWORDS: string[];
2
+ declare const NEPALI_STOPWORDS: string[];
3
+
4
+ interface KeyphraseOptions {
5
+ /** "en", "ne", or "auto" (default) — picks built-in stopwords. */
6
+ language?: "en" | "ne" | "auto";
7
+ /** How many keyphrases/tags to return. Default 10. */
8
+ topK?: number;
9
+ /** Longest phrase, in words. Default 4. */
10
+ maxWords?: number;
11
+ /** Replace the built-in stopwords entirely. */
12
+ stopwords?: string[];
13
+ /** Add to the built-in stopwords. */
14
+ extraStopwords?: string[];
15
+ /** Known entities (people/places), en and ne forms — always surfaced and boosted. */
16
+ gazetteer?: string[];
17
+ /** Category lexicons: { category: [terms] }. Returns a vote per category by term hits. */
18
+ categories?: Record<string, string[]>;
19
+ /** Minimum characters for a candidate word. Default 2. */
20
+ minWordLength?: number;
21
+ }
22
+ interface ScoredPhrase {
23
+ phrase: string;
24
+ score: number;
25
+ }
26
+ interface EntityHit {
27
+ text: string;
28
+ count: number;
29
+ }
30
+ interface CategoryVote {
31
+ category: string;
32
+ score: number;
33
+ }
34
+ interface KeyphraseResult {
35
+ /** Top keyphrases by RAKE degree/frequency score. */
36
+ phrases: ScoredPhrase[];
37
+ /** Flat, de-duplicated tag strings (top phrases, normalized). */
38
+ tags: string[];
39
+ /** Hashtags built from the top tags (CamelCase for Latin, Devanagari kept whole). */
40
+ hashtags: string[];
41
+ /** Candidate named entities: Latin Title-Case runs + gazetteer hits. */
42
+ entities: EntityHit[];
43
+ /** Category votes (only when `categories` is provided), highest first. */
44
+ categories: CategoryVote[];
45
+ /** Detected/So-used language. */
46
+ language: "en" | "ne";
47
+ }
48
+ /** Extract keyphrases, tags, hashtags, entities and category votes — no LLM. */
49
+ declare function keyphrase(text: string, options?: KeyphraseOptions): KeyphraseResult;
50
+ /** Build a hashtag: strip #, CamelCase Latin words, keep Devanagari, keep only letters/marks/numbers. */
51
+ declare function toHashtag(input: string): string;
52
+
53
+ export { type CategoryVote, ENGLISH_STOPWORDS, type EntityHit, type KeyphraseOptions, type KeyphraseResult, NEPALI_STOPWORDS, type ScoredPhrase, keyphrase, toHashtag };
@@ -0,0 +1,53 @@
1
+ declare const ENGLISH_STOPWORDS: string[];
2
+ declare const NEPALI_STOPWORDS: string[];
3
+
4
+ interface KeyphraseOptions {
5
+ /** "en", "ne", or "auto" (default) — picks built-in stopwords. */
6
+ language?: "en" | "ne" | "auto";
7
+ /** How many keyphrases/tags to return. Default 10. */
8
+ topK?: number;
9
+ /** Longest phrase, in words. Default 4. */
10
+ maxWords?: number;
11
+ /** Replace the built-in stopwords entirely. */
12
+ stopwords?: string[];
13
+ /** Add to the built-in stopwords. */
14
+ extraStopwords?: string[];
15
+ /** Known entities (people/places), en and ne forms — always surfaced and boosted. */
16
+ gazetteer?: string[];
17
+ /** Category lexicons: { category: [terms] }. Returns a vote per category by term hits. */
18
+ categories?: Record<string, string[]>;
19
+ /** Minimum characters for a candidate word. Default 2. */
20
+ minWordLength?: number;
21
+ }
22
+ interface ScoredPhrase {
23
+ phrase: string;
24
+ score: number;
25
+ }
26
+ interface EntityHit {
27
+ text: string;
28
+ count: number;
29
+ }
30
+ interface CategoryVote {
31
+ category: string;
32
+ score: number;
33
+ }
34
+ interface KeyphraseResult {
35
+ /** Top keyphrases by RAKE degree/frequency score. */
36
+ phrases: ScoredPhrase[];
37
+ /** Flat, de-duplicated tag strings (top phrases, normalized). */
38
+ tags: string[];
39
+ /** Hashtags built from the top tags (CamelCase for Latin, Devanagari kept whole). */
40
+ hashtags: string[];
41
+ /** Candidate named entities: Latin Title-Case runs + gazetteer hits. */
42
+ entities: EntityHit[];
43
+ /** Category votes (only when `categories` is provided), highest first. */
44
+ categories: CategoryVote[];
45
+ /** Detected/So-used language. */
46
+ language: "en" | "ne";
47
+ }
48
+ /** Extract keyphrases, tags, hashtags, entities and category votes — no LLM. */
49
+ declare function keyphrase(text: string, options?: KeyphraseOptions): KeyphraseResult;
50
+ /** Build a hashtag: strip #, CamelCase Latin words, keep Devanagari, keep only letters/marks/numbers. */
51
+ declare function toHashtag(input: string): string;
52
+
53
+ export { type CategoryVote, ENGLISH_STOPWORDS, type EntityHit, type KeyphraseOptions, type KeyphraseResult, NEPALI_STOPWORDS, type ScoredPhrase, keyphrase, toHashtag };
package/dist/index.js ADDED
@@ -0,0 +1,161 @@
1
+ // src/stopwords.ts
2
+ var ENGLISH_STOPWORDS = "a an and are as at be but by for from has have he her his i in is it its of on or that the their them they this to was were will with would you your we our us she him not no do does did done can could should may might must shall about after all also any because been before being between both during each few more most other over own same so some such than then there these those through under until up very what when where which who whom why how had having into out off down again further once here said says say according reported reports report told tuesday monday wednesday thursday friday saturday sunday am pm mr mrs ms dr new one two three per amid across".split(/\s+/);
3
+ var NEPALI_STOPWORDS = "\u0930 \u0915\u094B \u0915\u093E \u0915\u0940 \u092E\u093E \u0932\u0947 \u0939\u094B \u091B \u0925\u093F\u092F\u094B \u0925\u093F\u090F \u0939\u0941\u0928 \u0939\u0941\u0928\u094D\u091B \u092D\u0928\u0947 \u092D\u0928\u0940 \u092D\u0928\u094D\u0928\u0947 \u092A\u0928\u093F \u092F\u094B \u0924\u094D\u092F\u094B \u0924\u093F \u0924\u094D\u092F\u0938 \u092F\u0938 \u0917\u0930\u0947\u0915\u094B \u092D\u090F\u0915\u094B \u0932\u093E\u0917\u093F \u0924\u0925\u093E \u090F\u0935\u0902 \u0905\u0928\u0940 \u091C\u0938\u094D\u0924\u094B \u091C\u0938\u094D\u0924\u0948 \u0939\u094B\u0938\u094D \u091B\u0928\u094D \u0917\u0930\u094D\u0928 \u0917\u0930\u094D\u091B \u0917\u0930\u094D\u092F\u094B \u092D\u092F\u094B \u0905\u092C \u0938\u092C\u0948 \u0915\u0947\u0939\u0940 \u0906\u092B\u094D\u0928\u094B \u0909\u0928\u0940 \u0909\u0938\u0932\u0947 \u0909\u0928\u0932\u0947 \u092E\u093E\u0925\u093F \u0924\u0932 \u092D\u0928\u094D\u0926\u093E \u0938\u092E\u094D\u092C\u0928\u094D\u0927\u0940 \u092C\u093E\u091F \u0926\u0947\u0916\u093F \u0938\u092E\u094D\u092E".split(/\s+/).filter(Boolean);
4
+
5
+ // src/index.ts
6
+ var WORD_RE = /[\p{L}\p{N}][\p{L}\p{M}\p{N}‌‍]*/gu;
7
+ var DEVANAGARI_RE = /[ऀ-ॿ]/;
8
+ function detectLanguage(text) {
9
+ const deva = (text.match(/[ऀ-ॿ]/g) || []).length;
10
+ const latin = (text.match(/[A-Za-z]/g) || []).length;
11
+ return deva > latin ? "ne" : "en";
12
+ }
13
+ function keyphrase(text, options = {}) {
14
+ const language = options.language && options.language !== "auto" ? options.language : detectLanguage(text || "");
15
+ const topK = options.topK ?? 10;
16
+ const maxWords = options.maxWords ?? 4;
17
+ const minLen = options.minWordLength ?? 2;
18
+ const base = options.stopwords ?? (language === "ne" ? NEPALI_STOPWORDS : ENGLISH_STOPWORDS);
19
+ const stop = new Set([...base, ...options.extraStopwords ?? []].map((s) => s.toLowerCase()));
20
+ const gaz = (options.gazetteer ?? []).filter(Boolean);
21
+ const empty = { phrases: [], tags: [], hashtags: [], entities: [], categories: [], language };
22
+ if (!text || !text.trim()) return empty;
23
+ const candidates = [];
24
+ let current = [];
25
+ const tokenStream = tokenizeWithGaps(text);
26
+ for (const tok of tokenStream) {
27
+ if (tok.isBreak) {
28
+ if (current.length) candidates.push(current);
29
+ current = [];
30
+ continue;
31
+ }
32
+ const lw = tok.w.toLowerCase();
33
+ if (stop.has(lw) || tok.w.length < minLen || /^\d+$/.test(tok.w)) {
34
+ if (current.length) candidates.push(current);
35
+ current = [];
36
+ } else {
37
+ current.push(tok.w);
38
+ }
39
+ }
40
+ if (current.length) candidates.push(current);
41
+ const freq = /* @__PURE__ */ new Map();
42
+ const degree = /* @__PURE__ */ new Map();
43
+ for (const phrase of candidates) {
44
+ const deg = phrase.length - 1;
45
+ for (const w of phrase) {
46
+ const k = w.toLowerCase();
47
+ freq.set(k, (freq.get(k) ?? 0) + 1);
48
+ degree.set(k, (degree.get(k) ?? 0) + deg + 1);
49
+ }
50
+ }
51
+ const wordScore = (w) => {
52
+ const k = w.toLowerCase();
53
+ const f = freq.get(k) ?? 1;
54
+ return (degree.get(k) ?? f) / f;
55
+ };
56
+ const gazLower = new Set(gaz.map((g) => g.toLowerCase()));
57
+ const phraseScores = /* @__PURE__ */ new Map();
58
+ let order = 0;
59
+ for (const phrase of candidates) {
60
+ if (phrase.length === 0 || phrase.length > maxWords) continue;
61
+ const text2 = phrase.join(" ");
62
+ const key = text2.toLowerCase();
63
+ let score = phrase.reduce((s, w) => s + wordScore(w), 0);
64
+ if (gazLower.has(key) || gaz.some((g) => text2.includes(g))) score *= 1.5;
65
+ const prev = phraseScores.get(key);
66
+ if (prev) prev.score = Math.max(prev.score, score);
67
+ else phraseScores.set(key, { phrase: text2, score, order: order++ });
68
+ }
69
+ const phrases = [...phraseScores.values()].sort((a, b) => b.score - a.score || a.order - b.order).slice(0, topK).map((p) => ({ phrase: p.phrase, score: round(p.score) }));
70
+ const tags = phrases.map((p) => p.phrase);
71
+ const hashtags = dedupe(tags.map(toHashtag).filter((h) => h.length > 1));
72
+ const entities = extractEntities(text, gaz);
73
+ const categories = [];
74
+ if (options.categories) {
75
+ const lower = text.toLowerCase();
76
+ for (const [cat, terms] of Object.entries(options.categories)) {
77
+ let score = 0;
78
+ for (const term of terms) {
79
+ if (!term) continue;
80
+ if (DEVANAGARI_RE.test(term)) {
81
+ score += countOccurrences(text, term);
82
+ } else {
83
+ const re = new RegExp(`(?:^|[^a-z0-9])${escapeRe(term.toLowerCase())}(?:[^a-z0-9]|$)`, "g");
84
+ score += (lower.match(re) || []).length;
85
+ }
86
+ }
87
+ if (score > 0) categories.push({ category: cat, score });
88
+ }
89
+ categories.sort((a, b) => b.score - a.score || a.category.localeCompare(b.category));
90
+ }
91
+ return { phrases, tags, hashtags, entities, categories, language };
92
+ }
93
+ function tokenizeWithGaps(text) {
94
+ const out = [];
95
+ let last = 0;
96
+ let m;
97
+ WORD_RE.lastIndex = 0;
98
+ while ((m = WORD_RE.exec(text)) !== null) {
99
+ if (m.index > last) {
100
+ const gap = text.slice(last, m.index);
101
+ if (/[.!?;:,।॥\n\-–—/()\[\]"'“”]/.test(gap)) out.push({ isBreak: true });
102
+ }
103
+ out.push({ w: m[0], isBreak: false });
104
+ last = m.index + m[0].length;
105
+ }
106
+ return out;
107
+ }
108
+ function toHashtag(input) {
109
+ const cleaned = input.replace(/^#/, "");
110
+ const parts = cleaned.split(/[\s\-_/]+/).filter(Boolean);
111
+ const joined = parts.map((p) => /^[a-z]/.test(p) ? p.charAt(0).toUpperCase() + p.slice(1) : p).join("");
112
+ const kept = [...joined].filter((ch) => /[\p{L}\p{M}\p{N}]/u.test(ch)).join("");
113
+ return kept.length >= 2 && kept.length <= 30 ? "#" + kept : "";
114
+ }
115
+ function extractEntities(text, gaz) {
116
+ const counts = /* @__PURE__ */ new Map();
117
+ const re = /\b([A-Z][a-zA-Z]+(?:\s+[A-Z][a-zA-Z]+){0,3})\b/g;
118
+ let m;
119
+ while ((m = re.exec(text)) !== null) {
120
+ const e = m[1];
121
+ counts.set(e, (counts.get(e) ?? 0) + 1);
122
+ }
123
+ for (const g of gaz) {
124
+ if (!g) continue;
125
+ const c = countOccurrences(text, g);
126
+ if (c > 0) counts.set(g, Math.max(counts.get(g) ?? 0, c));
127
+ }
128
+ return [...counts.entries()].map(([t, c]) => ({ text: t, count: c })).sort((a, b) => b.count - a.count || a.text.localeCompare(b.text)).filter((e) => e.text.includes(" ") || gaz.includes(e.text) || e.count > 1 || /[ऀ-ॿ]/.test(e.text));
129
+ }
130
+ function countOccurrences(hay, needle) {
131
+ if (!needle) return 0;
132
+ let n = 0;
133
+ let i = hay.indexOf(needle);
134
+ while (i !== -1) {
135
+ n++;
136
+ i = hay.indexOf(needle, i + needle.length);
137
+ }
138
+ return n;
139
+ }
140
+ function dedupe(arr) {
141
+ const seen = /* @__PURE__ */ new Set();
142
+ const out = [];
143
+ for (const x of arr) {
144
+ const k = x.toLowerCase();
145
+ if (!seen.has(k)) {
146
+ seen.add(k);
147
+ out.push(x);
148
+ }
149
+ }
150
+ return out;
151
+ }
152
+ function escapeRe(s) {
153
+ return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
154
+ }
155
+ function round(n) {
156
+ return Math.round(n * 1e3) / 1e3;
157
+ }
158
+
159
+ export { ENGLISH_STOPWORDS, NEPALI_STOPWORDS, keyphrase, toHashtag };
160
+ //# sourceMappingURL=index.js.map
161
+ //# sourceMappingURL=index.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"sources":["../src/stopwords.ts","../src/index.ts"],"names":[],"mappings":";AACO,IAAM,iBAAA,GACX,wpBAAA,CAQA,KAAA,CAAM,KAAK;AAGN,IAAM,mBACX,4hCAAA,CAUA,KAAA,CAAM,KAAK,CAAA,CAAE,OAAO,OAAO;;;AC0B7B,IAAM,OAAA,GAAU,oCAAA;AAChB,IAAM,aAAA,GAAgB,OAAA;AAEtB,SAAS,eAAe,IAAA,EAA2B;AACjD,EAAA,MAAM,QAAQ,IAAA,CAAK,KAAA,CAAM,QAAQ,CAAA,IAAK,EAAC,EAAG,MAAA;AAC1C,EAAA,MAAM,SAAS,IAAA,CAAK,KAAA,CAAM,WAAW,CAAA,IAAK,EAAC,EAAG,MAAA;AAC9C,EAAA,OAAO,IAAA,GAAO,QAAQ,IAAA,GAAO,IAAA;AAC/B;AAWO,SAAS,SAAA,CAAU,IAAA,EAAc,OAAA,GAA4B,EAAC,EAAoB;AACvF,EAAA,MAAM,QAAA,GAAW,OAAA,CAAQ,QAAA,IAAY,OAAA,CAAQ,QAAA,KAAa,SAAS,OAAA,CAAQ,QAAA,GAAW,cAAA,CAAe,IAAA,IAAQ,EAAE,CAAA;AAC/G,EAAA,MAAM,IAAA,GAAO,QAAQ,IAAA,IAAQ,EAAA;AAC7B,EAAA,MAAM,QAAA,GAAW,QAAQ,QAAA,IAAY,CAAA;AACrC,EAAA,MAAM,MAAA,GAAS,QAAQ,aAAA,IAAiB,CAAA;AACxC,EAAA,MAAM,IAAA,GAAO,OAAA,CAAQ,SAAA,KAAc,QAAA,KAAa,OAAO,gBAAA,GAAmB,iBAAA,CAAA;AAC1E,EAAA,MAAM,OAAO,IAAI,GAAA,CAAI,CAAC,GAAG,IAAA,EAAM,GAAI,OAAA,CAAQ,cAAA,IAAkB,EAAG,EAAE,GAAA,CAAI,CAAC,MAAM,CAAA,CAAE,WAAA,EAAa,CAAC,CAAA;AAC7F,EAAA,MAAM,OAAO,OAAA,CAAQ,SAAA,IAAa,EAAC,EAAG,OAAO,OAAO,CAAA;AAEpD,EAAA,MAAM,QAAyB,EAAE,OAAA,EAAS,EAAC,EAAG,MAAM,EAAC,EAAG,QAAA,EAAU,IAAI,QAAA,EAAU,IAAI,UAAA,EAAY,IAAI,QAAA,EAAS;AAC7G,EAAA,IAAI,CAAC,IAAA,IAAQ,CAAC,IAAA,CAAK,IAAA,IAAQ,OAAO,KAAA;AAGlC,EAAA,MAAM,aAAyB,EAAC;AAChC,EAAA,IAAI,UAAoB,EAAC;AAEzB,EAAA,MAAM,WAAA,GAAc,iBAAiB,IAAI,CAAA;AACzC,EAAA,KAAA,MAAW,OAAO,WAAA,EAAa;AAC7B,IAAA,IAAI,IAAI,OAAA,EAAS;AACf,MAAA,IAAI,OAAA,CAAQ,MAAA,EAAQ,UAAA,CAAW,IAAA,CAAK,OAAO,CAAA;AAC3C,MAAA,OAAA,GAAU,EAAC;AACX,MAAA;AAAA,IACF;AACA,IAAA,MAAM,EAAA,GAAK,GAAA,CAAI,CAAA,CAAG,WAAA,EAAY;AAC9B,IAAA,IAAI,IAAA,CAAK,GAAA,CAAI,EAAE,CAAA,IAAK,GAAA,CAAI,CAAA,CAAG,MAAA,GAAS,MAAA,IAAU,OAAA,CAAQ,IAAA,CAAK,GAAA,CAAI,CAAE,CAAA,EAAG;AAClE,MAAA,IAAI,OAAA,CAAQ,MAAA,EAAQ,UAAA,CAAW,IAAA,CAAK,OAAO,CAAA;AAC3C,MAAA,OAAA,GAAU,EAAC;AAAA,IACb,CAAA,MAAO;AACL,MAAA,OAAA,CAAQ,IAAA,CAAK,IAAI,CAAE,CAAA;AAAA,IACrB;AAAA,EACF;AACA,EAAA,IAAI,OAAA,CAAQ,MAAA,EAAQ,UAAA,CAAW,IAAA,CAAK,OAAO,CAAA;AAG3C,EAAA,MAAM,IAAA,uBAAW,GAAA,EAAoB;AACrC,EAAA,MAAM,MAAA,uBAAa,GAAA,EAAoB;AACvC,EAAA,KAAA,MAAW,UAAU,UAAA,EAAY;AAC/B,IAAA,MAAM,GAAA,GAAM,OAAO,MAAA,GAAS,CAAA;AAC5B,IAAA,KAAA,MAAW,KAAK,MAAA,EAAQ;AACtB,MAAA,MAAM,CAAA,GAAI,EAAE,WAAA,EAAY;AACxB,MAAA,IAAA,CAAK,IAAI,CAAA,EAAA,CAAI,IAAA,CAAK,IAAI,CAAC,CAAA,IAAK,KAAK,CAAC,CAAA;AAClC,MAAA,MAAA,CAAO,GAAA,CAAI,IAAI,MAAA,CAAO,GAAA,CAAI,CAAC,CAAA,IAAK,CAAA,IAAK,MAAM,CAAC,CAAA;AAAA,IAC9C;AAAA,EACF;AACA,EAAA,MAAM,SAAA,GAAY,CAAC,CAAA,KAAc;AAC/B,IAAA,MAAM,CAAA,GAAI,EAAE,WAAA,EAAY;AACxB,IAAA,MAAM,CAAA,GAAI,IAAA,CAAK,GAAA,CAAI,CAAC,CAAA,IAAK,CAAA;AACzB,IAAA,OAAA,CAAQ,MAAA,CAAO,GAAA,CAAI,CAAC,CAAA,IAAK,CAAA,IAAK,CAAA;AAAA,EAChC,CAAA;AAGA,EAAA,MAAM,QAAA,GAAW,IAAI,GAAA,CAAI,GAAA,CAAI,GAAA,CAAI,CAAC,CAAA,KAAM,CAAA,CAAE,WAAA,EAAa,CAAC,CAAA;AACxD,EAAA,MAAM,YAAA,uBAAmB,GAAA,EAA8D;AACvF,EAAA,IAAI,KAAA,GAAQ,CAAA;AACZ,EAAA,KAAA,MAAW,UAAU,UAAA,EAAY;AAC/B,IAAA,IAAI,MAAA,CAAO,MAAA,KAAW,CAAA,IAAK,MAAA,CAAO,SAAS,QAAA,EAAU;AACrD,IAAA,MAAM,KAAA,GAAQ,MAAA,CAAO,IAAA,CAAK,GAAG,CAAA;AAC7B,IAAA,MAAM,GAAA,GAAM,MAAM,WAAA,EAAY;AAC9B,IAAA,IAAI,KAAA,GAAQ,MAAA,CAAO,MAAA,CAAO,CAAC,CAAA,EAAG,MAAM,CAAA,GAAI,SAAA,CAAU,CAAC,CAAA,EAAG,CAAC,CAAA;AACvD,IAAA,IAAI,QAAA,CAAS,GAAA,CAAI,GAAG,CAAA,IAAK,GAAA,CAAI,IAAA,CAAK,CAAC,CAAA,KAAM,KAAA,CAAM,QAAA,CAAS,CAAC,CAAC,GAAG,KAAA,IAAS,GAAA;AACtE,IAAA,MAAM,IAAA,GAAO,YAAA,CAAa,GAAA,CAAI,GAAG,CAAA;AACjC,IAAA,IAAI,MAAM,IAAA,CAAK,KAAA,GAAQ,KAAK,GAAA,CAAI,IAAA,CAAK,OAAO,KAAK,CAAA;AAAA,SAC5C,YAAA,CAAa,IAAI,GAAA,EAAK,EAAE,QAAQ,KAAA,EAAO,KAAA,EAAO,KAAA,EAAO,KAAA,EAAA,EAAS,CAAA;AAAA,EACrE;AACA,EAAA,MAAM,UAAU,CAAC,GAAG,YAAA,CAAa,MAAA,EAAQ,CAAA,CACtC,IAAA,CAAK,CAAC,CAAA,EAAG,MAAM,CAAA,CAAE,KAAA,GAAQ,CAAA,CAAE,KAAA,IAAS,EAAE,KAAA,GAAQ,CAAA,CAAE,KAAK,CAAA,CACrD,MAAM,CAAA,EAAG,IAAI,CAAA,CACb,GAAA,CAAI,CAAC,CAAA,MAAO,EAAE,MAAA,EAAQ,CAAA,CAAE,QAAQ,KAAA,EAAO,KAAA,CAAM,CAAA,CAAE,KAAK,GAAE,CAAE,CAAA;AAG3D,EAAA,MAAM,OAAO,OAAA,CAAQ,GAAA,CAAI,CAAC,CAAA,KAAM,EAAE,MAAM,CAAA;AACxC,EAAA,MAAM,QAAA,GAAW,MAAA,CAAO,IAAA,CAAK,GAAA,CAAI,SAAS,CAAA,CAAE,MAAA,CAAO,CAAC,CAAA,KAAM,CAAA,CAAE,MAAA,GAAS,CAAC,CAAC,CAAA;AAGvE,EAAA,MAAM,QAAA,GAAW,eAAA,CAAgB,IAAA,EAAM,GAAG,CAAA;AAG1C,EAAA,MAAM,aAA6B,EAAC;AACpC,EAAA,IAAI,QAAQ,UAAA,EAAY;AACtB,IAAA,MAAM,KAAA,GAAQ,KAAK,WAAA,EAAY;AAC/B,IAAA,KAAA,MAAW,CAAC,KAAK,KAAK,CAAA,IAAK,OAAO,OAAA,CAAQ,OAAA,CAAQ,UAAU,CAAA,EAAG;AAC7D,MAAA,IAAI,KAAA,GAAQ,CAAA;AACZ,MAAA,KAAA,MAAW,QAAQ,KAAA,EAAO;AACxB,QAAA,IAAI,CAAC,IAAA,EAAM;AACX,QAAA,IAAI,aAAA,CAAc,IAAA,CAAK,IAAI,CAAA,EAAG;AAC5B,UAAA,KAAA,IAAS,gBAAA,CAAiB,MAAM,IAAI,CAAA;AAAA,QACtC,CAAA,MAAO;AACL,UAAA,MAAM,EAAA,GAAK,IAAI,MAAA,CAAO,CAAA,eAAA,EAAkB,QAAA,CAAS,KAAK,WAAA,EAAa,CAAC,CAAA,eAAA,CAAA,EAAmB,GAAG,CAAA;AAC1F,UAAA,KAAA,IAAA,CAAU,KAAA,CAAM,KAAA,CAAM,EAAE,CAAA,IAAK,EAAC,EAAG,MAAA;AAAA,QACnC;AAAA,MACF;AACA,MAAA,IAAI,KAAA,GAAQ,GAAG,UAAA,CAAW,IAAA,CAAK,EAAE,QAAA,EAAU,GAAA,EAAK,OAAO,CAAA;AAAA,IACzD;AACA,IAAA,UAAA,CAAW,IAAA,CAAK,CAAC,CAAA,EAAG,CAAA,KAAM,CAAA,CAAE,KAAA,GAAQ,CAAA,CAAE,KAAA,IAAS,CAAA,CAAE,QAAA,CAAS,aAAA,CAAc,CAAA,CAAE,QAAQ,CAAC,CAAA;AAAA,EACrF;AAEA,EAAA,OAAO,EAAE,OAAA,EAAS,IAAA,EAAM,QAAA,EAAU,QAAA,EAAU,YAAY,QAAA,EAAS;AACnE;AAMA,SAAS,iBAAiB,IAAA,EAA2B;AACnD,EAAA,MAAM,MAAmB,EAAC;AAC1B,EAAA,IAAI,IAAA,GAAO,CAAA;AACX,EAAA,IAAI,CAAA;AACJ,EAAA,OAAA,CAAQ,SAAA,GAAY,CAAA;AACpB,EAAA,OAAA,CAAQ,CAAA,GAAI,OAAA,CAAQ,IAAA,CAAK,IAAI,OAAO,IAAA,EAAM;AACxC,IAAA,IAAI,CAAA,CAAE,QAAQ,IAAA,EAAM;AAElB,MAAA,MAAM,GAAA,GAAM,IAAA,CAAK,KAAA,CAAM,IAAA,EAAM,EAAE,KAAK,CAAA;AACpC,MAAA,IAAI,6BAAA,CAA8B,KAAK,GAAG,CAAA,MAAO,IAAA,CAAK,EAAE,OAAA,EAAS,IAAA,EAAM,CAAA;AAAA,IACzE;AACA,IAAA,GAAA,CAAI,IAAA,CAAK,EAAE,CAAA,EAAG,CAAA,CAAE,CAAC,CAAA,EAAG,OAAA,EAAS,OAAO,CAAA;AACpC,IAAA,IAAA,GAAO,CAAA,CAAE,KAAA,GAAQ,CAAA,CAAE,CAAC,CAAA,CAAE,MAAA;AAAA,EACxB;AACA,EAAA,OAAO,GAAA;AACT;AAGO,SAAS,UAAU,KAAA,EAAuB;AAC/C,EAAA,MAAM,OAAA,GAAU,KAAA,CAAM,OAAA,CAAQ,IAAA,EAAM,EAAE,CAAA;AACtC,EAAA,MAAM,QAAQ,OAAA,CAAQ,KAAA,CAAM,WAAW,CAAA,CAAE,OAAO,OAAO,CAAA;AACvD,EAAA,MAAM,MAAA,GAAS,MACZ,GAAA,CAAI,CAAC,MAAO,QAAA,CAAS,IAAA,CAAK,CAAC,CAAA,GAAI,CAAA,CAAE,MAAA,CAAO,CAAC,CAAA,CAAE,WAAA,KAAgB,CAAA,CAAE,KAAA,CAAM,CAAC,CAAA,GAAI,CAAE,CAAA,CAC1E,IAAA,CAAK,EAAE,CAAA;AACV,EAAA,MAAM,IAAA,GAAO,CAAC,GAAG,MAAM,EAAE,MAAA,CAAO,CAAC,EAAA,KAAO,oBAAA,CAAqB,IAAA,CAAK,EAAE,CAAC,CAAA,CAAE,KAAK,EAAE,CAAA;AAC9E,EAAA,OAAO,KAAK,MAAA,IAAU,CAAA,IAAK,KAAK,MAAA,IAAU,EAAA,GAAK,MAAM,IAAA,GAAO,EAAA;AAC9D;AAEA,SAAS,eAAA,CAAgB,MAAc,GAAA,EAA4B;AACjE,EAAA,MAAM,MAAA,uBAAa,GAAA,EAAoB;AAEvC,EAAA,MAAM,EAAA,GAAK,iDAAA;AACX,EAAA,IAAI,CAAA;AACJ,EAAA,OAAA,CAAQ,CAAA,GAAI,EAAA,CAAG,IAAA,CAAK,IAAI,OAAO,IAAA,EAAM;AACnC,IAAA,MAAM,CAAA,GAAI,EAAE,CAAC,CAAA;AAEb,IAAA,MAAA,CAAO,IAAI,CAAA,EAAA,CAAI,MAAA,CAAO,IAAI,CAAC,CAAA,IAAK,KAAK,CAAC,CAAA;AAAA,EACxC;AAEA,EAAA,KAAA,MAAW,KAAK,GAAA,EAAK;AACnB,IAAA,IAAI,CAAC,CAAA,EAAG;AACR,IAAA,MAAM,CAAA,GAAI,gBAAA,CAAiB,IAAA,EAAM,CAAC,CAAA;AAClC,IAAA,IAAI,CAAA,GAAI,CAAA,EAAG,MAAA,CAAO,GAAA,CAAI,CAAA,EAAG,IAAA,CAAK,GAAA,CAAI,MAAA,CAAO,GAAA,CAAI,CAAC,CAAA,IAAK,CAAA,EAAG,CAAC,CAAC,CAAA;AAAA,EAC1D;AACA,EAAA,OAAO,CAAC,GAAG,MAAA,CAAO,OAAA,EAAS,EACxB,GAAA,CAAI,CAAC,CAAC,CAAA,EAAG,CAAC,CAAA,MAAO,EAAE,IAAA,EAAM,CAAA,EAAG,KAAA,EAAO,CAAA,EAAE,CAAE,CAAA,CACvC,IAAA,CAAK,CAAC,CAAA,EAAG,CAAA,KAAM,CAAA,CAAE,KAAA,GAAQ,CAAA,CAAE,KAAA,IAAS,EAAE,IAAA,CAAK,aAAA,CAAc,CAAA,CAAE,IAAI,CAAC,CAAA,CAChE,OAAO,CAAC,CAAA,KAAM,CAAA,CAAE,IAAA,CAAK,QAAA,CAAS,GAAG,KAAK,GAAA,CAAI,QAAA,CAAS,CAAA,CAAE,IAAI,CAAA,IAAK,CAAA,CAAE,KAAA,GAAQ,CAAA,IAAK,OAAA,CAAQ,IAAA,CAAK,CAAA,CAAE,IAAI,CAAC,CAAA;AACtG;AAEA,SAAS,gBAAA,CAAiB,KAAa,MAAA,EAAwB;AAC7D,EAAA,IAAI,CAAC,QAAQ,OAAO,CAAA;AACpB,EAAA,IAAI,CAAA,GAAI,CAAA;AACR,EAAA,IAAI,CAAA,GAAI,GAAA,CAAI,OAAA,CAAQ,MAAM,CAAA;AAC1B,EAAA,OAAO,MAAM,EAAA,EAAI;AACf,IAAA,CAAA,EAAA;AACA,IAAA,CAAA,GAAI,GAAA,CAAI,OAAA,CAAQ,MAAA,EAAQ,CAAA,GAAI,OAAO,MAAM,CAAA;AAAA,EAC3C;AACA,EAAA,OAAO,CAAA;AACT;AACA,SAAS,OAAO,GAAA,EAAyB;AACvC,EAAA,MAAM,IAAA,uBAAW,GAAA,EAAY;AAC7B,EAAA,MAAM,MAAgB,EAAC;AACvB,EAAA,KAAA,MAAW,KAAK,GAAA,EAAK;AACnB,IAAA,MAAM,CAAA,GAAI,EAAE,WAAA,EAAY;AACxB,IAAA,IAAI,CAAC,IAAA,CAAK,GAAA,CAAI,CAAC,CAAA,EAAG;AAChB,MAAA,IAAA,CAAK,IAAI,CAAC,CAAA;AACV,MAAA,GAAA,CAAI,KAAK,CAAC,CAAA;AAAA,IACZ;AAAA,EACF;AACA,EAAA,OAAO,GAAA;AACT;AACA,SAAS,SAAS,CAAA,EAAmB;AACnC,EAAA,OAAO,CAAA,CAAE,OAAA,CAAQ,qBAAA,EAAuB,MAAM,CAAA;AAChD;AACA,SAAS,MAAM,CAAA,EAAmB;AAChC,EAAA,OAAO,IAAA,CAAK,KAAA,CAAM,CAAA,GAAI,GAAI,CAAA,GAAI,GAAA;AAChC","file":"index.js","sourcesContent":["// Compact, high-frequency stopword lists. Extend via options.extraStopwords.\nexport const ENGLISH_STOPWORDS: string[] = (\n \"a an and are as at be but by for from has have he her his i in is it its of on or \" +\n \"that the their them they this to was were will with would you your we our us she him \" +\n \"not no do does did done can could should may might must shall about after all also \" +\n \"any because been before being between both during each few more most other over own \" +\n \"same so some such than then there these those through under until up very what when \" +\n \"where which who whom why how had having into out off down again further once here \" +\n \"said says say according reported reports report told tuesday monday wednesday thursday \" +\n \"friday saturday sunday am pm mr mrs ms dr new one two three per amid across\"\n).split(/\\s+/);\n\n// Common Nepali (Devanagari) function words and news filler.\nexport const NEPALI_STOPWORDS: string[] = (\n \"र को का की मा ले हो \" +\n \"छ थियो थिए हुन हुन्छ \" +\n \"भने भनी भन्ने पनि यो \" +\n \"त्यो ति त्यस यस गरेको \" +\n \"भएको लागि तथा एवं अनी \" +\n \"जस्तो जस्तै होस् छन् \" +\n \"गर्न गर्छ गर्यो भयो \" +\n \"अब सबै केही आफ्नो उनी \" +\n \"उसले उनले माथि तल भन्दा \" +\n \"सम्बन्धी बाट देखि सम्म\"\n).split(/\\s+/).filter(Boolean);\n","import { ENGLISH_STOPWORDS, NEPALI_STOPWORDS } from \"./stopwords.js\";\n\nexport { ENGLISH_STOPWORDS, NEPALI_STOPWORDS } from \"./stopwords.js\";\n\nexport interface KeyphraseOptions {\n /** \"en\", \"ne\", or \"auto\" (default) — picks built-in stopwords. */\n language?: \"en\" | \"ne\" | \"auto\";\n /** How many keyphrases/tags to return. Default 10. */\n topK?: number;\n /** Longest phrase, in words. Default 4. */\n maxWords?: number;\n /** Replace the built-in stopwords entirely. */\n stopwords?: string[];\n /** Add to the built-in stopwords. */\n extraStopwords?: string[];\n /** Known entities (people/places), en and ne forms — always surfaced and boosted. */\n gazetteer?: string[];\n /** Category lexicons: { category: [terms] }. Returns a vote per category by term hits. */\n categories?: Record<string, string[]>;\n /** Minimum characters for a candidate word. Default 2. */\n minWordLength?: number;\n}\n\nexport interface ScoredPhrase {\n phrase: string;\n score: number;\n}\nexport interface EntityHit {\n text: string;\n count: number;\n}\nexport interface CategoryVote {\n category: string;\n score: number;\n}\nexport interface KeyphraseResult {\n /** Top keyphrases by RAKE degree/frequency score. */\n phrases: ScoredPhrase[];\n /** Flat, de-duplicated tag strings (top phrases, normalized). */\n tags: string[];\n /** Hashtags built from the top tags (CamelCase for Latin, Devanagari kept whole). */\n hashtags: string[];\n /** Candidate named entities: Latin Title-Case runs + gazetteer hits. */\n entities: EntityHit[];\n /** Category votes (only when `categories` is provided), highest first. */\n categories: CategoryVote[];\n /** Detected/So-used language. */\n language: \"en\" | \"ne\";\n}\n\nconst WORD_RE = /[\\p{L}\\p{N}][\\p{L}\\p{M}\\p{N}‌‍]*/gu;\nconst DEVANAGARI_RE = /[ऀ-ॿ]/;\n\nfunction detectLanguage(text: string): \"en\" | \"ne\" {\n const deva = (text.match(/[ऀ-ॿ]/g) || []).length;\n const latin = (text.match(/[A-Za-z]/g) || []).length;\n return deva > latin ? \"ne\" : \"en\";\n}\n\nfunction words(text: string): { w: string; start: number }[] {\n const out: { w: string; start: number }[] = [];\n let m: RegExpExecArray | null;\n WORD_RE.lastIndex = 0;\n while ((m = WORD_RE.exec(text)) !== null) out.push({ w: m[0], start: m.index });\n return out;\n}\n\n/** Extract keyphrases, tags, hashtags, entities and category votes — no LLM. */\nexport function keyphrase(text: string, options: KeyphraseOptions = {}): KeyphraseResult {\n const language = options.language && options.language !== \"auto\" ? options.language : detectLanguage(text || \"\");\n const topK = options.topK ?? 10;\n const maxWords = options.maxWords ?? 4;\n const minLen = options.minWordLength ?? 2;\n const base = options.stopwords ?? (language === \"ne\" ? NEPALI_STOPWORDS : ENGLISH_STOPWORDS);\n const stop = new Set([...base, ...(options.extraStopwords ?? [])].map((s) => s.toLowerCase()));\n const gaz = (options.gazetteer ?? []).filter(Boolean);\n\n const empty: KeyphraseResult = { phrases: [], tags: [], hashtags: [], entities: [], categories: [], language };\n if (!text || !text.trim()) return empty;\n\n // 1. RAKE: break into candidate phrases at stopwords and non-word chars.\n const candidates: string[][] = [];\n let current: string[] = [];\n // Walk the raw text so punctuation also breaks phrases.\n const tokenStream = tokenizeWithGaps(text);\n for (const tok of tokenStream) {\n if (tok.isBreak) {\n if (current.length) candidates.push(current);\n current = [];\n continue;\n }\n const lw = tok.w!.toLowerCase();\n if (stop.has(lw) || tok.w!.length < minLen || /^\\d+$/.test(tok.w!)) {\n if (current.length) candidates.push(current);\n current = [];\n } else {\n current.push(tok.w!);\n }\n }\n if (current.length) candidates.push(current);\n\n // 2. Word scores: degree / frequency (classic RAKE).\n const freq = new Map<string, number>();\n const degree = new Map<string, number>();\n for (const phrase of candidates) {\n const deg = phrase.length - 1;\n for (const w of phrase) {\n const k = w.toLowerCase();\n freq.set(k, (freq.get(k) ?? 0) + 1);\n degree.set(k, (degree.get(k) ?? 0) + deg + 1);\n }\n }\n const wordScore = (w: string) => {\n const k = w.toLowerCase();\n const f = freq.get(k) ?? 1;\n return (degree.get(k) ?? f) / f;\n };\n\n // 3. Phrase scores; keep ≤ maxWords; dedupe by lowercase text; boost gazetteer.\n const gazLower = new Set(gaz.map((g) => g.toLowerCase()));\n const phraseScores = new Map<string, { phrase: string; score: number; order: number }>();\n let order = 0;\n for (const phrase of candidates) {\n if (phrase.length === 0 || phrase.length > maxWords) continue;\n const text2 = phrase.join(\" \");\n const key = text2.toLowerCase();\n let score = phrase.reduce((s, w) => s + wordScore(w), 0);\n if (gazLower.has(key) || gaz.some((g) => text2.includes(g))) score *= 1.5;\n const prev = phraseScores.get(key);\n if (prev) prev.score = Math.max(prev.score, score);\n else phraseScores.set(key, { phrase: text2, score, order: order++ });\n }\n const phrases = [...phraseScores.values()]\n .sort((a, b) => b.score - a.score || a.order - b.order)\n .slice(0, topK)\n .map((p) => ({ phrase: p.phrase, score: round(p.score) }));\n\n // 4. Tags + hashtags.\n const tags = phrases.map((p) => p.phrase);\n const hashtags = dedupe(tags.map(toHashtag).filter((h) => h.length > 1));\n\n // 5. Entities: Latin Title-Case runs + gazetteer hits.\n const entities = extractEntities(text, gaz);\n\n // 6. Category votes.\n const categories: CategoryVote[] = [];\n if (options.categories) {\n const lower = text.toLowerCase();\n for (const [cat, terms] of Object.entries(options.categories)) {\n let score = 0;\n for (const term of terms) {\n if (!term) continue;\n if (DEVANAGARI_RE.test(term)) {\n score += countOccurrences(text, term);\n } else {\n const re = new RegExp(`(?:^|[^a-z0-9])${escapeRe(term.toLowerCase())}(?:[^a-z0-9]|$)`, \"g\");\n score += (lower.match(re) || []).length;\n }\n }\n if (score > 0) categories.push({ category: cat, score });\n }\n categories.sort((a, b) => b.score - a.score || a.category.localeCompare(b.category));\n }\n\n return { phrases, tags, hashtags, entities, categories, language };\n}\n\ninterface StreamTok {\n w?: string;\n isBreak: boolean;\n}\nfunction tokenizeWithGaps(text: string): StreamTok[] {\n const out: StreamTok[] = [];\n let last = 0;\n let m: RegExpExecArray | null;\n WORD_RE.lastIndex = 0;\n while ((m = WORD_RE.exec(text)) !== null) {\n if (m.index > last) {\n // any non-word gap that contains sentence punctuation is a hard break\n const gap = text.slice(last, m.index);\n if (/[.!?;:,।॥\\n\\-–—/()\\[\\]\"'“”]/.test(gap)) out.push({ isBreak: true });\n }\n out.push({ w: m[0], isBreak: false });\n last = m.index + m[0].length;\n }\n return out;\n}\n\n/** Build a hashtag: strip #, CamelCase Latin words, keep Devanagari, keep only letters/marks/numbers. */\nexport function toHashtag(input: string): string {\n const cleaned = input.replace(/^#/, \"\");\n const parts = cleaned.split(/[\\s\\-_/]+/).filter(Boolean);\n const joined = parts\n .map((p) => (/^[a-z]/.test(p) ? p.charAt(0).toUpperCase() + p.slice(1) : p))\n .join(\"\");\n const kept = [...joined].filter((ch) => /[\\p{L}\\p{M}\\p{N}]/u.test(ch)).join(\"\");\n return kept.length >= 2 && kept.length <= 30 ? \"#\" + kept : \"\";\n}\n\nfunction extractEntities(text: string, gaz: string[]): EntityHit[] {\n const counts = new Map<string, number>();\n // Latin Title-Case runs of 1–4 words.\n const re = /\\b([A-Z][a-zA-Z]+(?:\\s+[A-Z][a-zA-Z]+){0,3})\\b/g;\n let m: RegExpExecArray | null;\n while ((m = re.exec(text)) !== null) {\n const e = m[1]!;\n // skip a lone word that starts a sentence and is common (heuristic: keep multiword or gazetteer)\n counts.set(e, (counts.get(e) ?? 0) + 1);\n }\n // Gazetteer (incl. Devanagari) always counted.\n for (const g of gaz) {\n if (!g) continue;\n const c = countOccurrences(text, g);\n if (c > 0) counts.set(g, Math.max(counts.get(g) ?? 0, c));\n }\n return [...counts.entries()]\n .map(([t, c]) => ({ text: t, count: c }))\n .sort((a, b) => b.count - a.count || a.text.localeCompare(b.text))\n .filter((e) => e.text.includes(\" \") || gaz.includes(e.text) || e.count > 1 || /[ऀ-ॿ]/.test(e.text));\n}\n\nfunction countOccurrences(hay: string, needle: string): number {\n if (!needle) return 0;\n let n = 0;\n let i = hay.indexOf(needle);\n while (i !== -1) {\n n++;\n i = hay.indexOf(needle, i + needle.length);\n }\n return n;\n}\nfunction dedupe(arr: string[]): string[] {\n const seen = new Set<string>();\n const out: string[] = [];\n for (const x of arr) {\n const k = x.toLowerCase();\n if (!seen.has(k)) {\n seen.add(k);\n out.push(x);\n }\n }\n return out;\n}\nfunction escapeRe(s: string): string {\n return s.replace(/[.*+?^${}()|[\\]\\\\]/g, \"\\\\$&\");\n}\nfunction round(n: number): number {\n return Math.round(n * 1000) / 1000;\n}\n"]}
package/package.json ADDED
@@ -0,0 +1,26 @@
1
+ {
2
+ "name": "@lacspace/keyphrase",
3
+ "version": "1.0.0",
4
+ "description": "Extractive keyphrases, tags, hashtags, named entities and category votes — a zero-dependency RAKE + TF-IDF engine with built-in English and Nepali stopwords, Devanagari-aware, so you can stop asking an LLM to generate tags/hashtags/entities. Deterministic, isomorphic, typed.",
5
+ "type": "module",
6
+ "main": "./dist/index.cjs",
7
+ "module": "./dist/index.js",
8
+ "types": "./dist/index.d.ts",
9
+ "exports": {
10
+ ".": {
11
+ "import": { "types": "./dist/index.d.ts", "default": "./dist/index.js" },
12
+ "require": { "types": "./dist/index.d.cts", "default": "./dist/index.cjs" }
13
+ }
14
+ },
15
+ "files": ["dist"],
16
+ "sideEffects": false,
17
+ "scripts": { "build": "tsup", "prepublishOnly": "npm run build" },
18
+ "keywords": ["keyphrase", "keyword-extraction", "rake", "tfidf", "tags", "hashtags", "named-entity", "category", "extractive", "nepali", "devanagari", "stopwords", "zero-dependency", "isomorphic", "typescript"],
19
+ "author": "Lacspace <contact@lacspace.com>",
20
+ "license": "SEE LICENSE IN LICENSE",
21
+ "homepage": "https://developer.lacspace.com/packages/keyphrase",
22
+ "repository": { "type": "git", "url": "git+https://github.com/lacspace/npm-packages.git", "directory": "keyphrase" },
23
+ "bugs": { "url": "https://github.com/lacspace/npm-packages/issues" },
24
+ "engines": { "node": ">=18" },
25
+ "publishConfig": { "access": "public" }
26
+ }