@lacspace/condense 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +51 -0
- package/README.md +86 -0
- package/dist/index.cjs +255 -0
- package/dist/index.cjs.map +1 -0
- package/dist/index.d.cts +87 -0
- package/dist/index.d.ts +87 -0
- package/dist/index.js +252 -0
- package/dist/index.js.map +1 -0
- package/package.json +27 -0
package/LICENSE
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
Lacspace Free Licence
|
|
2
|
+
Version 1.0, August 2026
|
|
3
|
+
|
|
4
|
+
Copyright (c) 2026 Lacspace
|
|
5
|
+
|
|
6
|
+
PREAMBLE
|
|
7
|
+
|
|
8
|
+
This software is published by Lacspace under the Lacspace Free Licence — a free,
|
|
9
|
+
permissive licence that lets you use this software for any purpose, including in
|
|
10
|
+
commercial products and services, at no cost. It grants the same freedoms as
|
|
11
|
+
common permissive open-source licences; the only condition is that this notice
|
|
12
|
+
travels with the software. The canonical, always-current text of this licence is
|
|
13
|
+
maintained at https://lacspace.com/licenses/lacspace-free-1.0
|
|
14
|
+
|
|
15
|
+
GRANT OF RIGHTS
|
|
16
|
+
|
|
17
|
+
Permission is hereby granted, free of charge, to any person or organisation
|
|
18
|
+
obtaining a copy of this software and its associated documentation and data files
|
|
19
|
+
(the "Software"), to deal in the Software without restriction, including without
|
|
20
|
+
limitation the rights to use, copy, modify, merge, publish, distribute,
|
|
21
|
+
sublicense, and/or sell copies of the Software, and to permit persons to whom the
|
|
22
|
+
Software is furnished to do so, subject to the conditions below. These rights are
|
|
23
|
+
granted for any purpose, personal or commercial, and are perpetual, worldwide,
|
|
24
|
+
non-exclusive, and royalty-free.
|
|
25
|
+
|
|
26
|
+
CONDITIONS
|
|
27
|
+
|
|
28
|
+
The above copyright notice, this permission notice, and the name of this licence
|
|
29
|
+
("Lacspace Free Licence") shall be included in all copies or substantial portions
|
|
30
|
+
of the Software.
|
|
31
|
+
|
|
32
|
+
TRADEMARKS
|
|
33
|
+
|
|
34
|
+
This licence does not grant permission to use the trade names, trademarks, service
|
|
35
|
+
marks, logos, or product names of Lacspace, except as required to reproduce the
|
|
36
|
+
notice above or to describe the origin of the Software in a truthful manner.
|
|
37
|
+
|
|
38
|
+
DISCLAIMER OF WARRANTY AND LIMITATION OF LIABILITY
|
|
39
|
+
|
|
40
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
41
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
|
42
|
+
FOR A PARTICULAR PURPOSE, AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
|
43
|
+
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES, OR OTHER LIABILITY, WHETHER IN
|
|
44
|
+
AN ACTION OF CONTRACT, TORT, OR OTHERWISE, ARISING FROM, OUT OF, OR IN CONNECTION
|
|
45
|
+
WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
|
46
|
+
|
|
47
|
+
---
|
|
48
|
+
|
|
49
|
+
The Lacspace Free Licence is a source-available, permissive licence and is not (as
|
|
50
|
+
of this version) an OSI-approved licence. In substance it grants the same freedoms
|
|
51
|
+
as the MIT Licence. Learn more at https://lacspace.com/licenses
|
package/README.md
ADDED
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
# @lacspace/condense
|
|
2
|
+
|
|
3
|
+
**Feed the LLM a fraction of the words.** Turn several articles about the same story into one short, deduplicated, token-budgeted digest that keeps the numbers, quotes and named entities — so a model only has to rewrite what matters instead of re-reading six full sources. Purely extractive, deterministic, and zero external dependencies.
|
|
4
|
+
|
|
5
|
+
Built for newsrooms, RAG context assembly, and anywhere you're paying per token to summarise overlapping documents.
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
npm i @lacspace/condense
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
## Use
|
|
12
|
+
|
|
13
|
+
```ts
|
|
14
|
+
import { condense } from "@lacspace/condense";
|
|
15
|
+
|
|
16
|
+
const digest = condense(
|
|
17
|
+
[
|
|
18
|
+
{ text: kathmanduPostArticle, label: "Kathmandu Post" },
|
|
19
|
+
{ text: himalayanTimesArticle, label: "Himalayan Times" },
|
|
20
|
+
{ text: onlineKhabarArticle, label: "Online Khabar" },
|
|
21
|
+
],
|
|
22
|
+
{ tokenBudget: 1500, gazetteer: ["Nepal Rastra Bank", "नेपाल राष्ट्र बैंक"] },
|
|
23
|
+
);
|
|
24
|
+
|
|
25
|
+
digest.text;
|
|
26
|
+
// [S1 Kathmandu Post]
|
|
27
|
+
// Nepal's central bank cut the policy rate to 5.5 percent on Sunday. The governor said "inflation is easing".
|
|
28
|
+
//
|
|
29
|
+
// [S2 Himalayan Times]
|
|
30
|
+
// Analysts welcomed the decision.
|
|
31
|
+
//
|
|
32
|
+
// [S3 Online Khabar]
|
|
33
|
+
// Traders expect loans to get cheaper.
|
|
34
|
+
|
|
35
|
+
digest.tokens; // ≤ tokenBudget
|
|
36
|
+
digest.droppedDup; // near-duplicate sentences removed across sources
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
Then hand `digest.text` to your model. On a typical 6-source cluster this is ~50% of the input tokens of feeding the trimmed sources directly, with the facts preserved.
|
|
40
|
+
|
|
41
|
+
## What it does
|
|
42
|
+
|
|
43
|
+
- **Ranks sentences** by BM25 centrality across the whole cluster (via `@lacspace/rerank`).
|
|
44
|
+
- **Always keeps** each source's lede, any sentence with a number/currency/percentage, any quoted sentence, and any sentence naming a gazetteer entity — these are tagged in `reasons`.
|
|
45
|
+
- **Drops near-duplicates** (3-gram Jaccard) so five outlets saying the same figure appear once.
|
|
46
|
+
- **Groups per source** under a header line in the original source order, in each source's original reading order — so your prompt can still credit "according to S2".
|
|
47
|
+
- **Respects a token budget** (`@lacspace/tokenizer`) and a `maxSentencesPerSource` cap so one long article can't crowd out the rest.
|
|
48
|
+
- **Devanagari-aware**: splits on `।` and `॥`, detects `०-९` digits and `रु`/`नेरु` amounts, never splits a quote.
|
|
49
|
+
- **Deterministic**: identical input always yields identical output, so results are cacheable.
|
|
50
|
+
|
|
51
|
+
## API
|
|
52
|
+
|
|
53
|
+
```ts
|
|
54
|
+
condense(sources: CondenseSource[], options?: CondenseOptions): CondenseResult
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
```ts
|
|
58
|
+
interface CondenseSource { text: string; label?: string; url?: string; publishedAt?: string | Date; }
|
|
59
|
+
|
|
60
|
+
interface CondenseOptions {
|
|
61
|
+
tokenBudget?: number; // default 1500
|
|
62
|
+
maxSentencesPerSource?: number; // default 8
|
|
63
|
+
dedupeThreshold?: number; // 3-gram Jaccard ≥ this → drop as duplicate. default 0.8
|
|
64
|
+
keepLede?: boolean; // default true
|
|
65
|
+
gazetteer?: string[]; // entities to always keep + boost (pass en and ne forms)
|
|
66
|
+
model?: string; // token-counter model hint
|
|
67
|
+
countTokens?: (text: string) => number; // override token counting
|
|
68
|
+
header?: (s: { idx: number; label: string; url?: string }) => string;
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
interface CondensedSentence { text: string; sourceIdx: number; start: number; end: number; score: number; reasons: string[]; }
|
|
72
|
+
interface CondenseResult {
|
|
73
|
+
text: string; // grouped digest
|
|
74
|
+
sources: { idx; label; url?; sentences: CondensedSentence[] }[];
|
|
75
|
+
sentences: CondensedSentence[]; // flat, in output order
|
|
76
|
+
tokens: number; droppedDup: number; totalSentences: number;
|
|
77
|
+
}
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
Every kept sentence carries `sourceIdx` and the `{start, end}` char offsets in that source's original text — so an editor view or a plagiarism check can trace each line back to where it came from. `condense` never rewrites text; publishing generated prose is the model's job.
|
|
81
|
+
|
|
82
|
+
Also exported: `splitSentences(text)` — the Devanagari-aware sentence splitter with offsets, useful on its own.
|
|
83
|
+
|
|
84
|
+
## Licence
|
|
85
|
+
|
|
86
|
+
[Lacspace Free Licence v1.0](https://developer.lacspace.com/licenses/lacspace-free-1.0) — free for personal and commercial use.
|
package/dist/index.cjs
ADDED
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
'use strict';
|
|
2
|
+
|
|
3
|
+
var rerank = require('@lacspace/rerank');
|
|
4
|
+
var tokenizer = require('@lacspace/tokenizer');
|
|
5
|
+
|
|
6
|
+
// src/index.ts
|
|
7
|
+
|
|
8
|
+
// src/features.ts
|
|
9
|
+
var CURRENCY = /(?:रु|नेरु|रुपैयाँ|₹|\$|€|£|Rs\.?|USD|NPR|INR)/i;
|
|
10
|
+
var NUMBERISH = /[0-9०-९][0-9०-९.,%]*/;
|
|
11
|
+
var PERCENT = /[%]|प्रतिशत/;
|
|
12
|
+
var QUOTED = /"[^"]{3,}"|“[^”]{3,}”|‘[^’]{3,}’/;
|
|
13
|
+
function hasNumber(text) {
|
|
14
|
+
return NUMBERISH.test(text) || CURRENCY.test(text) || PERCENT.test(text);
|
|
15
|
+
}
|
|
16
|
+
function hasQuote(text) {
|
|
17
|
+
return QUOTED.test(text);
|
|
18
|
+
}
|
|
19
|
+
function hasEntity(text, gaz) {
|
|
20
|
+
if (!gaz || gaz.size === 0) return false;
|
|
21
|
+
const lower = text.toLowerCase();
|
|
22
|
+
for (const term of gaz) {
|
|
23
|
+
if (!term) continue;
|
|
24
|
+
if (/[^\u0000-\u007F]/.test(term)) {
|
|
25
|
+
if (text.includes(term)) return true;
|
|
26
|
+
} else if (new RegExp(`(?:^|[^a-z])${escapeRe(term.toLowerCase())}(?:[^a-z]|$)`).test(lower)) {
|
|
27
|
+
return true;
|
|
28
|
+
}
|
|
29
|
+
}
|
|
30
|
+
return false;
|
|
31
|
+
}
|
|
32
|
+
function escapeRe(s) {
|
|
33
|
+
return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
34
|
+
}
|
|
35
|
+
function shingles(text, size = 3) {
|
|
36
|
+
const words = text.toLowerCase().replace(/[^\p{L}\p{N}\s]/gu, " ").split(/\s+/).filter(Boolean);
|
|
37
|
+
const out = /* @__PURE__ */ new Set();
|
|
38
|
+
if (words.length < size) {
|
|
39
|
+
if (words.length) out.add(words.join(" "));
|
|
40
|
+
return out;
|
|
41
|
+
}
|
|
42
|
+
for (let i = 0; i + size <= words.length; i++) out.add(words.slice(i, i + size).join(" "));
|
|
43
|
+
return out;
|
|
44
|
+
}
|
|
45
|
+
function jaccard(a, b) {
|
|
46
|
+
if (a.size === 0 && b.size === 0) return 0;
|
|
47
|
+
let inter = 0;
|
|
48
|
+
const [small, large] = a.size <= b.size ? [a, b] : [b, a];
|
|
49
|
+
for (const t of small) if (large.has(t)) inter++;
|
|
50
|
+
return inter / (a.size + b.size - inter);
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
// src/sentences.ts
|
|
54
|
+
var TERMINATORS = /* @__PURE__ */ new Set([".", "!", "?", "\u0964", "\u0965", "\u2026"]);
|
|
55
|
+
var OPEN_QUOTES = /* @__PURE__ */ new Set(["\u201C", "\u2018", "\xAB"]);
|
|
56
|
+
var CLOSE_QUOTES = /* @__PURE__ */ new Set(["\u201D", "\u2019", "\xBB"]);
|
|
57
|
+
var ABBREV = /* @__PURE__ */ new Set([
|
|
58
|
+
"mr",
|
|
59
|
+
"mrs",
|
|
60
|
+
"ms",
|
|
61
|
+
"dr",
|
|
62
|
+
"prof",
|
|
63
|
+
"sr",
|
|
64
|
+
"jr",
|
|
65
|
+
"st",
|
|
66
|
+
"vs",
|
|
67
|
+
"etc",
|
|
68
|
+
"inc",
|
|
69
|
+
"ltd",
|
|
70
|
+
"co",
|
|
71
|
+
"corp",
|
|
72
|
+
"govt",
|
|
73
|
+
"gen",
|
|
74
|
+
"rep",
|
|
75
|
+
"sen",
|
|
76
|
+
"gov",
|
|
77
|
+
"no",
|
|
78
|
+
"vol",
|
|
79
|
+
"fig",
|
|
80
|
+
"al",
|
|
81
|
+
"rs",
|
|
82
|
+
"u.s",
|
|
83
|
+
"u.k",
|
|
84
|
+
"e.g",
|
|
85
|
+
"i.e",
|
|
86
|
+
"a.m",
|
|
87
|
+
"p.m"
|
|
88
|
+
]);
|
|
89
|
+
function isDigit(ch) {
|
|
90
|
+
return ch >= "0" && ch <= "9" || ch >= "\u0966" && ch <= "\u096F";
|
|
91
|
+
}
|
|
92
|
+
function splitSentences(text) {
|
|
93
|
+
const out = [];
|
|
94
|
+
if (!text) return out;
|
|
95
|
+
let quoteDepth = 0;
|
|
96
|
+
let doubleOpen = false;
|
|
97
|
+
let start = 0;
|
|
98
|
+
const n = text.length;
|
|
99
|
+
const push = (from, to) => {
|
|
100
|
+
const raw = text.slice(from, to);
|
|
101
|
+
const trimmedStart = raw.length - raw.trimStart().length;
|
|
102
|
+
const trimmedEnd = raw.length - raw.trimEnd().length;
|
|
103
|
+
const s = from + trimmedStart;
|
|
104
|
+
const e = to - trimmedEnd;
|
|
105
|
+
if (e > s) out.push({ text: text.slice(s, e), start: s, end: e });
|
|
106
|
+
};
|
|
107
|
+
for (let i = 0; i < n; i++) {
|
|
108
|
+
const ch = text[i];
|
|
109
|
+
if (ch === '"') {
|
|
110
|
+
doubleOpen = !doubleOpen;
|
|
111
|
+
continue;
|
|
112
|
+
}
|
|
113
|
+
if (OPEN_QUOTES.has(ch)) {
|
|
114
|
+
quoteDepth++;
|
|
115
|
+
continue;
|
|
116
|
+
}
|
|
117
|
+
if (CLOSE_QUOTES.has(ch)) {
|
|
118
|
+
if (quoteDepth > 0) quoteDepth--;
|
|
119
|
+
continue;
|
|
120
|
+
}
|
|
121
|
+
if (!TERMINATORS.has(ch)) continue;
|
|
122
|
+
if (quoteDepth > 0 || doubleOpen) continue;
|
|
123
|
+
if (ch === ".") {
|
|
124
|
+
const prev = text[i - 1];
|
|
125
|
+
const next = text[i + 1];
|
|
126
|
+
if (prev && next && isDigit(prev) && isDigit(next)) continue;
|
|
127
|
+
let j = i - 1;
|
|
128
|
+
while (j >= 0 && /[A-Za-z.]/.test(text[j])) j--;
|
|
129
|
+
const word = text.slice(j + 1, i).toLowerCase();
|
|
130
|
+
if (ABBREV.has(word)) continue;
|
|
131
|
+
}
|
|
132
|
+
let k = i + 1;
|
|
133
|
+
while (k < n && (TERMINATORS.has(text[k]) || CLOSE_QUOTES.has(text[k]) || text[k] === '"' || text[k] === ")" || text[k] === "]")) {
|
|
134
|
+
if (text[k] === '"') doubleOpen = !doubleOpen;
|
|
135
|
+
k++;
|
|
136
|
+
}
|
|
137
|
+
if (k >= n || /\s/.test(text[k])) {
|
|
138
|
+
push(start, k);
|
|
139
|
+
start = k;
|
|
140
|
+
i = k - 1;
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
if (start < n) push(start, n);
|
|
144
|
+
return out;
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
// src/index.ts
|
|
148
|
+
function condense(sources, options = {}) {
|
|
149
|
+
const tokenBudget = options.tokenBudget ?? 1500;
|
|
150
|
+
const maxPer = options.maxSentencesPerSource ?? 8;
|
|
151
|
+
const dedupe = options.dedupeThreshold ?? 0.8;
|
|
152
|
+
const keepLede = options.keepLede ?? true;
|
|
153
|
+
const count = options.countTokens ?? ((t) => tokenizer.countTokens(t, options.model));
|
|
154
|
+
const gaz = options.gazetteer && options.gazetteer.length ? new Set(options.gazetteer) : null;
|
|
155
|
+
const header = options.header ?? ((s) => `[S${s.idx + 1} ${s.label}]`);
|
|
156
|
+
const labels = sources.map((s, i) => s.label?.trim() || `S${i + 1}`);
|
|
157
|
+
const cands = [];
|
|
158
|
+
let order = 0;
|
|
159
|
+
sources.forEach((src, sIdx) => {
|
|
160
|
+
const sents = splitSentences(src.text ?? "");
|
|
161
|
+
sents.forEach((sen, localIdx) => {
|
|
162
|
+
cands.push({
|
|
163
|
+
text: sen.text,
|
|
164
|
+
sourceIdx: sIdx,
|
|
165
|
+
start: sen.start,
|
|
166
|
+
end: sen.end,
|
|
167
|
+
order: order++,
|
|
168
|
+
score: 0,
|
|
169
|
+
lede: localIdx === 0,
|
|
170
|
+
number: hasNumber(sen.text),
|
|
171
|
+
quote: hasQuote(sen.text),
|
|
172
|
+
entity: hasEntity(sen.text, gaz),
|
|
173
|
+
shingle: shingles(sen.text),
|
|
174
|
+
tokens: count(sen.text)
|
|
175
|
+
});
|
|
176
|
+
});
|
|
177
|
+
});
|
|
178
|
+
const totalSentences = cands.length;
|
|
179
|
+
if (totalSentences === 0) {
|
|
180
|
+
return { text: "", sources: [], sentences: [], tokens: 0, droppedDup: 0, totalSentences: 0 };
|
|
181
|
+
}
|
|
182
|
+
const clusterQuery = cands.map((c) => c.text).join(" ");
|
|
183
|
+
const scored = rerank.bm25(
|
|
184
|
+
clusterQuery,
|
|
185
|
+
cands.map((c) => ({ id: String(c.order), text: c.text }))
|
|
186
|
+
);
|
|
187
|
+
const scoreById = new Map(scored.map((s) => [s.id, s.rerankScore]));
|
|
188
|
+
for (const c of cands) c.score = scoreById.get(String(c.order)) ?? 0;
|
|
189
|
+
const priority = (c) => (keepLede && c.lede ? 4 : 0) + (c.quote ? 3 : 0) + (c.number ? 2 : 0) + (c.entity ? 1 : 0);
|
|
190
|
+
const ranked = [...cands].sort((a, b) => {
|
|
191
|
+
const pa = priority(a);
|
|
192
|
+
const pb = priority(b);
|
|
193
|
+
if (pa !== pb) return pb - pa;
|
|
194
|
+
if (b.score !== a.score) return b.score - a.score;
|
|
195
|
+
return a.order - b.order;
|
|
196
|
+
});
|
|
197
|
+
const kept = [];
|
|
198
|
+
const keptShingles = [];
|
|
199
|
+
const perSource = /* @__PURE__ */ new Map();
|
|
200
|
+
let used = 0;
|
|
201
|
+
let droppedDup = 0;
|
|
202
|
+
for (const c of ranked) {
|
|
203
|
+
let dup = false;
|
|
204
|
+
for (const ks of keptShingles) {
|
|
205
|
+
if (jaccard(c.shingle, ks) >= dedupe) {
|
|
206
|
+
dup = true;
|
|
207
|
+
break;
|
|
208
|
+
}
|
|
209
|
+
}
|
|
210
|
+
if (dup) {
|
|
211
|
+
droppedDup++;
|
|
212
|
+
continue;
|
|
213
|
+
}
|
|
214
|
+
if ((perSource.get(c.sourceIdx) ?? 0) >= maxPer) continue;
|
|
215
|
+
if (used + c.tokens > tokenBudget && kept.length > 0) continue;
|
|
216
|
+
kept.push(c);
|
|
217
|
+
keptShingles.push(c.shingle);
|
|
218
|
+
perSource.set(c.sourceIdx, (perSource.get(c.sourceIdx) ?? 0) + 1);
|
|
219
|
+
used += c.tokens;
|
|
220
|
+
}
|
|
221
|
+
const bySource = /* @__PURE__ */ new Map();
|
|
222
|
+
for (const c of kept) {
|
|
223
|
+
const arr = bySource.get(c.sourceIdx) ?? [];
|
|
224
|
+
arr.push(c);
|
|
225
|
+
bySource.set(c.sourceIdx, arr);
|
|
226
|
+
}
|
|
227
|
+
const toOut = (c) => {
|
|
228
|
+
const reasons = [];
|
|
229
|
+
if (keepLede && c.lede) reasons.push("lede");
|
|
230
|
+
if (c.quote) reasons.push("quote");
|
|
231
|
+
if (c.number) reasons.push("number");
|
|
232
|
+
if (c.entity) reasons.push("entity");
|
|
233
|
+
if (reasons.length === 0) reasons.push("rank");
|
|
234
|
+
return { text: c.text, sourceIdx: c.sourceIdx, start: c.start, end: c.end, score: c.score, reasons };
|
|
235
|
+
};
|
|
236
|
+
const outSources = [];
|
|
237
|
+
const flat = [];
|
|
238
|
+
const parts = [];
|
|
239
|
+
sources.forEach((src, sIdx) => {
|
|
240
|
+
const list = (bySource.get(sIdx) ?? []).sort((a, b) => a.start - b.start);
|
|
241
|
+
if (list.length === 0) return;
|
|
242
|
+
const outSents = list.map(toOut);
|
|
243
|
+
outSources.push({ idx: sIdx, label: labels[sIdx], url: src.url, sentences: outSents });
|
|
244
|
+
flat.push(...outSents);
|
|
245
|
+
parts.push(`${header({ idx: sIdx, label: labels[sIdx], url: src.url })}
|
|
246
|
+
${outSents.map((s) => s.text).join(" ")}`);
|
|
247
|
+
});
|
|
248
|
+
const text = parts.join("\n\n");
|
|
249
|
+
return { text, sources: outSources, sentences: flat, tokens: count(text), droppedDup, totalSentences };
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
exports.condense = condense;
|
|
253
|
+
exports.splitSentences = splitSentences;
|
|
254
|
+
//# sourceMappingURL=index.cjs.map
|
|
255
|
+
//# sourceMappingURL=index.cjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"sources":["../src/features.ts","../src/sentences.ts","../src/index.ts"],"names":["countTokens","bm25"],"mappings":";;;;;;;;AAEA,IAAM,QAAA,GAAW,iDAAA;AAEjB,IAAM,SAAA,GAAY,sBAAA;AAClB,IAAM,OAAA,GAAU,aAAA;AAEhB,IAAM,MAAA,GAAS,kCAAA;AAER,SAAS,UAAU,IAAA,EAAuB;AAC/C,EAAA,OAAO,SAAA,CAAU,IAAA,CAAK,IAAI,CAAA,IAAK,QAAA,CAAS,KAAK,IAAI,CAAA,IAAK,OAAA,CAAQ,IAAA,CAAK,IAAI,CAAA;AACzE;AAEO,SAAS,SAAS,IAAA,EAAuB;AAC9C,EAAA,OAAO,MAAA,CAAO,KAAK,IAAI,CAAA;AACzB;AAGO,SAAS,SAAA,CAAU,MAAc,GAAA,EAAkC;AACxE,EAAA,IAAI,CAAC,GAAA,IAAO,GAAA,CAAI,IAAA,KAAS,GAAG,OAAO,KAAA;AACnC,EAAA,MAAM,KAAA,GAAQ,KAAK,WAAA,EAAY;AAC/B,EAAA,KAAA,MAAW,QAAQ,GAAA,EAAK;AACtB,IAAA,IAAI,CAAC,IAAA,EAAM;AAEX,IAAA,IAAI,kBAAA,CAAmB,IAAA,CAAK,IAAI,CAAA,EAAG;AACjC,MAAA,IAAI,IAAA,CAAK,QAAA,CAAS,IAAI,CAAA,EAAG,OAAO,IAAA;AAAA,IAClC,CAAA,MAAA,IAAW,IAAI,MAAA,CAAO,CAAA,YAAA,EAAe,QAAA,CAAS,IAAA,CAAK,WAAA,EAAa,CAAC,CAAA,YAAA,CAAc,CAAA,CAAE,IAAA,CAAK,KAAK,CAAA,EAAG;AAC5F,MAAA,OAAO,IAAA;AAAA,IACT;AAAA,EACF;AACA,EAAA,OAAO,KAAA;AACT;AAEA,SAAS,SAAS,CAAA,EAAmB;AACnC,EAAA,OAAO,CAAA,CAAE,OAAA,CAAQ,qBAAA,EAAuB,MAAM,CAAA;AAChD;AAGO,SAAS,QAAA,CAAS,IAAA,EAAc,IAAA,GAAO,CAAA,EAAgB;AAC5D,EAAA,MAAM,KAAA,GAAQ,IAAA,CACX,WAAA,EAAY,CACZ,OAAA,CAAQ,mBAAA,EAAqB,GAAG,CAAA,CAChC,KAAA,CAAM,KAAK,CAAA,CACX,MAAA,CAAO,OAAO,CAAA;AACjB,EAAA,MAAM,GAAA,uBAAU,GAAA,EAAY;AAC5B,EAAA,IAAI,KAAA,CAAM,SAAS,IAAA,EAAM;AACvB,IAAA,IAAI,MAAM,MAAA,EAAQ,GAAA,CAAI,IAAI,KAAA,CAAM,IAAA,CAAK,GAAG,CAAC,CAAA;AACzC,IAAA,OAAO,GAAA;AAAA,EACT;AACA,EAAA,KAAA,IAAS,IAAI,CAAA,EAAG,CAAA,GAAI,IAAA,IAAQ,KAAA,CAAM,QAAQ,CAAA,EAAA,EAAK,GAAA,CAAI,GAAA,CAAI,KAAA,CAAM,MAAM,CAAA,EAAG,CAAA,GAAI,IAAI,CAAA,CAAE,IAAA,CAAK,GAAG,CAAC,CAAA;AACzF,EAAA,OAAO,GAAA;AACT;AAEO,SAAS,OAAA,CAAQ,GAAgB,CAAA,EAAwB;AAC9D,EAAA,IAAI,EAAE,IAAA,KAAS,CAAA,IAAK,CAAA,CAAE,IAAA,KAAS,GAAG,OAAO,CAAA;AACzC,EAAA,IAAI,KAAA,GAAQ,CAAA;AACZ,EAAA,MAAM,CAAC,KAAA,EAAO,KAAK,CAAA,GAAI,EAAE,IAAA,IAAQ,CAAA,CAAE,IAAA,GAAO,CAAC,CAAA,EAAG,CAAC,CAAA,GAAI,CAAC,GAAG,CAAC,CAAA;AACxD,EAAA,KAAA,MAAW,KAAK,KAAA,EAAO,IAAI,KAAA,CAAM,GAAA,CAAI,CAAC,CAAA,EAAG,KAAA,EAAA;AACzC,EAAA,OAAO,KAAA,IAAS,CAAA,CAAE,IAAA,GAAO,CAAA,CAAE,IAAA,GAAO,KAAA,CAAA;AACpC;;;ACjDA,IAAM,WAAA,mBAAc,IAAI,GAAA,CAAI,CAAC,GAAA,EAAK,KAAK,GAAA,EAAK,QAAA,EAAK,QAAA,EAAK,QAAG,CAAC,CAAA;AAE1D,IAAM,8BAAc,IAAI,GAAA,CAAI,CAAC,QAAA,EAAK,QAAA,EAAK,MAAG,CAAC,CAAA;AAC3C,IAAM,+BAAe,IAAI,GAAA,CAAI,CAAC,QAAA,EAAK,QAAA,EAAK,MAAG,CAAC,CAAA;AAG5C,IAAM,MAAA,uBAAa,GAAA,CAAI;AAAA,EACrB,IAAA;AAAA,EAAM,KAAA;AAAA,EAAO,IAAA;AAAA,EAAM,IAAA;AAAA,EAAM,MAAA;AAAA,EAAQ,IAAA;AAAA,EAAM,IAAA;AAAA,EAAM,IAAA;AAAA,EAAM,IAAA;AAAA,EAAM,KAAA;AAAA,EAAO,KAAA;AAAA,EAAO,KAAA;AAAA,EACvE,IAAA;AAAA,EAAM,MAAA;AAAA,EAAQ,MAAA;AAAA,EAAQ,KAAA;AAAA,EAAO,KAAA;AAAA,EAAO,KAAA;AAAA,EAAO,KAAA;AAAA,EAAO,IAAA;AAAA,EAAM,KAAA;AAAA,EAAO,KAAA;AAAA,EAAO,IAAA;AAAA,EACtE,IAAA;AAAA,EAAM,KAAA;AAAA,EAAO,KAAA;AAAA,EAAO,KAAA;AAAA,EAAO,KAAA;AAAA,EAAO,KAAA;AAAA,EAAO;AAC3C,CAAC,CAAA;AAED,SAAS,QAAQ,EAAA,EAAqB;AACpC,EAAA,OAAQ,MAAM,GAAA,IAAO,EAAA,IAAM,GAAA,IAAS,EAAA,IAAM,YAAO,EAAA,IAAM,QAAA;AACzD;AAQO,SAAS,eAAe,IAAA,EAA0B;AACvD,EAAA,MAAM,MAAkB,EAAC;AACzB,EAAA,IAAI,CAAC,MAAM,OAAO,GAAA;AAClB,EAAA,IAAI,UAAA,GAAa,CAAA;AACjB,EAAA,IAAI,UAAA,GAAa,KAAA;AACjB,EAAA,IAAI,KAAA,GAAQ,CAAA;AACZ,EAAA,MAAM,IAAI,IAAA,CAAK,MAAA;AAEf,EAAA,MAAM,IAAA,GAAO,CAAC,IAAA,EAAc,EAAA,KAAe;AACzC,IAAA,MAAM,GAAA,GAAM,IAAA,CAAK,KAAA,CAAM,IAAA,EAAM,EAAE,CAAA;AAC/B,IAAA,MAAM,YAAA,GAAe,GAAA,CAAI,MAAA,GAAS,GAAA,CAAI,WAAU,CAAE,MAAA;AAClD,IAAA,MAAM,UAAA,GAAa,GAAA,CAAI,MAAA,GAAS,GAAA,CAAI,SAAQ,CAAE,MAAA;AAC9C,IAAA,MAAM,IAAI,IAAA,GAAO,YAAA;AACjB,IAAA,MAAM,IAAI,EAAA,GAAK,UAAA;AACf,IAAA,IAAI,CAAA,GAAI,CAAA,EAAG,GAAA,CAAI,IAAA,CAAK,EAAE,IAAA,EAAM,IAAA,CAAK,KAAA,CAAM,CAAA,EAAG,CAAC,CAAA,EAAG,KAAA,EAAO,CAAA,EAAG,GAAA,EAAK,GAAG,CAAA;AAAA,EAClE,CAAA;AAEA,EAAA,KAAA,IAAS,CAAA,GAAI,CAAA,EAAG,CAAA,GAAI,CAAA,EAAG,CAAA,EAAA,EAAK;AAC1B,IAAA,MAAM,EAAA,GAAK,KAAK,CAAC,CAAA;AACjB,IAAA,IAAI,OAAO,GAAA,EAAK;AACd,MAAA,UAAA,GAAa,CAAC,UAAA;AACd,MAAA;AAAA,IACF;AACA,IAAA,IAAI,WAAA,CAAY,GAAA,CAAI,EAAE,CAAA,EAAG;AACvB,MAAA,UAAA,EAAA;AACA,MAAA;AAAA,IACF;AACA,IAAA,IAAI,YAAA,CAAa,GAAA,CAAI,EAAE,CAAA,EAAG;AACxB,MAAA,IAAI,aAAa,CAAA,EAAG,UAAA,EAAA;AACpB,MAAA;AAAA,IACF;AACA,IAAA,IAAI,CAAC,WAAA,CAAY,GAAA,CAAI,EAAE,CAAA,EAAG;AAC1B,IAAA,IAAI,UAAA,GAAa,KAAK,UAAA,EAAY;AAGlC,IAAA,IAAI,OAAO,GAAA,EAAK;AACd,MAAA,MAAM,IAAA,GAAO,IAAA,CAAK,CAAA,GAAI,CAAC,CAAA;AACvB,MAAA,MAAM,IAAA,GAAO,IAAA,CAAK,CAAA,GAAI,CAAC,CAAA;AACvB,MAAA,IAAI,QAAQ,IAAA,IAAQ,OAAA,CAAQ,IAAI,CAAA,IAAK,OAAA,CAAQ,IAAI,CAAA,EAAG;AAEpD,MAAA,IAAI,IAAI,CAAA,GAAI,CAAA;AACZ,MAAA,OAAO,KAAK,CAAA,IAAK,WAAA,CAAY,KAAK,IAAA,CAAK,CAAC,CAAE,CAAA,EAAG,CAAA,EAAA;AAC7C,MAAA,MAAM,OAAO,IAAA,CAAK,KAAA,CAAM,IAAI,CAAA,EAAG,CAAC,EAAE,WAAA,EAAY;AAC9C,MAAA,IAAI,MAAA,CAAO,GAAA,CAAI,IAAI,CAAA,EAAG;AAAA,IACxB;AAGA,IAAA,IAAI,IAAI,CAAA,GAAI,CAAA;AACZ,IAAA,OAAO,CAAA,GAAI,CAAA,KAAM,WAAA,CAAY,GAAA,CAAI,IAAA,CAAK,CAAC,CAAE,CAAA,IAAK,YAAA,CAAa,GAAA,CAAI,IAAA,CAAK,CAAC,CAAE,CAAA,IAAK,IAAA,CAAK,CAAC,CAAA,KAAM,GAAA,IAAO,IAAA,CAAK,CAAC,CAAA,KAAM,GAAA,IAAO,IAAA,CAAK,CAAC,CAAA,KAAM,GAAA,CAAA,EAAM;AAClI,MAAA,IAAI,IAAA,CAAK,CAAC,CAAA,KAAM,GAAA,eAAkB,CAAC,UAAA;AACnC,MAAA,CAAA,EAAA;AAAA,IACF;AAEA,IAAA,IAAI,KAAK,CAAA,IAAK,IAAA,CAAK,KAAK,IAAA,CAAK,CAAC,CAAE,CAAA,EAAG;AACjC,MAAA,IAAA,CAAK,OAAO,CAAC,CAAA;AACb,MAAA,KAAA,GAAQ,CAAA;AACR,MAAA,CAAA,GAAI,CAAA,GAAI,CAAA;AAAA,IACV;AAAA,EACF;AACA,EAAA,IAAI,KAAA,GAAQ,CAAA,EAAG,IAAA,CAAK,KAAA,EAAO,CAAC,CAAA;AAC5B,EAAA,OAAO,GAAA;AACT;;;ACDO,SAAS,QAAA,CAAS,OAAA,EAA2B,OAAA,GAA2B,EAAC,EAAmB;AACjG,EAAA,MAAM,WAAA,GAAc,QAAQ,WAAA,IAAe,IAAA;AAC3C,EAAA,MAAM,MAAA,GAAS,QAAQ,qBAAA,IAAyB,CAAA;AAChD,EAAA,MAAM,MAAA,GAAS,QAAQ,eAAA,IAAmB,GAAA;AAC1C,EAAA,MAAM,QAAA,GAAW,QAAQ,QAAA,IAAY,IAAA;AACrC,EAAA,MAAM,KAAA,GAAQ,QAAQ,WAAA,KAAgB,CAAC,MAAcA,qBAAA,CAAY,CAAA,EAAG,QAAQ,KAAK,CAAA,CAAA;AACjF,EAAA,MAAM,GAAA,GAAM,OAAA,CAAQ,SAAA,IAAa,OAAA,CAAQ,SAAA,CAAU,SAAS,IAAI,GAAA,CAAI,OAAA,CAAQ,SAAS,CAAA,GAAI,IAAA;AACzF,EAAA,MAAM,MAAA,GACJ,OAAA,CAAQ,MAAA,KAAW,CAAC,CAAA,KAAsC,CAAA,EAAA,EAAK,CAAA,CAAE,GAAA,GAAM,CAAC,CAAA,CAAA,EAAI,CAAA,CAAE,KAAK,CAAA,CAAA,CAAA,CAAA;AAErF,EAAA,MAAM,MAAA,GAAS,OAAA,CAAQ,GAAA,CAAI,CAAC,CAAA,EAAG,CAAA,KAAM,CAAA,CAAE,KAAA,EAAO,IAAA,EAAK,IAAK,CAAA,CAAA,EAAI,CAAA,GAAI,CAAC,CAAA,CAAE,CAAA;AAGnE,EAAA,MAAM,QAAqB,EAAC;AAC5B,EAAA,IAAI,KAAA,GAAQ,CAAA;AACZ,EAAA,OAAA,CAAQ,OAAA,CAAQ,CAAC,GAAA,EAAK,IAAA,KAAS;AAC7B,IAAA,MAAM,KAAA,GAAQ,cAAA,CAAe,GAAA,CAAI,IAAA,IAAQ,EAAE,CAAA;AAC3C,IAAA,KAAA,CAAM,OAAA,CAAQ,CAAC,GAAA,EAAK,QAAA,KAAa;AAC/B,MAAA,KAAA,CAAM,IAAA,CAAK;AAAA,QACT,MAAM,GAAA,CAAI,IAAA;AAAA,QACV,SAAA,EAAW,IAAA;AAAA,QACX,OAAO,GAAA,CAAI,KAAA;AAAA,QACX,KAAK,GAAA,CAAI,GAAA;AAAA,QACT,KAAA,EAAO,KAAA,EAAA;AAAA,QACP,KAAA,EAAO,CAAA;AAAA,QACP,MAAM,QAAA,KAAa,CAAA;AAAA,QACnB,MAAA,EAAQ,SAAA,CAAU,GAAA,CAAI,IAAI,CAAA;AAAA,QAC1B,KAAA,EAAO,QAAA,CAAS,GAAA,CAAI,IAAI,CAAA;AAAA,QACxB,MAAA,EAAQ,SAAA,CAAU,GAAA,CAAI,IAAA,EAAM,GAAG,CAAA;AAAA,QAC/B,OAAA,EAAS,QAAA,CAAS,GAAA,CAAI,IAAI,CAAA;AAAA,QAC1B,MAAA,EAAQ,KAAA,CAAM,GAAA,CAAI,IAAI;AAAA,OACvB,CAAA;AAAA,IACH,CAAC,CAAA;AAAA,EACH,CAAC,CAAA;AACD,EAAA,MAAM,iBAAiB,KAAA,CAAM,MAAA;AAC7B,EAAA,IAAI,mBAAmB,CAAA,EAAG;AACxB,IAAA,OAAO,EAAE,IAAA,EAAM,EAAA,EAAI,OAAA,EAAS,EAAC,EAAG,SAAA,EAAW,EAAC,EAAG,MAAA,EAAQ,CAAA,EAAG,UAAA,EAAY,CAAA,EAAG,gBAAgB,CAAA,EAAE;AAAA,EAC7F;AAGA,EAAA,MAAM,YAAA,GAAe,MAAM,GAAA,CAAI,CAAC,MAAM,CAAA,CAAE,IAAI,CAAA,CAAE,IAAA,CAAK,GAAG,CAAA;AACtD,EAAA,MAAM,MAAA,GAASC,WAAA;AAAA,IACb,YAAA;AAAA,IACA,KAAA,CAAM,GAAA,CAAI,CAAC,CAAA,MAAO,EAAE,EAAA,EAAI,MAAA,CAAO,CAAA,CAAE,KAAK,CAAA,EAAG,IAAA,EAAM,CAAA,CAAE,MAAK,CAAE;AAAA,GAC1D;AACA,EAAA,MAAM,SAAA,GAAY,IAAI,GAAA,CAAI,MAAA,CAAO,GAAA,CAAI,CAAC,CAAA,KAAM,CAAC,CAAA,CAAE,EAAA,EAAI,CAAA,CAAE,WAAW,CAAC,CAAC,CAAA;AAClE,EAAA,KAAA,MAAW,CAAA,IAAK,KAAA,EAAO,CAAA,CAAE,KAAA,GAAQ,SAAA,CAAU,IAAI,MAAA,CAAO,CAAA,CAAE,KAAK,CAAC,CAAA,IAAK,CAAA;AAInE,EAAA,MAAM,WAAW,CAAC,CAAA,KAAA,CACf,YAAY,CAAA,CAAE,IAAA,GAAO,IAAI,CAAA,KAAM,CAAA,CAAE,KAAA,GAAQ,CAAA,GAAI,MAAM,CAAA,CAAE,MAAA,GAAS,IAAI,CAAA,CAAA,IAAM,CAAA,CAAE,SAAS,CAAA,GAAI,CAAA,CAAA;AAC1F,EAAA,MAAM,MAAA,GAAS,CAAC,GAAG,KAAK,EAAE,IAAA,CAAK,CAAC,GAAG,CAAA,KAAM;AACvC,IAAA,MAAM,EAAA,GAAK,SAAS,CAAC,CAAA;AACrB,IAAA,MAAM,EAAA,GAAK,SAAS,CAAC,CAAA;AACrB,IAAA,IAAI,EAAA,KAAO,EAAA,EAAI,OAAO,EAAA,GAAK,EAAA;AAC3B,IAAA,IAAI,EAAE,KAAA,KAAU,CAAA,CAAE,OAAO,OAAO,CAAA,CAAE,QAAQ,CAAA,CAAE,KAAA;AAC5C,IAAA,OAAO,CAAA,CAAE,QAAQ,CAAA,CAAE,KAAA;AAAA,EACrB,CAAC,CAAA;AAGD,EAAA,MAAM,OAAoB,EAAC;AAC3B,EAAA,MAAM,eAA8B,EAAC;AACrC,EAAA,MAAM,SAAA,uBAAgB,GAAA,EAAoB;AAC1C,EAAA,IAAI,IAAA,GAAO,CAAA;AACX,EAAA,IAAI,UAAA,GAAa,CAAA;AAEjB,EAAA,KAAA,MAAW,KAAK,MAAA,EAAQ;AACtB,IAAA,IAAI,GAAA,GAAM,KAAA;AACV,IAAA,KAAA,MAAW,MAAM,YAAA,EAAc;AAC7B,MAAA,IAAI,OAAA,CAAQ,CAAA,CAAE,OAAA,EAAS,EAAE,KAAK,MAAA,EAAQ;AACpC,QAAA,GAAA,GAAM,IAAA;AACN,QAAA;AAAA,MACF;AAAA,IACF;AACA,IAAA,IAAI,GAAA,EAAK;AACP,MAAA,UAAA,EAAA;AACA,MAAA;AAAA,IACF;AACA,IAAA,IAAA,CAAK,UAAU,GAAA,CAAI,CAAA,CAAE,SAAS,CAAA,IAAK,MAAM,MAAA,EAAQ;AACjD,IAAA,IAAI,OAAO,CAAA,CAAE,MAAA,GAAS,WAAA,IAAe,IAAA,CAAK,SAAS,CAAA,EAAG;AACtD,IAAA,IAAA,CAAK,KAAK,CAAC,CAAA;AACX,IAAA,YAAA,CAAa,IAAA,CAAK,EAAE,OAAO,CAAA;AAC3B,IAAA,SAAA,CAAU,GAAA,CAAI,EAAE,SAAA,EAAA,CAAY,SAAA,CAAU,IAAI,CAAA,CAAE,SAAS,CAAA,IAAK,CAAA,IAAK,CAAC,CAAA;AAChE,IAAA,IAAA,IAAQ,CAAA,CAAE,MAAA;AAAA,EACZ;AAGA,EAAA,MAAM,QAAA,uBAAe,GAAA,EAAyB;AAC9C,EAAA,KAAA,MAAW,KAAK,IAAA,EAAM;AACpB,IAAA,MAAM,MAAM,QAAA,CAAS,GAAA,CAAI,CAAA,CAAE,SAAS,KAAK,EAAC;AAC1C,IAAA,GAAA,CAAI,KAAK,CAAC,CAAA;AACV,IAAA,QAAA,CAAS,GAAA,CAAI,CAAA,CAAE,SAAA,EAAW,GAAG,CAAA;AAAA,EAC/B;AAEA,EAAA,MAAM,KAAA,GAAQ,CAAC,CAAA,KAAoC;AACjD,IAAA,MAAM,UAAoB,EAAC;AAC3B,IAAA,IAAI,QAAA,IAAY,CAAA,CAAE,IAAA,EAAM,OAAA,CAAQ,KAAK,MAAM,CAAA;AAC3C,IAAA,IAAI,CAAA,CAAE,KAAA,EAAO,OAAA,CAAQ,IAAA,CAAK,OAAO,CAAA;AACjC,IAAA,IAAI,CAAA,CAAE,MAAA,EAAQ,OAAA,CAAQ,IAAA,CAAK,QAAQ,CAAA;AACnC,IAAA,IAAI,CAAA,CAAE,MAAA,EAAQ,OAAA,CAAQ,IAAA,CAAK,QAAQ,CAAA;AACnC,IAAA,IAAI,OAAA,CAAQ,MAAA,KAAW,CAAA,EAAG,OAAA,CAAQ,KAAK,MAAM,CAAA;AAC7C,IAAA,OAAO,EAAE,IAAA,EAAM,CAAA,CAAE,IAAA,EAAM,SAAA,EAAW,EAAE,SAAA,EAAW,KAAA,EAAO,CAAA,CAAE,KAAA,EAAO,KAAK,CAAA,CAAE,GAAA,EAAK,KAAA,EAAO,CAAA,CAAE,OAAO,OAAA,EAAQ;AAAA,EACrG,CAAA;AAEA,EAAA,MAAM,aAAgC,EAAC;AACvC,EAAA,MAAM,OAA4B,EAAC;AACnC,EAAA,MAAM,QAAkB,EAAC;AACzB,EAAA,OAAA,CAAQ,OAAA,CAAQ,CAAC,GAAA,EAAK,IAAA,KAAS;AAC7B,IAAA,MAAM,IAAA,GAAA,CAAQ,QAAA,CAAS,GAAA,CAAI,IAAI,KAAK,EAAC,EAAG,IAAA,CAAK,CAAC,CAAA,EAAG,CAAA,KAAM,CAAA,CAAE,KAAA,GAAQ,EAAE,KAAK,CAAA;AACxE,IAAA,IAAI,IAAA,CAAK,WAAW,CAAA,EAAG;AACvB,IAAA,MAAM,QAAA,GAAW,IAAA,CAAK,GAAA,CAAI,KAAK,CAAA;AAC/B,IAAA,UAAA,CAAW,IAAA,CAAK,EAAE,GAAA,EAAK,IAAA,EAAM,KAAA,EAAO,MAAA,CAAO,IAAI,CAAA,EAAI,GAAA,EAAK,GAAA,CAAI,GAAA,EAAK,SAAA,EAAW,UAAU,CAAA;AACtF,IAAA,IAAA,CAAK,IAAA,CAAK,GAAG,QAAQ,CAAA;AACrB,IAAA,KAAA,CAAM,IAAA,CAAK,CAAA,EAAG,MAAA,CAAO,EAAE,KAAK,IAAA,EAAM,KAAA,EAAO,MAAA,CAAO,IAAI,CAAA,EAAI,GAAA,EAAK,GAAA,CAAI,GAAA,EAAK,CAAC;AAAA,EAAK,QAAA,CAAS,GAAA,CAAI,CAAC,CAAA,KAAM,CAAA,CAAE,IAAI,CAAA,CAAE,IAAA,CAAK,GAAG,CAAC,CAAA,CAAE,CAAA;AAAA,EACrH,CAAC,CAAA;AAED,EAAA,MAAM,IAAA,GAAO,KAAA,CAAM,IAAA,CAAK,MAAM,CAAA;AAC9B,EAAA,OAAO,EAAE,IAAA,EAAM,OAAA,EAAS,UAAA,EAAY,SAAA,EAAW,IAAA,EAAM,MAAA,EAAQ,KAAA,CAAM,IAAI,CAAA,EAAG,UAAA,EAAY,cAAA,EAAe;AACvG","file":"index.cjs","sourcesContent":["// Zero-dep detectors for the sentence features we always want to keep.\n\nconst CURRENCY = /(?:रु|नेरु|रुपैयाँ|₹|\\$|€|£|Rs\\.?|USD|NPR|INR)/i;\n// A number: Latin or Devanagari digits, optionally with separators/percent.\nconst NUMBERISH = /[0-9०-९][0-9०-९.,%]*/;\nconst PERCENT = /[%]|प्रतिशत/;\n// A quotation: matched straight or curly quotes with content between.\nconst QUOTED = /\"[^\"]{3,}\"|“[^”]{3,}”|‘[^’]{3,}’/;\n\nexport function hasNumber(text: string): boolean {\n return NUMBERISH.test(text) || CURRENCY.test(text) || PERCENT.test(text);\n}\n\nexport function hasQuote(text: string): boolean {\n return QUOTED.test(text);\n}\n\n/** True if any gazetteer term appears in the text (case-insensitive, whole-token where Latin). */\nexport function hasEntity(text: string, gaz: Set<string> | null): boolean {\n if (!gaz || gaz.size === 0) return false;\n const lower = text.toLowerCase();\n for (const term of gaz) {\n if (!term) continue;\n // Devanagari has no case; Latin terms are matched with word boundaries.\n if (/[^\\u0000-\\u007F]/.test(term)) {\n if (text.includes(term)) return true;\n } else if (new RegExp(`(?:^|[^a-z])${escapeRe(term.toLowerCase())}(?:[^a-z]|$)`).test(lower)) {\n return true;\n }\n }\n return false;\n}\n\nfunction escapeRe(s: string): string {\n return s.replace(/[.*+?^${}()|[\\]\\\\]/g, \"\\\\$&\");\n}\n\n/** Word 3-gram shingles for near-duplicate detection. */\nexport function shingles(text: string, size = 3): Set<string> {\n const words = text\n .toLowerCase()\n .replace(/[^\\p{L}\\p{N}\\s]/gu, \" \")\n .split(/\\s+/)\n .filter(Boolean);\n const out = new Set<string>();\n if (words.length < size) {\n if (words.length) out.add(words.join(\" \"));\n return out;\n }\n for (let i = 0; i + size <= words.length; i++) out.add(words.slice(i, i + size).join(\" \"));\n return out;\n}\n\nexport function jaccard(a: Set<string>, b: Set<string>): number {\n if (a.size === 0 && b.size === 0) return 0;\n let inter = 0;\n const [small, large] = a.size <= b.size ? [a, b] : [b, a];\n for (const t of small) if (large.has(t)) inter++;\n return inter / (a.size + b.size - inter);\n}\n","/** A sentence with its character offsets in the original source text. */\nexport interface Sentence {\n text: string;\n /** Inclusive start offset in the source. */\n start: number;\n /** Exclusive end offset in the source. */\n end: number;\n}\n\n// Terminators: Latin . ! ? plus Devanagari danda । and double danda ॥ and the ellipsis.\nconst TERMINATORS = new Set([\".\", \"!\", \"?\", \"।\", \"॥\", \"…\"]);\n// Quote characters whose parity we track so we never split inside a quote.\nconst OPEN_QUOTES = new Set([\"“\", \"‘\", \"«\"]); // \" ' «\nconst CLOSE_QUOTES = new Set([\"”\", \"’\", \"»\"]); // \" ' »\n\n// Common abbreviations after which a period does NOT end a sentence.\nconst ABBREV = new Set([\n \"mr\", \"mrs\", \"ms\", \"dr\", \"prof\", \"sr\", \"jr\", \"st\", \"vs\", \"etc\", \"inc\", \"ltd\",\n \"co\", \"corp\", \"govt\", \"gen\", \"rep\", \"sen\", \"gov\", \"no\", \"vol\", \"fig\", \"al\",\n \"rs\", \"u.s\", \"u.k\", \"e.g\", \"i.e\", \"a.m\", \"p.m\",\n]);\n\nfunction isDigit(ch: string): boolean {\n return (ch >= \"0\" && ch <= \"9\") || (ch >= \"०\" && ch <= \"९\");\n}\n\n/**\n * Split `text` into sentences with character offsets. Handles Latin and\n * Devanagari terminators (। ॥), never splits inside a quotation, and does not\n * break on decimals (3.5), abbreviations (Dr.) or a terminator glued to a digit.\n * Deterministic: identical input always yields identical output.\n */\nexport function splitSentences(text: string): Sentence[] {\n const out: Sentence[] = [];\n if (!text) return out;\n let quoteDepth = 0;\n let doubleOpen = false;\n let start = 0;\n const n = text.length;\n\n const push = (from: number, to: number) => {\n const raw = text.slice(from, to);\n const trimmedStart = raw.length - raw.trimStart().length;\n const trimmedEnd = raw.length - raw.trimEnd().length;\n const s = from + trimmedStart;\n const e = to - trimmedEnd;\n if (e > s) out.push({ text: text.slice(s, e), start: s, end: e });\n };\n\n for (let i = 0; i < n; i++) {\n const ch = text[i]!;\n if (ch === '\"') {\n doubleOpen = !doubleOpen;\n continue;\n }\n if (OPEN_QUOTES.has(ch)) {\n quoteDepth++;\n continue;\n }\n if (CLOSE_QUOTES.has(ch)) {\n if (quoteDepth > 0) quoteDepth--;\n continue;\n }\n if (!TERMINATORS.has(ch)) continue;\n if (quoteDepth > 0 || doubleOpen) continue; // inside a quote — keep it whole\n\n // A period between digits (3.5) or in an abbreviation is not a break.\n if (ch === \".\") {\n const prev = text[i - 1];\n const next = text[i + 1];\n if (prev && next && isDigit(prev) && isDigit(next)) continue;\n // trailing abbreviation like \"Dr.\" — look back to the word\n let j = i - 1;\n while (j >= 0 && /[A-Za-z.]/.test(text[j]!)) j--;\n const word = text.slice(j + 1, i).toLowerCase();\n if (ABBREV.has(word)) continue;\n }\n\n // Consume any run of terminators/closing quotes/brackets.\n let k = i + 1;\n while (k < n && (TERMINATORS.has(text[k]!) || CLOSE_QUOTES.has(text[k]!) || text[k] === '\"' || text[k] === \")\" || text[k] === \"]\")) {\n if (text[k] === '\"') doubleOpen = !doubleOpen;\n k++;\n }\n // Must be followed by whitespace or end of text to count as a break.\n if (k >= n || /\\s/.test(text[k]!)) {\n push(start, k);\n start = k;\n i = k - 1;\n }\n }\n if (start < n) push(start, n);\n return out;\n}\n","import { bm25 } from \"@lacspace/rerank\";\nimport { countTokens } from \"@lacspace/tokenizer\";\nimport { hasEntity, hasNumber, hasQuote, jaccard, shingles } from \"./features.js\";\nimport { splitSentences } from \"./sentences.js\";\n\nexport { splitSentences } from \"./sentences.js\";\nexport type { Sentence } from \"./sentences.js\";\n\n/** One input article about the story. */\nexport interface CondenseSource {\n text: string;\n /** Short label used in the grouped output header, e.g. \"Kathmandu Post\". Defaults to \"S{n}\". */\n label?: string;\n url?: string;\n publishedAt?: string | Date;\n}\n\nexport interface CondenseOptions {\n /** Total token budget for the condensed digest. Default 1500. */\n tokenBudget?: number;\n /** Cap on kept sentences from any single source, so one long article can't crowd out the rest. Default 8. */\n maxSentencesPerSource?: number;\n /** Drop a sentence whose 3-gram Jaccard similarity to an already-kept sentence is ≥ this. Default 0.8. */\n dedupeThreshold?: number;\n /** Always keep each source's first sentence (the lede) regardless of score. Default true. */\n keepLede?: boolean;\n /** Named entities (people/places) to always keep and to boost — pass en and ne forms. */\n gazetteer?: string[];\n /** Model hint for token counting (passed to @lacspace/tokenizer). */\n model?: string;\n /** Override token counting entirely. Default: @lacspace/tokenizer countTokens. */\n countTokens?: (text: string) => number;\n /** Header format for each source group. Default `[S{n} {label}]`. */\n header?: (source: { idx: number; label: string; url?: string }) => string;\n}\n\nexport interface CondensedSentence {\n text: string;\n /** Index of the source this sentence came from (0-based). */\n sourceIdx: number;\n /** Character offsets in that source's original text. */\n start: number;\n end: number;\n /** BM25 centrality score within the cluster. */\n score: number;\n /** Why it was kept: any of \"lede\", \"number\", \"quote\", \"entity\", \"rank\". */\n reasons: string[];\n}\n\nexport interface CondensedSource {\n idx: number;\n label: string;\n url?: string;\n sentences: CondensedSentence[];\n}\n\nexport interface CondenseResult {\n /** The digest: kept sentences grouped per source (source order) under a header line. */\n text: string;\n /** Kept sentences grouped per source, each in the source's original order. */\n sources: CondensedSource[];\n /** All kept sentences, flat, in output (source-grouped) order. */\n sentences: CondensedSentence[];\n /** Token count of `text` (same counter used for the budget). */\n tokens: number;\n /** How many near-duplicate sentences were dropped. */\n droppedDup: number;\n /** Total sentences seen across all sources. */\n totalSentences: number;\n}\n\ninterface Candidate {\n text: string;\n sourceIdx: number;\n start: number;\n end: number;\n order: number; // global index for stable ties\n score: number;\n lede: boolean;\n number: boolean;\n quote: boolean;\n entity: boolean;\n shingle: Set<string>;\n tokens: number;\n}\n\n/**\n * Condense several sources on the same story into one short, deduplicated,\n * token-budgeted digest that keeps the numbers, quotes and named entities — so\n * an LLM only rewrites a fraction of the words. Purely extractive and\n * deterministic; no network, no model.\n */\nexport function condense(sources: CondenseSource[], options: CondenseOptions = {}): CondenseResult {\n const tokenBudget = options.tokenBudget ?? 1500;\n const maxPer = options.maxSentencesPerSource ?? 8;\n const dedupe = options.dedupeThreshold ?? 0.8;\n const keepLede = options.keepLede ?? true;\n const count = options.countTokens ?? ((t: string) => countTokens(t, options.model));\n const gaz = options.gazetteer && options.gazetteer.length ? new Set(options.gazetteer) : null;\n const header =\n options.header ?? ((s: { idx: number; label: string }) => `[S${s.idx + 1} ${s.label}]`);\n\n const labels = sources.map((s, i) => s.label?.trim() || `S${i + 1}`);\n\n // 1. Split every source into sentences with offsets.\n const cands: Candidate[] = [];\n let order = 0;\n sources.forEach((src, sIdx) => {\n const sents = splitSentences(src.text ?? \"\");\n sents.forEach((sen, localIdx) => {\n cands.push({\n text: sen.text,\n sourceIdx: sIdx,\n start: sen.start,\n end: sen.end,\n order: order++,\n score: 0,\n lede: localIdx === 0,\n number: hasNumber(sen.text),\n quote: hasQuote(sen.text),\n entity: hasEntity(sen.text, gaz),\n shingle: shingles(sen.text),\n tokens: count(sen.text),\n });\n });\n });\n const totalSentences = cands.length;\n if (totalSentences === 0) {\n return { text: \"\", sources: [], sentences: [], tokens: 0, droppedDup: 0, totalSentences: 0 };\n }\n\n // 2. Score sentence centrality with BM25 against the whole cluster.\n const clusterQuery = cands.map((c) => c.text).join(\" \");\n const scored = bm25(\n clusterQuery,\n cands.map((c) => ({ id: String(c.order), text: c.text })),\n );\n const scoreById = new Map(scored.map((s) => [s.id, s.rerankScore]));\n for (const c of cands) c.score = scoreById.get(String(c.order)) ?? 0;\n\n // 3. Priority: always-keep (lede/number/quote/entity) first, then by score.\n // Deterministic tie-break by global order.\n const priority = (c: Candidate) =>\n (keepLede && c.lede ? 4 : 0) + (c.quote ? 3 : 0) + (c.number ? 2 : 0) + (c.entity ? 1 : 0);\n const ranked = [...cands].sort((a, b) => {\n const pa = priority(a);\n const pb = priority(b);\n if (pa !== pb) return pb - pa;\n if (b.score !== a.score) return b.score - a.score;\n return a.order - b.order;\n });\n\n // 4. Greedy selection: dedupe against kept, respect per-source cap and budget.\n const kept: Candidate[] = [];\n const keptShingles: Set<string>[] = [];\n const perSource = new Map<number, number>();\n let used = 0;\n let droppedDup = 0;\n\n for (const c of ranked) {\n let dup = false;\n for (const ks of keptShingles) {\n if (jaccard(c.shingle, ks) >= dedupe) {\n dup = true;\n break;\n }\n }\n if (dup) {\n droppedDup++;\n continue;\n }\n if ((perSource.get(c.sourceIdx) ?? 0) >= maxPer) continue;\n if (used + c.tokens > tokenBudget && kept.length > 0) continue; // keep at least one\n kept.push(c);\n keptShingles.push(c.shingle);\n perSource.set(c.sourceIdx, (perSource.get(c.sourceIdx) ?? 0) + 1);\n used += c.tokens;\n }\n\n // 5. Group kept sentences per source, each in original reading order.\n const bySource = new Map<number, Candidate[]>();\n for (const c of kept) {\n const arr = bySource.get(c.sourceIdx) ?? [];\n arr.push(c);\n bySource.set(c.sourceIdx, arr);\n }\n\n const toOut = (c: Candidate): CondensedSentence => {\n const reasons: string[] = [];\n if (keepLede && c.lede) reasons.push(\"lede\");\n if (c.quote) reasons.push(\"quote\");\n if (c.number) reasons.push(\"number\");\n if (c.entity) reasons.push(\"entity\");\n if (reasons.length === 0) reasons.push(\"rank\");\n return { text: c.text, sourceIdx: c.sourceIdx, start: c.start, end: c.end, score: c.score, reasons };\n };\n\n const outSources: CondensedSource[] = [];\n const flat: CondensedSentence[] = [];\n const parts: string[] = [];\n sources.forEach((src, sIdx) => {\n const list = (bySource.get(sIdx) ?? []).sort((a, b) => a.start - b.start);\n if (list.length === 0) return;\n const outSents = list.map(toOut);\n outSources.push({ idx: sIdx, label: labels[sIdx]!, url: src.url, sentences: outSents });\n flat.push(...outSents);\n parts.push(`${header({ idx: sIdx, label: labels[sIdx]!, url: src.url })}\\n${outSents.map((s) => s.text).join(\" \")}`);\n });\n\n const text = parts.join(\"\\n\\n\");\n return { text, sources: outSources, sentences: flat, tokens: count(text), droppedDup, totalSentences };\n}\n"]}
|
package/dist/index.d.cts
ADDED
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
/** A sentence with its character offsets in the original source text. */
|
|
2
|
+
interface Sentence {
|
|
3
|
+
text: string;
|
|
4
|
+
/** Inclusive start offset in the source. */
|
|
5
|
+
start: number;
|
|
6
|
+
/** Exclusive end offset in the source. */
|
|
7
|
+
end: number;
|
|
8
|
+
}
|
|
9
|
+
/**
|
|
10
|
+
* Split `text` into sentences with character offsets. Handles Latin and
|
|
11
|
+
* Devanagari terminators (। ॥), never splits inside a quotation, and does not
|
|
12
|
+
* break on decimals (3.5), abbreviations (Dr.) or a terminator glued to a digit.
|
|
13
|
+
* Deterministic: identical input always yields identical output.
|
|
14
|
+
*/
|
|
15
|
+
declare function splitSentences(text: string): Sentence[];
|
|
16
|
+
|
|
17
|
+
/** One input article about the story. */
|
|
18
|
+
interface CondenseSource {
|
|
19
|
+
text: string;
|
|
20
|
+
/** Short label used in the grouped output header, e.g. "Kathmandu Post". Defaults to "S{n}". */
|
|
21
|
+
label?: string;
|
|
22
|
+
url?: string;
|
|
23
|
+
publishedAt?: string | Date;
|
|
24
|
+
}
|
|
25
|
+
interface CondenseOptions {
|
|
26
|
+
/** Total token budget for the condensed digest. Default 1500. */
|
|
27
|
+
tokenBudget?: number;
|
|
28
|
+
/** Cap on kept sentences from any single source, so one long article can't crowd out the rest. Default 8. */
|
|
29
|
+
maxSentencesPerSource?: number;
|
|
30
|
+
/** Drop a sentence whose 3-gram Jaccard similarity to an already-kept sentence is ≥ this. Default 0.8. */
|
|
31
|
+
dedupeThreshold?: number;
|
|
32
|
+
/** Always keep each source's first sentence (the lede) regardless of score. Default true. */
|
|
33
|
+
keepLede?: boolean;
|
|
34
|
+
/** Named entities (people/places) to always keep and to boost — pass en and ne forms. */
|
|
35
|
+
gazetteer?: string[];
|
|
36
|
+
/** Model hint for token counting (passed to @lacspace/tokenizer). */
|
|
37
|
+
model?: string;
|
|
38
|
+
/** Override token counting entirely. Default: @lacspace/tokenizer countTokens. */
|
|
39
|
+
countTokens?: (text: string) => number;
|
|
40
|
+
/** Header format for each source group. Default `[S{n} {label}]`. */
|
|
41
|
+
header?: (source: {
|
|
42
|
+
idx: number;
|
|
43
|
+
label: string;
|
|
44
|
+
url?: string;
|
|
45
|
+
}) => string;
|
|
46
|
+
}
|
|
47
|
+
interface CondensedSentence {
|
|
48
|
+
text: string;
|
|
49
|
+
/** Index of the source this sentence came from (0-based). */
|
|
50
|
+
sourceIdx: number;
|
|
51
|
+
/** Character offsets in that source's original text. */
|
|
52
|
+
start: number;
|
|
53
|
+
end: number;
|
|
54
|
+
/** BM25 centrality score within the cluster. */
|
|
55
|
+
score: number;
|
|
56
|
+
/** Why it was kept: any of "lede", "number", "quote", "entity", "rank". */
|
|
57
|
+
reasons: string[];
|
|
58
|
+
}
|
|
59
|
+
interface CondensedSource {
|
|
60
|
+
idx: number;
|
|
61
|
+
label: string;
|
|
62
|
+
url?: string;
|
|
63
|
+
sentences: CondensedSentence[];
|
|
64
|
+
}
|
|
65
|
+
interface CondenseResult {
|
|
66
|
+
/** The digest: kept sentences grouped per source (source order) under a header line. */
|
|
67
|
+
text: string;
|
|
68
|
+
/** Kept sentences grouped per source, each in the source's original order. */
|
|
69
|
+
sources: CondensedSource[];
|
|
70
|
+
/** All kept sentences, flat, in output (source-grouped) order. */
|
|
71
|
+
sentences: CondensedSentence[];
|
|
72
|
+
/** Token count of `text` (same counter used for the budget). */
|
|
73
|
+
tokens: number;
|
|
74
|
+
/** How many near-duplicate sentences were dropped. */
|
|
75
|
+
droppedDup: number;
|
|
76
|
+
/** Total sentences seen across all sources. */
|
|
77
|
+
totalSentences: number;
|
|
78
|
+
}
|
|
79
|
+
/**
|
|
80
|
+
* Condense several sources on the same story into one short, deduplicated,
|
|
81
|
+
* token-budgeted digest that keeps the numbers, quotes and named entities — so
|
|
82
|
+
* an LLM only rewrites a fraction of the words. Purely extractive and
|
|
83
|
+
* deterministic; no network, no model.
|
|
84
|
+
*/
|
|
85
|
+
declare function condense(sources: CondenseSource[], options?: CondenseOptions): CondenseResult;
|
|
86
|
+
|
|
87
|
+
export { type CondenseOptions, type CondenseResult, type CondenseSource, type CondensedSentence, type CondensedSource, type Sentence, condense, splitSentences };
|
package/dist/index.d.ts
ADDED
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
/** A sentence with its character offsets in the original source text. */
|
|
2
|
+
interface Sentence {
|
|
3
|
+
text: string;
|
|
4
|
+
/** Inclusive start offset in the source. */
|
|
5
|
+
start: number;
|
|
6
|
+
/** Exclusive end offset in the source. */
|
|
7
|
+
end: number;
|
|
8
|
+
}
|
|
9
|
+
/**
|
|
10
|
+
* Split `text` into sentences with character offsets. Handles Latin and
|
|
11
|
+
* Devanagari terminators (। ॥), never splits inside a quotation, and does not
|
|
12
|
+
* break on decimals (3.5), abbreviations (Dr.) or a terminator glued to a digit.
|
|
13
|
+
* Deterministic: identical input always yields identical output.
|
|
14
|
+
*/
|
|
15
|
+
declare function splitSentences(text: string): Sentence[];
|
|
16
|
+
|
|
17
|
+
/** One input article about the story. */
|
|
18
|
+
interface CondenseSource {
|
|
19
|
+
text: string;
|
|
20
|
+
/** Short label used in the grouped output header, e.g. "Kathmandu Post". Defaults to "S{n}". */
|
|
21
|
+
label?: string;
|
|
22
|
+
url?: string;
|
|
23
|
+
publishedAt?: string | Date;
|
|
24
|
+
}
|
|
25
|
+
interface CondenseOptions {
|
|
26
|
+
/** Total token budget for the condensed digest. Default 1500. */
|
|
27
|
+
tokenBudget?: number;
|
|
28
|
+
/** Cap on kept sentences from any single source, so one long article can't crowd out the rest. Default 8. */
|
|
29
|
+
maxSentencesPerSource?: number;
|
|
30
|
+
/** Drop a sentence whose 3-gram Jaccard similarity to an already-kept sentence is ≥ this. Default 0.8. */
|
|
31
|
+
dedupeThreshold?: number;
|
|
32
|
+
/** Always keep each source's first sentence (the lede) regardless of score. Default true. */
|
|
33
|
+
keepLede?: boolean;
|
|
34
|
+
/** Named entities (people/places) to always keep and to boost — pass en and ne forms. */
|
|
35
|
+
gazetteer?: string[];
|
|
36
|
+
/** Model hint for token counting (passed to @lacspace/tokenizer). */
|
|
37
|
+
model?: string;
|
|
38
|
+
/** Override token counting entirely. Default: @lacspace/tokenizer countTokens. */
|
|
39
|
+
countTokens?: (text: string) => number;
|
|
40
|
+
/** Header format for each source group. Default `[S{n} {label}]`. */
|
|
41
|
+
header?: (source: {
|
|
42
|
+
idx: number;
|
|
43
|
+
label: string;
|
|
44
|
+
url?: string;
|
|
45
|
+
}) => string;
|
|
46
|
+
}
|
|
47
|
+
interface CondensedSentence {
|
|
48
|
+
text: string;
|
|
49
|
+
/** Index of the source this sentence came from (0-based). */
|
|
50
|
+
sourceIdx: number;
|
|
51
|
+
/** Character offsets in that source's original text. */
|
|
52
|
+
start: number;
|
|
53
|
+
end: number;
|
|
54
|
+
/** BM25 centrality score within the cluster. */
|
|
55
|
+
score: number;
|
|
56
|
+
/** Why it was kept: any of "lede", "number", "quote", "entity", "rank". */
|
|
57
|
+
reasons: string[];
|
|
58
|
+
}
|
|
59
|
+
interface CondensedSource {
|
|
60
|
+
idx: number;
|
|
61
|
+
label: string;
|
|
62
|
+
url?: string;
|
|
63
|
+
sentences: CondensedSentence[];
|
|
64
|
+
}
|
|
65
|
+
interface CondenseResult {
|
|
66
|
+
/** The digest: kept sentences grouped per source (source order) under a header line. */
|
|
67
|
+
text: string;
|
|
68
|
+
/** Kept sentences grouped per source, each in the source's original order. */
|
|
69
|
+
sources: CondensedSource[];
|
|
70
|
+
/** All kept sentences, flat, in output (source-grouped) order. */
|
|
71
|
+
sentences: CondensedSentence[];
|
|
72
|
+
/** Token count of `text` (same counter used for the budget). */
|
|
73
|
+
tokens: number;
|
|
74
|
+
/** How many near-duplicate sentences were dropped. */
|
|
75
|
+
droppedDup: number;
|
|
76
|
+
/** Total sentences seen across all sources. */
|
|
77
|
+
totalSentences: number;
|
|
78
|
+
}
|
|
79
|
+
/**
|
|
80
|
+
* Condense several sources on the same story into one short, deduplicated,
|
|
81
|
+
* token-budgeted digest that keeps the numbers, quotes and named entities — so
|
|
82
|
+
* an LLM only rewrites a fraction of the words. Purely extractive and
|
|
83
|
+
* deterministic; no network, no model.
|
|
84
|
+
*/
|
|
85
|
+
declare function condense(sources: CondenseSource[], options?: CondenseOptions): CondenseResult;
|
|
86
|
+
|
|
87
|
+
export { type CondenseOptions, type CondenseResult, type CondenseSource, type CondensedSentence, type CondensedSource, type Sentence, condense, splitSentences };
|
package/dist/index.js
ADDED
|
@@ -0,0 +1,252 @@
|
|
|
1
|
+
import { bm25 } from '@lacspace/rerank';
|
|
2
|
+
import { countTokens } from '@lacspace/tokenizer';
|
|
3
|
+
|
|
4
|
+
// src/index.ts
|
|
5
|
+
|
|
6
|
+
// src/features.ts
|
|
7
|
+
var CURRENCY = /(?:रु|नेरु|रुपैयाँ|₹|\$|€|£|Rs\.?|USD|NPR|INR)/i;
|
|
8
|
+
var NUMBERISH = /[0-9०-९][0-9०-९.,%]*/;
|
|
9
|
+
var PERCENT = /[%]|प्रतिशत/;
|
|
10
|
+
var QUOTED = /"[^"]{3,}"|“[^”]{3,}”|‘[^’]{3,}’/;
|
|
11
|
+
function hasNumber(text) {
|
|
12
|
+
return NUMBERISH.test(text) || CURRENCY.test(text) || PERCENT.test(text);
|
|
13
|
+
}
|
|
14
|
+
function hasQuote(text) {
|
|
15
|
+
return QUOTED.test(text);
|
|
16
|
+
}
|
|
17
|
+
function hasEntity(text, gaz) {
|
|
18
|
+
if (!gaz || gaz.size === 0) return false;
|
|
19
|
+
const lower = text.toLowerCase();
|
|
20
|
+
for (const term of gaz) {
|
|
21
|
+
if (!term) continue;
|
|
22
|
+
if (/[^\u0000-\u007F]/.test(term)) {
|
|
23
|
+
if (text.includes(term)) return true;
|
|
24
|
+
} else if (new RegExp(`(?:^|[^a-z])${escapeRe(term.toLowerCase())}(?:[^a-z]|$)`).test(lower)) {
|
|
25
|
+
return true;
|
|
26
|
+
}
|
|
27
|
+
}
|
|
28
|
+
return false;
|
|
29
|
+
}
|
|
30
|
+
function escapeRe(s) {
|
|
31
|
+
return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
32
|
+
}
|
|
33
|
+
function shingles(text, size = 3) {
|
|
34
|
+
const words = text.toLowerCase().replace(/[^\p{L}\p{N}\s]/gu, " ").split(/\s+/).filter(Boolean);
|
|
35
|
+
const out = /* @__PURE__ */ new Set();
|
|
36
|
+
if (words.length < size) {
|
|
37
|
+
if (words.length) out.add(words.join(" "));
|
|
38
|
+
return out;
|
|
39
|
+
}
|
|
40
|
+
for (let i = 0; i + size <= words.length; i++) out.add(words.slice(i, i + size).join(" "));
|
|
41
|
+
return out;
|
|
42
|
+
}
|
|
43
|
+
function jaccard(a, b) {
|
|
44
|
+
if (a.size === 0 && b.size === 0) return 0;
|
|
45
|
+
let inter = 0;
|
|
46
|
+
const [small, large] = a.size <= b.size ? [a, b] : [b, a];
|
|
47
|
+
for (const t of small) if (large.has(t)) inter++;
|
|
48
|
+
return inter / (a.size + b.size - inter);
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
// src/sentences.ts
|
|
52
|
+
var TERMINATORS = /* @__PURE__ */ new Set([".", "!", "?", "\u0964", "\u0965", "\u2026"]);
|
|
53
|
+
var OPEN_QUOTES = /* @__PURE__ */ new Set(["\u201C", "\u2018", "\xAB"]);
|
|
54
|
+
var CLOSE_QUOTES = /* @__PURE__ */ new Set(["\u201D", "\u2019", "\xBB"]);
|
|
55
|
+
var ABBREV = /* @__PURE__ */ new Set([
|
|
56
|
+
"mr",
|
|
57
|
+
"mrs",
|
|
58
|
+
"ms",
|
|
59
|
+
"dr",
|
|
60
|
+
"prof",
|
|
61
|
+
"sr",
|
|
62
|
+
"jr",
|
|
63
|
+
"st",
|
|
64
|
+
"vs",
|
|
65
|
+
"etc",
|
|
66
|
+
"inc",
|
|
67
|
+
"ltd",
|
|
68
|
+
"co",
|
|
69
|
+
"corp",
|
|
70
|
+
"govt",
|
|
71
|
+
"gen",
|
|
72
|
+
"rep",
|
|
73
|
+
"sen",
|
|
74
|
+
"gov",
|
|
75
|
+
"no",
|
|
76
|
+
"vol",
|
|
77
|
+
"fig",
|
|
78
|
+
"al",
|
|
79
|
+
"rs",
|
|
80
|
+
"u.s",
|
|
81
|
+
"u.k",
|
|
82
|
+
"e.g",
|
|
83
|
+
"i.e",
|
|
84
|
+
"a.m",
|
|
85
|
+
"p.m"
|
|
86
|
+
]);
|
|
87
|
+
function isDigit(ch) {
|
|
88
|
+
return ch >= "0" && ch <= "9" || ch >= "\u0966" && ch <= "\u096F";
|
|
89
|
+
}
|
|
90
|
+
function splitSentences(text) {
|
|
91
|
+
const out = [];
|
|
92
|
+
if (!text) return out;
|
|
93
|
+
let quoteDepth = 0;
|
|
94
|
+
let doubleOpen = false;
|
|
95
|
+
let start = 0;
|
|
96
|
+
const n = text.length;
|
|
97
|
+
const push = (from, to) => {
|
|
98
|
+
const raw = text.slice(from, to);
|
|
99
|
+
const trimmedStart = raw.length - raw.trimStart().length;
|
|
100
|
+
const trimmedEnd = raw.length - raw.trimEnd().length;
|
|
101
|
+
const s = from + trimmedStart;
|
|
102
|
+
const e = to - trimmedEnd;
|
|
103
|
+
if (e > s) out.push({ text: text.slice(s, e), start: s, end: e });
|
|
104
|
+
};
|
|
105
|
+
for (let i = 0; i < n; i++) {
|
|
106
|
+
const ch = text[i];
|
|
107
|
+
if (ch === '"') {
|
|
108
|
+
doubleOpen = !doubleOpen;
|
|
109
|
+
continue;
|
|
110
|
+
}
|
|
111
|
+
if (OPEN_QUOTES.has(ch)) {
|
|
112
|
+
quoteDepth++;
|
|
113
|
+
continue;
|
|
114
|
+
}
|
|
115
|
+
if (CLOSE_QUOTES.has(ch)) {
|
|
116
|
+
if (quoteDepth > 0) quoteDepth--;
|
|
117
|
+
continue;
|
|
118
|
+
}
|
|
119
|
+
if (!TERMINATORS.has(ch)) continue;
|
|
120
|
+
if (quoteDepth > 0 || doubleOpen) continue;
|
|
121
|
+
if (ch === ".") {
|
|
122
|
+
const prev = text[i - 1];
|
|
123
|
+
const next = text[i + 1];
|
|
124
|
+
if (prev && next && isDigit(prev) && isDigit(next)) continue;
|
|
125
|
+
let j = i - 1;
|
|
126
|
+
while (j >= 0 && /[A-Za-z.]/.test(text[j])) j--;
|
|
127
|
+
const word = text.slice(j + 1, i).toLowerCase();
|
|
128
|
+
if (ABBREV.has(word)) continue;
|
|
129
|
+
}
|
|
130
|
+
let k = i + 1;
|
|
131
|
+
while (k < n && (TERMINATORS.has(text[k]) || CLOSE_QUOTES.has(text[k]) || text[k] === '"' || text[k] === ")" || text[k] === "]")) {
|
|
132
|
+
if (text[k] === '"') doubleOpen = !doubleOpen;
|
|
133
|
+
k++;
|
|
134
|
+
}
|
|
135
|
+
if (k >= n || /\s/.test(text[k])) {
|
|
136
|
+
push(start, k);
|
|
137
|
+
start = k;
|
|
138
|
+
i = k - 1;
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
if (start < n) push(start, n);
|
|
142
|
+
return out;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
// src/index.ts
|
|
146
|
+
function condense(sources, options = {}) {
|
|
147
|
+
const tokenBudget = options.tokenBudget ?? 1500;
|
|
148
|
+
const maxPer = options.maxSentencesPerSource ?? 8;
|
|
149
|
+
const dedupe = options.dedupeThreshold ?? 0.8;
|
|
150
|
+
const keepLede = options.keepLede ?? true;
|
|
151
|
+
const count = options.countTokens ?? ((t) => countTokens(t, options.model));
|
|
152
|
+
const gaz = options.gazetteer && options.gazetteer.length ? new Set(options.gazetteer) : null;
|
|
153
|
+
const header = options.header ?? ((s) => `[S${s.idx + 1} ${s.label}]`);
|
|
154
|
+
const labels = sources.map((s, i) => s.label?.trim() || `S${i + 1}`);
|
|
155
|
+
const cands = [];
|
|
156
|
+
let order = 0;
|
|
157
|
+
sources.forEach((src, sIdx) => {
|
|
158
|
+
const sents = splitSentences(src.text ?? "");
|
|
159
|
+
sents.forEach((sen, localIdx) => {
|
|
160
|
+
cands.push({
|
|
161
|
+
text: sen.text,
|
|
162
|
+
sourceIdx: sIdx,
|
|
163
|
+
start: sen.start,
|
|
164
|
+
end: sen.end,
|
|
165
|
+
order: order++,
|
|
166
|
+
score: 0,
|
|
167
|
+
lede: localIdx === 0,
|
|
168
|
+
number: hasNumber(sen.text),
|
|
169
|
+
quote: hasQuote(sen.text),
|
|
170
|
+
entity: hasEntity(sen.text, gaz),
|
|
171
|
+
shingle: shingles(sen.text),
|
|
172
|
+
tokens: count(sen.text)
|
|
173
|
+
});
|
|
174
|
+
});
|
|
175
|
+
});
|
|
176
|
+
const totalSentences = cands.length;
|
|
177
|
+
if (totalSentences === 0) {
|
|
178
|
+
return { text: "", sources: [], sentences: [], tokens: 0, droppedDup: 0, totalSentences: 0 };
|
|
179
|
+
}
|
|
180
|
+
const clusterQuery = cands.map((c) => c.text).join(" ");
|
|
181
|
+
const scored = bm25(
|
|
182
|
+
clusterQuery,
|
|
183
|
+
cands.map((c) => ({ id: String(c.order), text: c.text }))
|
|
184
|
+
);
|
|
185
|
+
const scoreById = new Map(scored.map((s) => [s.id, s.rerankScore]));
|
|
186
|
+
for (const c of cands) c.score = scoreById.get(String(c.order)) ?? 0;
|
|
187
|
+
const priority = (c) => (keepLede && c.lede ? 4 : 0) + (c.quote ? 3 : 0) + (c.number ? 2 : 0) + (c.entity ? 1 : 0);
|
|
188
|
+
const ranked = [...cands].sort((a, b) => {
|
|
189
|
+
const pa = priority(a);
|
|
190
|
+
const pb = priority(b);
|
|
191
|
+
if (pa !== pb) return pb - pa;
|
|
192
|
+
if (b.score !== a.score) return b.score - a.score;
|
|
193
|
+
return a.order - b.order;
|
|
194
|
+
});
|
|
195
|
+
const kept = [];
|
|
196
|
+
const keptShingles = [];
|
|
197
|
+
const perSource = /* @__PURE__ */ new Map();
|
|
198
|
+
let used = 0;
|
|
199
|
+
let droppedDup = 0;
|
|
200
|
+
for (const c of ranked) {
|
|
201
|
+
let dup = false;
|
|
202
|
+
for (const ks of keptShingles) {
|
|
203
|
+
if (jaccard(c.shingle, ks) >= dedupe) {
|
|
204
|
+
dup = true;
|
|
205
|
+
break;
|
|
206
|
+
}
|
|
207
|
+
}
|
|
208
|
+
if (dup) {
|
|
209
|
+
droppedDup++;
|
|
210
|
+
continue;
|
|
211
|
+
}
|
|
212
|
+
if ((perSource.get(c.sourceIdx) ?? 0) >= maxPer) continue;
|
|
213
|
+
if (used + c.tokens > tokenBudget && kept.length > 0) continue;
|
|
214
|
+
kept.push(c);
|
|
215
|
+
keptShingles.push(c.shingle);
|
|
216
|
+
perSource.set(c.sourceIdx, (perSource.get(c.sourceIdx) ?? 0) + 1);
|
|
217
|
+
used += c.tokens;
|
|
218
|
+
}
|
|
219
|
+
const bySource = /* @__PURE__ */ new Map();
|
|
220
|
+
for (const c of kept) {
|
|
221
|
+
const arr = bySource.get(c.sourceIdx) ?? [];
|
|
222
|
+
arr.push(c);
|
|
223
|
+
bySource.set(c.sourceIdx, arr);
|
|
224
|
+
}
|
|
225
|
+
const toOut = (c) => {
|
|
226
|
+
const reasons = [];
|
|
227
|
+
if (keepLede && c.lede) reasons.push("lede");
|
|
228
|
+
if (c.quote) reasons.push("quote");
|
|
229
|
+
if (c.number) reasons.push("number");
|
|
230
|
+
if (c.entity) reasons.push("entity");
|
|
231
|
+
if (reasons.length === 0) reasons.push("rank");
|
|
232
|
+
return { text: c.text, sourceIdx: c.sourceIdx, start: c.start, end: c.end, score: c.score, reasons };
|
|
233
|
+
};
|
|
234
|
+
const outSources = [];
|
|
235
|
+
const flat = [];
|
|
236
|
+
const parts = [];
|
|
237
|
+
sources.forEach((src, sIdx) => {
|
|
238
|
+
const list = (bySource.get(sIdx) ?? []).sort((a, b) => a.start - b.start);
|
|
239
|
+
if (list.length === 0) return;
|
|
240
|
+
const outSents = list.map(toOut);
|
|
241
|
+
outSources.push({ idx: sIdx, label: labels[sIdx], url: src.url, sentences: outSents });
|
|
242
|
+
flat.push(...outSents);
|
|
243
|
+
parts.push(`${header({ idx: sIdx, label: labels[sIdx], url: src.url })}
|
|
244
|
+
${outSents.map((s) => s.text).join(" ")}`);
|
|
245
|
+
});
|
|
246
|
+
const text = parts.join("\n\n");
|
|
247
|
+
return { text, sources: outSources, sentences: flat, tokens: count(text), droppedDup, totalSentences };
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
export { condense, splitSentences };
|
|
251
|
+
//# sourceMappingURL=index.js.map
|
|
252
|
+
//# sourceMappingURL=index.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"sources":["../src/features.ts","../src/sentences.ts","../src/index.ts"],"names":[],"mappings":";;;;;;AAEA,IAAM,QAAA,GAAW,iDAAA;AAEjB,IAAM,SAAA,GAAY,sBAAA;AAClB,IAAM,OAAA,GAAU,aAAA;AAEhB,IAAM,MAAA,GAAS,kCAAA;AAER,SAAS,UAAU,IAAA,EAAuB;AAC/C,EAAA,OAAO,SAAA,CAAU,IAAA,CAAK,IAAI,CAAA,IAAK,QAAA,CAAS,KAAK,IAAI,CAAA,IAAK,OAAA,CAAQ,IAAA,CAAK,IAAI,CAAA;AACzE;AAEO,SAAS,SAAS,IAAA,EAAuB;AAC9C,EAAA,OAAO,MAAA,CAAO,KAAK,IAAI,CAAA;AACzB;AAGO,SAAS,SAAA,CAAU,MAAc,GAAA,EAAkC;AACxE,EAAA,IAAI,CAAC,GAAA,IAAO,GAAA,CAAI,IAAA,KAAS,GAAG,OAAO,KAAA;AACnC,EAAA,MAAM,KAAA,GAAQ,KAAK,WAAA,EAAY;AAC/B,EAAA,KAAA,MAAW,QAAQ,GAAA,EAAK;AACtB,IAAA,IAAI,CAAC,IAAA,EAAM;AAEX,IAAA,IAAI,kBAAA,CAAmB,IAAA,CAAK,IAAI,CAAA,EAAG;AACjC,MAAA,IAAI,IAAA,CAAK,QAAA,CAAS,IAAI,CAAA,EAAG,OAAO,IAAA;AAAA,IAClC,CAAA,MAAA,IAAW,IAAI,MAAA,CAAO,CAAA,YAAA,EAAe,QAAA,CAAS,IAAA,CAAK,WAAA,EAAa,CAAC,CAAA,YAAA,CAAc,CAAA,CAAE,IAAA,CAAK,KAAK,CAAA,EAAG;AAC5F,MAAA,OAAO,IAAA;AAAA,IACT;AAAA,EACF;AACA,EAAA,OAAO,KAAA;AACT;AAEA,SAAS,SAAS,CAAA,EAAmB;AACnC,EAAA,OAAO,CAAA,CAAE,OAAA,CAAQ,qBAAA,EAAuB,MAAM,CAAA;AAChD;AAGO,SAAS,QAAA,CAAS,IAAA,EAAc,IAAA,GAAO,CAAA,EAAgB;AAC5D,EAAA,MAAM,KAAA,GAAQ,IAAA,CACX,WAAA,EAAY,CACZ,OAAA,CAAQ,mBAAA,EAAqB,GAAG,CAAA,CAChC,KAAA,CAAM,KAAK,CAAA,CACX,MAAA,CAAO,OAAO,CAAA;AACjB,EAAA,MAAM,GAAA,uBAAU,GAAA,EAAY;AAC5B,EAAA,IAAI,KAAA,CAAM,SAAS,IAAA,EAAM;AACvB,IAAA,IAAI,MAAM,MAAA,EAAQ,GAAA,CAAI,IAAI,KAAA,CAAM,IAAA,CAAK,GAAG,CAAC,CAAA;AACzC,IAAA,OAAO,GAAA;AAAA,EACT;AACA,EAAA,KAAA,IAAS,IAAI,CAAA,EAAG,CAAA,GAAI,IAAA,IAAQ,KAAA,CAAM,QAAQ,CAAA,EAAA,EAAK,GAAA,CAAI,GAAA,CAAI,KAAA,CAAM,MAAM,CAAA,EAAG,CAAA,GAAI,IAAI,CAAA,CAAE,IAAA,CAAK,GAAG,CAAC,CAAA;AACzF,EAAA,OAAO,GAAA;AACT;AAEO,SAAS,OAAA,CAAQ,GAAgB,CAAA,EAAwB;AAC9D,EAAA,IAAI,EAAE,IAAA,KAAS,CAAA,IAAK,CAAA,CAAE,IAAA,KAAS,GAAG,OAAO,CAAA;AACzC,EAAA,IAAI,KAAA,GAAQ,CAAA;AACZ,EAAA,MAAM,CAAC,KAAA,EAAO,KAAK,CAAA,GAAI,EAAE,IAAA,IAAQ,CAAA,CAAE,IAAA,GAAO,CAAC,CAAA,EAAG,CAAC,CAAA,GAAI,CAAC,GAAG,CAAC,CAAA;AACxD,EAAA,KAAA,MAAW,KAAK,KAAA,EAAO,IAAI,KAAA,CAAM,GAAA,CAAI,CAAC,CAAA,EAAG,KAAA,EAAA;AACzC,EAAA,OAAO,KAAA,IAAS,CAAA,CAAE,IAAA,GAAO,CAAA,CAAE,IAAA,GAAO,KAAA,CAAA;AACpC;;;ACjDA,IAAM,WAAA,mBAAc,IAAI,GAAA,CAAI,CAAC,GAAA,EAAK,KAAK,GAAA,EAAK,QAAA,EAAK,QAAA,EAAK,QAAG,CAAC,CAAA;AAE1D,IAAM,8BAAc,IAAI,GAAA,CAAI,CAAC,QAAA,EAAK,QAAA,EAAK,MAAG,CAAC,CAAA;AAC3C,IAAM,+BAAe,IAAI,GAAA,CAAI,CAAC,QAAA,EAAK,QAAA,EAAK,MAAG,CAAC,CAAA;AAG5C,IAAM,MAAA,uBAAa,GAAA,CAAI;AAAA,EACrB,IAAA;AAAA,EAAM,KAAA;AAAA,EAAO,IAAA;AAAA,EAAM,IAAA;AAAA,EAAM,MAAA;AAAA,EAAQ,IAAA;AAAA,EAAM,IAAA;AAAA,EAAM,IAAA;AAAA,EAAM,IAAA;AAAA,EAAM,KAAA;AAAA,EAAO,KAAA;AAAA,EAAO,KAAA;AAAA,EACvE,IAAA;AAAA,EAAM,MAAA;AAAA,EAAQ,MAAA;AAAA,EAAQ,KAAA;AAAA,EAAO,KAAA;AAAA,EAAO,KAAA;AAAA,EAAO,KAAA;AAAA,EAAO,IAAA;AAAA,EAAM,KAAA;AAAA,EAAO,KAAA;AAAA,EAAO,IAAA;AAAA,EACtE,IAAA;AAAA,EAAM,KAAA;AAAA,EAAO,KAAA;AAAA,EAAO,KAAA;AAAA,EAAO,KAAA;AAAA,EAAO,KAAA;AAAA,EAAO;AAC3C,CAAC,CAAA;AAED,SAAS,QAAQ,EAAA,EAAqB;AACpC,EAAA,OAAQ,MAAM,GAAA,IAAO,EAAA,IAAM,GAAA,IAAS,EAAA,IAAM,YAAO,EAAA,IAAM,QAAA;AACzD;AAQO,SAAS,eAAe,IAAA,EAA0B;AACvD,EAAA,MAAM,MAAkB,EAAC;AACzB,EAAA,IAAI,CAAC,MAAM,OAAO,GAAA;AAClB,EAAA,IAAI,UAAA,GAAa,CAAA;AACjB,EAAA,IAAI,UAAA,GAAa,KAAA;AACjB,EAAA,IAAI,KAAA,GAAQ,CAAA;AACZ,EAAA,MAAM,IAAI,IAAA,CAAK,MAAA;AAEf,EAAA,MAAM,IAAA,GAAO,CAAC,IAAA,EAAc,EAAA,KAAe;AACzC,IAAA,MAAM,GAAA,GAAM,IAAA,CAAK,KAAA,CAAM,IAAA,EAAM,EAAE,CAAA;AAC/B,IAAA,MAAM,YAAA,GAAe,GAAA,CAAI,MAAA,GAAS,GAAA,CAAI,WAAU,CAAE,MAAA;AAClD,IAAA,MAAM,UAAA,GAAa,GAAA,CAAI,MAAA,GAAS,GAAA,CAAI,SAAQ,CAAE,MAAA;AAC9C,IAAA,MAAM,IAAI,IAAA,GAAO,YAAA;AACjB,IAAA,MAAM,IAAI,EAAA,GAAK,UAAA;AACf,IAAA,IAAI,CAAA,GAAI,CAAA,EAAG,GAAA,CAAI,IAAA,CAAK,EAAE,IAAA,EAAM,IAAA,CAAK,KAAA,CAAM,CAAA,EAAG,CAAC,CAAA,EAAG,KAAA,EAAO,CAAA,EAAG,GAAA,EAAK,GAAG,CAAA;AAAA,EAClE,CAAA;AAEA,EAAA,KAAA,IAAS,CAAA,GAAI,CAAA,EAAG,CAAA,GAAI,CAAA,EAAG,CAAA,EAAA,EAAK;AAC1B,IAAA,MAAM,EAAA,GAAK,KAAK,CAAC,CAAA;AACjB,IAAA,IAAI,OAAO,GAAA,EAAK;AACd,MAAA,UAAA,GAAa,CAAC,UAAA;AACd,MAAA;AAAA,IACF;AACA,IAAA,IAAI,WAAA,CAAY,GAAA,CAAI,EAAE,CAAA,EAAG;AACvB,MAAA,UAAA,EAAA;AACA,MAAA;AAAA,IACF;AACA,IAAA,IAAI,YAAA,CAAa,GAAA,CAAI,EAAE,CAAA,EAAG;AACxB,MAAA,IAAI,aAAa,CAAA,EAAG,UAAA,EAAA;AACpB,MAAA;AAAA,IACF;AACA,IAAA,IAAI,CAAC,WAAA,CAAY,GAAA,CAAI,EAAE,CAAA,EAAG;AAC1B,IAAA,IAAI,UAAA,GAAa,KAAK,UAAA,EAAY;AAGlC,IAAA,IAAI,OAAO,GAAA,EAAK;AACd,MAAA,MAAM,IAAA,GAAO,IAAA,CAAK,CAAA,GAAI,CAAC,CAAA;AACvB,MAAA,MAAM,IAAA,GAAO,IAAA,CAAK,CAAA,GAAI,CAAC,CAAA;AACvB,MAAA,IAAI,QAAQ,IAAA,IAAQ,OAAA,CAAQ,IAAI,CAAA,IAAK,OAAA,CAAQ,IAAI,CAAA,EAAG;AAEpD,MAAA,IAAI,IAAI,CAAA,GAAI,CAAA;AACZ,MAAA,OAAO,KAAK,CAAA,IAAK,WAAA,CAAY,KAAK,IAAA,CAAK,CAAC,CAAE,CAAA,EAAG,CAAA,EAAA;AAC7C,MAAA,MAAM,OAAO,IAAA,CAAK,KAAA,CAAM,IAAI,CAAA,EAAG,CAAC,EAAE,WAAA,EAAY;AAC9C,MAAA,IAAI,MAAA,CAAO,GAAA,CAAI,IAAI,CAAA,EAAG;AAAA,IACxB;AAGA,IAAA,IAAI,IAAI,CAAA,GAAI,CAAA;AACZ,IAAA,OAAO,CAAA,GAAI,CAAA,KAAM,WAAA,CAAY,GAAA,CAAI,IAAA,CAAK,CAAC,CAAE,CAAA,IAAK,YAAA,CAAa,GAAA,CAAI,IAAA,CAAK,CAAC,CAAE,CAAA,IAAK,IAAA,CAAK,CAAC,CAAA,KAAM,GAAA,IAAO,IAAA,CAAK,CAAC,CAAA,KAAM,GAAA,IAAO,IAAA,CAAK,CAAC,CAAA,KAAM,GAAA,CAAA,EAAM;AAClI,MAAA,IAAI,IAAA,CAAK,CAAC,CAAA,KAAM,GAAA,eAAkB,CAAC,UAAA;AACnC,MAAA,CAAA,EAAA;AAAA,IACF;AAEA,IAAA,IAAI,KAAK,CAAA,IAAK,IAAA,CAAK,KAAK,IAAA,CAAK,CAAC,CAAE,CAAA,EAAG;AACjC,MAAA,IAAA,CAAK,OAAO,CAAC,CAAA;AACb,MAAA,KAAA,GAAQ,CAAA;AACR,MAAA,CAAA,GAAI,CAAA,GAAI,CAAA;AAAA,IACV;AAAA,EACF;AACA,EAAA,IAAI,KAAA,GAAQ,CAAA,EAAG,IAAA,CAAK,KAAA,EAAO,CAAC,CAAA;AAC5B,EAAA,OAAO,GAAA;AACT;;;ACDO,SAAS,QAAA,CAAS,OAAA,EAA2B,OAAA,GAA2B,EAAC,EAAmB;AACjG,EAAA,MAAM,WAAA,GAAc,QAAQ,WAAA,IAAe,IAAA;AAC3C,EAAA,MAAM,MAAA,GAAS,QAAQ,qBAAA,IAAyB,CAAA;AAChD,EAAA,MAAM,MAAA,GAAS,QAAQ,eAAA,IAAmB,GAAA;AAC1C,EAAA,MAAM,QAAA,GAAW,QAAQ,QAAA,IAAY,IAAA;AACrC,EAAA,MAAM,KAAA,GAAQ,QAAQ,WAAA,KAAgB,CAAC,MAAc,WAAA,CAAY,CAAA,EAAG,QAAQ,KAAK,CAAA,CAAA;AACjF,EAAA,MAAM,GAAA,GAAM,OAAA,CAAQ,SAAA,IAAa,OAAA,CAAQ,SAAA,CAAU,SAAS,IAAI,GAAA,CAAI,OAAA,CAAQ,SAAS,CAAA,GAAI,IAAA;AACzF,EAAA,MAAM,MAAA,GACJ,OAAA,CAAQ,MAAA,KAAW,CAAC,CAAA,KAAsC,CAAA,EAAA,EAAK,CAAA,CAAE,GAAA,GAAM,CAAC,CAAA,CAAA,EAAI,CAAA,CAAE,KAAK,CAAA,CAAA,CAAA,CAAA;AAErF,EAAA,MAAM,MAAA,GAAS,OAAA,CAAQ,GAAA,CAAI,CAAC,CAAA,EAAG,CAAA,KAAM,CAAA,CAAE,KAAA,EAAO,IAAA,EAAK,IAAK,CAAA,CAAA,EAAI,CAAA,GAAI,CAAC,CAAA,CAAE,CAAA;AAGnE,EAAA,MAAM,QAAqB,EAAC;AAC5B,EAAA,IAAI,KAAA,GAAQ,CAAA;AACZ,EAAA,OAAA,CAAQ,OAAA,CAAQ,CAAC,GAAA,EAAK,IAAA,KAAS;AAC7B,IAAA,MAAM,KAAA,GAAQ,cAAA,CAAe,GAAA,CAAI,IAAA,IAAQ,EAAE,CAAA;AAC3C,IAAA,KAAA,CAAM,OAAA,CAAQ,CAAC,GAAA,EAAK,QAAA,KAAa;AAC/B,MAAA,KAAA,CAAM,IAAA,CAAK;AAAA,QACT,MAAM,GAAA,CAAI,IAAA;AAAA,QACV,SAAA,EAAW,IAAA;AAAA,QACX,OAAO,GAAA,CAAI,KAAA;AAAA,QACX,KAAK,GAAA,CAAI,GAAA;AAAA,QACT,KAAA,EAAO,KAAA,EAAA;AAAA,QACP,KAAA,EAAO,CAAA;AAAA,QACP,MAAM,QAAA,KAAa,CAAA;AAAA,QACnB,MAAA,EAAQ,SAAA,CAAU,GAAA,CAAI,IAAI,CAAA;AAAA,QAC1B,KAAA,EAAO,QAAA,CAAS,GAAA,CAAI,IAAI,CAAA;AAAA,QACxB,MAAA,EAAQ,SAAA,CAAU,GAAA,CAAI,IAAA,EAAM,GAAG,CAAA;AAAA,QAC/B,OAAA,EAAS,QAAA,CAAS,GAAA,CAAI,IAAI,CAAA;AAAA,QAC1B,MAAA,EAAQ,KAAA,CAAM,GAAA,CAAI,IAAI;AAAA,OACvB,CAAA;AAAA,IACH,CAAC,CAAA;AAAA,EACH,CAAC,CAAA;AACD,EAAA,MAAM,iBAAiB,KAAA,CAAM,MAAA;AAC7B,EAAA,IAAI,mBAAmB,CAAA,EAAG;AACxB,IAAA,OAAO,EAAE,IAAA,EAAM,EAAA,EAAI,OAAA,EAAS,EAAC,EAAG,SAAA,EAAW,EAAC,EAAG,MAAA,EAAQ,CAAA,EAAG,UAAA,EAAY,CAAA,EAAG,gBAAgB,CAAA,EAAE;AAAA,EAC7F;AAGA,EAAA,MAAM,YAAA,GAAe,MAAM,GAAA,CAAI,CAAC,MAAM,CAAA,CAAE,IAAI,CAAA,CAAE,IAAA,CAAK,GAAG,CAAA;AACtD,EAAA,MAAM,MAAA,GAAS,IAAA;AAAA,IACb,YAAA;AAAA,IACA,KAAA,CAAM,GAAA,CAAI,CAAC,CAAA,MAAO,EAAE,EAAA,EAAI,MAAA,CAAO,CAAA,CAAE,KAAK,CAAA,EAAG,IAAA,EAAM,CAAA,CAAE,MAAK,CAAE;AAAA,GAC1D;AACA,EAAA,MAAM,SAAA,GAAY,IAAI,GAAA,CAAI,MAAA,CAAO,GAAA,CAAI,CAAC,CAAA,KAAM,CAAC,CAAA,CAAE,EAAA,EAAI,CAAA,CAAE,WAAW,CAAC,CAAC,CAAA;AAClE,EAAA,KAAA,MAAW,CAAA,IAAK,KAAA,EAAO,CAAA,CAAE,KAAA,GAAQ,SAAA,CAAU,IAAI,MAAA,CAAO,CAAA,CAAE,KAAK,CAAC,CAAA,IAAK,CAAA;AAInE,EAAA,MAAM,WAAW,CAAC,CAAA,KAAA,CACf,YAAY,CAAA,CAAE,IAAA,GAAO,IAAI,CAAA,KAAM,CAAA,CAAE,KAAA,GAAQ,CAAA,GAAI,MAAM,CAAA,CAAE,MAAA,GAAS,IAAI,CAAA,CAAA,IAAM,CAAA,CAAE,SAAS,CAAA,GAAI,CAAA,CAAA;AAC1F,EAAA,MAAM,MAAA,GAAS,CAAC,GAAG,KAAK,EAAE,IAAA,CAAK,CAAC,GAAG,CAAA,KAAM;AACvC,IAAA,MAAM,EAAA,GAAK,SAAS,CAAC,CAAA;AACrB,IAAA,MAAM,EAAA,GAAK,SAAS,CAAC,CAAA;AACrB,IAAA,IAAI,EAAA,KAAO,EAAA,EAAI,OAAO,EAAA,GAAK,EAAA;AAC3B,IAAA,IAAI,EAAE,KAAA,KAAU,CAAA,CAAE,OAAO,OAAO,CAAA,CAAE,QAAQ,CAAA,CAAE,KAAA;AAC5C,IAAA,OAAO,CAAA,CAAE,QAAQ,CAAA,CAAE,KAAA;AAAA,EACrB,CAAC,CAAA;AAGD,EAAA,MAAM,OAAoB,EAAC;AAC3B,EAAA,MAAM,eAA8B,EAAC;AACrC,EAAA,MAAM,SAAA,uBAAgB,GAAA,EAAoB;AAC1C,EAAA,IAAI,IAAA,GAAO,CAAA;AACX,EAAA,IAAI,UAAA,GAAa,CAAA;AAEjB,EAAA,KAAA,MAAW,KAAK,MAAA,EAAQ;AACtB,IAAA,IAAI,GAAA,GAAM,KAAA;AACV,IAAA,KAAA,MAAW,MAAM,YAAA,EAAc;AAC7B,MAAA,IAAI,OAAA,CAAQ,CAAA,CAAE,OAAA,EAAS,EAAE,KAAK,MAAA,EAAQ;AACpC,QAAA,GAAA,GAAM,IAAA;AACN,QAAA;AAAA,MACF;AAAA,IACF;AACA,IAAA,IAAI,GAAA,EAAK;AACP,MAAA,UAAA,EAAA;AACA,MAAA;AAAA,IACF;AACA,IAAA,IAAA,CAAK,UAAU,GAAA,CAAI,CAAA,CAAE,SAAS,CAAA,IAAK,MAAM,MAAA,EAAQ;AACjD,IAAA,IAAI,OAAO,CAAA,CAAE,MAAA,GAAS,WAAA,IAAe,IAAA,CAAK,SAAS,CAAA,EAAG;AACtD,IAAA,IAAA,CAAK,KAAK,CAAC,CAAA;AACX,IAAA,YAAA,CAAa,IAAA,CAAK,EAAE,OAAO,CAAA;AAC3B,IAAA,SAAA,CAAU,GAAA,CAAI,EAAE,SAAA,EAAA,CAAY,SAAA,CAAU,IAAI,CAAA,CAAE,SAAS,CAAA,IAAK,CAAA,IAAK,CAAC,CAAA;AAChE,IAAA,IAAA,IAAQ,CAAA,CAAE,MAAA;AAAA,EACZ;AAGA,EAAA,MAAM,QAAA,uBAAe,GAAA,EAAyB;AAC9C,EAAA,KAAA,MAAW,KAAK,IAAA,EAAM;AACpB,IAAA,MAAM,MAAM,QAAA,CAAS,GAAA,CAAI,CAAA,CAAE,SAAS,KAAK,EAAC;AAC1C,IAAA,GAAA,CAAI,KAAK,CAAC,CAAA;AACV,IAAA,QAAA,CAAS,GAAA,CAAI,CAAA,CAAE,SAAA,EAAW,GAAG,CAAA;AAAA,EAC/B;AAEA,EAAA,MAAM,KAAA,GAAQ,CAAC,CAAA,KAAoC;AACjD,IAAA,MAAM,UAAoB,EAAC;AAC3B,IAAA,IAAI,QAAA,IAAY,CAAA,CAAE,IAAA,EAAM,OAAA,CAAQ,KAAK,MAAM,CAAA;AAC3C,IAAA,IAAI,CAAA,CAAE,KAAA,EAAO,OAAA,CAAQ,IAAA,CAAK,OAAO,CAAA;AACjC,IAAA,IAAI,CAAA,CAAE,MAAA,EAAQ,OAAA,CAAQ,IAAA,CAAK,QAAQ,CAAA;AACnC,IAAA,IAAI,CAAA,CAAE,MAAA,EAAQ,OAAA,CAAQ,IAAA,CAAK,QAAQ,CAAA;AACnC,IAAA,IAAI,OAAA,CAAQ,MAAA,KAAW,CAAA,EAAG,OAAA,CAAQ,KAAK,MAAM,CAAA;AAC7C,IAAA,OAAO,EAAE,IAAA,EAAM,CAAA,CAAE,IAAA,EAAM,SAAA,EAAW,EAAE,SAAA,EAAW,KAAA,EAAO,CAAA,CAAE,KAAA,EAAO,KAAK,CAAA,CAAE,GAAA,EAAK,KAAA,EAAO,CAAA,CAAE,OAAO,OAAA,EAAQ;AAAA,EACrG,CAAA;AAEA,EAAA,MAAM,aAAgC,EAAC;AACvC,EAAA,MAAM,OAA4B,EAAC;AACnC,EAAA,MAAM,QAAkB,EAAC;AACzB,EAAA,OAAA,CAAQ,OAAA,CAAQ,CAAC,GAAA,EAAK,IAAA,KAAS;AAC7B,IAAA,MAAM,IAAA,GAAA,CAAQ,QAAA,CAAS,GAAA,CAAI,IAAI,KAAK,EAAC,EAAG,IAAA,CAAK,CAAC,CAAA,EAAG,CAAA,KAAM,CAAA,CAAE,KAAA,GAAQ,EAAE,KAAK,CAAA;AACxE,IAAA,IAAI,IAAA,CAAK,WAAW,CAAA,EAAG;AACvB,IAAA,MAAM,QAAA,GAAW,IAAA,CAAK,GAAA,CAAI,KAAK,CAAA;AAC/B,IAAA,UAAA,CAAW,IAAA,CAAK,EAAE,GAAA,EAAK,IAAA,EAAM,KAAA,EAAO,MAAA,CAAO,IAAI,CAAA,EAAI,GAAA,EAAK,GAAA,CAAI,GAAA,EAAK,SAAA,EAAW,UAAU,CAAA;AACtF,IAAA,IAAA,CAAK,IAAA,CAAK,GAAG,QAAQ,CAAA;AACrB,IAAA,KAAA,CAAM,IAAA,CAAK,CAAA,EAAG,MAAA,CAAO,EAAE,KAAK,IAAA,EAAM,KAAA,EAAO,MAAA,CAAO,IAAI,CAAA,EAAI,GAAA,EAAK,GAAA,CAAI,GAAA,EAAK,CAAC;AAAA,EAAK,QAAA,CAAS,GAAA,CAAI,CAAC,CAAA,KAAM,CAAA,CAAE,IAAI,CAAA,CAAE,IAAA,CAAK,GAAG,CAAC,CAAA,CAAE,CAAA;AAAA,EACrH,CAAC,CAAA;AAED,EAAA,MAAM,IAAA,GAAO,KAAA,CAAM,IAAA,CAAK,MAAM,CAAA;AAC9B,EAAA,OAAO,EAAE,IAAA,EAAM,OAAA,EAAS,UAAA,EAAY,SAAA,EAAW,IAAA,EAAM,MAAA,EAAQ,KAAA,CAAM,IAAI,CAAA,EAAG,UAAA,EAAY,cAAA,EAAe;AACvG","file":"index.js","sourcesContent":["// Zero-dep detectors for the sentence features we always want to keep.\n\nconst CURRENCY = /(?:रु|नेरु|रुपैयाँ|₹|\\$|€|£|Rs\\.?|USD|NPR|INR)/i;\n// A number: Latin or Devanagari digits, optionally with separators/percent.\nconst NUMBERISH = /[0-9०-९][0-9०-९.,%]*/;\nconst PERCENT = /[%]|प्रतिशत/;\n// A quotation: matched straight or curly quotes with content between.\nconst QUOTED = /\"[^\"]{3,}\"|“[^”]{3,}”|‘[^’]{3,}’/;\n\nexport function hasNumber(text: string): boolean {\n return NUMBERISH.test(text) || CURRENCY.test(text) || PERCENT.test(text);\n}\n\nexport function hasQuote(text: string): boolean {\n return QUOTED.test(text);\n}\n\n/** True if any gazetteer term appears in the text (case-insensitive, whole-token where Latin). */\nexport function hasEntity(text: string, gaz: Set<string> | null): boolean {\n if (!gaz || gaz.size === 0) return false;\n const lower = text.toLowerCase();\n for (const term of gaz) {\n if (!term) continue;\n // Devanagari has no case; Latin terms are matched with word boundaries.\n if (/[^\\u0000-\\u007F]/.test(term)) {\n if (text.includes(term)) return true;\n } else if (new RegExp(`(?:^|[^a-z])${escapeRe(term.toLowerCase())}(?:[^a-z]|$)`).test(lower)) {\n return true;\n }\n }\n return false;\n}\n\nfunction escapeRe(s: string): string {\n return s.replace(/[.*+?^${}()|[\\]\\\\]/g, \"\\\\$&\");\n}\n\n/** Word 3-gram shingles for near-duplicate detection. */\nexport function shingles(text: string, size = 3): Set<string> {\n const words = text\n .toLowerCase()\n .replace(/[^\\p{L}\\p{N}\\s]/gu, \" \")\n .split(/\\s+/)\n .filter(Boolean);\n const out = new Set<string>();\n if (words.length < size) {\n if (words.length) out.add(words.join(\" \"));\n return out;\n }\n for (let i = 0; i + size <= words.length; i++) out.add(words.slice(i, i + size).join(\" \"));\n return out;\n}\n\nexport function jaccard(a: Set<string>, b: Set<string>): number {\n if (a.size === 0 && b.size === 0) return 0;\n let inter = 0;\n const [small, large] = a.size <= b.size ? [a, b] : [b, a];\n for (const t of small) if (large.has(t)) inter++;\n return inter / (a.size + b.size - inter);\n}\n","/** A sentence with its character offsets in the original source text. */\nexport interface Sentence {\n text: string;\n /** Inclusive start offset in the source. */\n start: number;\n /** Exclusive end offset in the source. */\n end: number;\n}\n\n// Terminators: Latin . ! ? plus Devanagari danda । and double danda ॥ and the ellipsis.\nconst TERMINATORS = new Set([\".\", \"!\", \"?\", \"।\", \"॥\", \"…\"]);\n// Quote characters whose parity we track so we never split inside a quote.\nconst OPEN_QUOTES = new Set([\"“\", \"‘\", \"«\"]); // \" ' «\nconst CLOSE_QUOTES = new Set([\"”\", \"’\", \"»\"]); // \" ' »\n\n// Common abbreviations after which a period does NOT end a sentence.\nconst ABBREV = new Set([\n \"mr\", \"mrs\", \"ms\", \"dr\", \"prof\", \"sr\", \"jr\", \"st\", \"vs\", \"etc\", \"inc\", \"ltd\",\n \"co\", \"corp\", \"govt\", \"gen\", \"rep\", \"sen\", \"gov\", \"no\", \"vol\", \"fig\", \"al\",\n \"rs\", \"u.s\", \"u.k\", \"e.g\", \"i.e\", \"a.m\", \"p.m\",\n]);\n\nfunction isDigit(ch: string): boolean {\n return (ch >= \"0\" && ch <= \"9\") || (ch >= \"०\" && ch <= \"९\");\n}\n\n/**\n * Split `text` into sentences with character offsets. Handles Latin and\n * Devanagari terminators (। ॥), never splits inside a quotation, and does not\n * break on decimals (3.5), abbreviations (Dr.) or a terminator glued to a digit.\n * Deterministic: identical input always yields identical output.\n */\nexport function splitSentences(text: string): Sentence[] {\n const out: Sentence[] = [];\n if (!text) return out;\n let quoteDepth = 0;\n let doubleOpen = false;\n let start = 0;\n const n = text.length;\n\n const push = (from: number, to: number) => {\n const raw = text.slice(from, to);\n const trimmedStart = raw.length - raw.trimStart().length;\n const trimmedEnd = raw.length - raw.trimEnd().length;\n const s = from + trimmedStart;\n const e = to - trimmedEnd;\n if (e > s) out.push({ text: text.slice(s, e), start: s, end: e });\n };\n\n for (let i = 0; i < n; i++) {\n const ch = text[i]!;\n if (ch === '\"') {\n doubleOpen = !doubleOpen;\n continue;\n }\n if (OPEN_QUOTES.has(ch)) {\n quoteDepth++;\n continue;\n }\n if (CLOSE_QUOTES.has(ch)) {\n if (quoteDepth > 0) quoteDepth--;\n continue;\n }\n if (!TERMINATORS.has(ch)) continue;\n if (quoteDepth > 0 || doubleOpen) continue; // inside a quote — keep it whole\n\n // A period between digits (3.5) or in an abbreviation is not a break.\n if (ch === \".\") {\n const prev = text[i - 1];\n const next = text[i + 1];\n if (prev && next && isDigit(prev) && isDigit(next)) continue;\n // trailing abbreviation like \"Dr.\" — look back to the word\n let j = i - 1;\n while (j >= 0 && /[A-Za-z.]/.test(text[j]!)) j--;\n const word = text.slice(j + 1, i).toLowerCase();\n if (ABBREV.has(word)) continue;\n }\n\n // Consume any run of terminators/closing quotes/brackets.\n let k = i + 1;\n while (k < n && (TERMINATORS.has(text[k]!) || CLOSE_QUOTES.has(text[k]!) || text[k] === '\"' || text[k] === \")\" || text[k] === \"]\")) {\n if (text[k] === '\"') doubleOpen = !doubleOpen;\n k++;\n }\n // Must be followed by whitespace or end of text to count as a break.\n if (k >= n || /\\s/.test(text[k]!)) {\n push(start, k);\n start = k;\n i = k - 1;\n }\n }\n if (start < n) push(start, n);\n return out;\n}\n","import { bm25 } from \"@lacspace/rerank\";\nimport { countTokens } from \"@lacspace/tokenizer\";\nimport { hasEntity, hasNumber, hasQuote, jaccard, shingles } from \"./features.js\";\nimport { splitSentences } from \"./sentences.js\";\n\nexport { splitSentences } from \"./sentences.js\";\nexport type { Sentence } from \"./sentences.js\";\n\n/** One input article about the story. */\nexport interface CondenseSource {\n text: string;\n /** Short label used in the grouped output header, e.g. \"Kathmandu Post\". Defaults to \"S{n}\". */\n label?: string;\n url?: string;\n publishedAt?: string | Date;\n}\n\nexport interface CondenseOptions {\n /** Total token budget for the condensed digest. Default 1500. */\n tokenBudget?: number;\n /** Cap on kept sentences from any single source, so one long article can't crowd out the rest. Default 8. */\n maxSentencesPerSource?: number;\n /** Drop a sentence whose 3-gram Jaccard similarity to an already-kept sentence is ≥ this. Default 0.8. */\n dedupeThreshold?: number;\n /** Always keep each source's first sentence (the lede) regardless of score. Default true. */\n keepLede?: boolean;\n /** Named entities (people/places) to always keep and to boost — pass en and ne forms. */\n gazetteer?: string[];\n /** Model hint for token counting (passed to @lacspace/tokenizer). */\n model?: string;\n /** Override token counting entirely. Default: @lacspace/tokenizer countTokens. */\n countTokens?: (text: string) => number;\n /** Header format for each source group. Default `[S{n} {label}]`. */\n header?: (source: { idx: number; label: string; url?: string }) => string;\n}\n\nexport interface CondensedSentence {\n text: string;\n /** Index of the source this sentence came from (0-based). */\n sourceIdx: number;\n /** Character offsets in that source's original text. */\n start: number;\n end: number;\n /** BM25 centrality score within the cluster. */\n score: number;\n /** Why it was kept: any of \"lede\", \"number\", \"quote\", \"entity\", \"rank\". */\n reasons: string[];\n}\n\nexport interface CondensedSource {\n idx: number;\n label: string;\n url?: string;\n sentences: CondensedSentence[];\n}\n\nexport interface CondenseResult {\n /** The digest: kept sentences grouped per source (source order) under a header line. */\n text: string;\n /** Kept sentences grouped per source, each in the source's original order. */\n sources: CondensedSource[];\n /** All kept sentences, flat, in output (source-grouped) order. */\n sentences: CondensedSentence[];\n /** Token count of `text` (same counter used for the budget). */\n tokens: number;\n /** How many near-duplicate sentences were dropped. */\n droppedDup: number;\n /** Total sentences seen across all sources. */\n totalSentences: number;\n}\n\ninterface Candidate {\n text: string;\n sourceIdx: number;\n start: number;\n end: number;\n order: number; // global index for stable ties\n score: number;\n lede: boolean;\n number: boolean;\n quote: boolean;\n entity: boolean;\n shingle: Set<string>;\n tokens: number;\n}\n\n/**\n * Condense several sources on the same story into one short, deduplicated,\n * token-budgeted digest that keeps the numbers, quotes and named entities — so\n * an LLM only rewrites a fraction of the words. Purely extractive and\n * deterministic; no network, no model.\n */\nexport function condense(sources: CondenseSource[], options: CondenseOptions = {}): CondenseResult {\n const tokenBudget = options.tokenBudget ?? 1500;\n const maxPer = options.maxSentencesPerSource ?? 8;\n const dedupe = options.dedupeThreshold ?? 0.8;\n const keepLede = options.keepLede ?? true;\n const count = options.countTokens ?? ((t: string) => countTokens(t, options.model));\n const gaz = options.gazetteer && options.gazetteer.length ? new Set(options.gazetteer) : null;\n const header =\n options.header ?? ((s: { idx: number; label: string }) => `[S${s.idx + 1} ${s.label}]`);\n\n const labels = sources.map((s, i) => s.label?.trim() || `S${i + 1}`);\n\n // 1. Split every source into sentences with offsets.\n const cands: Candidate[] = [];\n let order = 0;\n sources.forEach((src, sIdx) => {\n const sents = splitSentences(src.text ?? \"\");\n sents.forEach((sen, localIdx) => {\n cands.push({\n text: sen.text,\n sourceIdx: sIdx,\n start: sen.start,\n end: sen.end,\n order: order++,\n score: 0,\n lede: localIdx === 0,\n number: hasNumber(sen.text),\n quote: hasQuote(sen.text),\n entity: hasEntity(sen.text, gaz),\n shingle: shingles(sen.text),\n tokens: count(sen.text),\n });\n });\n });\n const totalSentences = cands.length;\n if (totalSentences === 0) {\n return { text: \"\", sources: [], sentences: [], tokens: 0, droppedDup: 0, totalSentences: 0 };\n }\n\n // 2. Score sentence centrality with BM25 against the whole cluster.\n const clusterQuery = cands.map((c) => c.text).join(\" \");\n const scored = bm25(\n clusterQuery,\n cands.map((c) => ({ id: String(c.order), text: c.text })),\n );\n const scoreById = new Map(scored.map((s) => [s.id, s.rerankScore]));\n for (const c of cands) c.score = scoreById.get(String(c.order)) ?? 0;\n\n // 3. Priority: always-keep (lede/number/quote/entity) first, then by score.\n // Deterministic tie-break by global order.\n const priority = (c: Candidate) =>\n (keepLede && c.lede ? 4 : 0) + (c.quote ? 3 : 0) + (c.number ? 2 : 0) + (c.entity ? 1 : 0);\n const ranked = [...cands].sort((a, b) => {\n const pa = priority(a);\n const pb = priority(b);\n if (pa !== pb) return pb - pa;\n if (b.score !== a.score) return b.score - a.score;\n return a.order - b.order;\n });\n\n // 4. Greedy selection: dedupe against kept, respect per-source cap and budget.\n const kept: Candidate[] = [];\n const keptShingles: Set<string>[] = [];\n const perSource = new Map<number, number>();\n let used = 0;\n let droppedDup = 0;\n\n for (const c of ranked) {\n let dup = false;\n for (const ks of keptShingles) {\n if (jaccard(c.shingle, ks) >= dedupe) {\n dup = true;\n break;\n }\n }\n if (dup) {\n droppedDup++;\n continue;\n }\n if ((perSource.get(c.sourceIdx) ?? 0) >= maxPer) continue;\n if (used + c.tokens > tokenBudget && kept.length > 0) continue; // keep at least one\n kept.push(c);\n keptShingles.push(c.shingle);\n perSource.set(c.sourceIdx, (perSource.get(c.sourceIdx) ?? 0) + 1);\n used += c.tokens;\n }\n\n // 5. Group kept sentences per source, each in original reading order.\n const bySource = new Map<number, Candidate[]>();\n for (const c of kept) {\n const arr = bySource.get(c.sourceIdx) ?? [];\n arr.push(c);\n bySource.set(c.sourceIdx, arr);\n }\n\n const toOut = (c: Candidate): CondensedSentence => {\n const reasons: string[] = [];\n if (keepLede && c.lede) reasons.push(\"lede\");\n if (c.quote) reasons.push(\"quote\");\n if (c.number) reasons.push(\"number\");\n if (c.entity) reasons.push(\"entity\");\n if (reasons.length === 0) reasons.push(\"rank\");\n return { text: c.text, sourceIdx: c.sourceIdx, start: c.start, end: c.end, score: c.score, reasons };\n };\n\n const outSources: CondensedSource[] = [];\n const flat: CondensedSentence[] = [];\n const parts: string[] = [];\n sources.forEach((src, sIdx) => {\n const list = (bySource.get(sIdx) ?? []).sort((a, b) => a.start - b.start);\n if (list.length === 0) return;\n const outSents = list.map(toOut);\n outSources.push({ idx: sIdx, label: labels[sIdx]!, url: src.url, sentences: outSents });\n flat.push(...outSents);\n parts.push(`${header({ idx: sIdx, label: labels[sIdx]!, url: src.url })}\\n${outSents.map((s) => s.text).join(\" \")}`);\n });\n\n const text = parts.join(\"\\n\\n\");\n return { text, sources: outSources, sentences: flat, tokens: count(text), droppedDup, totalSentences };\n}\n"]}
|
package/package.json
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "@lacspace/condense",
|
|
3
|
+
"version": "1.0.0",
|
|
4
|
+
"description": "Extractive multi-source condenser — turn several articles on the same story into one short, deduplicated, token-budgeted digest that keeps the numbers, quotes and named entities, so an LLM only has to rewrite a fraction of the text. BM25 sentence ranking, near-duplicate removal, Devanagari-aware. Zero external deps, isomorphic, deterministic.",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"main": "./dist/index.cjs",
|
|
7
|
+
"module": "./dist/index.js",
|
|
8
|
+
"types": "./dist/index.d.ts",
|
|
9
|
+
"exports": {
|
|
10
|
+
".": {
|
|
11
|
+
"import": { "types": "./dist/index.d.ts", "default": "./dist/index.js" },
|
|
12
|
+
"require": { "types": "./dist/index.d.cts", "default": "./dist/index.cjs" }
|
|
13
|
+
}
|
|
14
|
+
},
|
|
15
|
+
"files": ["dist"],
|
|
16
|
+
"sideEffects": false,
|
|
17
|
+
"scripts": { "build": "tsup", "prepublishOnly": "npm run build" },
|
|
18
|
+
"keywords": ["condense", "summarize", "extractive-summary", "multi-document", "bm25", "token-budget", "dedupe", "llm", "rag", "prompt-compression", "devanagari", "nepali", "zero-dependency", "isomorphic", "typescript"],
|
|
19
|
+
"dependencies": { "@lacspace/rerank": "^1.0.0", "@lacspace/tokenizer": "^1.1.0" },
|
|
20
|
+
"author": "Lacspace <contact@lacspace.com>",
|
|
21
|
+
"license": "SEE LICENSE IN LICENSE",
|
|
22
|
+
"homepage": "https://developer.lacspace.com/packages/condense",
|
|
23
|
+
"repository": { "type": "git", "url": "git+https://github.com/lacspace/npm-packages.git", "directory": "condense" },
|
|
24
|
+
"bugs": { "url": "https://github.com/lacspace/npm-packages/issues" },
|
|
25
|
+
"engines": { "node": ">=18" },
|
|
26
|
+
"publishConfig": { "access": "public" }
|
|
27
|
+
}
|