wwbnlp 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +361 -0
- package/NOTICE.md +28 -0
- package/README.md +132 -0
- package/data/provenance.json +67 -0
- package/dist/data/models.json +70222 -0
- package/dist/src/core.d.ts +57 -0
- package/dist/src/core.js +223 -0
- package/dist/src/core.js.map +1 -0
- package/dist/src/index.d.ts +7 -0
- package/dist/src/index.js +17 -0
- package/dist/src/index.js.map +1 -0
- package/docs/api.md +92 -0
- package/docs/migration.md +124 -0
- package/docs/research.md +86 -0
- package/licenses/affect_intensity.txt +76 -0
- package/licenses/age_gender.txt +76 -0
- package/licenses/lex-helpers-MIT.txt +21 -0
- package/licenses/perma.txt +8 -0
- package/licenses/spanish_perma.txt +76 -0
- package/licenses/temporal_orientation.txt +76 -0
- package/licenses/weighted-lexica-MIT.txt +21 -0
- package/package.json +60 -0
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
/** Shared weighted-lexicon engine. No I/O, global state, or runtime dependencies. */
|
|
2
|
+
export type Encoding = "frequency" | "binary" | "percent";
|
|
3
|
+
export type Weights = Readonly<Record<string, Readonly<Record<string, number>>>>;
|
|
4
|
+
export interface LexiconDefinition {
|
|
5
|
+
readonly id: string;
|
|
6
|
+
readonly language?: string;
|
|
7
|
+
readonly categories: Weights;
|
|
8
|
+
readonly intercepts?: Readonly<Record<string, number>>;
|
|
9
|
+
readonly features?: Weights;
|
|
10
|
+
readonly ngrams?: readonly number[];
|
|
11
|
+
readonly encoding?: Encoding;
|
|
12
|
+
}
|
|
13
|
+
export interface Lexicon extends LexiconDefinition {
|
|
14
|
+
readonly intercepts: Readonly<Record<string, number>>;
|
|
15
|
+
readonly features: Weights;
|
|
16
|
+
readonly ngrams: readonly number[];
|
|
17
|
+
readonly encoding: Encoding;
|
|
18
|
+
}
|
|
19
|
+
export interface Options {
|
|
20
|
+
/** Frequency divides term occurrences by original token count, including unmatched tokens. */
|
|
21
|
+
readonly encoding?: Encoding;
|
|
22
|
+
/** Sizes include unigrams explicitly. An empty array disables lexical matching. */
|
|
23
|
+
readonly ngrams?: readonly number[];
|
|
24
|
+
readonly includeIntercept?: boolean;
|
|
25
|
+
readonly minWeight?: number;
|
|
26
|
+
readonly maxWeight?: number;
|
|
27
|
+
/** Round final scores only. Omit to retain full floating-point precision. */
|
|
28
|
+
readonly decimals?: number;
|
|
29
|
+
/** Measured nonlexical covariates; never infer these from a single text. */
|
|
30
|
+
readonly features?: Readonly<Record<string, number>>;
|
|
31
|
+
}
|
|
32
|
+
export interface Match {
|
|
33
|
+
readonly term: string;
|
|
34
|
+
readonly count: number;
|
|
35
|
+
readonly weight: number;
|
|
36
|
+
readonly contribution: number;
|
|
37
|
+
}
|
|
38
|
+
export interface Analysis {
|
|
39
|
+
readonly model: string;
|
|
40
|
+
readonly status: "ok" | "empty" | "no-matches";
|
|
41
|
+
readonly values: Readonly<Record<string, number | null>>;
|
|
42
|
+
readonly matches: Readonly<Record<string, readonly Match[]>>;
|
|
43
|
+
readonly featureContributions: Readonly<Record<string, Readonly<Record<string, number>>>>;
|
|
44
|
+
readonly info: {
|
|
45
|
+
readonly tokenCount: number;
|
|
46
|
+
readonly featureCount: number;
|
|
47
|
+
readonly matchedFeatureCount: number;
|
|
48
|
+
readonly uniqueMatchedTerms: number;
|
|
49
|
+
};
|
|
50
|
+
readonly warnings: readonly string[];
|
|
51
|
+
}
|
|
52
|
+
/** Copy and validate a custom lexicon once; returned data cannot be mutated. */
|
|
53
|
+
export declare function createLexicon(definition: LexiconDefinition): Lexicon;
|
|
54
|
+
/** Unicode-aware convenience tokenizer; supply exact study tokens when reproducing a pipeline. */
|
|
55
|
+
export declare function tokenize(text: string): string[];
|
|
56
|
+
/** Score a validated lexicon. Input token arrays are used exactly as supplied. */
|
|
57
|
+
export declare function score(input: string | readonly string[], lexicon: Lexicon, options?: Options): Analysis;
|
package/dist/src/core.js
ADDED
|
@@ -0,0 +1,223 @@
|
|
|
1
|
+
const own = (object, key) => Object.prototype.hasOwnProperty.call(object, key);
|
|
2
|
+
function finite(value, label) {
|
|
3
|
+
if (typeof value !== "number" || !Number.isFinite(value)) {
|
|
4
|
+
throw new TypeError(`${label} must be finite`);
|
|
5
|
+
}
|
|
6
|
+
}
|
|
7
|
+
function checkNgrams(value) {
|
|
8
|
+
if (!Array.isArray(value) ||
|
|
9
|
+
value.some((n) => !Number.isSafeInteger(n) || n < 1 || n > 8)) {
|
|
10
|
+
throw new RangeError("ngrams must contain integer sizes from 1 to 8");
|
|
11
|
+
}
|
|
12
|
+
}
|
|
13
|
+
function checkEncoding(value) {
|
|
14
|
+
if (!["frequency", "binary", "percent"].includes(value)) {
|
|
15
|
+
throw new RangeError("Unknown encoding");
|
|
16
|
+
}
|
|
17
|
+
}
|
|
18
|
+
/** Copy and validate a custom lexicon once; returned data cannot be mutated. */
|
|
19
|
+
export function createLexicon(definition) {
|
|
20
|
+
if (!definition || typeof definition.id !== "string" || !definition.id ||
|
|
21
|
+
!definition.categories || typeof definition.categories !== "object" ||
|
|
22
|
+
Array.isArray(definition.categories)) {
|
|
23
|
+
throw new TypeError("A lexicon needs an id and category weight maps");
|
|
24
|
+
}
|
|
25
|
+
const categories = Object
|
|
26
|
+
.create(null);
|
|
27
|
+
const intercepts = Object.create(null);
|
|
28
|
+
const features = Object
|
|
29
|
+
.create(null);
|
|
30
|
+
for (const [category, terms] of Object.entries(definition.categories)) {
|
|
31
|
+
if (!category || !terms || typeof terms !== "object" || Array.isArray(terms))
|
|
32
|
+
throw new TypeError("Invalid category");
|
|
33
|
+
for (const [term, weight] of Object.entries(terms)) {
|
|
34
|
+
if (!term)
|
|
35
|
+
throw new TypeError("Lexicon terms must not be empty");
|
|
36
|
+
finite(weight, `Weight for ${term}`);
|
|
37
|
+
}
|
|
38
|
+
categories[category] = Object.freeze({ ...terms });
|
|
39
|
+
const intercept = definition.intercepts && own(definition.intercepts, category)
|
|
40
|
+
? definition.intercepts[category]
|
|
41
|
+
: 0;
|
|
42
|
+
finite(intercept, `Intercept for ${category}`);
|
|
43
|
+
intercepts[category] = intercept;
|
|
44
|
+
const covariates = definition.features && own(definition.features, category)
|
|
45
|
+
? definition.features[category]
|
|
46
|
+
: {};
|
|
47
|
+
if (!covariates || typeof covariates !== "object" || Array.isArray(covariates))
|
|
48
|
+
throw new TypeError("Feature weights must be a map");
|
|
49
|
+
for (const [feature, weight] of Object.entries(covariates)) {
|
|
50
|
+
finite(weight, `Feature weight for ${feature}`);
|
|
51
|
+
}
|
|
52
|
+
features[category] = Object.freeze({ ...covariates });
|
|
53
|
+
}
|
|
54
|
+
if (!Object.keys(categories).length) {
|
|
55
|
+
throw new TypeError("A lexicon needs at least one category");
|
|
56
|
+
}
|
|
57
|
+
for (const key of Object.keys(definition.intercepts ?? {})) {
|
|
58
|
+
if (!own(categories, key)) {
|
|
59
|
+
throw new RangeError(`Unknown intercept category: ${key}`);
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
for (const key of Object.keys(definition.features ?? {})) {
|
|
63
|
+
if (!own(categories, key)) {
|
|
64
|
+
throw new RangeError(`Unknown feature category: ${key}`);
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
const ngrams = definition.ngrams ?? [1, 2, 3];
|
|
68
|
+
const encoding = definition.encoding ?? "frequency";
|
|
69
|
+
checkNgrams(ngrams);
|
|
70
|
+
checkEncoding(encoding);
|
|
71
|
+
return Object.freeze({
|
|
72
|
+
id: definition.id,
|
|
73
|
+
...(definition.language === undefined
|
|
74
|
+
? {}
|
|
75
|
+
: { language: definition.language }),
|
|
76
|
+
categories: Object.freeze(categories),
|
|
77
|
+
intercepts: Object.freeze(intercepts),
|
|
78
|
+
features: Object.freeze(features),
|
|
79
|
+
ngrams: Object.freeze([...new Set(ngrams)]),
|
|
80
|
+
encoding,
|
|
81
|
+
});
|
|
82
|
+
}
|
|
83
|
+
/** Unicode-aware convenience tokenizer; supply exact study tokens when reproducing a pipeline. */
|
|
84
|
+
export function tokenize(text) {
|
|
85
|
+
if (typeof text !== "string")
|
|
86
|
+
throw new TypeError("text must be a string");
|
|
87
|
+
return text.normalize("NFC").toLowerCase().replace(/[’‘]/gu, "'").match(/https?:\/\/[^\s]+|<3|[:;=8][\-o*']?[\)\]\(\[dp/\\]|[\)\]\(\[d][:;=8]|[#@][\p{L}\p{M}\p{N}_]+|\p{N}+(?:[.,:]\p{N}+)+|[\p{L}\p{M}\p{N}_]+(?:['-][\p{L}\p{M}\p{N}_]+)*|\.{3,}|\p{Extended_Pictographic}(?:\uFE0F|\p{M})*|[^\s]/gu) ?? [];
|
|
88
|
+
}
|
|
89
|
+
/** Score a validated lexicon. Input token arrays are used exactly as supplied. */
|
|
90
|
+
export function score(input, lexicon, options = {}) {
|
|
91
|
+
if (!options || typeof options !== "object" || Array.isArray(options)) {
|
|
92
|
+
throw new TypeError("options must be an object");
|
|
93
|
+
}
|
|
94
|
+
const allowed = [
|
|
95
|
+
"encoding",
|
|
96
|
+
"ngrams",
|
|
97
|
+
"includeIntercept",
|
|
98
|
+
"minWeight",
|
|
99
|
+
"maxWeight",
|
|
100
|
+
"decimals",
|
|
101
|
+
"features",
|
|
102
|
+
];
|
|
103
|
+
for (const key of Object.keys(options)) {
|
|
104
|
+
if (!allowed.includes(key)) {
|
|
105
|
+
throw new RangeError(`Unknown option: ${key}`);
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
const encoding = options.encoding ?? lexicon.encoding;
|
|
109
|
+
const ngrams = options.ngrams ?? lexicon.ngrams;
|
|
110
|
+
const min = options.minWeight ?? -Infinity, max = options.maxWeight ?? Infinity;
|
|
111
|
+
checkEncoding(encoding);
|
|
112
|
+
checkNgrams(ngrams);
|
|
113
|
+
if (typeof min !== "number" || typeof max !== "number" || Number.isNaN(min) ||
|
|
114
|
+
Number.isNaN(max) || min > max)
|
|
115
|
+
throw new RangeError("Invalid weight bounds");
|
|
116
|
+
if (options.includeIntercept !== undefined &&
|
|
117
|
+
typeof options.includeIntercept !== "boolean")
|
|
118
|
+
throw new TypeError("includeIntercept must be boolean");
|
|
119
|
+
if (options.decimals !== undefined &&
|
|
120
|
+
(!Number.isInteger(options.decimals) || options.decimals < 0 ||
|
|
121
|
+
options.decimals > 15))
|
|
122
|
+
throw new RangeError("decimals must be an integer from 0 to 15");
|
|
123
|
+
const tokens = typeof input === "string" ? tokenize(input) : input;
|
|
124
|
+
if (!Array.isArray(tokens) || tokens.some((t) => typeof t !== "string" || !t)) {
|
|
125
|
+
throw new TypeError("input must be text or an array of nonempty token strings");
|
|
126
|
+
}
|
|
127
|
+
const suppliedFeatures = options.features ?? {};
|
|
128
|
+
if (typeof suppliedFeatures !== "object" || Array.isArray(suppliedFeatures)) {
|
|
129
|
+
throw new TypeError("features must be an object");
|
|
130
|
+
}
|
|
131
|
+
const knownFeatures = new Set(Object.values(lexicon.features).flatMap((f) => Object.keys(f)));
|
|
132
|
+
for (const [key, value] of Object.entries(suppliedFeatures)) {
|
|
133
|
+
if (!knownFeatures.has(key)) {
|
|
134
|
+
throw new RangeError(`Unknown model feature: ${key}`);
|
|
135
|
+
}
|
|
136
|
+
finite(value, `Feature ${key}`);
|
|
137
|
+
}
|
|
138
|
+
const counts = new Map();
|
|
139
|
+
let featureCount = 0;
|
|
140
|
+
for (const n of new Set(ngrams)) {
|
|
141
|
+
for (let i = 0; i <= tokens.length - n; i++) {
|
|
142
|
+
const term = tokens.slice(i, i + n).join(" ");
|
|
143
|
+
counts.set(term, (counts.get(term) ?? 0) + 1);
|
|
144
|
+
featureCount++;
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
const values = Object.create(null);
|
|
148
|
+
const matches = Object.create(null);
|
|
149
|
+
const featureContributions = Object
|
|
150
|
+
.create(null);
|
|
151
|
+
const matchedTerms = new Set();
|
|
152
|
+
let hasEvidence = false;
|
|
153
|
+
const warnings = encoding === "percent"
|
|
154
|
+
? []
|
|
155
|
+
: [...knownFeatures].filter((f) => !own(suppliedFeatures, f)).map((f) => `Structural feature ${f} was not supplied; its contribution is omitted.`);
|
|
156
|
+
for (const [category, weights] of Object.entries(lexicon.categories)) {
|
|
157
|
+
let total = encoding === "percent" || options.includeIntercept === false
|
|
158
|
+
? 0
|
|
159
|
+
: lexicon.intercepts[category];
|
|
160
|
+
const categoryMatches = [];
|
|
161
|
+
const covariates = Object.create(null);
|
|
162
|
+
let matchedCount = 0;
|
|
163
|
+
for (const [term, count] of counts) {
|
|
164
|
+
if (!own(weights, term))
|
|
165
|
+
continue;
|
|
166
|
+
const weight = weights[term];
|
|
167
|
+
if (weight < min || weight > max)
|
|
168
|
+
continue;
|
|
169
|
+
const contribution = encoding === "binary"
|
|
170
|
+
? weight
|
|
171
|
+
: encoding === "percent"
|
|
172
|
+
? count / featureCount
|
|
173
|
+
: weight * (count / tokens.length);
|
|
174
|
+
total += contribution;
|
|
175
|
+
matchedCount += count;
|
|
176
|
+
categoryMatches.push({ term, count, weight, contribution });
|
|
177
|
+
matchedTerms.add(term);
|
|
178
|
+
}
|
|
179
|
+
if (encoding === "percent") {
|
|
180
|
+
total = featureCount ? matchedCount / featureCount : 0;
|
|
181
|
+
}
|
|
182
|
+
if (encoding !== "percent" && tokens.length) {
|
|
183
|
+
for (const [feature, weight] of Object.entries(lexicon.features[category])) {
|
|
184
|
+
if (own(suppliedFeatures, feature)) {
|
|
185
|
+
const contribution = suppliedFeatures[feature] * weight;
|
|
186
|
+
covariates[feature] = contribution;
|
|
187
|
+
total += contribution;
|
|
188
|
+
}
|
|
189
|
+
}
|
|
190
|
+
}
|
|
191
|
+
const evidence = tokens.length > 0 &&
|
|
192
|
+
(categoryMatches.length > 0 || Object.keys(covariates).length > 0);
|
|
193
|
+
hasEvidence ||= evidence;
|
|
194
|
+
if (!Number.isFinite(total)) {
|
|
195
|
+
throw new RangeError(`Score overflow for ${category}`);
|
|
196
|
+
}
|
|
197
|
+
values[category] = evidence
|
|
198
|
+
? options.decimals === undefined
|
|
199
|
+
? total
|
|
200
|
+
: Number(total.toFixed(options.decimals))
|
|
201
|
+
: null;
|
|
202
|
+
matches[category] = categoryMatches.sort((a, b) => b.count - a.count || a.term.localeCompare(b.term, "en"));
|
|
203
|
+
featureContributions[category] = covariates;
|
|
204
|
+
}
|
|
205
|
+
let matchedFeatureCount = 0;
|
|
206
|
+
for (const term of matchedTerms)
|
|
207
|
+
matchedFeatureCount += counts.get(term);
|
|
208
|
+
return {
|
|
209
|
+
model: lexicon.id,
|
|
210
|
+
status: !tokens.length ? "empty" : hasEvidence ? "ok" : "no-matches",
|
|
211
|
+
values,
|
|
212
|
+
matches,
|
|
213
|
+
featureContributions,
|
|
214
|
+
info: {
|
|
215
|
+
tokenCount: tokens.length,
|
|
216
|
+
featureCount,
|
|
217
|
+
matchedFeatureCount,
|
|
218
|
+
uniqueMatchedTerms: matchedTerms.size,
|
|
219
|
+
},
|
|
220
|
+
warnings,
|
|
221
|
+
};
|
|
222
|
+
}
|
|
223
|
+
//# sourceMappingURL=core.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"core.js","sourceRoot":"","sources":["../../src/core.ts"],"names":[],"mappings":"AAuDA,MAAM,GAAG,GAAG,CAAC,MAAc,EAAE,GAAW,EAAE,EAAE,CAC1C,MAAM,CAAC,SAAS,CAAC,cAAc,CAAC,IAAI,CAAC,MAAM,EAAE,GAAG,CAAC,CAAC;AACpD,SAAS,MAAM,CAAC,KAAc,EAAE,KAAa;IAC3C,IAAI,OAAO,KAAK,KAAK,QAAQ,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,KAAK,CAAC,EAAE,CAAC;QACzD,MAAM,IAAI,SAAS,CAAC,GAAG,KAAK,iBAAiB,CAAC,CAAC;IACjD,CAAC;AACH,CAAC;AACD,SAAS,WAAW,CAAC,KAAwB;IAC3C,IACE,CAAC,KAAK,CAAC,OAAO,CAAC,KAAK,CAAC;QACrB,KAAK,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,MAAM,CAAC,aAAa,CAAC,CAAC,CAAC,IAAI,CAAC,GAAG,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,EAC7D,CAAC;QACD,MAAM,IAAI,UAAU,CAAC,+CAA+C,CAAC,CAAC;IACxE,CAAC;AACH,CAAC;AACD,SAAS,aAAa,CAAC,KAAc;IACnC,IAAI,CAAC,CAAC,WAAW,EAAE,QAAQ,EAAE,SAAS,CAAC,CAAC,QAAQ,CAAC,KAAe,CAAC,EAAE,CAAC;QAClE,MAAM,IAAI,UAAU,CAAC,kBAAkB,CAAC,CAAC;IAC3C,CAAC;AACH,CAAC;AACD,gFAAgF;AAChF,MAAM,UAAU,aAAa,CAAC,UAA6B;IACzD,IACE,CAAC,UAAU,IAAI,OAAO,UAAU,CAAC,EAAE,KAAK,QAAQ,IAAI,CAAC,UAAU,CAAC,EAAE;QAClE,CAAC,UAAU,CAAC,UAAU,IAAI,OAAO,UAAU,CAAC,UAAU,KAAK,QAAQ;QACnE,KAAK,CAAC,OAAO,CAAC,UAAU,CAAC,UAAU,CAAC,EACpC,CAAC;QACD,MAAM,IAAI,SAAS,CAAC,gDAAgD,CAAC,CAAC;IACxE,CAAC;IACD,MAAM,UAAU,GAAqD,MAAM;SACxE,MAAM,CAAC,IAAI,CAAC,CAAC;IAChB,MAAM,UAAU,GAA2B,MAAM,CAAC,MAAM,CAAC,IAAI,CAAC,CAAC;IAC/D,MAAM,QAAQ,GAAqD,MAAM;SACtE,MAAM,CAAC,IAAI,CAAC,CAAC;IAChB,KAAK,MAAM,CAAC,QAAQ,EAAE,KAAK,CAAC,IAAI,MAAM,CAAC,OAAO,CAAC,UAAU,CAAC,UAAU,CAAC,EAAE,CAAC;QACtE,IACE,CAAC,QAAQ,IAAI,CAAC,KAAK,IAAI,OAAO,KAAK,KAAK,QAAQ,IAAI,KAAK,CAAC,OAAO,CAAC,KAAK,CAAC;YACxE,MAAM,IAAI,SAAS,CAAC,kBAAkB,CAAC,CAAC;QAC1C,KAAK,MAAM,CAAC,IAAI,EAAE,MAAM,CAAC,IAAI,MAAM,CAAC,OAAO,CAAC,KAAK,CAAC,EAAE,CAAC;YACnD,IAAI,CAAC,IAAI;gBAAE,MAAM,IAAI,SAAS,CAAC,iCAAiC,CAAC,CAAC;YAClE,MAAM,CAAC,MAAM,EAAE,cAAc,IAAI,EAAE,CAAC,CAAC;QACvC,CAAC;QACD,UAAU,CAAC,QAAQ,CAAC,GAAG,MAAM,CAAC,MAAM,CAAC,EAAE,GAAG,KAAK,EAAE,CAAC,CAAC;QACnD,MAAM,SAAS,GACb,UAAU,CAAC,UAAU,IAAI,GAAG,CAAC,UAAU,CAAC,UAAU,EAAE,QAAQ,CAAC;YAC3D,CAAC,CAAC,UAAU,CAAC,UAAU,CAAC,QAAQ,CAAC;YACjC,CAAC,CAAC,CAAC,CAAC;QACR,MAAM,CAAC,SAAS,EAAE,iBAAiB,QAAQ,EAAE,CAAC,CAAC;QAC/C,UAAU,CAAC,QAAQ,CAAC,GAAG,SAAS,CAAC;QACjC,MAAM,UAAU,GAAG,UAAU,CAAC,QAAQ,IAAI,GAAG,CAAC,UAAU,CAAC,QAAQ,EAAE,QAAQ,CAAC;YAC1E,CAAC,CAAC,UAAU,CAAC,QAAQ,CAAC,QAAQ,CAAE;YAChC,CAAC,CAAC,EAAE,CAAC;QACP,IACE,CAAC,UAAU,IAAI,OAAO,UAAU,KAAK,QAAQ,IAAI,KAAK,CAAC,OAAO,CAAC,UAAU,CAAC;YAC1E,MAAM,IAAI,SAAS,CAAC,+BAA+B,CAAC,CAAC;QACvD,KAAK,MAAM,CAAC,OAAO,EAAE,MAAM,CAAC,IAAI,MAAM,CAAC,OAAO,CAAC,UAAU,CAAC,EAAE,CAAC;YAC3D,MAAM,CAAC,MAAM,EAAE,sBAAsB,OAAO,EAAE,CAAC,CAAC;QAClD,CAAC;QACD,QAAQ,CAAC,QAAQ,CAAC,GAAG,MAAM,CAAC,MAAM,CAAC,EAAE,GAAG,UAAU,EAAE,CAAC,CAAC;IACxD,CAAC;IACD,IAAI,CAAC,MAAM,CAAC,IAAI,CAAC,UAAU,CAAC,CAAC,MAAM,EAAE,CAAC;QACpC,MAAM,IAAI,SAAS,CAAC,uCAAuC,CAAC,CAAC;IAC/D,CAAC;IACD,KAAK,MAAM,GAAG,IAAI,MAAM,CAAC,IAAI,CAAC,UAAU,CAAC,UAAU,IAAI,EAAE,CAAC,EAAE,CAAC;QAC3D,IAAI,CAAC,GAAG,CAAC,UAAU,EAAE,GAAG,CAAC,EAAE,CAAC;YAC1B,MAAM,IAAI,UAAU,CAAC,+BAA+B,GAAG,EAAE,CAAC,CAAC;QAC7D,CAAC;IACH,CAAC;IACD,KAAK,MAAM,GAAG,IAAI,MAAM,CAAC,IAAI,CAAC,UAAU,CAAC,QAAQ,IAAI,EAAE,CAAC,EAAE,CAAC;QACzD,IAAI,CAAC,GAAG,CAAC,UAAU,EAAE,GAAG,CAAC,EAAE,CAAC;YAC1B,MAAM,IAAI,UAAU,CAAC,6BAA6B,GAAG,EAAE,CAAC,CAAC;QAC3D,CAAC;IACH,CAAC;IACD,MAAM,MAAM,GAAG,UAAU,CAAC,MAAM,IAAI,CAAC,CAAC,EAAE,CAAC,EAAE,CAAC,CAAC,CAAC;IAC9C,MAAM,QAAQ,GAAG,UAAU,CAAC,QAAQ,IAAI,WAAW,CAAC;IACpD,WAAW,CAAC,MAAM,CAAC,CAAC;IACpB,aAAa,CAAC,QAAQ,CAAC,CAAC;IACxB,OAAO,MAAM,CAAC,MAAM,CAAC;QACnB,EAAE,EAAE,UAAU,CAAC,EAAE;QACjB,GAAG,CAAC,UAAU,CAAC,QAAQ,KAAK,SAAS;YACnC,CAAC,CAAC,EAAE;YACJ,CAAC,CAAC,EAAE,QAAQ,EAAE,UAAU,CAAC,QAAQ,EAAE,CAAC;QACtC,UAAU,EAAE,MAAM,CAAC,MAAM,CAAC,UAAU,CAAC;QACrC,UAAU,EAAE,MAAM,CAAC,MAAM,CAAC,UAAU,CAAC;QACrC,QAAQ,EAAE,MAAM,CAAC,MAAM,CAAC,QAAQ,CAAC;QACjC,MAAM,EAAE,MAAM,CAAC,MAAM,CAAC,CAAC,GAAG,IAAI,GAAG,CAAC,MAAM,CAAC,CAAC,CAAC;QAC3C,QAAQ;KACT,CAAC,CAAC;AACL,CAAC;AACD,kGAAkG;AAClG,MAAM,UAAU,QAAQ,CAAC,IAAY;IACnC,IAAI,OAAO,IAAI,KAAK,QAAQ;QAAE,MAAM,IAAI,SAAS,CAAC,uBAAuB,CAAC,CAAC;IAC3E,OAAO,IAAI,CAAC,SAAS,CAAC,KAAK,CAAC,CAAC,WAAW,EAAE,CAAC,OAAO,CAAC,QAAQ,EAAE,GAAG,CAAC,CAAC,KAAK,CACrE,+NAA+N,CAChO,IAAI,EAAE,CAAC;AACV,CAAC;AACD,kFAAkF;AAClF,MAAM,UAAU,KAAK,CACnB,KAAiC,EACjC,OAAgB,EAChB,OAAO,GAAY,EAAE;IAErB,IAAI,CAAC,OAAO,IAAI,OAAO,OAAO,KAAK,QAAQ,IAAI,KAAK,CAAC,OAAO,CAAC,OAAO,CAAC,EAAE,CAAC;QACtE,MAAM,IAAI,SAAS,CAAC,2BAA2B,CAAC,CAAC;IACnD,CAAC;IACD,MAAM,OAAO,GAAG;QACd,UAAU;QACV,QAAQ;QACR,kBAAkB;QAClB,WAAW;QACX,WAAW;QACX,UAAU;QACV,UAAU;KACX,CAAC;IACF,KAAK,MAAM,GAAG,IAAI,MAAM,CAAC,IAAI,CAAC,OAAO,CAAC,EAAE,CAAC;QACvC,IAAI,CAAC,OAAO,CAAC,QAAQ,CAAC,GAAG,CAAC,EAAE,CAAC;YAC3B,MAAM,IAAI,UAAU,CAAC,mBAAmB,GAAG,EAAE,CAAC,CAAC;QACjD,CAAC;IACH,CAAC;IACD,MAAM,QAAQ,GAAG,OAAO,CAAC,QAAQ,IAAI,OAAO,CAAC,QAAQ,CAAC;IACtD,MAAM,MAAM,GAAG,OAAO,CAAC,MAAM,IAAI,OAAO,CAAC,MAAM,CAAC;IAChD,MAAM,GAAG,GAAG,OAAO,CAAC,SAAS,IAAI,CAAC,QAAQ,EACxC,GAAG,GAAG,OAAO,CAAC,SAAS,IAAI,QAAQ,CAAC;IACtC,aAAa,CAAC,QAAQ,CAAC,CAAC;IACxB,WAAW,CAAC,MAAM,CAAC,CAAC;IACpB,IACE,OAAO,GAAG,KAAK,QAAQ,IAAI,OAAO,GAAG,KAAK,QAAQ,IAAI,MAAM,CAAC,KAAK,CAAC,GAAG,CAAC;QACvE,MAAM,CAAC,KAAK,CAAC,GAAG,CAAC,IAAI,GAAG,GAAG,GAAG;QAC9B,MAAM,IAAI,UAAU,CAAC,uBAAuB,CAAC,CAAC;IAChD,IACE,OAAO,CAAC,gBAAgB,KAAK,SAAS;QACtC,OAAO,OAAO,CAAC,gBAAgB,KAAK,SAAS;QAC7C,MAAM,IAAI,SAAS,CAAC,kCAAkC,CAAC,CAAC;IAC1D,IACE,OAAO,CAAC,QAAQ,KAAK,SAAS;QAC9B,CAAC,CAAC,MAAM,CAAC,SAAS,CAAC,OAAO,CAAC,QAAQ,CAAC,IAAI,OAAO,CAAC,QAAQ,GAAG,CAAC;YAC1D,OAAO,CAAC,QAAQ,GAAG,EAAE,CAAC;QACxB,MAAM,IAAI,UAAU,CAAC,0CAA0C,CAAC,CAAC;IACnE,MAAM,MAAM,GAAG,OAAO,KAAK,KAAK,QAAQ,CAAC,CAAC,CAAC,QAAQ,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,KAAK,CAAC;IACnE,IACE,CAAC,KAAK,CAAC,OAAO,CAAC,MAAM,CAAC,IAAI,MAAM,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,OAAO,CAAC,KAAK,QAAQ,IAAI,CAAC,CAAC,CAAC,EACzE,CAAC;QACD,MAAM,IAAI,SAAS,CACjB,0DAA0D,CAC3D,CAAC;IACJ,CAAC;IACD,MAAM,gBAAgB,GAAG,OAAO,CAAC,QAAQ,IAAI,EAAE,CAAC;IAChD,IAAI,OAAO,gBAAgB,KAAK,QAAQ,IAAI,KAAK,CAAC,OAAO,CAAC,gBAAgB,CAAC,EAAE,CAAC;QAC5E,MAAM,IAAI,SAAS,CAAC,4BAA4B,CAAC,CAAC;IACpD,CAAC;IACD,MAAM,aAAa,GAAG,IAAI,GAAG,CAC3B,MAAM,CAAC,MAAM,CAAC,OAAO,CAAC,QAAQ,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,MAAM,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC,CAC/D,CAAC;IACF,KAAK,MAAM,CAAC,GAAG,EAAE,KAAK,CAAC,IAAI,MAAM,CAAC,OAAO,CAAC,gBAAgB,CAAC,EAAE,CAAC;QAC5D,IAAI,CAAC,aAAa,CAAC,GAAG,CAAC,GAAG,CAAC,EAAE,CAAC;YAC5B,MAAM,IAAI,UAAU,CAAC,0BAA0B,GAAG,EAAE,CAAC,CAAC;QACxD,CAAC;QACD,MAAM,CAAC,KAAK,EAAE,WAAW,GAAG,EAAE,CAAC,CAAC;IAClC,CAAC;IACD,MAAM,MAAM,GAAG,IAAI,GAAG,EAAkB,CAAC;IACzC,IAAI,YAAY,GAAG,CAAC,CAAC;IACrB,KAAK,MAAM,CAAC,IAAI,IAAI,GAAG,CAAC,MAAM,CAAC,EAAE,CAAC;QAChC,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,IAAI,MAAM,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC;YAC5C,MAAM,IAAI,GAAG,MAAM,CAAC,KAAK,CAAC,CAAC,EAAE,CAAC,GAAG,CAAC,CAAC,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC;YAC9C,MAAM,CAAC,GAAG,CAAC,IAAI,EAAE,CAAC,MAAM,CAAC,GAAG,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC;YAC9C,YAAY,EAAE,CAAC;QACjB,CAAC;IACH,CAAC;IACD,MAAM,MAAM,GAAkC,MAAM,CAAC,MAAM,CAAC,IAAI,CAAC,CAAC;IAClE,MAAM,OAAO,GAA4B,MAAM,CAAC,MAAM,CAAC,IAAI,CAAC,CAAC;IAC7D,MAAM,oBAAoB,GAA2C,MAAM;SACxE,MAAM,CAAC,IAAI,CAAC,CAAC;IAChB,MAAM,YAAY,GAAG,IAAI,GAAG,EAAU,CAAC;IACvC,IAAI,WAAW,GAAG,KAAK,CAAC;IACxB,MAAM,QAAQ,GAAG,QAAQ,KAAK,SAAS;QACrC,CAAC,CAAC,EAAE;QACJ,CAAC,CAAC,CAAC,GAAG,aAAa,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,GAAG,CAAC,gBAAgB,EAAE,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CACtE,sBAAsB,CAAC,iDAAiD,CACzE,CAAC;IACJ,KAAK,MAAM,CAAC,QAAQ,EAAE,OAAO,CAAC,IAAI,MAAM,CAAC,OAAO,CAAC,OAAO,CAAC,UAAU,CAAC,EAAE,CAAC;QACrE,IAAI,KAAK,GAAG,QAAQ,KAAK,SAAS,IAAI,OAAO,CAAC,gBAAgB,KAAK,KAAK;YACtE,CAAC,CAAC,CAAC;YACH,CAAC,CAAC,OAAO,CAAC,UAAU,CAAC,QAAQ,CAAE,CAAC;QAClC,MAAM,eAAe,GAAY,EAAE,CAAC;QACpC,MAAM,UAAU,GAA2B,MAAM,CAAC,MAAM,CAAC,IAAI,CAAC,CAAC;QAC/D,IAAI,YAAY,GAAG,CAAC,CAAC;QACrB,KAAK,MAAM,CAAC,IAAI,EAAE,KAAK,CAAC,IAAI,MAAM,EAAE,CAAC;YACnC,IAAI,CAAC,GAAG,CAAC,OAAO,EAAE,IAAI,CAAC;gBAAE,SAAS;YAClC,MAAM,MAAM,GAAG,OAAO,CAAC,IAAI,CAAE,CAAC;YAC9B,IAAI,MAAM,GAAG,GAAG,IAAI,MAAM,GAAG,GAAG;gBAAE,SAAS;YAC3C,MAAM,YAAY,GAAG,QAAQ,KAAK,QAAQ;gBACxC,CAAC,CAAC,MAAM;gBACR,CAAC,CAAC,QAAQ,KAAK,SAAS;oBACxB,CAAC,CAAC,KAAK,GAAG,YAAY;oBACtB,CAAC,CAAC,MAAM,GAAG,CAAC,KAAK,GAAG,MAAM,CAAC,MAAM,CAAC,CAAC;YACrC,KAAK,IAAI,YAAY,CAAC;YACtB,YAAY,IAAI,KAAK,CAAC;YACtB,eAAe,CAAC,IAAI,CAAC,EAAE,IAAI,EAAE,KAAK,EAAE,MAAM,EAAE,YAAY,EAAE,CAAC,CAAC;YAC5D,YAAY,CAAC,GAAG,CAAC,IAAI,CAAC,CAAC;QACzB,CAAC;QACD,IAAI,QAAQ,KAAK,SAAS,EAAE,CAAC;YAC3B,KAAK,GAAG,YAAY,CAAC,CAAC,CAAC,YAAY,GAAG,YAAY,CAAC,CAAC,CAAC,CAAC,CAAC;QACzD,CAAC;QACD,IAAI,QAAQ,KAAK,SAAS,IAAI,MAAM,CAAC,MAAM,EAAE,CAAC;YAC5C,KACE,MAAM,CAAC,OAAO,EAAE,MAAM,CAAC,IAAI,MAAM,CAAC,OAAO,CAAC,OAAO,CAAC,QAAQ,CAAC,QAAQ,CAAE,CAAC,EACtE,CAAC;gBACD,IAAI,GAAG,CAAC,gBAAgB,EAAE,OAAO,CAAC,EAAE,CAAC;oBACnC,MAAM,YAAY,GAAG,gBAAgB,CAAC,OAAO,CAAE,GAAG,MAAM,CAAC;oBACzD,UAAU,CAAC,OAAO,CAAC,GAAG,YAAY,CAAC;oBACnC,KAAK,IAAI,YAAY,CAAC;gBACxB,CAAC;YACH,CAAC;QACH,CAAC;QACD,MAAM,QAAQ,GAAG,MAAM,CAAC,MAAM,GAAG,CAAC;YAChC,CAAC,eAAe,CAAC,MAAM,GAAG,CAAC,IAAI,MAAM,CAAC,IAAI,CAAC,UAAU,CAAC,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC;QACrE,WAAW,KAAK,QAAQ,CAAC;QACzB,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,KAAK,CAAC,EAAE,CAAC;YAC5B,MAAM,IAAI,UAAU,CAAC,sBAAsB,QAAQ,EAAE,CAAC,CAAC;QACzD,CAAC;QACD,MAAM,CAAC,QAAQ,CAAC,GAAG,QAAQ;YACzB,CAAC,CAAC,OAAO,CAAC,QAAQ,KAAK,SAAS;gBAC9B,CAAC,CAAC,KAAK;gBACP,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC,OAAO,CAAC,OAAO,CAAC,QAAQ,CAAC,CAAC;YAC3C,CAAC,CAAC,IAAI,CAAC;QACT,OAAO,CAAC,QAAQ,CAAC,GAAG,eAAe,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAChD,CAAC,CAAC,KAAK,GAAG,CAAC,CAAC,KAAK,IAAI,CAAC,CAAC,IAAI,CAAC,aAAa,CAAC,CAAC,CAAC,IAAI,EAAE,IAAI,CAAC,CACxD,CAAC;QACF,oBAAoB,CAAC,QAAQ,CAAC,GAAG,UAAU,CAAC;IAC9C,CAAC;IACD,IAAI,mBAAmB,GAAG,CAAC,CAAC;IAC5B,KAAK,MAAM,IAAI,IAAI,YAAY;QAAE,mBAAmB,IAAI,MAAM,CAAC,GAAG,CAAC,IAAI,CAAE,CAAC;IAC1E,OAAO;QACL,KAAK,EAAE,OAAO,CAAC,EAAE;QACjB,MAAM,EAAE,CAAC,MAAM,CAAC,MAAM,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,WAAW,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,YAAY;QACpE,MAAM;QACN,OAAO;QACP,oBAAoB;QACpB,IAAI,EAAE;YACJ,UAAU,EAAE,MAAM,CAAC,MAAM;YACzB,YAAY;YACZ,mBAAmB;YACnB,kBAAkB,EAAE,YAAY,CAAC,IAAI;SACtC;QACD,QAAQ;KACT,CAAC;AACJ,CAAC","sourcesContent":["/** Shared weighted-lexicon engine. No I/O, global state, or runtime dependencies. */\nexport type Encoding = \"frequency\" | \"binary\" | \"percent\";\nexport type Weights = Readonly<\n Record<string, Readonly<Record<string, number>>>\n>;\nexport interface LexiconDefinition {\n readonly id: string;\n readonly language?: string;\n readonly categories: Weights;\n readonly intercepts?: Readonly<Record<string, number>>;\n readonly features?: Weights;\n readonly ngrams?: readonly number[];\n readonly encoding?: Encoding;\n}\nexport interface Lexicon extends LexiconDefinition {\n readonly intercepts: Readonly<Record<string, number>>;\n readonly features: Weights;\n readonly ngrams: readonly number[];\n readonly encoding: Encoding;\n}\nexport interface Options {\n /** Frequency divides term occurrences by original token count, including unmatched tokens. */\n readonly encoding?: Encoding;\n /** Sizes include unigrams explicitly. An empty array disables lexical matching. */\n readonly ngrams?: readonly number[];\n readonly includeIntercept?: boolean;\n readonly minWeight?: number;\n readonly maxWeight?: number;\n /** Round final scores only. Omit to retain full floating-point precision. */\n readonly decimals?: number;\n /** Measured nonlexical covariates; never infer these from a single text. */\n readonly features?: Readonly<Record<string, number>>;\n}\nexport interface Match {\n readonly term: string;\n readonly count: number;\n readonly weight: number;\n readonly contribution: number;\n}\nexport interface Analysis {\n readonly model: string;\n readonly status: \"ok\" | \"empty\" | \"no-matches\";\n readonly values: Readonly<Record<string, number | null>>;\n readonly matches: Readonly<Record<string, readonly Match[]>>;\n readonly featureContributions: Readonly<\n Record<string, Readonly<Record<string, number>>>\n >;\n readonly info: {\n readonly tokenCount: number;\n readonly featureCount: number;\n readonly matchedFeatureCount: number;\n readonly uniqueMatchedTerms: number;\n };\n readonly warnings: readonly string[];\n}\nconst own = (object: object, key: string) =>\n Object.prototype.hasOwnProperty.call(object, key);\nfunction finite(value: unknown, label: string): asserts value is number {\n if (typeof value !== \"number\" || !Number.isFinite(value)) {\n throw new TypeError(`${label} must be finite`);\n }\n}\nfunction checkNgrams(value: readonly number[]): void {\n if (\n !Array.isArray(value) ||\n value.some((n) => !Number.isSafeInteger(n) || n < 1 || n > 8)\n ) {\n throw new RangeError(\"ngrams must contain integer sizes from 1 to 8\");\n }\n}\nfunction checkEncoding(value: unknown): asserts value is Encoding {\n if (![\"frequency\", \"binary\", \"percent\"].includes(value as string)) {\n throw new RangeError(\"Unknown encoding\");\n }\n}\n/** Copy and validate a custom lexicon once; returned data cannot be mutated. */\nexport function createLexicon(definition: LexiconDefinition): Lexicon {\n if (\n !definition || typeof definition.id !== \"string\" || !definition.id ||\n !definition.categories || typeof definition.categories !== \"object\" ||\n Array.isArray(definition.categories)\n ) {\n throw new TypeError(\"A lexicon needs an id and category weight maps\");\n }\n const categories: Record<string, Readonly<Record<string, number>>> = Object\n .create(null);\n const intercepts: Record<string, number> = Object.create(null);\n const features: Record<string, Readonly<Record<string, number>>> = Object\n .create(null);\n for (const [category, terms] of Object.entries(definition.categories)) {\n if (\n !category || !terms || typeof terms !== \"object\" || Array.isArray(terms)\n ) throw new TypeError(\"Invalid category\");\n for (const [term, weight] of Object.entries(terms)) {\n if (!term) throw new TypeError(\"Lexicon terms must not be empty\");\n finite(weight, `Weight for ${term}`);\n }\n categories[category] = Object.freeze({ ...terms });\n const intercept =\n definition.intercepts && own(definition.intercepts, category)\n ? definition.intercepts[category]\n : 0;\n finite(intercept, `Intercept for ${category}`);\n intercepts[category] = intercept;\n const covariates = definition.features && own(definition.features, category)\n ? definition.features[category]!\n : {};\n if (\n !covariates || typeof covariates !== \"object\" || Array.isArray(covariates)\n ) throw new TypeError(\"Feature weights must be a map\");\n for (const [feature, weight] of Object.entries(covariates)) {\n finite(weight, `Feature weight for ${feature}`);\n }\n features[category] = Object.freeze({ ...covariates });\n }\n if (!Object.keys(categories).length) {\n throw new TypeError(\"A lexicon needs at least one category\");\n }\n for (const key of Object.keys(definition.intercepts ?? {})) {\n if (!own(categories, key)) {\n throw new RangeError(`Unknown intercept category: ${key}`);\n }\n }\n for (const key of Object.keys(definition.features ?? {})) {\n if (!own(categories, key)) {\n throw new RangeError(`Unknown feature category: ${key}`);\n }\n }\n const ngrams = definition.ngrams ?? [1, 2, 3];\n const encoding = definition.encoding ?? \"frequency\";\n checkNgrams(ngrams);\n checkEncoding(encoding);\n return Object.freeze({\n id: definition.id,\n ...(definition.language === undefined\n ? {}\n : { language: definition.language }),\n categories: Object.freeze(categories),\n intercepts: Object.freeze(intercepts),\n features: Object.freeze(features),\n ngrams: Object.freeze([...new Set(ngrams)]),\n encoding,\n });\n}\n/** Unicode-aware convenience tokenizer; supply exact study tokens when reproducing a pipeline. */\nexport function tokenize(text: string): string[] {\n if (typeof text !== \"string\") throw new TypeError(\"text must be a string\");\n return text.normalize(\"NFC\").toLowerCase().replace(/[’‘]/gu, \"'\").match(\n /https?:\\/\\/[^\\s]+|<3|[:;=8][\\-o*']?[\\)\\]\\(\\[dp/\\\\]|[\\)\\]\\(\\[d][:;=8]|[#@][\\p{L}\\p{M}\\p{N}_]+|\\p{N}+(?:[.,:]\\p{N}+)+|[\\p{L}\\p{M}\\p{N}_]+(?:['-][\\p{L}\\p{M}\\p{N}_]+)*|\\.{3,}|\\p{Extended_Pictographic}(?:\\uFE0F|\\p{M})*|[^\\s]/gu,\n ) ?? [];\n}\n/** Score a validated lexicon. Input token arrays are used exactly as supplied. */\nexport function score(\n input: string | readonly string[],\n lexicon: Lexicon,\n options: Options = {},\n): Analysis {\n if (!options || typeof options !== \"object\" || Array.isArray(options)) {\n throw new TypeError(\"options must be an object\");\n }\n const allowed = [\n \"encoding\",\n \"ngrams\",\n \"includeIntercept\",\n \"minWeight\",\n \"maxWeight\",\n \"decimals\",\n \"features\",\n ];\n for (const key of Object.keys(options)) {\n if (!allowed.includes(key)) {\n throw new RangeError(`Unknown option: ${key}`);\n }\n }\n const encoding = options.encoding ?? lexicon.encoding;\n const ngrams = options.ngrams ?? lexicon.ngrams;\n const min = options.minWeight ?? -Infinity,\n max = options.maxWeight ?? Infinity;\n checkEncoding(encoding);\n checkNgrams(ngrams);\n if (\n typeof min !== \"number\" || typeof max !== \"number\" || Number.isNaN(min) ||\n Number.isNaN(max) || min > max\n ) throw new RangeError(\"Invalid weight bounds\");\n if (\n options.includeIntercept !== undefined &&\n typeof options.includeIntercept !== \"boolean\"\n ) throw new TypeError(\"includeIntercept must be boolean\");\n if (\n options.decimals !== undefined &&\n (!Number.isInteger(options.decimals) || options.decimals < 0 ||\n options.decimals > 15)\n ) throw new RangeError(\"decimals must be an integer from 0 to 15\");\n const tokens = typeof input === \"string\" ? tokenize(input) : input;\n if (\n !Array.isArray(tokens) || tokens.some((t) => typeof t !== \"string\" || !t)\n ) {\n throw new TypeError(\n \"input must be text or an array of nonempty token strings\",\n );\n }\n const suppliedFeatures = options.features ?? {};\n if (typeof suppliedFeatures !== \"object\" || Array.isArray(suppliedFeatures)) {\n throw new TypeError(\"features must be an object\");\n }\n const knownFeatures = new Set(\n Object.values(lexicon.features).flatMap((f) => Object.keys(f)),\n );\n for (const [key, value] of Object.entries(suppliedFeatures)) {\n if (!knownFeatures.has(key)) {\n throw new RangeError(`Unknown model feature: ${key}`);\n }\n finite(value, `Feature ${key}`);\n }\n const counts = new Map<string, number>();\n let featureCount = 0;\n for (const n of new Set(ngrams)) {\n for (let i = 0; i <= tokens.length - n; i++) {\n const term = tokens.slice(i, i + n).join(\" \");\n counts.set(term, (counts.get(term) ?? 0) + 1);\n featureCount++;\n }\n }\n const values: Record<string, number | null> = Object.create(null);\n const matches: Record<string, Match[]> = Object.create(null);\n const featureContributions: Record<string, Record<string, number>> = Object\n .create(null);\n const matchedTerms = new Set<string>();\n let hasEvidence = false;\n const warnings = encoding === \"percent\"\n ? []\n : [...knownFeatures].filter((f) => !own(suppliedFeatures, f)).map((f) =>\n `Structural feature ${f} was not supplied; its contribution is omitted.`\n );\n for (const [category, weights] of Object.entries(lexicon.categories)) {\n let total = encoding === \"percent\" || options.includeIntercept === false\n ? 0\n : lexicon.intercepts[category]!;\n const categoryMatches: Match[] = [];\n const covariates: Record<string, number> = Object.create(null);\n let matchedCount = 0;\n for (const [term, count] of counts) {\n if (!own(weights, term)) continue;\n const weight = weights[term]!;\n if (weight < min || weight > max) continue;\n const contribution = encoding === \"binary\"\n ? weight\n : encoding === \"percent\"\n ? count / featureCount\n : weight * (count / tokens.length);\n total += contribution;\n matchedCount += count;\n categoryMatches.push({ term, count, weight, contribution });\n matchedTerms.add(term);\n }\n if (encoding === \"percent\") {\n total = featureCount ? matchedCount / featureCount : 0;\n }\n if (encoding !== \"percent\" && tokens.length) {\n for (\n const [feature, weight] of Object.entries(lexicon.features[category]!)\n ) {\n if (own(suppliedFeatures, feature)) {\n const contribution = suppliedFeatures[feature]! * weight;\n covariates[feature] = contribution;\n total += contribution;\n }\n }\n }\n const evidence = tokens.length > 0 &&\n (categoryMatches.length > 0 || Object.keys(covariates).length > 0);\n hasEvidence ||= evidence;\n if (!Number.isFinite(total)) {\n throw new RangeError(`Score overflow for ${category}`);\n }\n values[category] = evidence\n ? options.decimals === undefined\n ? total\n : Number(total.toFixed(options.decimals))\n : null;\n matches[category] = categoryMatches.sort((a, b) =>\n b.count - a.count || a.term.localeCompare(b.term, \"en\")\n );\n featureContributions[category] = covariates;\n }\n let matchedFeatureCount = 0;\n for (const term of matchedTerms) matchedFeatureCount += counts.get(term)!;\n return {\n model: lexicon.id,\n status: !tokens.length ? \"empty\" : hasEvidence ? \"ok\" : \"no-matches\",\n values,\n matches,\n featureContributions,\n info: {\n tokenCount: tokens.length,\n featureCount,\n matchedFeatureCount,\n uniqueMatchedTerms: matchedTerms.size,\n },\n warnings,\n };\n}\n"]}
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
import { type Analysis, type Lexicon, type Options } from "./core.ts";
|
|
2
|
+
export * from "./core.ts";
|
|
3
|
+
export type ModelId = "affect" | "age" | "gender" | "temporal" | "perma" | "permaEs" | "bigFive" | "darkTriad" | "optimism";
|
|
4
|
+
/** Immutable model definitions, including weights and research intercepts. */
|
|
5
|
+
export declare const models: Readonly<Record<ModelId, Lexicon>>;
|
|
6
|
+
/** Select a bundled research lexicon or supply a validated custom lexicon. */
|
|
7
|
+
export declare function analyse(input: string | readonly string[], model: ModelId | Lexicon, options?: Options): Analysis;
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
import definitions from "../data/models.json" with { type: "json" };
|
|
2
|
+
import { createLexicon, score, } from "./core.js";
|
|
3
|
+
export * from "./core.js";
|
|
4
|
+
/** Immutable model definitions, including weights and research intercepts. */
|
|
5
|
+
export const models = Object.freeze(Object.fromEntries(Object.entries(definitions).map(([id, definition]) => [id, createLexicon(definition)])));
|
|
6
|
+
/** Select a bundled research lexicon or supply a validated custom lexicon. */
|
|
7
|
+
export function analyse(input, model, options = {}) {
|
|
8
|
+
const lexicon = typeof model === "string"
|
|
9
|
+
? Object.prototype.hasOwnProperty.call(models, model)
|
|
10
|
+
? models[model]
|
|
11
|
+
: undefined
|
|
12
|
+
: model;
|
|
13
|
+
if (!lexicon)
|
|
14
|
+
throw new RangeError(`Unknown model: ${String(model)}`);
|
|
15
|
+
return score(input, lexicon, options);
|
|
16
|
+
}
|
|
17
|
+
//# sourceMappingURL=index.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"index.js","sourceRoot":"","sources":["../../src/index.ts"],"names":[],"mappings":"AAAA,OAAO,WAAW,MAAM,qBAAqB,CAAC,OAAO,IAAI,EAAE,MAAM,EAAE,CAAC;AACpE,OAAO,EAEL,aAAa,EAIb,KAAK,GACN,MAAM,WAAW,CAAC;AACnB,cAAc,WAAW,CAAC;AAW1B,8EAA8E;AAC9E,MAAM,CAAC,MAAM,MAAM,GAAuC,MAAM,CAAC,MAAM,CACrE,MAAM,CAAC,WAAW,CAChB,MAAM,CAAC,OAAO,CAAC,WAAW,CAAC,CAAC,GAAG,CAAC,CAC9B,CAAC,EAAE,EAAE,UAAU,CAAC,EAChB,EAAE,CAAC,CAAC,EAAE,EAAE,aAAa,CAAC,UAA+B,CAAC,CAAC,CAAC,CAC/B,CAC9B,CAAC;AACF,8EAA8E;AAC9E,MAAM,UAAU,OAAO,CACrB,KAAiC,EACjC,KAAwB,EACxB,OAAO,GAAY,EAAE;IAErB,MAAM,OAAO,GAAG,OAAO,KAAK,KAAK,QAAQ;QACvC,CAAC,CAAC,MAAM,CAAC,SAAS,CAAC,cAAc,CAAC,IAAI,CAAC,MAAM,EAAE,KAAK,CAAC;YACnD,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC;YACf,CAAC,CAAC,SAAS;QACb,CAAC,CAAC,KAAK,CAAC;IACV,IAAI,CAAC,OAAO;QAAE,MAAM,IAAI,UAAU,CAAC,kBAAkB,MAAM,CAAC,KAAK,CAAC,EAAE,CAAC,CAAC;IACtE,OAAO,KAAK,CAAC,KAAK,EAAE,OAAO,EAAE,OAAO,CAAC,CAAC;AACxC,CAAC","sourcesContent":["import definitions from \"../data/models.json\" with { type: \"json\" };\nimport {\n type Analysis,\n createLexicon,\n type Lexicon,\n type LexiconDefinition,\n type Options,\n score,\n} from \"./core.ts\";\nexport * from \"./core.ts\";\nexport type ModelId =\n | \"affect\"\n | \"age\"\n | \"gender\"\n | \"temporal\"\n | \"perma\"\n | \"permaEs\"\n | \"bigFive\"\n | \"darkTriad\"\n | \"optimism\";\n/** Immutable model definitions, including weights and research intercepts. */\nexport const models: Readonly<Record<ModelId, Lexicon>> = Object.freeze(\n Object.fromEntries(\n Object.entries(definitions).map((\n [id, definition],\n ) => [id, createLexicon(definition as LexiconDefinition)]),\n ) as Record<ModelId, Lexicon>,\n);\n/** Select a bundled research lexicon or supply a validated custom lexicon. */\nexport function analyse(\n input: string | readonly string[],\n model: ModelId | Lexicon,\n options: Options = {},\n): Analysis {\n const lexicon = typeof model === \"string\"\n ? Object.prototype.hasOwnProperty.call(models, model)\n ? models[model]\n : undefined\n : model;\n if (!lexicon) throw new RangeError(`Unknown model: ${String(model)}`);\n return score(input, lexicon, options);\n}\n"]}
|
package/docs/api.md
ADDED
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
# API
|
|
2
|
+
|
|
3
|
+
## `analyse(input, model, options?)`
|
|
4
|
+
|
|
5
|
+
Import from `wwbnlp`. `input` is a string or readonly array of nonempty token
|
|
6
|
+
strings. `model` is one of the nine model IDs in the README, or a lexicon
|
|
7
|
+
returned by `createLexicon`. Invalid model names, inputs and options throw; they
|
|
8
|
+
never fall back to another model or encoding.
|
|
9
|
+
|
|
10
|
+
## `createLexicon(definition)` and `score(input, lexicon, options?)`
|
|
11
|
+
|
|
12
|
+
Import from `wwbnlp/core` to use your own data without loading the bundled
|
|
13
|
+
lexica:
|
|
14
|
+
|
|
15
|
+
```js
|
|
16
|
+
import { createLexicon, score } from "wwbnlp/core";
|
|
17
|
+
|
|
18
|
+
const lexicon = createLexicon({
|
|
19
|
+
id: "example",
|
|
20
|
+
categories: { SENTIMENT: { good: 2, bad: -2, "really good": 3 } },
|
|
21
|
+
intercepts: { SENTIMENT: 0.5 },
|
|
22
|
+
ngrams: [1, 2],
|
|
23
|
+
encoding: "frequency",
|
|
24
|
+
});
|
|
25
|
+
|
|
26
|
+
console.log(score(["really", "good", "good"], lexicon).values);
|
|
27
|
+
// { SENTIMENT: 2.833333333333333 }
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
Definitions are validated, copied and frozen. Every category needs a
|
|
31
|
+
term-to-finite-weight map. Intercepts default to zero. Optional structural
|
|
32
|
+
feature weights use the same category map shape, keyed by feature names. Unknown
|
|
33
|
+
intercept/feature categories are rejected. Custom definitions default to
|
|
34
|
+
frequency encoding and `[1, 2, 3]` ngrams.
|
|
35
|
+
|
|
36
|
+
`score` and `analyse` return `Analysis` with category maps and explicit status;
|
|
37
|
+
see the README. A category without a lexical match or supplied structural
|
|
38
|
+
feature returns `null`, even if it has an intercept. Empty input returns `empty`
|
|
39
|
+
and all-null values. Mixed evidence can produce both numbers and nulls in the
|
|
40
|
+
same result.
|
|
41
|
+
|
|
42
|
+
## Options
|
|
43
|
+
|
|
44
|
+
| Option | Default | Meaning |
|
|
45
|
+
| ------------------------ | -------------------------- | ----------------------------------------------------------------- |
|
|
46
|
+
| `encoding` | Model default | `frequency`, `binary`, or `percent` |
|
|
47
|
+
| `ngrams` | Model vocabulary sizes | Unique integer sizes 1–8; `[]` disables matching |
|
|
48
|
+
| `includeIntercept` | `true` | Include the model intercept, except in percent mode |
|
|
49
|
+
| `minWeight`, `maxWeight` | Negative/positive infinity | Inclusive term-weight filters |
|
|
50
|
+
| `decimals` | Omitted | Round final category scores only, 0–15 decimal places |
|
|
51
|
+
| `features` | `{}` | Finite measured structural values; keys must belong to this model |
|
|
52
|
+
|
|
53
|
+
Unrecognized options throw, so legacy options cannot be silently ignored.
|
|
54
|
+
Options, definitions and input arrays are never modified.
|
|
55
|
+
|
|
56
|
+
Frequency mode divides each matched feature’s occurrence count by the original
|
|
57
|
+
token count, before adding its weight and intercept. Adding bigrams does not
|
|
58
|
+
inflate that denominator. Binary mode counts unique terms once. Percent mode
|
|
59
|
+
measures coverage of generated candidate features; repeated terms count, weights
|
|
60
|
+
do not, and the result stays between zero and one.
|
|
61
|
+
|
|
62
|
+
`matches` sort by descending occurrence count, then term. Contributions retain
|
|
63
|
+
full precision even when final values are rounded. `matchedFeatureCount` counts
|
|
64
|
+
each matched occurrence once across all categories. It includes ngrams, so it is
|
|
65
|
+
not a count of distinct word positions.
|
|
66
|
+
|
|
67
|
+
Structural values are multiplied by their weights and added once per category.
|
|
68
|
+
They are independent of lexical `minWeight`/`maxWeight` filters and are ignored
|
|
69
|
+
in percent mode. Missing structural values produce warnings in weighted modes.
|
|
70
|
+
Supplied zeros are measurements and count as evidence. Supply the same feature
|
|
71
|
+
definitions and units as the original research pipeline; the engine does not
|
|
72
|
+
invent them.
|
|
73
|
+
|
|
74
|
+
## `tokenize(text)`
|
|
75
|
+
|
|
76
|
+
Returns the documented convenience token stream. Supply exact study tokens to
|
|
77
|
+
`score` or `analyse` when controlling preprocessing. There is no automatic
|
|
78
|
+
British-to-American translation, HTML entity decoding or language detection.
|
|
79
|
+
Spanish uses the explicit `permaEs` model.
|
|
80
|
+
|
|
81
|
+
## `models`
|
|
82
|
+
|
|
83
|
+
Readonly registry of bundled model definitions. Categories, coefficients,
|
|
84
|
+
intercepts, structural features and default ngram sizes are available for
|
|
85
|
+
inspection. Importing `wwbnlp/core` avoids loading this registry.
|
|
86
|
+
|
|
87
|
+
## Runtime support
|
|
88
|
+
|
|
89
|
+
ESM, typed declarations, Node.js 22.12+ native `require()` of ESM, and Deno 2
|
|
90
|
+
with the built entry point. JSON modules use import attributes. Older
|
|
91
|
+
Node/CommonJS applications must upgrade or use dynamic import where supported.
|
|
92
|
+
No browser support claim is made without testing a specific bundler/runtime.
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
# Migrating to wwbnlp
|
|
2
|
+
|
|
3
|
+
The existing npm releases remain available. Archiving GitHub repositories does
|
|
4
|
+
not unpublish npm versions, and this consolidation does not replace existing
|
|
5
|
+
dependencies automatically. The legacy npm account is
|
|
6
|
+
[phughes](https://www.npmjs.com/~phughes).
|
|
7
|
+
|
|
8
|
+
The new [wwbnlp](https://www.npmjs.com/package/wwbnlp) package is maintained by
|
|
9
|
+
[phughesmcr](https://www.npmjs.com/~phughesmcr). Install it with
|
|
10
|
+
`npm install wwbnlp` and migrate explicitly using the model IDs below.
|
|
11
|
+
|
|
12
|
+
## Imports and model IDs
|
|
13
|
+
|
|
14
|
+
| Existing package | Last observed npm version | New call |
|
|
15
|
+
| ------------------ | -------------------------------------- | ------------------------------------------------------ |
|
|
16
|
+
| affectimo | 3.1.0 | `analyse(text, 'affect')` |
|
|
17
|
+
| bigfive | 3.0.0 | `analyse(text, 'bigFive')` |
|
|
18
|
+
| darktriad | 3.1.1 | `analyse(text, 'darkTriad')` |
|
|
19
|
+
| optimismo | 4.0.1 | `analyse(text, 'optimism')` |
|
|
20
|
+
| predictage | 4.0.1 | `analyse(text, 'age')` |
|
|
21
|
+
| predictgender | 4.0.0 | `analyse(text, 'gender')` |
|
|
22
|
+
| prospectimo | 3.0.0 | `analyse(text, 'temporal')` |
|
|
23
|
+
| wellbeing_analysis | 4.0.0 | `analyse(text, 'perma')` or `analyse(text, 'permaEs')` |
|
|
24
|
+
| lex-helpers | 0.6.0 | `createLexicon` and `score` from `wwbnlp/core` |
|
|
25
|
+
| weighted-lexica | Not found under this unscoped npm name | `createLexicon` and `score` from `wwbnlp/core` |
|
|
26
|
+
|
|
27
|
+
Registry versions were checked on 4 October 2026. `weighted-lexica` may have
|
|
28
|
+
been published under a different name or scope; a 404 does not establish that no
|
|
29
|
+
release ever existed.
|
|
30
|
+
|
|
31
|
+
```js
|
|
32
|
+
// Before
|
|
33
|
+
const affectimo = require("affectimo");
|
|
34
|
+
const values = affectimo(text);
|
|
35
|
+
|
|
36
|
+
// After
|
|
37
|
+
import { analyse } from "wwbnlp";
|
|
38
|
+
const result = analyse(text, "affect");
|
|
39
|
+
const valuesAfter = result.values;
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
Native CommonJS works on Node 22.12+:
|
|
43
|
+
|
|
44
|
+
```js
|
|
45
|
+
const { analyse } = require("wwbnlp");
|
|
46
|
+
const result = analyse(text, "affect");
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
## Intentional changes
|
|
50
|
+
|
|
51
|
+
This is a new API, not a drop-in wrapper or promise of numerically identical
|
|
52
|
+
output.
|
|
53
|
+
|
|
54
|
+
- Always read `result.values`, `result.matches` and `result.info`. There are no
|
|
55
|
+
`output: 'lex'/'matches'/'full'` modes.
|
|
56
|
+
- No data or no matching evidence yields explicit status and null category
|
|
57
|
+
values. Intercepts alone cannot turn unknown input into a prediction.
|
|
58
|
+
- Affect, temporal orientation and PERMA now default to relative frequency,
|
|
59
|
+
rather than the old binary sums. Big Five and optimism retain binary sums as
|
|
60
|
+
association measures. Request an encoding explicitly for comparisons.
|
|
61
|
+
- Canonical WWBP CSV weights and intercepts replace rounded or damaged legacy
|
|
62
|
+
copies where available. Spanish accents are restored, and `permaEs` selects
|
|
63
|
+
Spanish directly. The old wellbeing module mistakenly inspected `output`
|
|
64
|
+
rather than `lang` when selecting Spanish.
|
|
65
|
+
- Matching uses one token stream for all ngrams. All vocabulary ngram sizes are
|
|
66
|
+
enabled by default; the old age/gender wrappers ignored phrase terms. To
|
|
67
|
+
compare unigrams, set `ngrams: [1]` and supply controlled tokens.
|
|
68
|
+
- English PERMA’s structural feature weights are retained and missing inputs are
|
|
69
|
+
reported. Lexical-only results should not be described as the complete
|
|
70
|
+
original trained model.
|
|
71
|
+
- Big Five retains historical association weights except an invalid empty-string
|
|
72
|
+
O term. Scores have no calibrated personality scale.
|
|
73
|
+
- Optimism is an experimental composition of future terms and affect weights. It
|
|
74
|
+
is not a separately trained optimism predictor.
|
|
75
|
+
- Gender returns the historical classifier margin, without converting it into an
|
|
76
|
+
assertion about identity. The historical sign convention was negative/positive
|
|
77
|
+
for the source’s male/female labels.
|
|
78
|
+
- Temporal scores are not probabilities. Any ranking must handle nulls and ties
|
|
79
|
+
explicitly; there is no automatic “Unknown” or orientation label.
|
|
80
|
+
- Results are synchronous with no async dependency or mutable global state.
|
|
81
|
+
Caller options are never mutated.
|
|
82
|
+
|
|
83
|
+
## Options
|
|
84
|
+
|
|
85
|
+
| Previous option | New equivalent |
|
|
86
|
+
| --------------------- | ---------------------------------------------------------------------------- |
|
|
87
|
+
| `encoding: 'freq'` | `encoding: 'frequency'` |
|
|
88
|
+
| `encoding: 'binary'` | Same unique-term sum semantics |
|
|
89
|
+
| `encoding: 'percent'` | Candidate-feature coverage fraction; no intercept |
|
|
90
|
+
| `nGrams: [2, 3]` | `ngrams: [1, 2, 3]` (unigrams are explicit) |
|
|
91
|
+
| `nGrams: [0]` | `ngrams: [1]` to keep unigrams only |
|
|
92
|
+
| `noInt: true` | `includeIntercept: false` |
|
|
93
|
+
| `min`, `max` | `minWeight`, `maxWeight` |
|
|
94
|
+
| `places` | `decimals` (final scores only) |
|
|
95
|
+
| `logs`, `suppressLog` | Removed; the library does not log |
|
|
96
|
+
| `sortBy` | Sort `result.matches[category]` in your application |
|
|
97
|
+
| `wcGrams` | Removed; weighted frequency uses original tokens |
|
|
98
|
+
| `locale` | Normalize spelling in your preprocessing if needed; no automatic translation |
|
|
99
|
+
| `lang: 'spanish'` | Model ID `permaEs` |
|
|
100
|
+
|
|
101
|
+
## Utility packages
|
|
102
|
+
|
|
103
|
+
Replace `lexFrequencyPipeline(lex, intercept)` with a lexicon configured for
|
|
104
|
+
`[1]` and `frequency`, then read `score(tokens, lexicon).values.CATEGORY`. The
|
|
105
|
+
old `lexBinaryPipeline` divided unique weights by the number of unique tokens;
|
|
106
|
+
the new `binary` encoding sums unique weights without that division. Use
|
|
107
|
+
explicit arithmetic if you need the old normalized-unique formula.
|
|
108
|
+
|
|
109
|
+
The previous `Term`/`Category`/`Lexicon` object graph is replaced by immutable
|
|
110
|
+
category-to-weight maps. Build a new definition to change vocabulary rather than
|
|
111
|
+
maintaining bidirectional mutable associations. The new engine includes category
|
|
112
|
+
intercepts; the old `weighted-lexica.analyse()` implementation created an empty
|
|
113
|
+
intercept map and never filled it.
|
|
114
|
+
|
|
115
|
+
## Maintainer release sequence
|
|
116
|
+
|
|
117
|
+
1. Run tests, source verification and package inspection; review the tarball.
|
|
118
|
+
2. Publish `wwbnlp` from the `phughesmcr` npm account, after authentication and
|
|
119
|
+
any npm-required account confirmation.
|
|
120
|
+
3. Verify the installed registry package and its bundled installation
|
|
121
|
+
instructions.
|
|
122
|
+
4. If retiring the old npm names, add a deprecation message linking to this
|
|
123
|
+
guide. Deprecation is a separate registry change; it has not been applied by
|
|
124
|
+
archiving repositories. Keep old versions available for reproducibility.
|