wwbnlp 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,57 @@
1
+ /** Shared weighted-lexicon engine. No I/O, global state, or runtime dependencies. */
2
+ export type Encoding = "frequency" | "binary" | "percent";
3
+ export type Weights = Readonly<Record<string, Readonly<Record<string, number>>>>;
4
+ export interface LexiconDefinition {
5
+ readonly id: string;
6
+ readonly language?: string;
7
+ readonly categories: Weights;
8
+ readonly intercepts?: Readonly<Record<string, number>>;
9
+ readonly features?: Weights;
10
+ readonly ngrams?: readonly number[];
11
+ readonly encoding?: Encoding;
12
+ }
13
+ export interface Lexicon extends LexiconDefinition {
14
+ readonly intercepts: Readonly<Record<string, number>>;
15
+ readonly features: Weights;
16
+ readonly ngrams: readonly number[];
17
+ readonly encoding: Encoding;
18
+ }
19
+ export interface Options {
20
+ /** Frequency divides term occurrences by original token count, including unmatched tokens. */
21
+ readonly encoding?: Encoding;
22
+ /** Sizes include unigrams explicitly. An empty array disables lexical matching. */
23
+ readonly ngrams?: readonly number[];
24
+ readonly includeIntercept?: boolean;
25
+ readonly minWeight?: number;
26
+ readonly maxWeight?: number;
27
+ /** Round final scores only. Omit to retain full floating-point precision. */
28
+ readonly decimals?: number;
29
+ /** Measured nonlexical covariates; never infer these from a single text. */
30
+ readonly features?: Readonly<Record<string, number>>;
31
+ }
32
+ export interface Match {
33
+ readonly term: string;
34
+ readonly count: number;
35
+ readonly weight: number;
36
+ readonly contribution: number;
37
+ }
38
+ export interface Analysis {
39
+ readonly model: string;
40
+ readonly status: "ok" | "empty" | "no-matches";
41
+ readonly values: Readonly<Record<string, number | null>>;
42
+ readonly matches: Readonly<Record<string, readonly Match[]>>;
43
+ readonly featureContributions: Readonly<Record<string, Readonly<Record<string, number>>>>;
44
+ readonly info: {
45
+ readonly tokenCount: number;
46
+ readonly featureCount: number;
47
+ readonly matchedFeatureCount: number;
48
+ readonly uniqueMatchedTerms: number;
49
+ };
50
+ readonly warnings: readonly string[];
51
+ }
52
+ /** Copy and validate a custom lexicon once; returned data cannot be mutated. */
53
+ export declare function createLexicon(definition: LexiconDefinition): Lexicon;
54
+ /** Unicode-aware convenience tokenizer; supply exact study tokens when reproducing a pipeline. */
55
+ export declare function tokenize(text: string): string[];
56
+ /** Score a validated lexicon. Input token arrays are used exactly as supplied. */
57
+ export declare function score(input: string | readonly string[], lexicon: Lexicon, options?: Options): Analysis;
@@ -0,0 +1,223 @@
1
+ const own = (object, key) => Object.prototype.hasOwnProperty.call(object, key);
2
+ function finite(value, label) {
3
+ if (typeof value !== "number" || !Number.isFinite(value)) {
4
+ throw new TypeError(`${label} must be finite`);
5
+ }
6
+ }
7
+ function checkNgrams(value) {
8
+ if (!Array.isArray(value) ||
9
+ value.some((n) => !Number.isSafeInteger(n) || n < 1 || n > 8)) {
10
+ throw new RangeError("ngrams must contain integer sizes from 1 to 8");
11
+ }
12
+ }
13
+ function checkEncoding(value) {
14
+ if (!["frequency", "binary", "percent"].includes(value)) {
15
+ throw new RangeError("Unknown encoding");
16
+ }
17
+ }
18
+ /** Copy and validate a custom lexicon once; returned data cannot be mutated. */
19
+ export function createLexicon(definition) {
20
+ if (!definition || typeof definition.id !== "string" || !definition.id ||
21
+ !definition.categories || typeof definition.categories !== "object" ||
22
+ Array.isArray(definition.categories)) {
23
+ throw new TypeError("A lexicon needs an id and category weight maps");
24
+ }
25
+ const categories = Object
26
+ .create(null);
27
+ const intercepts = Object.create(null);
28
+ const features = Object
29
+ .create(null);
30
+ for (const [category, terms] of Object.entries(definition.categories)) {
31
+ if (!category || !terms || typeof terms !== "object" || Array.isArray(terms))
32
+ throw new TypeError("Invalid category");
33
+ for (const [term, weight] of Object.entries(terms)) {
34
+ if (!term)
35
+ throw new TypeError("Lexicon terms must not be empty");
36
+ finite(weight, `Weight for ${term}`);
37
+ }
38
+ categories[category] = Object.freeze({ ...terms });
39
+ const intercept = definition.intercepts && own(definition.intercepts, category)
40
+ ? definition.intercepts[category]
41
+ : 0;
42
+ finite(intercept, `Intercept for ${category}`);
43
+ intercepts[category] = intercept;
44
+ const covariates = definition.features && own(definition.features, category)
45
+ ? definition.features[category]
46
+ : {};
47
+ if (!covariates || typeof covariates !== "object" || Array.isArray(covariates))
48
+ throw new TypeError("Feature weights must be a map");
49
+ for (const [feature, weight] of Object.entries(covariates)) {
50
+ finite(weight, `Feature weight for ${feature}`);
51
+ }
52
+ features[category] = Object.freeze({ ...covariates });
53
+ }
54
+ if (!Object.keys(categories).length) {
55
+ throw new TypeError("A lexicon needs at least one category");
56
+ }
57
+ for (const key of Object.keys(definition.intercepts ?? {})) {
58
+ if (!own(categories, key)) {
59
+ throw new RangeError(`Unknown intercept category: ${key}`);
60
+ }
61
+ }
62
+ for (const key of Object.keys(definition.features ?? {})) {
63
+ if (!own(categories, key)) {
64
+ throw new RangeError(`Unknown feature category: ${key}`);
65
+ }
66
+ }
67
+ const ngrams = definition.ngrams ?? [1, 2, 3];
68
+ const encoding = definition.encoding ?? "frequency";
69
+ checkNgrams(ngrams);
70
+ checkEncoding(encoding);
71
+ return Object.freeze({
72
+ id: definition.id,
73
+ ...(definition.language === undefined
74
+ ? {}
75
+ : { language: definition.language }),
76
+ categories: Object.freeze(categories),
77
+ intercepts: Object.freeze(intercepts),
78
+ features: Object.freeze(features),
79
+ ngrams: Object.freeze([...new Set(ngrams)]),
80
+ encoding,
81
+ });
82
+ }
83
+ /** Unicode-aware convenience tokenizer; supply exact study tokens when reproducing a pipeline. */
84
+ export function tokenize(text) {
85
+ if (typeof text !== "string")
86
+ throw new TypeError("text must be a string");
87
+ return text.normalize("NFC").toLowerCase().replace(/[’‘]/gu, "'").match(/https?:\/\/[^\s]+|<3|[:;=8][\-o*']?[\)\]\(\[dp/\\]|[\)\]\(\[d][:;=8]|[#@][\p{L}\p{M}\p{N}_]+|\p{N}+(?:[.,:]\p{N}+)+|[\p{L}\p{M}\p{N}_]+(?:['-][\p{L}\p{M}\p{N}_]+)*|\.{3,}|\p{Extended_Pictographic}(?:\uFE0F|\p{M})*|[^\s]/gu) ?? [];
88
+ }
89
+ /** Score a validated lexicon. Input token arrays are used exactly as supplied. */
90
+ export function score(input, lexicon, options = {}) {
91
+ if (!options || typeof options !== "object" || Array.isArray(options)) {
92
+ throw new TypeError("options must be an object");
93
+ }
94
+ const allowed = [
95
+ "encoding",
96
+ "ngrams",
97
+ "includeIntercept",
98
+ "minWeight",
99
+ "maxWeight",
100
+ "decimals",
101
+ "features",
102
+ ];
103
+ for (const key of Object.keys(options)) {
104
+ if (!allowed.includes(key)) {
105
+ throw new RangeError(`Unknown option: ${key}`);
106
+ }
107
+ }
108
+ const encoding = options.encoding ?? lexicon.encoding;
109
+ const ngrams = options.ngrams ?? lexicon.ngrams;
110
+ const min = options.minWeight ?? -Infinity, max = options.maxWeight ?? Infinity;
111
+ checkEncoding(encoding);
112
+ checkNgrams(ngrams);
113
+ if (typeof min !== "number" || typeof max !== "number" || Number.isNaN(min) ||
114
+ Number.isNaN(max) || min > max)
115
+ throw new RangeError("Invalid weight bounds");
116
+ if (options.includeIntercept !== undefined &&
117
+ typeof options.includeIntercept !== "boolean")
118
+ throw new TypeError("includeIntercept must be boolean");
119
+ if (options.decimals !== undefined &&
120
+ (!Number.isInteger(options.decimals) || options.decimals < 0 ||
121
+ options.decimals > 15))
122
+ throw new RangeError("decimals must be an integer from 0 to 15");
123
+ const tokens = typeof input === "string" ? tokenize(input) : input;
124
+ if (!Array.isArray(tokens) || tokens.some((t) => typeof t !== "string" || !t)) {
125
+ throw new TypeError("input must be text or an array of nonempty token strings");
126
+ }
127
+ const suppliedFeatures = options.features ?? {};
128
+ if (typeof suppliedFeatures !== "object" || Array.isArray(suppliedFeatures)) {
129
+ throw new TypeError("features must be an object");
130
+ }
131
+ const knownFeatures = new Set(Object.values(lexicon.features).flatMap((f) => Object.keys(f)));
132
+ for (const [key, value] of Object.entries(suppliedFeatures)) {
133
+ if (!knownFeatures.has(key)) {
134
+ throw new RangeError(`Unknown model feature: ${key}`);
135
+ }
136
+ finite(value, `Feature ${key}`);
137
+ }
138
+ const counts = new Map();
139
+ let featureCount = 0;
140
+ for (const n of new Set(ngrams)) {
141
+ for (let i = 0; i <= tokens.length - n; i++) {
142
+ const term = tokens.slice(i, i + n).join(" ");
143
+ counts.set(term, (counts.get(term) ?? 0) + 1);
144
+ featureCount++;
145
+ }
146
+ }
147
+ const values = Object.create(null);
148
+ const matches = Object.create(null);
149
+ const featureContributions = Object
150
+ .create(null);
151
+ const matchedTerms = new Set();
152
+ let hasEvidence = false;
153
+ const warnings = encoding === "percent"
154
+ ? []
155
+ : [...knownFeatures].filter((f) => !own(suppliedFeatures, f)).map((f) => `Structural feature ${f} was not supplied; its contribution is omitted.`);
156
+ for (const [category, weights] of Object.entries(lexicon.categories)) {
157
+ let total = encoding === "percent" || options.includeIntercept === false
158
+ ? 0
159
+ : lexicon.intercepts[category];
160
+ const categoryMatches = [];
161
+ const covariates = Object.create(null);
162
+ let matchedCount = 0;
163
+ for (const [term, count] of counts) {
164
+ if (!own(weights, term))
165
+ continue;
166
+ const weight = weights[term];
167
+ if (weight < min || weight > max)
168
+ continue;
169
+ const contribution = encoding === "binary"
170
+ ? weight
171
+ : encoding === "percent"
172
+ ? count / featureCount
173
+ : weight * (count / tokens.length);
174
+ total += contribution;
175
+ matchedCount += count;
176
+ categoryMatches.push({ term, count, weight, contribution });
177
+ matchedTerms.add(term);
178
+ }
179
+ if (encoding === "percent") {
180
+ total = featureCount ? matchedCount / featureCount : 0;
181
+ }
182
+ if (encoding !== "percent" && tokens.length) {
183
+ for (const [feature, weight] of Object.entries(lexicon.features[category])) {
184
+ if (own(suppliedFeatures, feature)) {
185
+ const contribution = suppliedFeatures[feature] * weight;
186
+ covariates[feature] = contribution;
187
+ total += contribution;
188
+ }
189
+ }
190
+ }
191
+ const evidence = tokens.length > 0 &&
192
+ (categoryMatches.length > 0 || Object.keys(covariates).length > 0);
193
+ hasEvidence ||= evidence;
194
+ if (!Number.isFinite(total)) {
195
+ throw new RangeError(`Score overflow for ${category}`);
196
+ }
197
+ values[category] = evidence
198
+ ? options.decimals === undefined
199
+ ? total
200
+ : Number(total.toFixed(options.decimals))
201
+ : null;
202
+ matches[category] = categoryMatches.sort((a, b) => b.count - a.count || a.term.localeCompare(b.term, "en"));
203
+ featureContributions[category] = covariates;
204
+ }
205
+ let matchedFeatureCount = 0;
206
+ for (const term of matchedTerms)
207
+ matchedFeatureCount += counts.get(term);
208
+ return {
209
+ model: lexicon.id,
210
+ status: !tokens.length ? "empty" : hasEvidence ? "ok" : "no-matches",
211
+ values,
212
+ matches,
213
+ featureContributions,
214
+ info: {
215
+ tokenCount: tokens.length,
216
+ featureCount,
217
+ matchedFeatureCount,
218
+ uniqueMatchedTerms: matchedTerms.size,
219
+ },
220
+ warnings,
221
+ };
222
+ }
223
+ //# sourceMappingURL=core.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"core.js","sourceRoot":"","sources":["../../src/core.ts"],"names":[],"mappings":"AAuDA,MAAM,GAAG,GAAG,CAAC,MAAc,EAAE,GAAW,EAAE,EAAE,CAC1C,MAAM,CAAC,SAAS,CAAC,cAAc,CAAC,IAAI,CAAC,MAAM,EAAE,GAAG,CAAC,CAAC;AACpD,SAAS,MAAM,CAAC,KAAc,EAAE,KAAa;IAC3C,IAAI,OAAO,KAAK,KAAK,QAAQ,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,KAAK,CAAC,EAAE,CAAC;QACzD,MAAM,IAAI,SAAS,CAAC,GAAG,KAAK,iBAAiB,CAAC,CAAC;IACjD,CAAC;AACH,CAAC;AACD,SAAS,WAAW,CAAC,KAAwB;IAC3C,IACE,CAAC,KAAK,CAAC,OAAO,CAAC,KAAK,CAAC;QACrB,KAAK,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,MAAM,CAAC,aAAa,CAAC,CAAC,CAAC,IAAI,CAAC,GAAG,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,EAC7D,CAAC;QACD,MAAM,IAAI,UAAU,CAAC,+CAA+C,CAAC,CAAC;IACxE,CAAC;AACH,CAAC;AACD,SAAS,aAAa,CAAC,KAAc;IACnC,IAAI,CAAC,CAAC,WAAW,EAAE,QAAQ,EAAE,SAAS,CAAC,CAAC,QAAQ,CAAC,KAAe,CAAC,EAAE,CAAC;QAClE,MAAM,IAAI,UAAU,CAAC,kBAAkB,CAAC,CAAC;IAC3C,CAAC;AACH,CAAC;AACD,gFAAgF;AAChF,MAAM,UAAU,aAAa,CAAC,UAA6B;IACzD,IACE,CAAC,UAAU,IAAI,OAAO,UAAU,CAAC,EAAE,KAAK,QAAQ,IAAI,CAAC,UAAU,CAAC,EAAE;QAClE,CAAC,UAAU,CAAC,UAAU,IAAI,OAAO,UAAU,CAAC,UAAU,KAAK,QAAQ;QACnE,KAAK,CAAC,OAAO,CAAC,UAAU,CAAC,UAAU,CAAC,EACpC,CAAC;QACD,MAAM,IAAI,SAAS,CAAC,gDAAgD,CAAC,CAAC;IACxE,CAAC;IACD,MAAM,UAAU,GAAqD,MAAM;SACxE,MAAM,CAAC,IAAI,CAAC,CAAC;IAChB,MAAM,UAAU,GAA2B,MAAM,CAAC,MAAM,CAAC,IAAI,CAAC,CAAC;IAC/D,MAAM,QAAQ,GAAqD,MAAM;SACtE,MAAM,CAAC,IAAI,CAAC,CAAC;IAChB,KAAK,MAAM,CAAC,QAAQ,EAAE,KAAK,CAAC,IAAI,MAAM,CAAC,OAAO,CAAC,UAAU,CAAC,UAAU,CAAC,EAAE,CAAC;QACtE,IACE,CAAC,QAAQ,IAAI,CAAC,KAAK,IAAI,OAAO,KAAK,KAAK,QAAQ,IAAI,KAAK,CAAC,OAAO,CAAC,KAAK,CAAC;YACxE,MAAM,IAAI,SAAS,CAAC,kBAAkB,CAAC,CAAC;QAC1C,KAAK,MAAM,CAAC,IAAI,EAAE,MAAM,CAAC,IAAI,MAAM,CAAC,OAAO,CAAC,KAAK,CAAC,EAAE,CAAC;YACnD,IAAI,CAAC,IAAI;gBAAE,MAAM,IAAI,SAAS,CAAC,iCAAiC,CAAC,CAAC;YAClE,MAAM,CAAC,MAAM,EAAE,cAAc,IAAI,EAAE,CAAC,CAAC;QACvC,CAAC;QACD,UAAU,CAAC,QAAQ,CAAC,GAAG,MAAM,CAAC,MAAM,CAAC,EAAE,GAAG,KAAK,EAAE,CAAC,CAAC;QACnD,MAAM,SAAS,GACb,UAAU,CAAC,UAAU,IAAI,GAAG,CAAC,UAAU,CAAC,UAAU,EAAE,QAAQ,CAAC;YAC3D,CAAC,CAAC,UAAU,CAAC,UAAU,CAAC,QAAQ,CAAC;YACjC,CAAC,CAAC,CAAC,CAAC;QACR,MAAM,CAAC,SAAS,EAAE,iBAAiB,QAAQ,EAAE,CAAC,CAAC;QAC/C,UAAU,CAAC,QAAQ,CAAC,GAAG,SAAS,CAAC;QACjC,MAAM,UAAU,GAAG,UAAU,CAAC,QAAQ,IAAI,GAAG,CAAC,UAAU,CAAC,QAAQ,EAAE,QAAQ,CAAC;YAC1E,CAAC,CAAC,UAAU,CAAC,QAAQ,CAAC,QAAQ,CAAE;YAChC,CAAC,CAAC,EAAE,CAAC;QACP,IACE,CAAC,UAAU,IAAI,OAAO,UAAU,KAAK,QAAQ,IAAI,KAAK,CAAC,OAAO,CAAC,UAAU,CAAC;YAC1E,MAAM,IAAI,SAAS,CAAC,+BAA+B,CAAC,CAAC;QACvD,KAAK,MAAM,CAAC,OAAO,EAAE,MAAM,CAAC,IAAI,MAAM,CAAC,OAAO,CAAC,UAAU,CAAC,EAAE,CAAC;YAC3D,MAAM,CAAC,MAAM,EAAE,sBAAsB,OAAO,EAAE,CAAC,CAAC;QAClD,CAAC;QACD,QAAQ,CAAC,QAAQ,CAAC,GAAG,MAAM,CAAC,MAAM,CAAC,EAAE,GAAG,UAAU,EAAE,CAAC,CAAC;IACxD,CAAC;IACD,IAAI,CAAC,MAAM,CAAC,IAAI,CAAC,UAAU,CAAC,CAAC,MAAM,EAAE,CAAC;QACpC,MAAM,IAAI,SAAS,CAAC,uCAAuC,CAAC,CAAC;IAC/D,CAAC;IACD,KAAK,MAAM,GAAG,IAAI,MAAM,CAAC,IAAI,CAAC,UAAU,CAAC,UAAU,IAAI,EAAE,CAAC,EAAE,CAAC;QAC3D,IAAI,CAAC,GAAG,CAAC,UAAU,EAAE,GAAG,CAAC,EAAE,CAAC;YAC1B,MAAM,IAAI,UAAU,CAAC,+BAA+B,GAAG,EAAE,CAAC,CAAC;QAC7D,CAAC;IACH,CAAC;IACD,KAAK,MAAM,GAAG,IAAI,MAAM,CAAC,IAAI,CAAC,UAAU,CAAC,QAAQ,IAAI,EAAE,CAAC,EAAE,CAAC;QACzD,IAAI,CAAC,GAAG,CAAC,UAAU,EAAE,GAAG,CAAC,EAAE,CAAC;YAC1B,MAAM,IAAI,UAAU,CAAC,6BAA6B,GAAG,EAAE,CAAC,CAAC;QAC3D,CAAC;IACH,CAAC;IACD,MAAM,MAAM,GAAG,UAAU,CAAC,MAAM,IAAI,CAAC,CAAC,EAAE,CAAC,EAAE,CAAC,CAAC,CAAC;IAC9C,MAAM,QAAQ,GAAG,UAAU,CAAC,QAAQ,IAAI,WAAW,CAAC;IACpD,WAAW,CAAC,MAAM,CAAC,CAAC;IACpB,aAAa,CAAC,QAAQ,CAAC,CAAC;IACxB,OAAO,MAAM,CAAC,MAAM,CAAC;QACnB,EAAE,EAAE,UAAU,CAAC,EAAE;QACjB,GAAG,CAAC,UAAU,CAAC,QAAQ,KAAK,SAAS;YACnC,CAAC,CAAC,EAAE;YACJ,CAAC,CAAC,EAAE,QAAQ,EAAE,UAAU,CAAC,QAAQ,EAAE,CAAC;QACtC,UAAU,EAAE,MAAM,CAAC,MAAM,CAAC,UAAU,CAAC;QACrC,UAAU,EAAE,MAAM,CAAC,MAAM,CAAC,UAAU,CAAC;QACrC,QAAQ,EAAE,MAAM,CAAC,MAAM,CAAC,QAAQ,CAAC;QACjC,MAAM,EAAE,MAAM,CAAC,MAAM,CAAC,CAAC,GAAG,IAAI,GAAG,CAAC,MAAM,CAAC,CAAC,CAAC;QAC3C,QAAQ;KACT,CAAC,CAAC;AACL,CAAC;AACD,kGAAkG;AAClG,MAAM,UAAU,QAAQ,CAAC,IAAY;IACnC,IAAI,OAAO,IAAI,KAAK,QAAQ;QAAE,MAAM,IAAI,SAAS,CAAC,uBAAuB,CAAC,CAAC;IAC3E,OAAO,IAAI,CAAC,SAAS,CAAC,KAAK,CAAC,CAAC,WAAW,EAAE,CAAC,OAAO,CAAC,QAAQ,EAAE,GAAG,CAAC,CAAC,KAAK,CACrE,+NAA+N,CAChO,IAAI,EAAE,CAAC;AACV,CAAC;AACD,kFAAkF;AAClF,MAAM,UAAU,KAAK,CACnB,KAAiC,EACjC,OAAgB,EAChB,OAAO,GAAY,EAAE;IAErB,IAAI,CAAC,OAAO,IAAI,OAAO,OAAO,KAAK,QAAQ,IAAI,KAAK,CAAC,OAAO,CAAC,OAAO,CAAC,EAAE,CAAC;QACtE,MAAM,IAAI,SAAS,CAAC,2BAA2B,CAAC,CAAC;IACnD,CAAC;IACD,MAAM,OAAO,GAAG;QACd,UAAU;QACV,QAAQ;QACR,kBAAkB;QAClB,WAAW;QACX,WAAW;QACX,UAAU;QACV,UAAU;KACX,CAAC;IACF,KAAK,MAAM,GAAG,IAAI,MAAM,CAAC,IAAI,CAAC,OAAO,CAAC,EAAE,CAAC;QACvC,IAAI,CAAC,OAAO,CAAC,QAAQ,CAAC,GAAG,CAAC,EAAE,CAAC;YAC3B,MAAM,IAAI,UAAU,CAAC,mBAAmB,GAAG,EAAE,CAAC,CAAC;QACjD,CAAC;IACH,CAAC;IACD,MAAM,QAAQ,GAAG,OAAO,CAAC,QAAQ,IAAI,OAAO,CAAC,QAAQ,CAAC;IACtD,MAAM,MAAM,GAAG,OAAO,CAAC,MAAM,IAAI,OAAO,CAAC,MAAM,CAAC;IAChD,MAAM,GAAG,GAAG,OAAO,CAAC,SAAS,IAAI,CAAC,QAAQ,EACxC,GAAG,GAAG,OAAO,CAAC,SAAS,IAAI,QAAQ,CAAC;IACtC,aAAa,CAAC,QAAQ,CAAC,CAAC;IACxB,WAAW,CAAC,MAAM,CAAC,CAAC;IACpB,IACE,OAAO,GAAG,KAAK,QAAQ,IAAI,OAAO,GAAG,KAAK,QAAQ,IAAI,MAAM,CAAC,KAAK,CAAC,GAAG,CAAC;QACvE,MAAM,CAAC,KAAK,CAAC,GAAG,CAAC,IAAI,GAAG,GAAG,GAAG;QAC9B,MAAM,IAAI,UAAU,CAAC,uBAAuB,CAAC,CAAC;IAChD,IACE,OAAO,CAAC,gBAAgB,KAAK,SAAS;QACtC,OAAO,OAAO,CAAC,gBAAgB,KAAK,SAAS;QAC7C,MAAM,IAAI,SAAS,CAAC,kCAAkC,CAAC,CAAC;IAC1D,IACE,OAAO,CAAC,QAAQ,KAAK,SAAS;QAC9B,CAAC,CAAC,MAAM,CAAC,SAAS,CAAC,OAAO,CAAC,QAAQ,CAAC,IAAI,OAAO,CAAC,QAAQ,GAAG,CAAC;YAC1D,OAAO,CAAC,QAAQ,GAAG,EAAE,CAAC;QACxB,MAAM,IAAI,UAAU,CAAC,0CAA0C,CAAC,CAAC;IACnE,MAAM,MAAM,GAAG,OAAO,KAAK,KAAK,QAAQ,CAAC,CAAC,CAAC,QAAQ,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,KAAK,CAAC;IACnE,IACE,CAAC,KAAK,CAAC,OAAO,CAAC,MAAM,CAAC,IAAI,MAAM,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,OAAO,CAAC,KAAK,QAAQ,IAAI,CAAC,CAAC,CAAC,EACzE,CAAC;QACD,MAAM,IAAI,SAAS,CACjB,0DAA0D,CAC3D,CAAC;IACJ,CAAC;IACD,MAAM,gBAAgB,GAAG,OAAO,CAAC,QAAQ,IAAI,EAAE,CAAC;IAChD,IAAI,OAAO,gBAAgB,KAAK,QAAQ,IAAI,KAAK,CAAC,OAAO,CAAC,gBAAgB,CAAC,EAAE,CAAC;QAC5E,MAAM,IAAI,SAAS,CAAC,4BAA4B,CAAC,CAAC;IACpD,CAAC;IACD,MAAM,aAAa,GAAG,IAAI,GAAG,CAC3B,MAAM,CAAC,MAAM,CAAC,OAAO,CAAC,QAAQ,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,MAAM,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC,CAC/D,CAAC;IACF,KAAK,MAAM,CAAC,GAAG,EAAE,KAAK,CAAC,IAAI,MAAM,CAAC,OAAO,CAAC,gBAAgB,CAAC,EAAE,CAAC;QAC5D,IAAI,CAAC,aAAa,CAAC,GAAG,CAAC,GAAG,CAAC,EAAE,CAAC;YAC5B,MAAM,IAAI,UAAU,CAAC,0BAA0B,GAAG,EAAE,CAAC,CAAC;QACxD,CAAC;QACD,MAAM,CAAC,KAAK,EAAE,WAAW,GAAG,EAAE,CAAC,CAAC;IAClC,CAAC;IACD,MAAM,MAAM,GAAG,IAAI,GAAG,EAAkB,CAAC;IACzC,IAAI,YAAY,GAAG,CAAC,CAAC;IACrB,KAAK,MAAM,CAAC,IAAI,IAAI,GAAG,CAAC,MAAM,CAAC,EAAE,CAAC;QAChC,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,IAAI,MAAM,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC;YAC5C,MAAM,IAAI,GAAG,MAAM,CAAC,KAAK,CAAC,CAAC,EAAE,CAAC,GAAG,CAAC,CAAC,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC;YAC9C,MAAM,CAAC,GAAG,CAAC,IAAI,EAAE,CAAC,MAAM,CAAC,GAAG,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC;YAC9C,YAAY,EAAE,CAAC;QACjB,CAAC;IACH,CAAC;IACD,MAAM,MAAM,GAAkC,MAAM,CAAC,MAAM,CAAC,IAAI,CAAC,CAAC;IAClE,MAAM,OAAO,GAA4B,MAAM,CAAC,MAAM,CAAC,IAAI,CAAC,CAAC;IAC7D,MAAM,oBAAoB,GAA2C,MAAM;SACxE,MAAM,CAAC,IAAI,CAAC,CAAC;IAChB,MAAM,YAAY,GAAG,IAAI,GAAG,EAAU,CAAC;IACvC,IAAI,WAAW,GAAG,KAAK,CAAC;IACxB,MAAM,QAAQ,GAAG,QAAQ,KAAK,SAAS;QACrC,CAAC,CAAC,EAAE;QACJ,CAAC,CAAC,CAAC,GAAG,aAAa,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,GAAG,CAAC,gBAAgB,EAAE,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CACtE,sBAAsB,CAAC,iDAAiD,CACzE,CAAC;IACJ,KAAK,MAAM,CAAC,QAAQ,EAAE,OAAO,CAAC,IAAI,MAAM,CAAC,OAAO,CAAC,OAAO,CAAC,UAAU,CAAC,EAAE,CAAC;QACrE,IAAI,KAAK,GAAG,QAAQ,KAAK,SAAS,IAAI,OAAO,CAAC,gBAAgB,KAAK,KAAK;YACtE,CAAC,CAAC,CAAC;YACH,CAAC,CAAC,OAAO,CAAC,UAAU,CAAC,QAAQ,CAAE,CAAC;QAClC,MAAM,eAAe,GAAY,EAAE,CAAC;QACpC,MAAM,UAAU,GAA2B,MAAM,CAAC,MAAM,CAAC,IAAI,CAAC,CAAC;QAC/D,IAAI,YAAY,GAAG,CAAC,CAAC;QACrB,KAAK,MAAM,CAAC,IAAI,EAAE,KAAK,CAAC,IAAI,MAAM,EAAE,CAAC;YACnC,IAAI,CAAC,GAAG,CAAC,OAAO,EAAE,IAAI,CAAC;gBAAE,SAAS;YAClC,MAAM,MAAM,GAAG,OAAO,CAAC,IAAI,CAAE,CAAC;YAC9B,IAAI,MAAM,GAAG,GAAG,IAAI,MAAM,GAAG,GAAG;gBAAE,SAAS;YAC3C,MAAM,YAAY,GAAG,QAAQ,KAAK,QAAQ;gBACxC,CAAC,CAAC,MAAM;gBACR,CAAC,CAAC,QAAQ,KAAK,SAAS;oBACxB,CAAC,CAAC,KAAK,GAAG,YAAY;oBACtB,CAAC,CAAC,MAAM,GAAG,CAAC,KAAK,GAAG,MAAM,CAAC,MAAM,CAAC,CAAC;YACrC,KAAK,IAAI,YAAY,CAAC;YACtB,YAAY,IAAI,KAAK,CAAC;YACtB,eAAe,CAAC,IAAI,CAAC,EAAE,IAAI,EAAE,KAAK,EAAE,MAAM,EAAE,YAAY,EAAE,CAAC,CAAC;YAC5D,YAAY,CAAC,GAAG,CAAC,IAAI,CAAC,CAAC;QACzB,CAAC;QACD,IAAI,QAAQ,KAAK,SAAS,EAAE,CAAC;YAC3B,KAAK,GAAG,YAAY,CAAC,CAAC,CAAC,YAAY,GAAG,YAAY,CAAC,CAAC,CAAC,CAAC,CAAC;QACzD,CAAC;QACD,IAAI,QAAQ,KAAK,SAAS,IAAI,MAAM,CAAC,MAAM,EAAE,CAAC;YAC5C,KACE,MAAM,CAAC,OAAO,EAAE,MAAM,CAAC,IAAI,MAAM,CAAC,OAAO,CAAC,OAAO,CAAC,QAAQ,CAAC,QAAQ,CAAE,CAAC,EACtE,CAAC;gBACD,IAAI,GAAG,CAAC,gBAAgB,EAAE,OAAO,CAAC,EAAE,CAAC;oBACnC,MAAM,YAAY,GAAG,gBAAgB,CAAC,OAAO,CAAE,GAAG,MAAM,CAAC;oBACzD,UAAU,CAAC,OAAO,CAAC,GAAG,YAAY,CAAC;oBACnC,KAAK,IAAI,YAAY,CAAC;gBACxB,CAAC;YACH,CAAC;QACH,CAAC;QACD,MAAM,QAAQ,GAAG,MAAM,CAAC,MAAM,GAAG,CAAC;YAChC,CAAC,eAAe,CAAC,MAAM,GAAG,CAAC,IAAI,MAAM,CAAC,IAAI,CAAC,UAAU,CAAC,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC;QACrE,WAAW,KAAK,QAAQ,CAAC;QACzB,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,KAAK,CAAC,EAAE,CAAC;YAC5B,MAAM,IAAI,UAAU,CAAC,sBAAsB,QAAQ,EAAE,CAAC,CAAC;QACzD,CAAC;QACD,MAAM,CAAC,QAAQ,CAAC,GAAG,QAAQ;YACzB,CAAC,CAAC,OAAO,CAAC,QAAQ,KAAK,SAAS;gBAC9B,CAAC,CAAC,KAAK;gBACP,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC,OAAO,CAAC,OAAO,CAAC,QAAQ,CAAC,CAAC;YAC3C,CAAC,CAAC,IAAI,CAAC;QACT,OAAO,CAAC,QAAQ,CAAC,GAAG,eAAe,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAChD,CAAC,CAAC,KAAK,GAAG,CAAC,CAAC,KAAK,IAAI,CAAC,CAAC,IAAI,CAAC,aAAa,CAAC,CAAC,CAAC,IAAI,EAAE,IAAI,CAAC,CACxD,CAAC;QACF,oBAAoB,CAAC,QAAQ,CAAC,GAAG,UAAU,CAAC;IAC9C,CAAC;IACD,IAAI,mBAAmB,GAAG,CAAC,CAAC;IAC5B,KAAK,MAAM,IAAI,IAAI,YAAY;QAAE,mBAAmB,IAAI,MAAM,CAAC,GAAG,CAAC,IAAI,CAAE,CAAC;IAC1E,OAAO;QACL,KAAK,EAAE,OAAO,CAAC,EAAE;QACjB,MAAM,EAAE,CAAC,MAAM,CAAC,MAAM,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,WAAW,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,YAAY;QACpE,MAAM;QACN,OAAO;QACP,oBAAoB;QACpB,IAAI,EAAE;YACJ,UAAU,EAAE,MAAM,CAAC,MAAM;YACzB,YAAY;YACZ,mBAAmB;YACnB,kBAAkB,EAAE,YAAY,CAAC,IAAI;SACtC;QACD,QAAQ;KACT,CAAC;AACJ,CAAC","sourcesContent":["/** Shared weighted-lexicon engine. No I/O, global state, or runtime dependencies. */\nexport type Encoding = \"frequency\" | \"binary\" | \"percent\";\nexport type Weights = Readonly<\n Record<string, Readonly<Record<string, number>>>\n>;\nexport interface LexiconDefinition {\n readonly id: string;\n readonly language?: string;\n readonly categories: Weights;\n readonly intercepts?: Readonly<Record<string, number>>;\n readonly features?: Weights;\n readonly ngrams?: readonly number[];\n readonly encoding?: Encoding;\n}\nexport interface Lexicon extends LexiconDefinition {\n readonly intercepts: Readonly<Record<string, number>>;\n readonly features: Weights;\n readonly ngrams: readonly number[];\n readonly encoding: Encoding;\n}\nexport interface Options {\n /** Frequency divides term occurrences by original token count, including unmatched tokens. */\n readonly encoding?: Encoding;\n /** Sizes include unigrams explicitly. An empty array disables lexical matching. */\n readonly ngrams?: readonly number[];\n readonly includeIntercept?: boolean;\n readonly minWeight?: number;\n readonly maxWeight?: number;\n /** Round final scores only. Omit to retain full floating-point precision. */\n readonly decimals?: number;\n /** Measured nonlexical covariates; never infer these from a single text. */\n readonly features?: Readonly<Record<string, number>>;\n}\nexport interface Match {\n readonly term: string;\n readonly count: number;\n readonly weight: number;\n readonly contribution: number;\n}\nexport interface Analysis {\n readonly model: string;\n readonly status: \"ok\" | \"empty\" | \"no-matches\";\n readonly values: Readonly<Record<string, number | null>>;\n readonly matches: Readonly<Record<string, readonly Match[]>>;\n readonly featureContributions: Readonly<\n Record<string, Readonly<Record<string, number>>>\n >;\n readonly info: {\n readonly tokenCount: number;\n readonly featureCount: number;\n readonly matchedFeatureCount: number;\n readonly uniqueMatchedTerms: number;\n };\n readonly warnings: readonly string[];\n}\nconst own = (object: object, key: string) =>\n Object.prototype.hasOwnProperty.call(object, key);\nfunction finite(value: unknown, label: string): asserts value is number {\n if (typeof value !== \"number\" || !Number.isFinite(value)) {\n throw new TypeError(`${label} must be finite`);\n }\n}\nfunction checkNgrams(value: readonly number[]): void {\n if (\n !Array.isArray(value) ||\n value.some((n) => !Number.isSafeInteger(n) || n < 1 || n > 8)\n ) {\n throw new RangeError(\"ngrams must contain integer sizes from 1 to 8\");\n }\n}\nfunction checkEncoding(value: unknown): asserts value is Encoding {\n if (![\"frequency\", \"binary\", \"percent\"].includes(value as string)) {\n throw new RangeError(\"Unknown encoding\");\n }\n}\n/** Copy and validate a custom lexicon once; returned data cannot be mutated. */\nexport function createLexicon(definition: LexiconDefinition): Lexicon {\n if (\n !definition || typeof definition.id !== \"string\" || !definition.id ||\n !definition.categories || typeof definition.categories !== \"object\" ||\n Array.isArray(definition.categories)\n ) {\n throw new TypeError(\"A lexicon needs an id and category weight maps\");\n }\n const categories: Record<string, Readonly<Record<string, number>>> = Object\n .create(null);\n const intercepts: Record<string, number> = Object.create(null);\n const features: Record<string, Readonly<Record<string, number>>> = Object\n .create(null);\n for (const [category, terms] of Object.entries(definition.categories)) {\n if (\n !category || !terms || typeof terms !== \"object\" || Array.isArray(terms)\n ) throw new TypeError(\"Invalid category\");\n for (const [term, weight] of Object.entries(terms)) {\n if (!term) throw new TypeError(\"Lexicon terms must not be empty\");\n finite(weight, `Weight for ${term}`);\n }\n categories[category] = Object.freeze({ ...terms });\n const intercept =\n definition.intercepts && own(definition.intercepts, category)\n ? definition.intercepts[category]\n : 0;\n finite(intercept, `Intercept for ${category}`);\n intercepts[category] = intercept;\n const covariates = definition.features && own(definition.features, category)\n ? definition.features[category]!\n : {};\n if (\n !covariates || typeof covariates !== \"object\" || Array.isArray(covariates)\n ) throw new TypeError(\"Feature weights must be a map\");\n for (const [feature, weight] of Object.entries(covariates)) {\n finite(weight, `Feature weight for ${feature}`);\n }\n features[category] = Object.freeze({ ...covariates });\n }\n if (!Object.keys(categories).length) {\n throw new TypeError(\"A lexicon needs at least one category\");\n }\n for (const key of Object.keys(definition.intercepts ?? {})) {\n if (!own(categories, key)) {\n throw new RangeError(`Unknown intercept category: ${key}`);\n }\n }\n for (const key of Object.keys(definition.features ?? {})) {\n if (!own(categories, key)) {\n throw new RangeError(`Unknown feature category: ${key}`);\n }\n }\n const ngrams = definition.ngrams ?? [1, 2, 3];\n const encoding = definition.encoding ?? \"frequency\";\n checkNgrams(ngrams);\n checkEncoding(encoding);\n return Object.freeze({\n id: definition.id,\n ...(definition.language === undefined\n ? {}\n : { language: definition.language }),\n categories: Object.freeze(categories),\n intercepts: Object.freeze(intercepts),\n features: Object.freeze(features),\n ngrams: Object.freeze([...new Set(ngrams)]),\n encoding,\n });\n}\n/** Unicode-aware convenience tokenizer; supply exact study tokens when reproducing a pipeline. */\nexport function tokenize(text: string): string[] {\n if (typeof text !== \"string\") throw new TypeError(\"text must be a string\");\n return text.normalize(\"NFC\").toLowerCase().replace(/[’‘]/gu, \"'\").match(\n /https?:\\/\\/[^\\s]+|<3|[:;=8][\\-o*']?[\\)\\]\\(\\[dp/\\\\]|[\\)\\]\\(\\[d][:;=8]|[#@][\\p{L}\\p{M}\\p{N}_]+|\\p{N}+(?:[.,:]\\p{N}+)+|[\\p{L}\\p{M}\\p{N}_]+(?:['-][\\p{L}\\p{M}\\p{N}_]+)*|\\.{3,}|\\p{Extended_Pictographic}(?:\\uFE0F|\\p{M})*|[^\\s]/gu,\n ) ?? [];\n}\n/** Score a validated lexicon. Input token arrays are used exactly as supplied. */\nexport function score(\n input: string | readonly string[],\n lexicon: Lexicon,\n options: Options = {},\n): Analysis {\n if (!options || typeof options !== \"object\" || Array.isArray(options)) {\n throw new TypeError(\"options must be an object\");\n }\n const allowed = [\n \"encoding\",\n \"ngrams\",\n \"includeIntercept\",\n \"minWeight\",\n \"maxWeight\",\n \"decimals\",\n \"features\",\n ];\n for (const key of Object.keys(options)) {\n if (!allowed.includes(key)) {\n throw new RangeError(`Unknown option: ${key}`);\n }\n }\n const encoding = options.encoding ?? lexicon.encoding;\n const ngrams = options.ngrams ?? lexicon.ngrams;\n const min = options.minWeight ?? -Infinity,\n max = options.maxWeight ?? Infinity;\n checkEncoding(encoding);\n checkNgrams(ngrams);\n if (\n typeof min !== \"number\" || typeof max !== \"number\" || Number.isNaN(min) ||\n Number.isNaN(max) || min > max\n ) throw new RangeError(\"Invalid weight bounds\");\n if (\n options.includeIntercept !== undefined &&\n typeof options.includeIntercept !== \"boolean\"\n ) throw new TypeError(\"includeIntercept must be boolean\");\n if (\n options.decimals !== undefined &&\n (!Number.isInteger(options.decimals) || options.decimals < 0 ||\n options.decimals > 15)\n ) throw new RangeError(\"decimals must be an integer from 0 to 15\");\n const tokens = typeof input === \"string\" ? tokenize(input) : input;\n if (\n !Array.isArray(tokens) || tokens.some((t) => typeof t !== \"string\" || !t)\n ) {\n throw new TypeError(\n \"input must be text or an array of nonempty token strings\",\n );\n }\n const suppliedFeatures = options.features ?? {};\n if (typeof suppliedFeatures !== \"object\" || Array.isArray(suppliedFeatures)) {\n throw new TypeError(\"features must be an object\");\n }\n const knownFeatures = new Set(\n Object.values(lexicon.features).flatMap((f) => Object.keys(f)),\n );\n for (const [key, value] of Object.entries(suppliedFeatures)) {\n if (!knownFeatures.has(key)) {\n throw new RangeError(`Unknown model feature: ${key}`);\n }\n finite(value, `Feature ${key}`);\n }\n const counts = new Map<string, number>();\n let featureCount = 0;\n for (const n of new Set(ngrams)) {\n for (let i = 0; i <= tokens.length - n; i++) {\n const term = tokens.slice(i, i + n).join(\" \");\n counts.set(term, (counts.get(term) ?? 0) + 1);\n featureCount++;\n }\n }\n const values: Record<string, number | null> = Object.create(null);\n const matches: Record<string, Match[]> = Object.create(null);\n const featureContributions: Record<string, Record<string, number>> = Object\n .create(null);\n const matchedTerms = new Set<string>();\n let hasEvidence = false;\n const warnings = encoding === \"percent\"\n ? []\n : [...knownFeatures].filter((f) => !own(suppliedFeatures, f)).map((f) =>\n `Structural feature ${f} was not supplied; its contribution is omitted.`\n );\n for (const [category, weights] of Object.entries(lexicon.categories)) {\n let total = encoding === \"percent\" || options.includeIntercept === false\n ? 0\n : lexicon.intercepts[category]!;\n const categoryMatches: Match[] = [];\n const covariates: Record<string, number> = Object.create(null);\n let matchedCount = 0;\n for (const [term, count] of counts) {\n if (!own(weights, term)) continue;\n const weight = weights[term]!;\n if (weight < min || weight > max) continue;\n const contribution = encoding === \"binary\"\n ? weight\n : encoding === \"percent\"\n ? count / featureCount\n : weight * (count / tokens.length);\n total += contribution;\n matchedCount += count;\n categoryMatches.push({ term, count, weight, contribution });\n matchedTerms.add(term);\n }\n if (encoding === \"percent\") {\n total = featureCount ? matchedCount / featureCount : 0;\n }\n if (encoding !== \"percent\" && tokens.length) {\n for (\n const [feature, weight] of Object.entries(lexicon.features[category]!)\n ) {\n if (own(suppliedFeatures, feature)) {\n const contribution = suppliedFeatures[feature]! * weight;\n covariates[feature] = contribution;\n total += contribution;\n }\n }\n }\n const evidence = tokens.length > 0 &&\n (categoryMatches.length > 0 || Object.keys(covariates).length > 0);\n hasEvidence ||= evidence;\n if (!Number.isFinite(total)) {\n throw new RangeError(`Score overflow for ${category}`);\n }\n values[category] = evidence\n ? options.decimals === undefined\n ? total\n : Number(total.toFixed(options.decimals))\n : null;\n matches[category] = categoryMatches.sort((a, b) =>\n b.count - a.count || a.term.localeCompare(b.term, \"en\")\n );\n featureContributions[category] = covariates;\n }\n let matchedFeatureCount = 0;\n for (const term of matchedTerms) matchedFeatureCount += counts.get(term)!;\n return {\n model: lexicon.id,\n status: !tokens.length ? \"empty\" : hasEvidence ? \"ok\" : \"no-matches\",\n values,\n matches,\n featureContributions,\n info: {\n tokenCount: tokens.length,\n featureCount,\n matchedFeatureCount,\n uniqueMatchedTerms: matchedTerms.size,\n },\n warnings,\n };\n}\n"]}
@@ -0,0 +1,7 @@
1
+ import { type Analysis, type Lexicon, type Options } from "./core.ts";
2
+ export * from "./core.ts";
3
+ export type ModelId = "affect" | "age" | "gender" | "temporal" | "perma" | "permaEs" | "bigFive" | "darkTriad" | "optimism";
4
+ /** Immutable model definitions, including weights and research intercepts. */
5
+ export declare const models: Readonly<Record<ModelId, Lexicon>>;
6
+ /** Select a bundled research lexicon or supply a validated custom lexicon. */
7
+ export declare function analyse(input: string | readonly string[], model: ModelId | Lexicon, options?: Options): Analysis;
@@ -0,0 +1,17 @@
1
+ import definitions from "../data/models.json" with { type: "json" };
2
+ import { createLexicon, score, } from "./core.js";
3
+ export * from "./core.js";
4
+ /** Immutable model definitions, including weights and research intercepts. */
5
+ export const models = Object.freeze(Object.fromEntries(Object.entries(definitions).map(([id, definition]) => [id, createLexicon(definition)])));
6
+ /** Select a bundled research lexicon or supply a validated custom lexicon. */
7
+ export function analyse(input, model, options = {}) {
8
+ const lexicon = typeof model === "string"
9
+ ? Object.prototype.hasOwnProperty.call(models, model)
10
+ ? models[model]
11
+ : undefined
12
+ : model;
13
+ if (!lexicon)
14
+ throw new RangeError(`Unknown model: ${String(model)}`);
15
+ return score(input, lexicon, options);
16
+ }
17
+ //# sourceMappingURL=index.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"index.js","sourceRoot":"","sources":["../../src/index.ts"],"names":[],"mappings":"AAAA,OAAO,WAAW,MAAM,qBAAqB,CAAC,OAAO,IAAI,EAAE,MAAM,EAAE,CAAC;AACpE,OAAO,EAEL,aAAa,EAIb,KAAK,GACN,MAAM,WAAW,CAAC;AACnB,cAAc,WAAW,CAAC;AAW1B,8EAA8E;AAC9E,MAAM,CAAC,MAAM,MAAM,GAAuC,MAAM,CAAC,MAAM,CACrE,MAAM,CAAC,WAAW,CAChB,MAAM,CAAC,OAAO,CAAC,WAAW,CAAC,CAAC,GAAG,CAAC,CAC9B,CAAC,EAAE,EAAE,UAAU,CAAC,EAChB,EAAE,CAAC,CAAC,EAAE,EAAE,aAAa,CAAC,UAA+B,CAAC,CAAC,CAAC,CAC/B,CAC9B,CAAC;AACF,8EAA8E;AAC9E,MAAM,UAAU,OAAO,CACrB,KAAiC,EACjC,KAAwB,EACxB,OAAO,GAAY,EAAE;IAErB,MAAM,OAAO,GAAG,OAAO,KAAK,KAAK,QAAQ;QACvC,CAAC,CAAC,MAAM,CAAC,SAAS,CAAC,cAAc,CAAC,IAAI,CAAC,MAAM,EAAE,KAAK,CAAC;YACnD,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC;YACf,CAAC,CAAC,SAAS;QACb,CAAC,CAAC,KAAK,CAAC;IACV,IAAI,CAAC,OAAO;QAAE,MAAM,IAAI,UAAU,CAAC,kBAAkB,MAAM,CAAC,KAAK,CAAC,EAAE,CAAC,CAAC;IACtE,OAAO,KAAK,CAAC,KAAK,EAAE,OAAO,EAAE,OAAO,CAAC,CAAC;AACxC,CAAC","sourcesContent":["import definitions from \"../data/models.json\" with { type: \"json\" };\nimport {\n type Analysis,\n createLexicon,\n type Lexicon,\n type LexiconDefinition,\n type Options,\n score,\n} from \"./core.ts\";\nexport * from \"./core.ts\";\nexport type ModelId =\n | \"affect\"\n | \"age\"\n | \"gender\"\n | \"temporal\"\n | \"perma\"\n | \"permaEs\"\n | \"bigFive\"\n | \"darkTriad\"\n | \"optimism\";\n/** Immutable model definitions, including weights and research intercepts. */\nexport const models: Readonly<Record<ModelId, Lexicon>> = Object.freeze(\n Object.fromEntries(\n Object.entries(definitions).map((\n [id, definition],\n ) => [id, createLexicon(definition as LexiconDefinition)]),\n ) as Record<ModelId, Lexicon>,\n);\n/** Select a bundled research lexicon or supply a validated custom lexicon. */\nexport function analyse(\n input: string | readonly string[],\n model: ModelId | Lexicon,\n options: Options = {},\n): Analysis {\n const lexicon = typeof model === \"string\"\n ? Object.prototype.hasOwnProperty.call(models, model)\n ? models[model]\n : undefined\n : model;\n if (!lexicon) throw new RangeError(`Unknown model: ${String(model)}`);\n return score(input, lexicon, options);\n}\n"]}
package/docs/api.md ADDED
@@ -0,0 +1,92 @@
1
+ # API
2
+
3
+ ## `analyse(input, model, options?)`
4
+
5
+ Import from `wwbnlp`. `input` is a string or readonly array of nonempty token
6
+ strings. `model` is one of the nine model IDs in the README, or a lexicon
7
+ returned by `createLexicon`. Invalid model names, inputs and options throw; they
8
+ never fall back to another model or encoding.
9
+
10
+ ## `createLexicon(definition)` and `score(input, lexicon, options?)`
11
+
12
+ Import from `wwbnlp/core` to use your own data without loading the bundled
13
+ lexica:
14
+
15
+ ```js
16
+ import { createLexicon, score } from "wwbnlp/core";
17
+
18
+ const lexicon = createLexicon({
19
+ id: "example",
20
+ categories: { SENTIMENT: { good: 2, bad: -2, "really good": 3 } },
21
+ intercepts: { SENTIMENT: 0.5 },
22
+ ngrams: [1, 2],
23
+ encoding: "frequency",
24
+ });
25
+
26
+ console.log(score(["really", "good", "good"], lexicon).values);
27
+ // { SENTIMENT: 2.833333333333333 }
28
+ ```
29
+
30
+ Definitions are validated, copied and frozen. Every category needs a
31
+ term-to-finite-weight map. Intercepts default to zero. Optional structural
32
+ feature weights use the same category map shape, keyed by feature names. Unknown
33
+ intercept/feature categories are rejected. Custom definitions default to
34
+ frequency encoding and `[1, 2, 3]` ngrams.
35
+
36
+ `score` and `analyse` return `Analysis` with category maps and explicit status;
37
+ see the README. A category without a lexical match or supplied structural
38
+ feature returns `null`, even if it has an intercept. Empty input returns `empty`
39
+ and all-null values. Mixed evidence can produce both numbers and nulls in the
40
+ same result.
41
+
42
+ ## Options
43
+
44
+ | Option | Default | Meaning |
45
+ | ------------------------ | -------------------------- | ----------------------------------------------------------------- |
46
+ | `encoding` | Model default | `frequency`, `binary`, or `percent` |
47
+ | `ngrams` | Model vocabulary sizes | Unique integer sizes 1–8; `[]` disables matching |
48
+ | `includeIntercept` | `true` | Include the model intercept, except in percent mode |
49
+ | `minWeight`, `maxWeight` | Negative/positive infinity | Inclusive term-weight filters |
50
+ | `decimals` | Omitted | Round final category scores only, 0–15 decimal places |
51
+ | `features` | `{}` | Finite measured structural values; keys must belong to this model |
52
+
53
+ Unrecognized options throw, so legacy options cannot be silently ignored.
54
+ Options, definitions and input arrays are never modified.
55
+
56
+ Frequency mode divides each matched feature’s occurrence count by the original
57
+ token count, before adding its weight and intercept. Adding bigrams does not
58
+ inflate that denominator. Binary mode counts unique terms once. Percent mode
59
+ measures coverage of generated candidate features; repeated terms count, weights
60
+ do not, and the result stays between zero and one.
61
+
62
+ `matches` sort by descending occurrence count, then term. Contributions retain
63
+ full precision even when final values are rounded. `matchedFeatureCount` counts
64
+ each matched occurrence once across all categories. It includes ngrams, so it is
65
+ not a count of distinct word positions.
66
+
67
+ Structural values are multiplied by their weights and added once per category.
68
+ They are independent of lexical `minWeight`/`maxWeight` filters and are ignored
69
+ in percent mode. Missing structural values produce warnings in weighted modes.
70
+ Supplied zeros are measurements and count as evidence. Supply the same feature
71
+ definitions and units as the original research pipeline; the engine does not
72
+ invent them.
73
+
74
+ ## `tokenize(text)`
75
+
76
+ Returns the documented convenience token stream. Supply exact study tokens to
77
+ `score` or `analyse` when controlling preprocessing. There is no automatic
78
+ British-to-American translation, HTML entity decoding or language detection.
79
+ Spanish uses the explicit `permaEs` model.
80
+
81
+ ## `models`
82
+
83
+ Readonly registry of bundled model definitions. Categories, coefficients,
84
+ intercepts, structural features and default ngram sizes are available for
85
+ inspection. Importing `wwbnlp/core` avoids loading this registry.
86
+
87
+ ## Runtime support
88
+
89
+ ESM, typed declarations, Node.js 22.12+ native `require()` of ESM, and Deno 2
90
+ with the built entry point. JSON modules use import attributes. Older
91
+ Node/CommonJS applications must upgrade or use dynamic import where supported.
92
+ No browser support claim is made without testing a specific bundler/runtime.
@@ -0,0 +1,124 @@
1
+ # Migrating to wwbnlp
2
+
3
+ The existing npm releases remain available. Archiving GitHub repositories does
4
+ not unpublish npm versions, and this consolidation does not replace existing
5
+ dependencies automatically. The legacy npm account is
6
+ [phughes](https://www.npmjs.com/~phughes).
7
+
8
+ The new [wwbnlp](https://www.npmjs.com/package/wwbnlp) package is maintained by
9
+ [phughesmcr](https://www.npmjs.com/~phughesmcr). Install it with
10
+ `npm install wwbnlp` and migrate explicitly using the model IDs below.
11
+
12
+ ## Imports and model IDs
13
+
14
+ | Existing package | Last observed npm version | New call |
15
+ | ------------------ | -------------------------------------- | ------------------------------------------------------ |
16
+ | affectimo | 3.1.0 | `analyse(text, 'affect')` |
17
+ | bigfive | 3.0.0 | `analyse(text, 'bigFive')` |
18
+ | darktriad | 3.1.1 | `analyse(text, 'darkTriad')` |
19
+ | optimismo | 4.0.1 | `analyse(text, 'optimism')` |
20
+ | predictage | 4.0.1 | `analyse(text, 'age')` |
21
+ | predictgender | 4.0.0 | `analyse(text, 'gender')` |
22
+ | prospectimo | 3.0.0 | `analyse(text, 'temporal')` |
23
+ | wellbeing_analysis | 4.0.0 | `analyse(text, 'perma')` or `analyse(text, 'permaEs')` |
24
+ | lex-helpers | 0.6.0 | `createLexicon` and `score` from `wwbnlp/core` |
25
+ | weighted-lexica | Not found under this unscoped npm name | `createLexicon` and `score` from `wwbnlp/core` |
26
+
27
+ Registry versions were checked on 4 October 2026. `weighted-lexica` may have
28
+ been published under a different name or scope; a 404 does not establish that no
29
+ release ever existed.
30
+
31
+ ```js
32
+ // Before
33
+ const affectimo = require("affectimo");
34
+ const values = affectimo(text);
35
+
36
+ // After
37
+ import { analyse } from "wwbnlp";
38
+ const result = analyse(text, "affect");
39
+ const valuesAfter = result.values;
40
+ ```
41
+
42
+ Native CommonJS works on Node 22.12+:
43
+
44
+ ```js
45
+ const { analyse } = require("wwbnlp");
46
+ const result = analyse(text, "affect");
47
+ ```
48
+
49
+ ## Intentional changes
50
+
51
+ This is a new API, not a drop-in wrapper or promise of numerically identical
52
+ output.
53
+
54
+ - Always read `result.values`, `result.matches` and `result.info`. There are no
55
+ `output: 'lex'/'matches'/'full'` modes.
56
+ - No data or no matching evidence yields explicit status and null category
57
+ values. Intercepts alone cannot turn unknown input into a prediction.
58
+ - Affect, temporal orientation and PERMA now default to relative frequency,
59
+ rather than the old binary sums. Big Five and optimism retain binary sums as
60
+ association measures. Request an encoding explicitly for comparisons.
61
+ - Canonical WWBP CSV weights and intercepts replace rounded or damaged legacy
62
+ copies where available. Spanish accents are restored, and `permaEs` selects
63
+ Spanish directly. The old wellbeing module mistakenly inspected `output`
64
+ rather than `lang` when selecting Spanish.
65
+ - Matching uses one token stream for all ngrams. All vocabulary ngram sizes are
66
+ enabled by default; the old age/gender wrappers ignored phrase terms. To
67
+ compare unigrams, set `ngrams: [1]` and supply controlled tokens.
68
+ - English PERMA’s structural feature weights are retained and missing inputs are
69
+ reported. Lexical-only results should not be described as the complete
70
+ original trained model.
71
+ - Big Five retains historical association weights except an invalid empty-string
72
+ O term. Scores have no calibrated personality scale.
73
+ - Optimism is an experimental composition of future terms and affect weights. It
74
+ is not a separately trained optimism predictor.
75
+ - Gender returns the historical classifier margin, without converting it into an
76
+ assertion about identity. The historical sign convention was negative/positive
77
+ for the source’s male/female labels.
78
+ - Temporal scores are not probabilities. Any ranking must handle nulls and ties
79
+ explicitly; there is no automatic “Unknown” or orientation label.
80
+ - Results are synchronous with no async dependency or mutable global state.
81
+ Caller options are never mutated.
82
+
83
+ ## Options
84
+
85
+ | Previous option | New equivalent |
86
+ | --------------------- | ---------------------------------------------------------------------------- |
87
+ | `encoding: 'freq'` | `encoding: 'frequency'` |
88
+ | `encoding: 'binary'` | Same unique-term sum semantics |
89
+ | `encoding: 'percent'` | Candidate-feature coverage fraction; no intercept |
90
+ | `nGrams: [2, 3]` | `ngrams: [1, 2, 3]` (unigrams are explicit) |
91
+ | `nGrams: [0]` | `ngrams: [1]` to keep unigrams only |
92
+ | `noInt: true` | `includeIntercept: false` |
93
+ | `min`, `max` | `minWeight`, `maxWeight` |
94
+ | `places` | `decimals` (final scores only) |
95
+ | `logs`, `suppressLog` | Removed; the library does not log |
96
+ | `sortBy` | Sort `result.matches[category]` in your application |
97
+ | `wcGrams` | Removed; weighted frequency uses original tokens |
98
+ | `locale` | Normalize spelling in your preprocessing if needed; no automatic translation |
99
+ | `lang: 'spanish'` | Model ID `permaEs` |
100
+
101
+ ## Utility packages
102
+
103
+ Replace `lexFrequencyPipeline(lex, intercept)` with a lexicon configured for
104
+ `[1]` and `frequency`, then read `score(tokens, lexicon).values.CATEGORY`. The
105
+ old `lexBinaryPipeline` divided unique weights by the number of unique tokens;
106
+ the new `binary` encoding sums unique weights without that division. Use
107
+ explicit arithmetic if you need the old normalized-unique formula.
108
+
109
+ The previous `Term`/`Category`/`Lexicon` object graph is replaced by immutable
110
+ category-to-weight maps. Build a new definition to change vocabulary rather than
111
+ maintaining bidirectional mutable associations. The new engine includes category
112
+ intercepts; the old `weighted-lexica.analyse()` implementation created an empty
113
+ intercept map and never filled it.
114
+
115
+ ## Maintainer release sequence
116
+
117
+ 1. Run tests, source verification and package inspection; review the tarball.
118
+ 2. Publish `wwbnlp` from the `phughesmcr` npm account, after authentication and
119
+ any npm-required account confirmation.
120
+ 3. Verify the installed registry package and its bundled installation
121
+ instructions.
122
+ 4. If retiring the old npm names, add a deprecation message linking to this
123
+ guide. Deprecation is a separate registry change; it has not been applied by
124
+ archiving repositories. Keep old versions available for reproducibility.