@willwade/aac-processors 0.3.1 → 0.3.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,281 @@
1
+ /**
2
+ * Linguistic Competence Metrics
3
+ *
4
+ * Privacy-preserving analysis of AAC *spoken output* (phrase history), based on:
5
+ * - Niemeijer, Sheldon & Hillary Zisk (2025), "Measuring AAC user linguistic
6
+ * competence: A novel approach", AssistiveWare (Communication Matters handout).
7
+ * - Frisch, Wade et al. (2026), "It's Complicated: On the Design and Evaluation
8
+ * of AI-Powered AAC Interfaces", arXiv:2606.24854.
9
+ *
10
+ * DESIGN (read me):
11
+ * - **Source-agnostic.** The only input is `{ text, timestampMs }[]`. It does
12
+ * not know or care whether the speech history came from Grid 3, Snap,
13
+ * TouchChat, OBF/OBFL logs, or anything else. See `historyEntriesToCompetence*
14
+ * Utterances` (in history.ts) to adapt any `HistoryEntry[]` source.
15
+ * - **Language-agnostic core.** This module contains NO word lists. Language-
16
+ * specific resources (a closed-class word set, an inflection classifier) are
17
+ * INJECTED via `LanguageResources`. When a resource is missing for a
18
+ * language, the affected measure is reported as `unavailable` with a reason
19
+ * and a warning is raised — never silently wrong.
20
+ * - **Pure / no I/O.** No filesystem, no platform APIs. Runs anywhere (browser
21
+ * included) and emits only aggregate statistics (never the raw text).
22
+ *
23
+ * The four dimensions of linguistic competence (Light, 1989) and the measures we
24
+ * use for each, following the AssistiveWare findings:
25
+ *
26
+ * Semantic -> MATTR-30 lexical diversity (always available)
27
+ * Syntactic -> preposition/conjunction diversity (needs closedClassWords)
28
+ * Morphological -> inflected-form diversity (needs classifyInflection)
29
+ * Phonological -> proportion of unique words in a dictionary (needs a dictionary)
30
+ *
31
+ * All diversity measures use 30-word moving-average windows (Covington & McFall,
32
+ * 2010), making them sample-length independent and usable for the tiny, highly
33
+ * variable samples typical of AAC. MLU is intentionally NOT a headline (it
34
+ * conflates linguistic/operational/strategic/social competence in AAC); it is
35
+ * reported only as a distribution.
36
+ */
37
+ /** A single spoken utterance with a production timestamp (epoch ms). */
38
+ export interface CompetenceUtterance {
39
+ text: string;
40
+ timestampMs: number;
41
+ }
42
+ /** A tokenised word stream produced in chronological order. */
43
+ export type WordStream = string[];
44
+ /** Coarse inflection category used by the (injected) morphology classifier. */
45
+ export type InflectionCategory = 'base' | 'plural' | 'possessive' | 'past' | 'progressive' | 'comparative' | 'superlative' | 'adverb';
46
+ /**
47
+ * Language-specific resources, injected by the caller. Providing none leaves the
48
+ * language-specific measures unavailable (with explicit warnings) — the semantic
49
+ * measure still works for any language.
50
+ */
51
+ export interface LanguageResources {
52
+ /** Closed-class words (prepositions, conjunctions, ...) for syntactic diversity. */
53
+ closedClassWords?: Set<string>;
54
+ /** Maps a lowercased word to an inflection category for morphological diversity. */
55
+ classifyInflection?: (word: string) => InflectionCategory;
56
+ }
57
+ export interface DiversityOptions {
58
+ /** Moving-average window size in words. The papers use 30. */
59
+ windowSize?: number;
60
+ /** Closed-class word set (for the syntactic measure). */
61
+ closedClassWords?: Set<string>;
62
+ /** Inflection classifier (for the morphological measure). */
63
+ classifyInflection?: (word: string) => InflectionCategory;
64
+ }
65
+ export interface DiversityResult {
66
+ /** Median of the per-window values (the headline figure, per the paper). */
67
+ median: number | null;
68
+ mean: number | null;
69
+ /** Number of windows that contributed a value (after any skipping). */
70
+ nWindows: number;
71
+ /** The window size used. */
72
+ windowSize: number;
73
+ /** Present when the measure could not be computed (e.g. missing language data). */
74
+ unavailable?: string;
75
+ }
76
+ /**
77
+ * Tokenise raw text into a lowercased word stream.
78
+ * Keeps intra-word apostrophes (don't, children's) but drops leading/trailing
79
+ * punctuation and pure whitespace. Accented characters are preserved (\p{L}).
80
+ */
81
+ export declare function tokenize(text: string): WordStream;
82
+ /**
83
+ * Moving-Average Type-Token Ratio (Covington & McFall, 2010).
84
+ *
85
+ * Slides a fixed-size window across the word stream and computes the TTR for
86
+ * each window. The median across windows is sample-length independent, which is
87
+ * exactly why it is preferred over plain TTR for highly variable AAC samples.
88
+ */
89
+ export declare function movingAverageTTR(words: WordStream, windowSize?: number): DiversityResult;
90
+ /** Convenience: MATTR-30 lexical diversity (the semantic headline). */
91
+ export declare function lexicalDiversity(words: WordStream, windowSize?: number): DiversityResult;
92
+ /**
93
+ * Closed-class diversity (generalises AssistiveWare's MA-UPC-TWR-30).
94
+ *
95
+ * For each window, compute the type-token ratio restricted to the supplied
96
+ * closed-class words (prepositions + conjunctions in the original paper). Windows
97
+ * containing none are skipped. The caller supplies the set via
98
+ * `closedClassWords`, so this works for any language without hardcoding here.
99
+ */
100
+ export declare function syntacticDiversity(words: WordStream, options?: DiversityOptions): DiversityResult;
101
+ /**
102
+ * Morphological diversity (proxy for MA-UMORPH-TLWR-30).
103
+ *
104
+ * Within each window, words the supplied classifier marks as inflected (any
105
+ * category other than "base") contribute their surface forms to a type-token
106
+ * ratio. Windows with no inflected words are skipped. The classifier is injected
107
+ * (`classifyInflection`) so the heuristic lives with the caller, per language.
108
+ *
109
+ * Caveat (per AssistiveWare): for symbol-supported AAC, pre-stored morphology
110
+ * buttons ("finished", "is", ...) heavily affect this measure — interpret trends
111
+ * rather than absolutes.
112
+ */
113
+ export declare function morphologicalDiversity(words: WordStream, options?: DiversityOptions): DiversityResult;
114
+ /**
115
+ * Proportion of unique alphabetic words present in the supplied dictionary.
116
+ * Unique words only, so it is not skewed by repetition or repeated misspellings.
117
+ * Returns null (unavailable) if no dictionary is provided.
118
+ */
119
+ export declare function spellingValidity(words: WordStream, dictionary?: Set<string>): number | null;
120
+ export interface DistributionStats {
121
+ median: number | null;
122
+ mean: number | null;
123
+ p25: number | null;
124
+ p75: number | null;
125
+ n: number;
126
+ }
127
+ export interface ActivityStats {
128
+ utterances: number;
129
+ words: number;
130
+ /** Distinct tokens — vocabulary breadth (count only, never the words). */
131
+ uniqueWords: number;
132
+ activeDays: number;
133
+ wordsPerUtterance: DistributionStats;
134
+ }
135
+ /** Compute engagement/activity stats for a set of utterances. */
136
+ export declare function summarizeActivity(utterances: CompetenceUtterance[]): ActivityStats;
137
+ export interface MonthBin {
138
+ /** Calendar month in local time, "YYYY-MM". */
139
+ month: string;
140
+ utterances: number;
141
+ words: number;
142
+ uniqueWords: number;
143
+ activeDays: number;
144
+ wordsPerUtterance: DistributionStats;
145
+ /** Semantic — the headline measure. */
146
+ lexicalDiversity: DiversityResult;
147
+ /** Syntactic. null/unavailable when no closed-class data for the language. */
148
+ syntacticDiversity: DiversityResult;
149
+ /** Morphological (proxy). null/unavailable when no classifier for the language. */
150
+ morphologicalDiversity: DiversityResult;
151
+ /** Phonological — only when a dictionary is supplied. */
152
+ spellingValidity: number | null;
153
+ /** True when the month has too little data to trust the diversity figures. */
154
+ suppressed: boolean;
155
+ suppressReason: string | null;
156
+ }
157
+ export interface TrendResult {
158
+ metric: string;
159
+ /** Slope of the weighted linear regression, in metric units per month. */
160
+ slopePerMonth: number | null;
161
+ firstHalf: number | null;
162
+ secondHalf: number | null;
163
+ /** secondHalf - firstHalf. */
164
+ delta: number | null;
165
+ direction: 'up' | 'down' | 'flat' | 'unknown';
166
+ }
167
+ export interface DimensionSupport {
168
+ available: boolean;
169
+ reason?: string;
170
+ }
171
+ export interface PagesetSummary {
172
+ label: string;
173
+ gridsetIncluded: boolean;
174
+ analysisVersion?: string;
175
+ totalBoards: number;
176
+ totalButtons: number;
177
+ totalWords: number;
178
+ grid: {
179
+ rows: number;
180
+ columns: number;
181
+ };
182
+ effort: DistributionStats;
183
+ hasDynamicPrediction: boolean;
184
+ spellingEffort: {
185
+ base: number | null;
186
+ perLetter: number | null;
187
+ };
188
+ error?: string;
189
+ }
190
+ export interface UserSettingsSummary {
191
+ startupGridSet: string | null;
192
+ onlineAiToolsOptIn: boolean | null;
193
+ accessMethods: string[];
194
+ personalisation: {
195
+ pronunciations: number;
196
+ capitalisations: number;
197
+ abbreviationExpansions: number;
198
+ smallWords: number;
199
+ };
200
+ }
201
+ export interface CompetenceReport {
202
+ schema: string;
203
+ generatedAt: string;
204
+ privacy: {
205
+ rawUtterancesIncluded: boolean;
206
+ wordListsIncluded: boolean;
207
+ fringeWordFrequencyIncluded: boolean;
208
+ minAggregationWindowDays: number;
209
+ notes: string[];
210
+ };
211
+ source: {
212
+ platform: string;
213
+ langCode?: string;
214
+ userLabel?: string;
215
+ dbPathIncluded: boolean;
216
+ };
217
+ config: {
218
+ months: number;
219
+ windowSize: number;
220
+ lang: string;
221
+ minWordsPerMonth: number;
222
+ dictionaryProvided: boolean;
223
+ };
224
+ overall: {
225
+ windowStart: string;
226
+ windowEnd: string;
227
+ totalUtterances: number;
228
+ totalWords: number;
229
+ monthsCovered: number;
230
+ monthsSuppressed: number;
231
+ };
232
+ timeline: MonthBin[];
233
+ trend: TrendResult;
234
+ /** Per-dimension availability for the detected language (no silent degradation). */
235
+ support: {
236
+ lang: string;
237
+ semantic: DimensionSupport;
238
+ syntactic: DimensionSupport;
239
+ morphological: DimensionSupport;
240
+ phonological: DimensionSupport;
241
+ };
242
+ /** Human-readable notes about anything skipped or approximate. */
243
+ warnings: string[];
244
+ /** Structural metrics for the user's default gridset (null if unavailable). */
245
+ pageset?: PagesetSummary | null;
246
+ /** System-configuration context (access method, AI opt-in, startup gridset). */
247
+ userSettings?: UserSettingsSummary | null;
248
+ }
249
+ export interface TimelineOptions {
250
+ /** How many trailing months to analyse. Default 12. */
251
+ months?: number;
252
+ /** Moving-average window. Default 30. */
253
+ windowSize?: number;
254
+ /** Language code. Default "en". Used for reporting only. */
255
+ lang?: string;
256
+ /** Months with fewer than this many words are flagged suppressed. Default 150. */
257
+ minWordsPerMonth?: number;
258
+ /** Optional dictionary Set for the spelling measure. */
259
+ dictionary?: Set<string>;
260
+ /** Language-specific resources (closed-class words, inflection classifier). */
261
+ resources?: LanguageResources;
262
+ /** Epoch ms for "now". Defaults to Date.now(). Mainly for tests. */
263
+ now?: number;
264
+ /** Platform label for the report. Default "Grid3". */
265
+ platform?: string;
266
+ userLabel?: string;
267
+ langCode?: string;
268
+ dbPathIncluded?: boolean;
269
+ }
270
+ /**
271
+ * Analyse a corpus of utterances as a longitudinal competence report.
272
+ *
273
+ * Utterances are filtered to the trailing `months` window, binned by calendar
274
+ * month, and each bin is scored on the four competence dimensions plus activity.
275
+ * Language-specific measures require matching `resources`; missing resources are
276
+ * reported under `support` and `warnings` rather than silently dropped.
277
+ *
278
+ * No raw text, word list, or fringe-vocabulary frequency is included in the
279
+ * returned report — only aggregate statistics.
280
+ */
281
+ export declare function analyzeTimeline(utterances: CompetenceUtterance[], options?: TimelineOptions): CompetenceReport;