@willwade/aac-processors 0.3.1 → 0.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -126,6 +126,7 @@ export interface AACPage {
126
126
  descriptionHtml?: string;
127
127
  images?: any[];
128
128
  sounds?: any[];
129
+ wordListItems?: AACWordListItem[];
129
130
  semantic_ids?: string[];
130
131
  clone_ids?: string[];
131
132
  scanningConfig?: ScanningConfig;
@@ -0,0 +1,288 @@
1
+ /**
2
+ * Linguistic Competence Metrics
3
+ *
4
+ * Privacy-preserving analysis of AAC *spoken output* (phrase history), based on:
5
+ * - Niemeijer, Sheldon & Hillary Zisk (2025), "Measuring AAC user linguistic
6
+ * competence: A novel approach", AssistiveWare (Communication Matters handout).
7
+ * - Frisch, Wade et al. (2026), "It's Complicated: On the Design and Evaluation
8
+ * of AI-Powered AAC Interfaces", arXiv:2606.24854.
9
+ *
10
+ * DESIGN (read me):
11
+ * - **Source-agnostic.** The only input is `{ text, timestampMs }[]`. It does
12
+ * not know or care whether the speech history came from Grid 3, Snap,
13
+ * TouchChat, OBF/OBFL logs, or anything else. See `historyEntriesToCompetence*
14
+ * Utterances` (in history.ts) to adapt any `HistoryEntry[]` source.
15
+ * - **Language-agnostic core.** This module contains NO word lists. Language-
16
+ * specific resources (a closed-class word set, an inflection classifier) are
17
+ * INJECTED via `LanguageResources`. When a resource is missing for a
18
+ * language, the affected measure is reported as `unavailable` with a reason
19
+ * and a warning is raised — never silently wrong.
20
+ * - **Pure / no I/O.** No filesystem, no platform APIs. Runs anywhere (browser
21
+ * included) and emits only aggregate statistics (never the raw text).
22
+ *
23
+ * The four dimensions of linguistic competence (Light, 1989) and the measures we
24
+ * use for each, following the AssistiveWare findings:
25
+ *
26
+ * Semantic -> MATTR-30 lexical diversity (always available)
27
+ * Syntactic -> preposition/conjunction diversity (needs closedClassWords)
28
+ * Morphological -> inflected-form diversity (needs classifyInflection)
29
+ * Phonological -> proportion of unique words in a dictionary (needs a dictionary)
30
+ *
31
+ * All diversity measures use 30-word moving-average windows (Covington & McFall,
32
+ * 2010), making them sample-length independent and usable for the tiny, highly
33
+ * variable samples typical of AAC. MLU is intentionally NOT a headline (it
34
+ * conflates linguistic/operational/strategic/social competence in AAC); it is
35
+ * reported only as a distribution.
36
+ */
37
+ import type { VocabularySummary } from './metrics/vocabularyDump';
38
+ /** A single spoken utterance with a production timestamp (epoch ms). */
39
+ export interface CompetenceUtterance {
40
+ text: string;
41
+ timestampMs: number;
42
+ }
43
+ /** A tokenised word stream produced in chronological order. */
44
+ export type WordStream = string[];
45
+ /** Coarse inflection category used by the (injected) morphology classifier. */
46
+ export type InflectionCategory = 'base' | 'plural' | 'possessive' | 'past' | 'progressive' | 'comparative' | 'superlative' | 'adverb';
47
+ /**
48
+ * Language-specific resources, injected by the caller. Providing none leaves the
49
+ * language-specific measures unavailable (with explicit warnings) — the semantic
50
+ * measure still works for any language.
51
+ */
52
+ export interface LanguageResources {
53
+ /** Closed-class words (prepositions, conjunctions, ...) for syntactic diversity. */
54
+ closedClassWords?: Set<string>;
55
+ /** Maps a lowercased word to an inflection category for morphological diversity. */
56
+ classifyInflection?: (word: string) => InflectionCategory;
57
+ }
58
+ export interface DiversityOptions {
59
+ /** Moving-average window size in words. The papers use 30. */
60
+ windowSize?: number;
61
+ /** Closed-class word set (for the syntactic measure). */
62
+ closedClassWords?: Set<string>;
63
+ /** Inflection classifier (for the morphological measure). */
64
+ classifyInflection?: (word: string) => InflectionCategory;
65
+ }
66
+ export interface DiversityResult {
67
+ /** Median of the per-window values (the headline figure, per the paper). */
68
+ median: number | null;
69
+ mean: number | null;
70
+ /** Number of windows that contributed a value (after any skipping). */
71
+ nWindows: number;
72
+ /** The window size used. */
73
+ windowSize: number;
74
+ /** Present when the measure could not be computed (e.g. missing language data). */
75
+ unavailable?: string;
76
+ }
77
+ /**
78
+ * Tokenise raw text into a lowercased word stream.
79
+ * Keeps intra-word apostrophes (don't, children's) but drops leading/trailing
80
+ * punctuation and pure whitespace. Accented characters are preserved (\p{L}).
81
+ */
82
+ export declare function tokenize(text: string): WordStream;
83
+ /**
84
+ * Moving-Average Type-Token Ratio (Covington & McFall, 2010).
85
+ *
86
+ * Slides a fixed-size window across the word stream and computes the TTR for
87
+ * each window. The median across windows is sample-length independent, which is
88
+ * exactly why it is preferred over plain TTR for highly variable AAC samples.
89
+ */
90
+ export declare function movingAverageTTR(words: WordStream, windowSize?: number): DiversityResult;
91
+ /** Convenience: MATTR-30 lexical diversity (the semantic headline). */
92
+ export declare function lexicalDiversity(words: WordStream, windowSize?: number): DiversityResult;
93
+ /**
94
+ * Closed-class diversity (generalises AssistiveWare's MA-UPC-TWR-30).
95
+ *
96
+ * For each window, compute the type-token ratio restricted to the supplied
97
+ * closed-class words (prepositions + conjunctions in the original paper). Windows
98
+ * containing none are skipped. The caller supplies the set via
99
+ * `closedClassWords`, so this works for any language without hardcoding here.
100
+ */
101
+ export declare function syntacticDiversity(words: WordStream, options?: DiversityOptions): DiversityResult;
102
+ /**
103
+ * Morphological diversity (proxy for MA-UMORPH-TLWR-30).
104
+ *
105
+ * Within each window, words the supplied classifier marks as inflected (any
106
+ * category other than "base") contribute their surface forms to a type-token
107
+ * ratio. Windows with no inflected words are skipped. The classifier is injected
108
+ * (`classifyInflection`) so the heuristic lives with the caller, per language.
109
+ *
110
+ * Caveat (per AssistiveWare): for symbol-supported AAC, pre-stored morphology
111
+ * buttons ("finished", "is", ...) heavily affect this measure — interpret trends
112
+ * rather than absolutes.
113
+ */
114
+ export declare function morphologicalDiversity(words: WordStream, options?: DiversityOptions): DiversityResult;
115
+ /**
116
+ * Proportion of unique alphabetic words present in the supplied dictionary.
117
+ * Unique words only, so it is not skewed by repetition or repeated misspellings.
118
+ * Returns null (unavailable) if no dictionary is provided.
119
+ */
120
+ export declare function spellingValidity(words: WordStream, dictionary?: Set<string>): number | null;
121
+ export interface DistributionStats {
122
+ median: number | null;
123
+ mean: number | null;
124
+ p25: number | null;
125
+ p75: number | null;
126
+ n: number;
127
+ }
128
+ export interface ActivityStats {
129
+ utterances: number;
130
+ words: number;
131
+ /** Distinct tokens — vocabulary breadth (count only, never the words). */
132
+ uniqueWords: number;
133
+ activeDays: number;
134
+ wordsPerUtterance: DistributionStats;
135
+ }
136
+ /** Compute engagement/activity stats for a set of utterances. */
137
+ export declare function summarizeActivity(utterances: CompetenceUtterance[]): ActivityStats;
138
+ export interface MonthBin {
139
+ /** Calendar month in local time, "YYYY-MM". */
140
+ month: string;
141
+ utterances: number;
142
+ words: number;
143
+ uniqueWords: number;
144
+ activeDays: number;
145
+ wordsPerUtterance: DistributionStats;
146
+ /** Semantic — the headline measure. */
147
+ lexicalDiversity: DiversityResult;
148
+ /** Syntactic. null/unavailable when no closed-class data for the language. */
149
+ syntacticDiversity: DiversityResult;
150
+ /** Morphological (proxy). null/unavailable when no classifier for the language. */
151
+ morphologicalDiversity: DiversityResult;
152
+ /** Phonological — only when a dictionary is supplied. */
153
+ spellingValidity: number | null;
154
+ /** True when the month has too little data to trust the diversity figures. */
155
+ suppressed: boolean;
156
+ suppressReason: string | null;
157
+ }
158
+ export interface TrendResult {
159
+ metric: string;
160
+ /** Slope of the weighted linear regression, in metric units per month. */
161
+ slopePerMonth: number | null;
162
+ firstHalf: number | null;
163
+ secondHalf: number | null;
164
+ /** secondHalf - firstHalf. */
165
+ delta: number | null;
166
+ direction: 'up' | 'down' | 'flat' | 'unknown';
167
+ }
168
+ export interface DimensionSupport {
169
+ available: boolean;
170
+ reason?: string;
171
+ }
172
+ export interface PagesetSummary {
173
+ label: string;
174
+ gridsetIncluded: boolean;
175
+ analysisVersion?: string;
176
+ totalBoards: number;
177
+ totalButtons: number;
178
+ totalWords: number;
179
+ grid: {
180
+ rows: number;
181
+ columns: number;
182
+ };
183
+ effort: DistributionStats;
184
+ hasDynamicPrediction: boolean;
185
+ spellingEffort: {
186
+ base: number | null;
187
+ perLetter: number | null;
188
+ };
189
+ /**
190
+ * Counts-only vocabulary inventory (buttons, wordlists, prediction
191
+ * dictionaries, smart-grammar word forms) — see dumpVocabulary().
192
+ * Null/undefined when the caller did not compute it. Contains counts only.
193
+ */
194
+ vocabulary?: VocabularySummary | null;
195
+ error?: string;
196
+ }
197
+ export interface UserSettingsSummary {
198
+ startupGridSet: string | null;
199
+ onlineAiToolsOptIn: boolean | null;
200
+ accessMethods: string[];
201
+ personalisation: {
202
+ pronunciations: number;
203
+ capitalisations: number;
204
+ abbreviationExpansions: number;
205
+ smallWords: number;
206
+ };
207
+ }
208
+ export interface CompetenceReport {
209
+ schema: string;
210
+ generatedAt: string;
211
+ privacy: {
212
+ rawUtterancesIncluded: boolean;
213
+ wordListsIncluded: boolean;
214
+ fringeWordFrequencyIncluded: boolean;
215
+ minAggregationWindowDays: number;
216
+ notes: string[];
217
+ };
218
+ source: {
219
+ platform: string;
220
+ langCode?: string;
221
+ userLabel?: string;
222
+ dbPathIncluded: boolean;
223
+ };
224
+ config: {
225
+ months: number;
226
+ windowSize: number;
227
+ lang: string;
228
+ minWordsPerMonth: number;
229
+ dictionaryProvided: boolean;
230
+ };
231
+ overall: {
232
+ windowStart: string;
233
+ windowEnd: string;
234
+ totalUtterances: number;
235
+ totalWords: number;
236
+ monthsCovered: number;
237
+ monthsSuppressed: number;
238
+ };
239
+ timeline: MonthBin[];
240
+ trend: TrendResult;
241
+ /** Per-dimension availability for the detected language (no silent degradation). */
242
+ support: {
243
+ lang: string;
244
+ semantic: DimensionSupport;
245
+ syntactic: DimensionSupport;
246
+ morphological: DimensionSupport;
247
+ phonological: DimensionSupport;
248
+ };
249
+ /** Human-readable notes about anything skipped or approximate. */
250
+ warnings: string[];
251
+ /** Structural metrics for the user's default gridset (null if unavailable). */
252
+ pageset?: PagesetSummary | null;
253
+ /** System-configuration context (access method, AI opt-in, startup gridset). */
254
+ userSettings?: UserSettingsSummary | null;
255
+ }
256
+ export interface TimelineOptions {
257
+ /** How many trailing months to analyse. Default 12. */
258
+ months?: number;
259
+ /** Moving-average window. Default 30. */
260
+ windowSize?: number;
261
+ /** Language code. Default "en". Used for reporting only. */
262
+ lang?: string;
263
+ /** Months with fewer than this many words are flagged suppressed. Default 150. */
264
+ minWordsPerMonth?: number;
265
+ /** Optional dictionary Set for the spelling measure. */
266
+ dictionary?: Set<string>;
267
+ /** Language-specific resources (closed-class words, inflection classifier). */
268
+ resources?: LanguageResources;
269
+ /** Epoch ms for "now". Defaults to Date.now(). Mainly for tests. */
270
+ now?: number;
271
+ /** Platform label for the report. Default "Grid3". */
272
+ platform?: string;
273
+ userLabel?: string;
274
+ langCode?: string;
275
+ dbPathIncluded?: boolean;
276
+ }
277
+ /**
278
+ * Analyse a corpus of utterances as a longitudinal competence report.
279
+ *
280
+ * Utterances are filtered to the trailing `months` window, binned by calendar
281
+ * month, and each bin is scored on the four competence dimensions plus activity.
282
+ * Language-specific measures require matching `resources`; missing resources are
283
+ * reported under `support` and `warnings` rather than silently dropped.
284
+ *
285
+ * No raw text, word list, or fringe-vocabulary frequency is included in the
286
+ * returned report — only aggregate statistics.
287
+ */
288
+ export declare function analyzeTimeline(utterances: CompetenceUtterance[], options?: TimelineOptions): CompetenceReport;