@willwade/aac-processors 0.3.1 → 0.3.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -5,3 +5,4 @@
5
5
  * This is separate from pageset metrics.
6
6
  */
7
7
  export * from './utilities/analytics/history';
8
+ export * from './utilities/analytics/competence';
package/dist/analytics.js CHANGED
@@ -21,3 +21,4 @@ var __exportStar = (this && this.__exportStar) || function(m, exports) {
21
21
  };
22
22
  Object.defineProperty(exports, "__esModule", { value: true });
23
23
  __exportStar(require("./utilities/analytics/history"), exports);
24
+ __exportStar(require("./utilities/analytics/competence"), exports);
@@ -0,0 +1,510 @@
1
+ /**
2
+ * Linguistic Competence Metrics
3
+ *
4
+ * Privacy-preserving analysis of AAC *spoken output* (phrase history), based on:
5
+ * - Niemeijer, Sheldon & Hillary Zisk (2025), "Measuring AAC user linguistic
6
+ * competence: A novel approach", AssistiveWare (Communication Matters handout).
7
+ * - Frisch, Wade et al. (2026), "It's Complicated: On the Design and Evaluation
8
+ * of AI-Powered AAC Interfaces", arXiv:2606.24854.
9
+ *
10
+ * DESIGN (read me):
11
+ * - **Source-agnostic.** The only input is `{ text, timestampMs }[]`. It does
12
+ * not know or care whether the speech history came from Grid 3, Snap,
13
+ * TouchChat, OBF/OBFL logs, or anything else. See `historyEntriesToCompetence*
14
+ * Utterances` (in history.ts) to adapt any `HistoryEntry[]` source.
15
+ * - **Language-agnostic core.** This module contains NO word lists. Language-
16
+ * specific resources (a closed-class word set, an inflection classifier) are
17
+ * INJECTED via `LanguageResources`. When a resource is missing for a
18
+ * language, the affected measure is reported as `unavailable` with a reason
19
+ * and a warning is raised — never silently wrong.
20
+ * - **Pure / no I/O.** No filesystem, no platform APIs. Runs anywhere (browser
21
+ * included) and emits only aggregate statistics (never the raw text).
22
+ *
23
+ * The four dimensions of linguistic competence (Light, 1989) and the measures we
24
+ * use for each, following the AssistiveWare findings:
25
+ *
26
+ * Semantic -> MATTR-30 lexical diversity (always available)
27
+ * Syntactic -> preposition/conjunction diversity (needs closedClassWords)
28
+ * Morphological -> inflected-form diversity (needs classifyInflection)
29
+ * Phonological -> proportion of unique words in a dictionary (needs a dictionary)
30
+ *
31
+ * All diversity measures use 30-word moving-average windows (Covington & McFall,
32
+ * 2010), making them sample-length independent and usable for the tiny, highly
33
+ * variable samples typical of AAC. MLU is intentionally NOT a headline (it
34
+ * conflates linguistic/operational/strategic/social competence in AAC); it is
35
+ * reported only as a distribution.
36
+ */
37
+ /* ------------------------------------------------------------------ *
38
+ * Small statistics helpers
39
+ * ------------------------------------------------------------------ */
40
+ function mean(values) {
41
+ if (values.length === 0)
42
+ return null;
43
+ let sum = 0;
44
+ for (const v of values)
45
+ sum += v;
46
+ return sum / values.length;
47
+ }
48
+ function median(values) {
49
+ if (values.length === 0)
50
+ return null;
51
+ const sorted = [...values].sort((a, b) => a - b);
52
+ const mid = Math.floor(sorted.length / 2);
53
+ return sorted.length % 2 === 0 ? (sorted[mid - 1] + sorted[mid]) / 2 : sorted[mid];
54
+ }
55
+ function quantile(values, q) {
56
+ if (values.length === 0)
57
+ return null;
58
+ const sorted = [...values].sort((a, b) => a - b);
59
+ const pos = (sorted.length - 1) * q;
60
+ const base = Math.floor(pos);
61
+ const rest = pos - base;
62
+ if (sorted[base + 1] !== undefined) {
63
+ return sorted[base] + rest * (sorted[base + 1] - sorted[base]);
64
+ }
65
+ return sorted[base];
66
+ }
67
+ /* ------------------------------------------------------------------ *
68
+ * Tokenisation
69
+ * ------------------------------------------------------------------ */
70
+ const TOKEN_RE = /[\p{L}\p{N}]+(?:['’][\p{L}\p{N}]+)?/gu;
71
+ /**
72
+ * Tokenise raw text into a lowercased word stream.
73
+ * Keeps intra-word apostrophes (don't, children's) but drops leading/trailing
74
+ * punctuation and pure whitespace. Accented characters are preserved (\p{L}).
75
+ */
76
+ export function tokenize(text) {
77
+ if (!text)
78
+ return [];
79
+ const out = [];
80
+ let m;
81
+ TOKEN_RE.lastIndex = 0;
82
+ while ((m = TOKEN_RE.exec(text)) !== null) {
83
+ let tok = m[0].toLowerCase();
84
+ tok = tok.replace(/^['’]+|['’]+$/g, '');
85
+ if (tok.length > 0)
86
+ out.push(tok);
87
+ }
88
+ return out;
89
+ }
90
+ /* ------------------------------------------------------------------ *
91
+ * Semantic competence: lexical diversity (MATTR-30) — always available
92
+ * ------------------------------------------------------------------ */
93
+ function segmentTTR(segment) {
94
+ if (segment.length === 0)
95
+ return 0;
96
+ return new Set(segment).size / segment.length;
97
+ }
98
+ /**
99
+ * Moving-Average Type-Token Ratio (Covington & McFall, 2010).
100
+ *
101
+ * Slides a fixed-size window across the word stream and computes the TTR for
102
+ * each window. The median across windows is sample-length independent, which is
103
+ * exactly why it is preferred over plain TTR for highly variable AAC samples.
104
+ */
105
+ export function movingAverageTTR(words, windowSize = 30) {
106
+ const n = words.length;
107
+ const w = Math.max(1, Math.floor(windowSize));
108
+ if (n === 0)
109
+ return { median: null, mean: null, nWindows: 0, windowSize: w };
110
+ if (n < w) {
111
+ const v = segmentTTR(words);
112
+ return { median: v, mean: v, nWindows: 1, windowSize: w };
113
+ }
114
+ const values = [];
115
+ for (let i = 0; i <= n - w; i++) {
116
+ values.push(segmentTTR(words.slice(i, i + w)));
117
+ }
118
+ return {
119
+ median: median(values),
120
+ mean: mean(values),
121
+ nWindows: values.length,
122
+ windowSize: w,
123
+ };
124
+ }
125
+ /** Convenience: MATTR-30 lexical diversity (the semantic headline). */
126
+ export function lexicalDiversity(words, windowSize = 30) {
127
+ return movingAverageTTR(words, windowSize);
128
+ }
129
+ /* ------------------------------------------------------------------ *
130
+ * Syntactic competence: closed-class diversity (MA-UPC-TWR-30)
131
+ * ------------------------------------------------------------------ */
132
+ /**
133
+ * Closed-class diversity (generalises AssistiveWare's MA-UPC-TWR-30).
134
+ *
135
+ * For each window, compute the type-token ratio restricted to the supplied
136
+ * closed-class words (prepositions + conjunctions in the original paper). Windows
137
+ * containing none are skipped. The caller supplies the set via
138
+ * `closedClassWords`, so this works for any language without hardcoding here.
139
+ */
140
+ export function syntacticDiversity(words, options = {}) {
141
+ const w = Math.max(1, Math.floor(options.windowSize ?? 30));
142
+ const closed = options.closedClassWords;
143
+ if (!closed || closed.size === 0) {
144
+ return {
145
+ median: null,
146
+ mean: null,
147
+ nWindows: 0,
148
+ windowSize: w,
149
+ unavailable: 'no closed-class word data provided for this language',
150
+ };
151
+ }
152
+ const n = words.length;
153
+ if (n === 0)
154
+ return { median: null, mean: null, nWindows: 0, windowSize: w };
155
+ const values = [];
156
+ const scan = (segment) => {
157
+ const cc = segment.filter((tok) => closed.has(tok));
158
+ if (cc.length > 0) {
159
+ values.push(new Set(cc).size / cc.length);
160
+ }
161
+ };
162
+ if (n < w) {
163
+ scan(words);
164
+ }
165
+ else {
166
+ for (let i = 0; i <= n - w; i++) {
167
+ scan(words.slice(i, i + w));
168
+ }
169
+ }
170
+ if (values.length === 0) {
171
+ return {
172
+ median: null,
173
+ mean: null,
174
+ nWindows: 0,
175
+ windowSize: w,
176
+ unavailable: 'no closed-class words found in the sample',
177
+ };
178
+ }
179
+ return {
180
+ median: median(values),
181
+ mean: mean(values),
182
+ nWindows: values.length,
183
+ windowSize: w,
184
+ };
185
+ }
186
+ /* ------------------------------------------------------------------ *
187
+ * Morphological competence: inflected-form diversity (MA-UMORPH-TLWR-30 proxy)
188
+ * ------------------------------------------------------------------ */
189
+ /**
190
+ * Morphological diversity (proxy for MA-UMORPH-TLWR-30).
191
+ *
192
+ * Within each window, words the supplied classifier marks as inflected (any
193
+ * category other than "base") contribute their surface forms to a type-token
194
+ * ratio. Windows with no inflected words are skipped. The classifier is injected
195
+ * (`classifyInflection`) so the heuristic lives with the caller, per language.
196
+ *
197
+ * Caveat (per AssistiveWare): for symbol-supported AAC, pre-stored morphology
198
+ * buttons ("finished", "is", ...) heavily affect this measure — interpret trends
199
+ * rather than absolutes.
200
+ */
201
+ export function morphologicalDiversity(words, options = {}) {
202
+ const w = Math.max(1, Math.floor(options.windowSize ?? 30));
203
+ const classify = options.classifyInflection;
204
+ if (!classify) {
205
+ return {
206
+ median: null,
207
+ mean: null,
208
+ nWindows: 0,
209
+ windowSize: w,
210
+ unavailable: 'no inflection classifier provided for this language',
211
+ };
212
+ }
213
+ const n = words.length;
214
+ if (n === 0)
215
+ return { median: null, mean: null, nWindows: 0, windowSize: w };
216
+ const values = [];
217
+ const scan = (segment) => {
218
+ const inflected = segment.filter((tok) => classify(tok) !== 'base');
219
+ if (inflected.length > 0) {
220
+ values.push(new Set(inflected).size / inflected.length);
221
+ }
222
+ };
223
+ if (n < w) {
224
+ scan(words);
225
+ }
226
+ else {
227
+ for (let i = 0; i <= n - w; i++) {
228
+ scan(words.slice(i, i + w));
229
+ }
230
+ }
231
+ if (values.length === 0) {
232
+ return {
233
+ median: null,
234
+ mean: null,
235
+ nWindows: 0,
236
+ windowSize: w,
237
+ unavailable: 'no inflected word forms found in the sample',
238
+ };
239
+ }
240
+ return {
241
+ median: median(values),
242
+ mean: mean(values),
243
+ nWindows: values.length,
244
+ windowSize: w,
245
+ };
246
+ }
247
+ /* ------------------------------------------------------------------ *
248
+ * Phonological competence: spelling validity (optional, weak)
249
+ * ------------------------------------------------------------------ */
250
+ /**
251
+ * Proportion of unique alphabetic words present in the supplied dictionary.
252
+ * Unique words only, so it is not skewed by repetition or repeated misspellings.
253
+ * Returns null (unavailable) if no dictionary is provided.
254
+ */
255
+ export function spellingValidity(words, dictionary) {
256
+ if (!dictionary || dictionary.size === 0)
257
+ return null;
258
+ const unique = new Set(words);
259
+ let checked = 0;
260
+ let correct = 0;
261
+ for (const tok of unique) {
262
+ if (!/^\p{L}{2,}$/u.test(tok))
263
+ continue;
264
+ checked++;
265
+ if (dictionary.has(tok))
266
+ correct++;
267
+ }
268
+ return checked === 0 ? null : correct / checked;
269
+ }
270
+ function distribution(values) {
271
+ return {
272
+ median: median(values),
273
+ mean: mean(values),
274
+ p25: quantile(values, 0.25),
275
+ p75: quantile(values, 0.75),
276
+ n: values.length,
277
+ };
278
+ }
279
+ /** Compute engagement/activity stats for a set of utterances. */
280
+ export function summarizeActivity(utterances) {
281
+ const days = new Set();
282
+ let totalWords = 0;
283
+ const wpu = [];
284
+ for (const u of utterances) {
285
+ const toks = tokenize(u.text);
286
+ totalWords += toks.length;
287
+ wpu.push(toks.length);
288
+ const day = Math.floor(u.timestampMs / 86400000);
289
+ days.add(day);
290
+ }
291
+ return {
292
+ utterances: utterances.length,
293
+ words: totalWords,
294
+ uniqueWords: 0,
295
+ activeDays: days.size,
296
+ wordsPerUtterance: distribution(wpu),
297
+ };
298
+ }
299
+ function monthKey(timestampMs) {
300
+ const d = new Date(timestampMs);
301
+ const y = d.getFullYear();
302
+ const m = String(d.getMonth() + 1).padStart(2, '0');
303
+ return `${y}-${m}`;
304
+ }
305
+ function weightedSlope(points) {
306
+ const usable = points.filter((p) => p.y !== null && isFinite(p.y));
307
+ if (usable.length < 2)
308
+ return null;
309
+ let sw = 0;
310
+ let swx = 0;
311
+ let swy = 0;
312
+ let swxx = 0;
313
+ let swxy = 0;
314
+ for (const p of usable) {
315
+ const wgt = Math.max(p.w, 1);
316
+ sw += wgt;
317
+ swx += wgt * p.x;
318
+ swy += wgt * p.y;
319
+ swxx += wgt * p.x * p.x;
320
+ swxy += wgt * p.x * p.y;
321
+ }
322
+ const denom = sw * swxx - swx * swx;
323
+ if (denom === 0)
324
+ return null;
325
+ return (sw * swxy - swx * swy) / denom;
326
+ }
327
+ /**
328
+ * Analyse a corpus of utterances as a longitudinal competence report.
329
+ *
330
+ * Utterances are filtered to the trailing `months` window, binned by calendar
331
+ * month, and each bin is scored on the four competence dimensions plus activity.
332
+ * Language-specific measures require matching `resources`; missing resources are
333
+ * reported under `support` and `warnings` rather than silently dropped.
334
+ *
335
+ * No raw text, word list, or fringe-vocabulary frequency is included in the
336
+ * returned report — only aggregate statistics.
337
+ */
338
+ export function analyzeTimeline(utterances, options = {}) {
339
+ const months = Math.max(1, Math.floor(options.months ?? 12));
340
+ const windowSize = Math.max(1, Math.floor(options.windowSize ?? 30));
341
+ const lang = options.lang ?? 'en';
342
+ const minWordsPerMonth = Math.max(0, Math.floor(options.minWordsPerMonth ?? 150));
343
+ const dictionary = options.dictionary;
344
+ const resources = options.resources ?? {};
345
+ const now = options.now ?? Date.now();
346
+ // ---- Filter to the trailing N months ----------------------------------
347
+ const windowMs = months * 31 * 86400000;
348
+ const windowStartMs = now - windowMs;
349
+ const inWindow = utterances.filter((u) => u.timestampMs <= now && u.timestampMs > windowStartMs && u.text && u.text.trim().length > 0);
350
+ // ---- Bin by month -----------------------------------------------------
351
+ const bins = new Map();
352
+ for (const u of inWindow) {
353
+ const key = monthKey(u.timestampMs);
354
+ const arr = bins.get(key) ?? [];
355
+ arr.push(u);
356
+ bins.set(key, arr);
357
+ }
358
+ const sortedKeys = [...bins.keys()].sort();
359
+ const timeline = [];
360
+ for (const key of sortedKeys) {
361
+ const monthUtts = (bins.get(key) ?? []).slice().sort((a, b) => a.timestampMs - b.timestampMs);
362
+ const stream = [];
363
+ for (const u of monthUtts)
364
+ stream.push(...tokenize(u.text));
365
+ const activity = summarizeActivity(monthUtts);
366
+ activity.uniqueWords = new Set(stream).size;
367
+ const suppressed = stream.length < minWordsPerMonth;
368
+ const suppressReason = suppressed
369
+ ? `fewer than ${minWordsPerMonth} words (${stream.length})`
370
+ : null;
371
+ const lex = suppressed
372
+ ? { median: null, mean: null, nWindows: 0, windowSize }
373
+ : lexicalDiversity(stream, windowSize);
374
+ const syn = suppressed
375
+ ? { median: null, mean: null, nWindows: 0, windowSize }
376
+ : syntacticDiversity(stream, { windowSize, closedClassWords: resources.closedClassWords });
377
+ const mor = suppressed
378
+ ? { median: null, mean: null, nWindows: 0, windowSize }
379
+ : morphologicalDiversity(stream, {
380
+ windowSize,
381
+ classifyInflection: resources.classifyInflection,
382
+ });
383
+ const spell = suppressed ? null : spellingValidity(stream, dictionary);
384
+ timeline.push({
385
+ month: key,
386
+ utterances: activity.utterances,
387
+ words: activity.words,
388
+ uniqueWords: activity.uniqueWords,
389
+ activeDays: activity.activeDays,
390
+ wordsPerUtterance: activity.wordsPerUtterance,
391
+ lexicalDiversity: lex,
392
+ syntacticDiversity: syn,
393
+ morphologicalDiversity: mor,
394
+ spellingValidity: spell,
395
+ suppressed,
396
+ suppressReason,
397
+ });
398
+ }
399
+ // ---- Support + warnings (no silent language degradation) --------------
400
+ const warnings = [];
401
+ const hasCC = !!resources.closedClassWords && resources.closedClassWords.size > 0;
402
+ const hasMorph = !!resources.classifyInflection;
403
+ if (!hasCC) {
404
+ warnings.push(`Syntactic diversity unavailable: no closed-class word data for language '${lang}'. ` +
405
+ `Provide resources.closedClassWords to enable it.`);
406
+ }
407
+ if (!hasMorph) {
408
+ warnings.push(`Morphological diversity unavailable: no inflection classifier for language '${lang}'. ` +
409
+ `Provide resources.classifyInflection to enable it.`);
410
+ }
411
+ if (!dictionary) {
412
+ warnings.push('Spelling validity unavailable: no dictionary provided.');
413
+ }
414
+ const suppCount = timeline.filter((b) => b.suppressed).length;
415
+ if (suppCount > 0) {
416
+ warnings.push(`${suppCount} month(s) suppressed for having fewer than ${minWordsPerMonth} words.`);
417
+ }
418
+ // ---- Trend on the headline lexical-diversity median -------------------
419
+ const points = [];
420
+ for (let i = 0; i < timeline.length; i++) {
421
+ const b = timeline[i];
422
+ if (b.suppressed)
423
+ continue;
424
+ const y = b.lexicalDiversity.median;
425
+ if (y === null)
426
+ continue;
427
+ points.push({ x: i, y, w: b.words });
428
+ }
429
+ const slope = weightedSlope(points);
430
+ const validYs = points.map((p) => p.y);
431
+ const half = Math.floor(validYs.length / 2);
432
+ let firstHalf = null;
433
+ let secondHalf = null;
434
+ if (validYs.length >= 2) {
435
+ const fh = validYs.slice(0, Math.max(1, half));
436
+ const sh = validYs.slice(Math.max(1, half));
437
+ firstHalf = mean(fh);
438
+ secondHalf = mean(sh);
439
+ }
440
+ const delta = firstHalf !== null && secondHalf !== null ? secondHalf - firstHalf : null;
441
+ let direction = 'unknown';
442
+ if (slope !== null) {
443
+ if (Math.abs(slope) < 0.0005)
444
+ direction = 'flat';
445
+ else
446
+ direction = slope > 0 ? 'up' : 'down';
447
+ }
448
+ else if (delta !== null) {
449
+ if (Math.abs(delta) < 0.005)
450
+ direction = 'flat';
451
+ else
452
+ direction = delta > 0 ? 'up' : 'down';
453
+ }
454
+ const totalUtts = timeline.reduce((s, b) => s + b.utterances, 0);
455
+ const totalWords = timeline.reduce((s, b) => s + b.words, 0);
456
+ const dim = (available, reason) => available ? { available: true } : { available: false, reason };
457
+ return {
458
+ schema: 'aac-competence-report/v1',
459
+ generatedAt: new Date(now).toISOString(),
460
+ privacy: {
461
+ rawUtterancesIncluded: false,
462
+ wordListsIncluded: false,
463
+ fringeWordFrequencyIncluded: false,
464
+ minAggregationWindowDays: 31,
465
+ notes: [
466
+ 'All metrics computed locally; only aggregate statistics are emitted.',
467
+ 'Utterances are binned by calendar month so no pattern can be tied to a specific day or time.',
468
+ 'No word lists or fringe-vocabulary frequencies are included (per AssistiveWare privacy guidance).',
469
+ ],
470
+ },
471
+ source: {
472
+ platform: options.platform ?? 'Grid3',
473
+ langCode: options.langCode,
474
+ userLabel: options.userLabel,
475
+ dbPathIncluded: options.dbPathIncluded === true,
476
+ },
477
+ config: {
478
+ months,
479
+ windowSize,
480
+ lang,
481
+ minWordsPerMonth,
482
+ dictionaryProvided: !!dictionary,
483
+ },
484
+ overall: {
485
+ windowStart: sortedKeys[0] ?? monthKey(windowStartMs),
486
+ windowEnd: sortedKeys[sortedKeys.length - 1] ?? monthKey(now),
487
+ totalUtterances: totalUtts,
488
+ totalWords: totalWords,
489
+ monthsCovered: timeline.length,
490
+ monthsSuppressed: suppCount,
491
+ },
492
+ timeline,
493
+ trend: {
494
+ metric: 'lexicalDiversity.median',
495
+ slopePerMonth: slope,
496
+ firstHalf,
497
+ secondHalf,
498
+ delta,
499
+ direction,
500
+ },
501
+ support: {
502
+ lang,
503
+ semantic: dim(true),
504
+ syntactic: dim(hasCC, hasCC ? undefined : 'no closed-class data for this language'),
505
+ morphological: dim(hasMorph, hasMorph ? undefined : 'no inflection classifier for this language'),
506
+ phonological: dim(!!dictionary, dictionary ? undefined : 'no dictionary provided'),
507
+ },
508
+ warnings,
509
+ };
510
+ }
@@ -2,6 +2,28 @@ import { dotNetTicksToDate } from '../../utils/dotnetTicks';
2
2
  import { findGrid3Users, readAllGrid3History as readAllGrid3HistoryImpl, readGrid3History as readGrid3HistoryImpl, readGrid3HistoryForUser as readGrid3HistoryForUserImpl, } from '../../processors/gridset/helpers';
3
3
  import { findSnapUsers, readSnapUsage as readSnapUsageImpl, readSnapUsageForUser as readSnapUsageForUserImpl, } from '../../processors/snap/helpers';
4
4
  export { dotNetTicksToDate };
5
+ /**
6
+ * Adapt any `HistoryEntry[]` (Grid 3, Snap, OBF/OBFL logs, ...) into the generic
7
+ * utterance stream consumed by the linguistic-competence engine
8
+ * (`analyzeTimeline`). Each occurrence of each phrase becomes one utterance,
9
+ * timestamped by its occurrence time. This keeps the competence metrics fully
10
+ * source-agnostic: anything the library can read as history can be analysed.
11
+ */
12
+ export function historyEntriesToCompetenceUtterances(entries) {
13
+ const out = [];
14
+ for (const e of entries) {
15
+ const text = e.content;
16
+ if (!text || text.trim().length === 0)
17
+ continue;
18
+ const occs = e.occurrences ?? [];
19
+ for (const occ of occs) {
20
+ if (!occ.timestamp)
21
+ continue;
22
+ out.push({ text, timestampMs: occ.timestamp.getTime() });
23
+ }
24
+ }
25
+ return out;
26
+ }
5
27
  const generateUuid = () => {
6
28
  if (typeof globalThis.crypto?.randomUUID === 'function') {
7
29
  return globalThis.crypto.randomUUID();