@willwade/aac-processors 0.3.1 → 0.3.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/analytics.d.ts +1 -0
- package/dist/analytics.js +1 -0
- package/dist/browser/utilities/analytics/competence.js +510 -0
- package/dist/browser/utilities/analytics/history.js +22 -0
- package/dist/utilities/analytics/competence.d.ts +281 -0
- package/dist/utilities/analytics/competence.js +520 -0
- package/dist/utilities/analytics/history.d.ts +9 -0
- package/dist/utilities/analytics/history.js +23 -0
- package/dist/utilities/analytics/index.d.ts +1 -0
- package/dist/utilities/analytics/index.js +2 -0
- package/package.json +1 -1
|
@@ -0,0 +1,520 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* Linguistic Competence Metrics
|
|
4
|
+
*
|
|
5
|
+
* Privacy-preserving analysis of AAC *spoken output* (phrase history), based on:
|
|
6
|
+
* - Niemeijer, Sheldon & Hillary Zisk (2025), "Measuring AAC user linguistic
|
|
7
|
+
* competence: A novel approach", AssistiveWare (Communication Matters handout).
|
|
8
|
+
* - Frisch, Wade et al. (2026), "It's Complicated: On the Design and Evaluation
|
|
9
|
+
* of AI-Powered AAC Interfaces", arXiv:2606.24854.
|
|
10
|
+
*
|
|
11
|
+
* DESIGN (read me):
|
|
12
|
+
* - **Source-agnostic.** The only input is `{ text, timestampMs }[]`. It does
|
|
13
|
+
* not know or care whether the speech history came from Grid 3, Snap,
|
|
14
|
+
* TouchChat, OBF/OBFL logs, or anything else. See `historyEntriesToCompetence*
|
|
15
|
+
* Utterances` (in history.ts) to adapt any `HistoryEntry[]` source.
|
|
16
|
+
* - **Language-agnostic core.** This module contains NO word lists. Language-
|
|
17
|
+
* specific resources (a closed-class word set, an inflection classifier) are
|
|
18
|
+
* INJECTED via `LanguageResources`. When a resource is missing for a
|
|
19
|
+
* language, the affected measure is reported as `unavailable` with a reason
|
|
20
|
+
* and a warning is raised — never silently wrong.
|
|
21
|
+
* - **Pure / no I/O.** No filesystem, no platform APIs. Runs anywhere (browser
|
|
22
|
+
* included) and emits only aggregate statistics (never the raw text).
|
|
23
|
+
*
|
|
24
|
+
* The four dimensions of linguistic competence (Light, 1989) and the measures we
|
|
25
|
+
* use for each, following the AssistiveWare findings:
|
|
26
|
+
*
|
|
27
|
+
* Semantic -> MATTR-30 lexical diversity (always available)
|
|
28
|
+
* Syntactic -> preposition/conjunction diversity (needs closedClassWords)
|
|
29
|
+
* Morphological -> inflected-form diversity (needs classifyInflection)
|
|
30
|
+
* Phonological -> proportion of unique words in a dictionary (needs a dictionary)
|
|
31
|
+
*
|
|
32
|
+
* All diversity measures use 30-word moving-average windows (Covington & McFall,
|
|
33
|
+
* 2010), making them sample-length independent and usable for the tiny, highly
|
|
34
|
+
* variable samples typical of AAC. MLU is intentionally NOT a headline (it
|
|
35
|
+
* conflates linguistic/operational/strategic/social competence in AAC); it is
|
|
36
|
+
* reported only as a distribution.
|
|
37
|
+
*/
|
|
38
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
39
|
+
exports.tokenize = tokenize;
|
|
40
|
+
exports.movingAverageTTR = movingAverageTTR;
|
|
41
|
+
exports.lexicalDiversity = lexicalDiversity;
|
|
42
|
+
exports.syntacticDiversity = syntacticDiversity;
|
|
43
|
+
exports.morphologicalDiversity = morphologicalDiversity;
|
|
44
|
+
exports.spellingValidity = spellingValidity;
|
|
45
|
+
exports.summarizeActivity = summarizeActivity;
|
|
46
|
+
exports.analyzeTimeline = analyzeTimeline;
|
|
47
|
+
/* ------------------------------------------------------------------ *
|
|
48
|
+
* Small statistics helpers
|
|
49
|
+
* ------------------------------------------------------------------ */
|
|
50
|
+
function mean(values) {
|
|
51
|
+
if (values.length === 0)
|
|
52
|
+
return null;
|
|
53
|
+
let sum = 0;
|
|
54
|
+
for (const v of values)
|
|
55
|
+
sum += v;
|
|
56
|
+
return sum / values.length;
|
|
57
|
+
}
|
|
58
|
+
function median(values) {
|
|
59
|
+
if (values.length === 0)
|
|
60
|
+
return null;
|
|
61
|
+
const sorted = [...values].sort((a, b) => a - b);
|
|
62
|
+
const mid = Math.floor(sorted.length / 2);
|
|
63
|
+
return sorted.length % 2 === 0 ? (sorted[mid - 1] + sorted[mid]) / 2 : sorted[mid];
|
|
64
|
+
}
|
|
65
|
+
function quantile(values, q) {
|
|
66
|
+
if (values.length === 0)
|
|
67
|
+
return null;
|
|
68
|
+
const sorted = [...values].sort((a, b) => a - b);
|
|
69
|
+
const pos = (sorted.length - 1) * q;
|
|
70
|
+
const base = Math.floor(pos);
|
|
71
|
+
const rest = pos - base;
|
|
72
|
+
if (sorted[base + 1] !== undefined) {
|
|
73
|
+
return sorted[base] + rest * (sorted[base + 1] - sorted[base]);
|
|
74
|
+
}
|
|
75
|
+
return sorted[base];
|
|
76
|
+
}
|
|
77
|
+
/* ------------------------------------------------------------------ *
|
|
78
|
+
* Tokenisation
|
|
79
|
+
* ------------------------------------------------------------------ */
|
|
80
|
+
const TOKEN_RE = /[\p{L}\p{N}]+(?:['’][\p{L}\p{N}]+)?/gu;
|
|
81
|
+
/**
|
|
82
|
+
* Tokenise raw text into a lowercased word stream.
|
|
83
|
+
* Keeps intra-word apostrophes (don't, children's) but drops leading/trailing
|
|
84
|
+
* punctuation and pure whitespace. Accented characters are preserved (\p{L}).
|
|
85
|
+
*/
|
|
86
|
+
function tokenize(text) {
|
|
87
|
+
if (!text)
|
|
88
|
+
return [];
|
|
89
|
+
const out = [];
|
|
90
|
+
let m;
|
|
91
|
+
TOKEN_RE.lastIndex = 0;
|
|
92
|
+
while ((m = TOKEN_RE.exec(text)) !== null) {
|
|
93
|
+
let tok = m[0].toLowerCase();
|
|
94
|
+
tok = tok.replace(/^['’]+|['’]+$/g, '');
|
|
95
|
+
if (tok.length > 0)
|
|
96
|
+
out.push(tok);
|
|
97
|
+
}
|
|
98
|
+
return out;
|
|
99
|
+
}
|
|
100
|
+
/* ------------------------------------------------------------------ *
|
|
101
|
+
* Semantic competence: lexical diversity (MATTR-30) — always available
|
|
102
|
+
* ------------------------------------------------------------------ */
|
|
103
|
+
function segmentTTR(segment) {
|
|
104
|
+
if (segment.length === 0)
|
|
105
|
+
return 0;
|
|
106
|
+
return new Set(segment).size / segment.length;
|
|
107
|
+
}
|
|
108
|
+
/**
|
|
109
|
+
* Moving-Average Type-Token Ratio (Covington & McFall, 2010).
|
|
110
|
+
*
|
|
111
|
+
* Slides a fixed-size window across the word stream and computes the TTR for
|
|
112
|
+
* each window. The median across windows is sample-length independent, which is
|
|
113
|
+
* exactly why it is preferred over plain TTR for highly variable AAC samples.
|
|
114
|
+
*/
|
|
115
|
+
function movingAverageTTR(words, windowSize = 30) {
|
|
116
|
+
const n = words.length;
|
|
117
|
+
const w = Math.max(1, Math.floor(windowSize));
|
|
118
|
+
if (n === 0)
|
|
119
|
+
return { median: null, mean: null, nWindows: 0, windowSize: w };
|
|
120
|
+
if (n < w) {
|
|
121
|
+
const v = segmentTTR(words);
|
|
122
|
+
return { median: v, mean: v, nWindows: 1, windowSize: w };
|
|
123
|
+
}
|
|
124
|
+
const values = [];
|
|
125
|
+
for (let i = 0; i <= n - w; i++) {
|
|
126
|
+
values.push(segmentTTR(words.slice(i, i + w)));
|
|
127
|
+
}
|
|
128
|
+
return {
|
|
129
|
+
median: median(values),
|
|
130
|
+
mean: mean(values),
|
|
131
|
+
nWindows: values.length,
|
|
132
|
+
windowSize: w,
|
|
133
|
+
};
|
|
134
|
+
}
|
|
135
|
+
/** Convenience: MATTR-30 lexical diversity (the semantic headline). */
|
|
136
|
+
function lexicalDiversity(words, windowSize = 30) {
|
|
137
|
+
return movingAverageTTR(words, windowSize);
|
|
138
|
+
}
|
|
139
|
+
/* ------------------------------------------------------------------ *
|
|
140
|
+
* Syntactic competence: closed-class diversity (MA-UPC-TWR-30)
|
|
141
|
+
* ------------------------------------------------------------------ */
|
|
142
|
+
/**
|
|
143
|
+
* Closed-class diversity (generalises AssistiveWare's MA-UPC-TWR-30).
|
|
144
|
+
*
|
|
145
|
+
* For each window, compute the type-token ratio restricted to the supplied
|
|
146
|
+
* closed-class words (prepositions + conjunctions in the original paper). Windows
|
|
147
|
+
* containing none are skipped. The caller supplies the set via
|
|
148
|
+
* `closedClassWords`, so this works for any language without hardcoding here.
|
|
149
|
+
*/
|
|
150
|
+
function syntacticDiversity(words, options = {}) {
|
|
151
|
+
const w = Math.max(1, Math.floor(options.windowSize ?? 30));
|
|
152
|
+
const closed = options.closedClassWords;
|
|
153
|
+
if (!closed || closed.size === 0) {
|
|
154
|
+
return {
|
|
155
|
+
median: null,
|
|
156
|
+
mean: null,
|
|
157
|
+
nWindows: 0,
|
|
158
|
+
windowSize: w,
|
|
159
|
+
unavailable: 'no closed-class word data provided for this language',
|
|
160
|
+
};
|
|
161
|
+
}
|
|
162
|
+
const n = words.length;
|
|
163
|
+
if (n === 0)
|
|
164
|
+
return { median: null, mean: null, nWindows: 0, windowSize: w };
|
|
165
|
+
const values = [];
|
|
166
|
+
const scan = (segment) => {
|
|
167
|
+
const cc = segment.filter((tok) => closed.has(tok));
|
|
168
|
+
if (cc.length > 0) {
|
|
169
|
+
values.push(new Set(cc).size / cc.length);
|
|
170
|
+
}
|
|
171
|
+
};
|
|
172
|
+
if (n < w) {
|
|
173
|
+
scan(words);
|
|
174
|
+
}
|
|
175
|
+
else {
|
|
176
|
+
for (let i = 0; i <= n - w; i++) {
|
|
177
|
+
scan(words.slice(i, i + w));
|
|
178
|
+
}
|
|
179
|
+
}
|
|
180
|
+
if (values.length === 0) {
|
|
181
|
+
return {
|
|
182
|
+
median: null,
|
|
183
|
+
mean: null,
|
|
184
|
+
nWindows: 0,
|
|
185
|
+
windowSize: w,
|
|
186
|
+
unavailable: 'no closed-class words found in the sample',
|
|
187
|
+
};
|
|
188
|
+
}
|
|
189
|
+
return {
|
|
190
|
+
median: median(values),
|
|
191
|
+
mean: mean(values),
|
|
192
|
+
nWindows: values.length,
|
|
193
|
+
windowSize: w,
|
|
194
|
+
};
|
|
195
|
+
}
|
|
196
|
+
/* ------------------------------------------------------------------ *
|
|
197
|
+
* Morphological competence: inflected-form diversity (MA-UMORPH-TLWR-30 proxy)
|
|
198
|
+
* ------------------------------------------------------------------ */
|
|
199
|
+
/**
|
|
200
|
+
* Morphological diversity (proxy for MA-UMORPH-TLWR-30).
|
|
201
|
+
*
|
|
202
|
+
* Within each window, words the supplied classifier marks as inflected (any
|
|
203
|
+
* category other than "base") contribute their surface forms to a type-token
|
|
204
|
+
* ratio. Windows with no inflected words are skipped. The classifier is injected
|
|
205
|
+
* (`classifyInflection`) so the heuristic lives with the caller, per language.
|
|
206
|
+
*
|
|
207
|
+
* Caveat (per AssistiveWare): for symbol-supported AAC, pre-stored morphology
|
|
208
|
+
* buttons ("finished", "is", ...) heavily affect this measure — interpret trends
|
|
209
|
+
* rather than absolutes.
|
|
210
|
+
*/
|
|
211
|
+
function morphologicalDiversity(words, options = {}) {
|
|
212
|
+
const w = Math.max(1, Math.floor(options.windowSize ?? 30));
|
|
213
|
+
const classify = options.classifyInflection;
|
|
214
|
+
if (!classify) {
|
|
215
|
+
return {
|
|
216
|
+
median: null,
|
|
217
|
+
mean: null,
|
|
218
|
+
nWindows: 0,
|
|
219
|
+
windowSize: w,
|
|
220
|
+
unavailable: 'no inflection classifier provided for this language',
|
|
221
|
+
};
|
|
222
|
+
}
|
|
223
|
+
const n = words.length;
|
|
224
|
+
if (n === 0)
|
|
225
|
+
return { median: null, mean: null, nWindows: 0, windowSize: w };
|
|
226
|
+
const values = [];
|
|
227
|
+
const scan = (segment) => {
|
|
228
|
+
const inflected = segment.filter((tok) => classify(tok) !== 'base');
|
|
229
|
+
if (inflected.length > 0) {
|
|
230
|
+
values.push(new Set(inflected).size / inflected.length);
|
|
231
|
+
}
|
|
232
|
+
};
|
|
233
|
+
if (n < w) {
|
|
234
|
+
scan(words);
|
|
235
|
+
}
|
|
236
|
+
else {
|
|
237
|
+
for (let i = 0; i <= n - w; i++) {
|
|
238
|
+
scan(words.slice(i, i + w));
|
|
239
|
+
}
|
|
240
|
+
}
|
|
241
|
+
if (values.length === 0) {
|
|
242
|
+
return {
|
|
243
|
+
median: null,
|
|
244
|
+
mean: null,
|
|
245
|
+
nWindows: 0,
|
|
246
|
+
windowSize: w,
|
|
247
|
+
unavailable: 'no inflected word forms found in the sample',
|
|
248
|
+
};
|
|
249
|
+
}
|
|
250
|
+
return {
|
|
251
|
+
median: median(values),
|
|
252
|
+
mean: mean(values),
|
|
253
|
+
nWindows: values.length,
|
|
254
|
+
windowSize: w,
|
|
255
|
+
};
|
|
256
|
+
}
|
|
257
|
+
/* ------------------------------------------------------------------ *
|
|
258
|
+
* Phonological competence: spelling validity (optional, weak)
|
|
259
|
+
* ------------------------------------------------------------------ */
|
|
260
|
+
/**
|
|
261
|
+
* Proportion of unique alphabetic words present in the supplied dictionary.
|
|
262
|
+
* Unique words only, so it is not skewed by repetition or repeated misspellings.
|
|
263
|
+
* Returns null (unavailable) if no dictionary is provided.
|
|
264
|
+
*/
|
|
265
|
+
function spellingValidity(words, dictionary) {
|
|
266
|
+
if (!dictionary || dictionary.size === 0)
|
|
267
|
+
return null;
|
|
268
|
+
const unique = new Set(words);
|
|
269
|
+
let checked = 0;
|
|
270
|
+
let correct = 0;
|
|
271
|
+
for (const tok of unique) {
|
|
272
|
+
if (!/^\p{L}{2,}$/u.test(tok))
|
|
273
|
+
continue;
|
|
274
|
+
checked++;
|
|
275
|
+
if (dictionary.has(tok))
|
|
276
|
+
correct++;
|
|
277
|
+
}
|
|
278
|
+
return checked === 0 ? null : correct / checked;
|
|
279
|
+
}
|
|
280
|
+
function distribution(values) {
|
|
281
|
+
return {
|
|
282
|
+
median: median(values),
|
|
283
|
+
mean: mean(values),
|
|
284
|
+
p25: quantile(values, 0.25),
|
|
285
|
+
p75: quantile(values, 0.75),
|
|
286
|
+
n: values.length,
|
|
287
|
+
};
|
|
288
|
+
}
|
|
289
|
+
/** Compute engagement/activity stats for a set of utterances. */
|
|
290
|
+
function summarizeActivity(utterances) {
|
|
291
|
+
const days = new Set();
|
|
292
|
+
let totalWords = 0;
|
|
293
|
+
const wpu = [];
|
|
294
|
+
for (const u of utterances) {
|
|
295
|
+
const toks = tokenize(u.text);
|
|
296
|
+
totalWords += toks.length;
|
|
297
|
+
wpu.push(toks.length);
|
|
298
|
+
const day = Math.floor(u.timestampMs / 86400000);
|
|
299
|
+
days.add(day);
|
|
300
|
+
}
|
|
301
|
+
return {
|
|
302
|
+
utterances: utterances.length,
|
|
303
|
+
words: totalWords,
|
|
304
|
+
uniqueWords: 0,
|
|
305
|
+
activeDays: days.size,
|
|
306
|
+
wordsPerUtterance: distribution(wpu),
|
|
307
|
+
};
|
|
308
|
+
}
|
|
309
|
+
function monthKey(timestampMs) {
|
|
310
|
+
const d = new Date(timestampMs);
|
|
311
|
+
const y = d.getFullYear();
|
|
312
|
+
const m = String(d.getMonth() + 1).padStart(2, '0');
|
|
313
|
+
return `${y}-${m}`;
|
|
314
|
+
}
|
|
315
|
+
function weightedSlope(points) {
|
|
316
|
+
const usable = points.filter((p) => p.y !== null && isFinite(p.y));
|
|
317
|
+
if (usable.length < 2)
|
|
318
|
+
return null;
|
|
319
|
+
let sw = 0;
|
|
320
|
+
let swx = 0;
|
|
321
|
+
let swy = 0;
|
|
322
|
+
let swxx = 0;
|
|
323
|
+
let swxy = 0;
|
|
324
|
+
for (const p of usable) {
|
|
325
|
+
const wgt = Math.max(p.w, 1);
|
|
326
|
+
sw += wgt;
|
|
327
|
+
swx += wgt * p.x;
|
|
328
|
+
swy += wgt * p.y;
|
|
329
|
+
swxx += wgt * p.x * p.x;
|
|
330
|
+
swxy += wgt * p.x * p.y;
|
|
331
|
+
}
|
|
332
|
+
const denom = sw * swxx - swx * swx;
|
|
333
|
+
if (denom === 0)
|
|
334
|
+
return null;
|
|
335
|
+
return (sw * swxy - swx * swy) / denom;
|
|
336
|
+
}
|
|
337
|
+
/**
|
|
338
|
+
* Analyse a corpus of utterances as a longitudinal competence report.
|
|
339
|
+
*
|
|
340
|
+
* Utterances are filtered to the trailing `months` window, binned by calendar
|
|
341
|
+
* month, and each bin is scored on the four competence dimensions plus activity.
|
|
342
|
+
* Language-specific measures require matching `resources`; missing resources are
|
|
343
|
+
* reported under `support` and `warnings` rather than silently dropped.
|
|
344
|
+
*
|
|
345
|
+
* No raw text, word list, or fringe-vocabulary frequency is included in the
|
|
346
|
+
* returned report — only aggregate statistics.
|
|
347
|
+
*/
|
|
348
|
+
function analyzeTimeline(utterances, options = {}) {
|
|
349
|
+
const months = Math.max(1, Math.floor(options.months ?? 12));
|
|
350
|
+
const windowSize = Math.max(1, Math.floor(options.windowSize ?? 30));
|
|
351
|
+
const lang = options.lang ?? 'en';
|
|
352
|
+
const minWordsPerMonth = Math.max(0, Math.floor(options.minWordsPerMonth ?? 150));
|
|
353
|
+
const dictionary = options.dictionary;
|
|
354
|
+
const resources = options.resources ?? {};
|
|
355
|
+
const now = options.now ?? Date.now();
|
|
356
|
+
// ---- Filter to the trailing N months ----------------------------------
|
|
357
|
+
const windowMs = months * 31 * 86400000;
|
|
358
|
+
const windowStartMs = now - windowMs;
|
|
359
|
+
const inWindow = utterances.filter((u) => u.timestampMs <= now && u.timestampMs > windowStartMs && u.text && u.text.trim().length > 0);
|
|
360
|
+
// ---- Bin by month -----------------------------------------------------
|
|
361
|
+
const bins = new Map();
|
|
362
|
+
for (const u of inWindow) {
|
|
363
|
+
const key = monthKey(u.timestampMs);
|
|
364
|
+
const arr = bins.get(key) ?? [];
|
|
365
|
+
arr.push(u);
|
|
366
|
+
bins.set(key, arr);
|
|
367
|
+
}
|
|
368
|
+
const sortedKeys = [...bins.keys()].sort();
|
|
369
|
+
const timeline = [];
|
|
370
|
+
for (const key of sortedKeys) {
|
|
371
|
+
const monthUtts = (bins.get(key) ?? []).slice().sort((a, b) => a.timestampMs - b.timestampMs);
|
|
372
|
+
const stream = [];
|
|
373
|
+
for (const u of monthUtts)
|
|
374
|
+
stream.push(...tokenize(u.text));
|
|
375
|
+
const activity = summarizeActivity(monthUtts);
|
|
376
|
+
activity.uniqueWords = new Set(stream).size;
|
|
377
|
+
const suppressed = stream.length < minWordsPerMonth;
|
|
378
|
+
const suppressReason = suppressed
|
|
379
|
+
? `fewer than ${minWordsPerMonth} words (${stream.length})`
|
|
380
|
+
: null;
|
|
381
|
+
const lex = suppressed
|
|
382
|
+
? { median: null, mean: null, nWindows: 0, windowSize }
|
|
383
|
+
: lexicalDiversity(stream, windowSize);
|
|
384
|
+
const syn = suppressed
|
|
385
|
+
? { median: null, mean: null, nWindows: 0, windowSize }
|
|
386
|
+
: syntacticDiversity(stream, { windowSize, closedClassWords: resources.closedClassWords });
|
|
387
|
+
const mor = suppressed
|
|
388
|
+
? { median: null, mean: null, nWindows: 0, windowSize }
|
|
389
|
+
: morphologicalDiversity(stream, {
|
|
390
|
+
windowSize,
|
|
391
|
+
classifyInflection: resources.classifyInflection,
|
|
392
|
+
});
|
|
393
|
+
const spell = suppressed ? null : spellingValidity(stream, dictionary);
|
|
394
|
+
timeline.push({
|
|
395
|
+
month: key,
|
|
396
|
+
utterances: activity.utterances,
|
|
397
|
+
words: activity.words,
|
|
398
|
+
uniqueWords: activity.uniqueWords,
|
|
399
|
+
activeDays: activity.activeDays,
|
|
400
|
+
wordsPerUtterance: activity.wordsPerUtterance,
|
|
401
|
+
lexicalDiversity: lex,
|
|
402
|
+
syntacticDiversity: syn,
|
|
403
|
+
morphologicalDiversity: mor,
|
|
404
|
+
spellingValidity: spell,
|
|
405
|
+
suppressed,
|
|
406
|
+
suppressReason,
|
|
407
|
+
});
|
|
408
|
+
}
|
|
409
|
+
// ---- Support + warnings (no silent language degradation) --------------
|
|
410
|
+
const warnings = [];
|
|
411
|
+
const hasCC = !!resources.closedClassWords && resources.closedClassWords.size > 0;
|
|
412
|
+
const hasMorph = !!resources.classifyInflection;
|
|
413
|
+
if (!hasCC) {
|
|
414
|
+
warnings.push(`Syntactic diversity unavailable: no closed-class word data for language '${lang}'. ` +
|
|
415
|
+
`Provide resources.closedClassWords to enable it.`);
|
|
416
|
+
}
|
|
417
|
+
if (!hasMorph) {
|
|
418
|
+
warnings.push(`Morphological diversity unavailable: no inflection classifier for language '${lang}'. ` +
|
|
419
|
+
`Provide resources.classifyInflection to enable it.`);
|
|
420
|
+
}
|
|
421
|
+
if (!dictionary) {
|
|
422
|
+
warnings.push('Spelling validity unavailable: no dictionary provided.');
|
|
423
|
+
}
|
|
424
|
+
const suppCount = timeline.filter((b) => b.suppressed).length;
|
|
425
|
+
if (suppCount > 0) {
|
|
426
|
+
warnings.push(`${suppCount} month(s) suppressed for having fewer than ${minWordsPerMonth} words.`);
|
|
427
|
+
}
|
|
428
|
+
// ---- Trend on the headline lexical-diversity median -------------------
|
|
429
|
+
const points = [];
|
|
430
|
+
for (let i = 0; i < timeline.length; i++) {
|
|
431
|
+
const b = timeline[i];
|
|
432
|
+
if (b.suppressed)
|
|
433
|
+
continue;
|
|
434
|
+
const y = b.lexicalDiversity.median;
|
|
435
|
+
if (y === null)
|
|
436
|
+
continue;
|
|
437
|
+
points.push({ x: i, y, w: b.words });
|
|
438
|
+
}
|
|
439
|
+
const slope = weightedSlope(points);
|
|
440
|
+
const validYs = points.map((p) => p.y);
|
|
441
|
+
const half = Math.floor(validYs.length / 2);
|
|
442
|
+
let firstHalf = null;
|
|
443
|
+
let secondHalf = null;
|
|
444
|
+
if (validYs.length >= 2) {
|
|
445
|
+
const fh = validYs.slice(0, Math.max(1, half));
|
|
446
|
+
const sh = validYs.slice(Math.max(1, half));
|
|
447
|
+
firstHalf = mean(fh);
|
|
448
|
+
secondHalf = mean(sh);
|
|
449
|
+
}
|
|
450
|
+
const delta = firstHalf !== null && secondHalf !== null ? secondHalf - firstHalf : null;
|
|
451
|
+
let direction = 'unknown';
|
|
452
|
+
if (slope !== null) {
|
|
453
|
+
if (Math.abs(slope) < 0.0005)
|
|
454
|
+
direction = 'flat';
|
|
455
|
+
else
|
|
456
|
+
direction = slope > 0 ? 'up' : 'down';
|
|
457
|
+
}
|
|
458
|
+
else if (delta !== null) {
|
|
459
|
+
if (Math.abs(delta) < 0.005)
|
|
460
|
+
direction = 'flat';
|
|
461
|
+
else
|
|
462
|
+
direction = delta > 0 ? 'up' : 'down';
|
|
463
|
+
}
|
|
464
|
+
const totalUtts = timeline.reduce((s, b) => s + b.utterances, 0);
|
|
465
|
+
const totalWords = timeline.reduce((s, b) => s + b.words, 0);
|
|
466
|
+
const dim = (available, reason) => available ? { available: true } : { available: false, reason };
|
|
467
|
+
return {
|
|
468
|
+
schema: 'aac-competence-report/v1',
|
|
469
|
+
generatedAt: new Date(now).toISOString(),
|
|
470
|
+
privacy: {
|
|
471
|
+
rawUtterancesIncluded: false,
|
|
472
|
+
wordListsIncluded: false,
|
|
473
|
+
fringeWordFrequencyIncluded: false,
|
|
474
|
+
minAggregationWindowDays: 31,
|
|
475
|
+
notes: [
|
|
476
|
+
'All metrics computed locally; only aggregate statistics are emitted.',
|
|
477
|
+
'Utterances are binned by calendar month so no pattern can be tied to a specific day or time.',
|
|
478
|
+
'No word lists or fringe-vocabulary frequencies are included (per AssistiveWare privacy guidance).',
|
|
479
|
+
],
|
|
480
|
+
},
|
|
481
|
+
source: {
|
|
482
|
+
platform: options.platform ?? 'Grid3',
|
|
483
|
+
langCode: options.langCode,
|
|
484
|
+
userLabel: options.userLabel,
|
|
485
|
+
dbPathIncluded: options.dbPathIncluded === true,
|
|
486
|
+
},
|
|
487
|
+
config: {
|
|
488
|
+
months,
|
|
489
|
+
windowSize,
|
|
490
|
+
lang,
|
|
491
|
+
minWordsPerMonth,
|
|
492
|
+
dictionaryProvided: !!dictionary,
|
|
493
|
+
},
|
|
494
|
+
overall: {
|
|
495
|
+
windowStart: sortedKeys[0] ?? monthKey(windowStartMs),
|
|
496
|
+
windowEnd: sortedKeys[sortedKeys.length - 1] ?? monthKey(now),
|
|
497
|
+
totalUtterances: totalUtts,
|
|
498
|
+
totalWords: totalWords,
|
|
499
|
+
monthsCovered: timeline.length,
|
|
500
|
+
monthsSuppressed: suppCount,
|
|
501
|
+
},
|
|
502
|
+
timeline,
|
|
503
|
+
trend: {
|
|
504
|
+
metric: 'lexicalDiversity.median',
|
|
505
|
+
slopePerMonth: slope,
|
|
506
|
+
firstHalf,
|
|
507
|
+
secondHalf,
|
|
508
|
+
delta,
|
|
509
|
+
direction,
|
|
510
|
+
},
|
|
511
|
+
support: {
|
|
512
|
+
lang,
|
|
513
|
+
semantic: dim(true),
|
|
514
|
+
syntactic: dim(hasCC, hasCC ? undefined : 'no closed-class data for this language'),
|
|
515
|
+
morphological: dim(hasMorph, hasMorph ? undefined : 'no inflection classifier for this language'),
|
|
516
|
+
phonological: dim(!!dictionary, dictionary ? undefined : 'no dictionary provided'),
|
|
517
|
+
},
|
|
518
|
+
warnings,
|
|
519
|
+
};
|
|
520
|
+
}
|
|
@@ -2,6 +2,7 @@ import { dotNetTicksToDate } from '../../utils/dotnetTicks';
|
|
|
2
2
|
import { Grid3UserPath } from '../../processors/gridset/helpers';
|
|
3
3
|
import { SnapUserInfo } from '../../processors/snap/helpers';
|
|
4
4
|
import { AACSemanticCategory, AACSemanticIntent } from '../../core/treeStructure';
|
|
5
|
+
import type { CompetenceUtterance } from './competence';
|
|
5
6
|
export type HistorySource = 'Grid' | 'Snap' | 'OBL' | string;
|
|
6
7
|
export interface HistoryOccurrence {
|
|
7
8
|
timestamp: Date;
|
|
@@ -36,6 +37,14 @@ export interface HistoryEntry {
|
|
|
36
37
|
platform?: HistoryPlatformExtras;
|
|
37
38
|
}
|
|
38
39
|
export { dotNetTicksToDate };
|
|
40
|
+
/**
|
|
41
|
+
* Adapt any `HistoryEntry[]` (Grid 3, Snap, OBF/OBFL logs, ...) into the generic
|
|
42
|
+
* utterance stream consumed by the linguistic-competence engine
|
|
43
|
+
* (`analyzeTimeline`). Each occurrence of each phrase becomes one utterance,
|
|
44
|
+
* timestamped by its occurrence time. This keeps the competence metrics fully
|
|
45
|
+
* source-agnostic: anything the library can read as history can be analysed.
|
|
46
|
+
*/
|
|
47
|
+
export declare function historyEntriesToCompetenceUtterances(entries: HistoryEntry[]): CompetenceUtterance[];
|
|
39
48
|
export interface BatonExportMetadata {
|
|
40
49
|
timestamp: string;
|
|
41
50
|
latitude?: number | null;
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
3
|
exports.dotNetTicksToDate = void 0;
|
|
4
|
+
exports.historyEntriesToCompetenceUtterances = historyEntriesToCompetenceUtterances;
|
|
4
5
|
exports.exportHistoryToBaton = exportHistoryToBaton;
|
|
5
6
|
exports.readGrid3History = readGrid3History;
|
|
6
7
|
exports.readGrid3HistoryForUser = readGrid3HistoryForUser;
|
|
@@ -14,6 +15,28 @@ const dotnetTicks_1 = require("../../utils/dotnetTicks");
|
|
|
14
15
|
Object.defineProperty(exports, "dotNetTicksToDate", { enumerable: true, get: function () { return dotnetTicks_1.dotNetTicksToDate; } });
|
|
15
16
|
const helpers_1 = require("../../processors/gridset/helpers");
|
|
16
17
|
const helpers_2 = require("../../processors/snap/helpers");
|
|
18
|
+
/**
|
|
19
|
+
* Adapt any `HistoryEntry[]` (Grid 3, Snap, OBF/OBFL logs, ...) into the generic
|
|
20
|
+
* utterance stream consumed by the linguistic-competence engine
|
|
21
|
+
* (`analyzeTimeline`). Each occurrence of each phrase becomes one utterance,
|
|
22
|
+
* timestamped by its occurrence time. This keeps the competence metrics fully
|
|
23
|
+
* source-agnostic: anything the library can read as history can be analysed.
|
|
24
|
+
*/
|
|
25
|
+
function historyEntriesToCompetenceUtterances(entries) {
|
|
26
|
+
const out = [];
|
|
27
|
+
for (const e of entries) {
|
|
28
|
+
const text = e.content;
|
|
29
|
+
if (!text || text.trim().length === 0)
|
|
30
|
+
continue;
|
|
31
|
+
const occs = e.occurrences ?? [];
|
|
32
|
+
for (const occ of occs) {
|
|
33
|
+
if (!occ.timestamp)
|
|
34
|
+
continue;
|
|
35
|
+
out.push({ text, timestampMs: occ.timestamp.getTime() });
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
return out;
|
|
39
|
+
}
|
|
17
40
|
const generateUuid = () => {
|
|
18
41
|
if (typeof globalThis.crypto?.randomUUID === 'function') {
|
|
19
42
|
return globalThis.crypto.randomUUID();
|
|
@@ -21,6 +21,7 @@ export { VocabularyAnalyzer } from './metrics/vocabulary';
|
|
|
21
21
|
export { SentenceAnalyzer } from './metrics/sentence';
|
|
22
22
|
export { ComparisonAnalyzer } from './metrics/comparison';
|
|
23
23
|
export { ReferenceLoader } from './reference';
|
|
24
|
+
export * from './competence';
|
|
24
25
|
/**
|
|
25
26
|
* Get the default reference data path
|
|
26
27
|
*/
|
|
@@ -52,6 +52,8 @@ var comparison_1 = require("./metrics/comparison");
|
|
|
52
52
|
Object.defineProperty(exports, "ComparisonAnalyzer", { enumerable: true, get: function () { return comparison_1.ComparisonAnalyzer; } });
|
|
53
53
|
var reference_1 = require("./reference");
|
|
54
54
|
Object.defineProperty(exports, "ReferenceLoader", { enumerable: true, get: function () { return reference_1.ReferenceLoader; } });
|
|
55
|
+
// Export linguistic-competence measures (privacy-preserving spoken-output analysis)
|
|
56
|
+
__exportStar(require("./competence"), exports);
|
|
55
57
|
/**
|
|
56
58
|
* Get the default reference data path
|
|
57
59
|
*/
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@willwade/aac-processors",
|
|
3
|
-
"version": "0.3.
|
|
3
|
+
"version": "0.3.2",
|
|
4
4
|
"description": "A comprehensive TypeScript library for processing AAC (Augmentative and Alternative Communication) file formats with translation support",
|
|
5
5
|
"main": "dist/index.js",
|
|
6
6
|
"browser": "dist/browser/index.browser.js",
|