@willwade/aac-processors 0.3.2 → 0.3.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -102,6 +102,12 @@ function getProcessor(filePathOrExtension, options) {
102
102
  const extension = filePathOrExtension.includes('.')
103
103
  ? filePathOrExtension.substring(filePathOrExtension.lastIndexOf('.'))
104
104
  : filePathOrExtension;
105
+ // GoTalk NOW share exports keep their book suffix before the archive
106
+ // extension (e.g. "MyBook.gotalk-book.zip"), so match on the name, not the
107
+ // final extension.
108
+ if (/\.gotalk-book(\.|$)/i.test(filePathOrExtension)) {
109
+ return new gotalkNowProcessor_1.GotalkNowProcessor(options);
110
+ }
105
111
  switch (extension.toLowerCase()) {
106
112
  case '.dot':
107
113
  return new dotProcessor_1.DotProcessor(options);
@@ -152,6 +158,7 @@ function getSupportedExtensions() {
152
158
  '.plist',
153
159
  '.grd',
154
160
  '.gtbz',
161
+ '.gotalk-book',
155
162
  ];
156
163
  }
157
164
  /**
package/dist/metrics.d.ts CHANGED
@@ -10,6 +10,8 @@ export * from './utilities/analytics/metrics/obl-types';
10
10
  export { OblUtil, OblAnonymizer } from './utilities/analytics/metrics/obl';
11
11
  export { MetricsCalculator } from './utilities/analytics/metrics/core';
12
12
  export { VocabularyAnalyzer } from './utilities/analytics/metrics/vocabulary';
13
+ export { dumpVocabulary } from './utilities/analytics/metrics/vocabularyDump';
14
+ export type { VocabularyDump, VocabularyDumpOptions, VocabularySourceCounts, VocabularySummary, } from './utilities/analytics/metrics/vocabularyDump';
13
15
  export { SentenceAnalyzer } from './utilities/analytics/metrics/sentence';
14
16
  export { ComparisonAnalyzer } from './utilities/analytics/metrics/comparison';
15
17
  export { MorphologyEngine } from './utilities/analytics/morphology';
package/dist/metrics.js CHANGED
@@ -20,7 +20,7 @@ var __exportStar = (this && this.__exportStar) || function(m, exports) {
20
20
  for (var p in m) if (p !== "default" && !Object.prototype.hasOwnProperty.call(exports, p)) __createBinding(exports, m, p);
21
21
  };
22
22
  Object.defineProperty(exports, "__esModule", { value: true });
23
- exports.loadReferenceDataFromUrl = exports.createBrowserReferenceLoader = exports.InMemoryReferenceLoader = exports.ReferenceLoader = exports.WordFormGenerator = exports.MorphologyEngine = exports.ComparisonAnalyzer = exports.SentenceAnalyzer = exports.VocabularyAnalyzer = exports.MetricsCalculator = exports.OblAnonymizer = exports.OblUtil = void 0;
23
+ exports.loadReferenceDataFromUrl = exports.createBrowserReferenceLoader = exports.InMemoryReferenceLoader = exports.ReferenceLoader = exports.WordFormGenerator = exports.MorphologyEngine = exports.ComparisonAnalyzer = exports.SentenceAnalyzer = exports.dumpVocabulary = exports.VocabularyAnalyzer = exports.MetricsCalculator = exports.OblAnonymizer = exports.OblUtil = void 0;
24
24
  __exportStar(require("./utilities/analytics/metrics/types"), exports);
25
25
  __exportStar(require("./utilities/analytics/metrics/effort"), exports);
26
26
  __exportStar(require("./utilities/analytics/metrics/obl-types"), exports);
@@ -31,6 +31,8 @@ var core_1 = require("./utilities/analytics/metrics/core");
31
31
  Object.defineProperty(exports, "MetricsCalculator", { enumerable: true, get: function () { return core_1.MetricsCalculator; } });
32
32
  var vocabulary_1 = require("./utilities/analytics/metrics/vocabulary");
33
33
  Object.defineProperty(exports, "VocabularyAnalyzer", { enumerable: true, get: function () { return vocabulary_1.VocabularyAnalyzer; } });
34
+ var vocabularyDump_1 = require("./utilities/analytics/metrics/vocabularyDump");
35
+ Object.defineProperty(exports, "dumpVocabulary", { enumerable: true, get: function () { return vocabularyDump_1.dumpVocabulary; } });
34
36
  var sentence_1 = require("./utilities/analytics/metrics/sentence");
35
37
  Object.defineProperty(exports, "SentenceAnalyzer", { enumerable: true, get: function () { return sentence_1.SentenceAnalyzer; } });
36
38
  var comparison_1 = require("./utilities/analytics/metrics/comparison");
@@ -17,6 +17,13 @@
17
17
  * - Backup-<n>.zip – rolling backups
18
18
  * - REQUIRESREGEN – marker file
19
19
  *
20
+ * Two on-disk variants exist and both are accepted:
21
+ * - `.gtbz` app backups: plists at the ZIP root.
22
+ * - `.gotalk-book` share exports (often `.zip`-suffixed): everything nested
23
+ * under a `<BookName>.gotalk-book/` folder, possibly alongside `__MACOSX/`
24
+ * resource-fork entries (ignored). Button images are pre-rendered
25
+ * snapshots named `<page>-<button>-<buttonCount>.png`.
26
+ *
20
27
  * Colours and fonts in PageData are NSKeyedArchiver-encoded binary plists
21
28
  * carried inside `<data>` tags. They are treated as opaque blobs and preserved
22
29
  * verbatim so that only text fields are touched during translation.
@@ -32,6 +39,13 @@ declare class GotalkNowProcessor extends BaseProcessor {
32
39
  };
33
40
  constructor(options?: ProcessorOptions);
34
41
  private openZip;
42
+ /**
43
+ * Resolve a plist entry name inside the archive, tolerating both layout
44
+ * variants: `.gtbz` (plists at ZIP root) and `.gotalk-book` share exports
45
+ * (plists nested under a `<BookName>.gotalk-book/` folder). `__MACOSX`
46
+ * resource-fork entries are ignored. Returns the full entry name, or null.
47
+ */
48
+ private resolveEntryName;
35
49
  private readPlistFromZip;
36
50
  extractTexts(filePathOrBuffer: ProcessorInput): Promise<string[]>;
37
51
  loadIntoTree(filePathOrBuffer: ProcessorInput): Promise<AACTree>;
@@ -18,6 +18,13 @@
18
18
  * - Backup-<n>.zip – rolling backups
19
19
  * - REQUIRESREGEN – marker file
20
20
  *
21
+ * Two on-disk variants exist and both are accepted:
22
+ * - `.gtbz` app backups: plists at the ZIP root.
23
+ * - `.gotalk-book` share exports (often `.zip`-suffixed): everything nested
24
+ * under a `<BookName>.gotalk-book/` folder, possibly alongside `__MACOSX/`
25
+ * resource-fork entries (ignored). Button images are pre-rendered
26
+ * snapshots named `<page>-<button>-<buttonCount>.png`.
27
+ *
21
28
  * Colours and fonts in PageData are NSKeyedArchiver-encoded binary plists
22
29
  * carried inside `<data>` tags. They are treated as opaque blobs and preserved
23
30
  * verbatim so that only text fields are touched during translation.
@@ -45,7 +52,8 @@ function buildSemanticAction(rawButton, pageId, buttonIndex) {
45
52
  switch (buttonType) {
46
53
  case 'Jump': {
47
54
  const jumpTo = actionData.JumpTo;
48
- const target = jumpTo !== undefined ? String(jumpTo) : undefined;
55
+ // GoTalk treats 0 / -1 / "" as "no jump configured".
56
+ const target = !isNoJumpTarget(jumpTo) ? String(jumpTo) : undefined;
49
57
  return {
50
58
  semanticAction: {
51
59
  category: treeStructure_1.AACSemanticCategory.NAVIGATION,
@@ -54,7 +62,7 @@ function buildSemanticAction(rawButton, pageId, buttonIndex) {
54
62
  platformData: {
55
63
  gotalkNow: { buttonType: 'Jump', jumpTo, jumpBook: actionData.JumpBook },
56
64
  },
57
- fallback: { type: 'NAVIGATE', targetPageId: target },
65
+ fallback: target ? { type: 'NAVIGATE', targetPageId: target } : { type: 'ACTION' },
58
66
  },
59
67
  message: '',
60
68
  targetPageId: target,
@@ -163,12 +171,62 @@ function buildSemanticAction(rawButton, pageId, buttonIndex) {
163
171
  }
164
172
  }
165
173
  /**
166
- * Derive a square-ish grid layout from the declared button count.
167
- * GoTalk NOW uses fixed layouts (1, 4, 9, 16, 25, 36 …).
174
+ * Derive a GoTalk NOW layout from the declared button count.
175
+ * GoTalk NOW uses fixed square layouts (1, 4, 9, 16, 25, 36 …); when the
176
+ * count is not a perfect square (alternate/custom layouts), fall back to a
177
+ * near-square rectangle: cols = ceil(sqrt(n)), rows = ceil(n / cols).
168
178
  */
169
179
  function gridDimensionsFromButtonCount(count) {
170
- const side = Math.max(1, Math.ceil(Math.sqrt(Math.max(count, 1))));
171
- return { rows: side, cols: side };
180
+ const n = Math.max(count, 1);
181
+ const side = Math.ceil(Math.sqrt(n));
182
+ return { rows: Math.ceil(n / side), cols: side };
183
+ }
184
+ /** GoTalk treats JumpTo 0 / -1 / "" as "no jump". */
185
+ function isNoJumpTarget(value) {
186
+ return (value === undefined ||
187
+ value === null ||
188
+ value === 0 ||
189
+ value === -1 ||
190
+ value === '' ||
191
+ value === '0' ||
192
+ value === '-1');
193
+ }
194
+ /** Normalise a zip or iOS-container path to a lowercase basename. */
195
+ function basenameLower(p) {
196
+ if (!p)
197
+ return null;
198
+ const base = p.replace(/\\/g, '/').split('/').pop() || '';
199
+ const trimmed = base.trim().toLowerCase();
200
+ return trimmed.length > 0 ? trimmed : null;
201
+ }
202
+ /**
203
+ * Resolve a bundled image for a button against the archive's file list.
204
+ *
205
+ * `Location`/`QuickRecoveryPath` values are iOS app-container paths that never
206
+ * match archive entry names verbatim, so candidates are matched by basename
207
+ * (case-insensitive), with extension guessing when the name has none.
208
+ */
209
+ function findZipImageEntry(zipBasenames, candidates) {
210
+ const tried = new Set();
211
+ for (const candidate of candidates) {
212
+ const base = basenameLower(candidate);
213
+ if (!base || tried.has(base))
214
+ continue;
215
+ tried.add(base);
216
+ // 1. Exact basename match.
217
+ const direct = zipBasenames.get(base);
218
+ if (direct) {
219
+ const ext = base.includes('.') ? base.slice(base.lastIndexOf('.')) : '.png';
220
+ return { entry: direct, ext };
221
+ }
222
+ // 2. Try common image extensions.
223
+ for (const ext of ['.png', '.jpg', '.jpeg', '.bmp']) {
224
+ const hit = zipBasenames.get(base + ext);
225
+ if (hit)
226
+ return { entry: hit, ext };
227
+ }
228
+ }
229
+ return null;
172
230
  }
173
231
  // ---------------------------------------------------------------------------
174
232
  // Processor
@@ -187,14 +245,27 @@ class GotalkNowProcessor extends baseProcessor_1.BaseProcessor {
187
245
  const zip = await this.options.zipAdapter(filePathOrBuffer, this.options.fileAdapter);
188
246
  return zip;
189
247
  }
190
- async readPlistFromZip(zip, name, fallback) {
191
- const files = zip.listFiles();
192
- if (!files.includes(name)) {
248
+ /**
249
+ * Resolve a plist entry name inside the archive, tolerating both layout
250
+ * variants: `.gtbz` (plists at ZIP root) and `.gotalk-book` share exports
251
+ * (plists nested under a `<BookName>.gotalk-book/` folder). `__MACOSX`
252
+ * resource-fork entries are ignored. Returns the full entry name, or null.
253
+ */
254
+ resolveEntryName(files, basename, prefix) {
255
+ const rooted = `${prefix}${basename}`;
256
+ if (files.includes(rooted))
257
+ return rooted;
258
+ // Nested variant: find the (non-__MACOSX) entry ending in /<basename>.
259
+ const matches = files.filter((f) => f !== `__MACOSX` && !f.startsWith('__MACOSX/') && f.endsWith(`/${basename}`));
260
+ return matches.length > 0 ? matches[0] : null;
261
+ }
262
+ async readPlistFromZip(zip, name, resolvedName, fallback) {
263
+ if (!resolvedName) {
193
264
  if (fallback)
194
265
  return fallback();
195
266
  throw new Error(`GoTalk NOW archive is missing ${name}`);
196
267
  }
197
- const bytes = await zip.readFile(name);
268
+ const bytes = await zip.readFile(resolvedName);
198
269
  const text = typeof Buffer !== 'undefined' && Buffer.isBuffer(bytes)
199
270
  ? bytes.toString('utf8')
200
271
  : new TextDecoder().decode(bytes);
@@ -227,9 +298,13 @@ class GotalkNowProcessor extends baseProcessor_1.BaseProcessor {
227
298
  let bookInfoValue;
228
299
  try {
229
300
  zip = await this.openZip(filePathOrBuffer);
230
- pageDataValue = await this.readPlistFromZip(zip, 'PageData.plist');
231
- pageOrderValue = await this.readPlistFromZip(zip, 'PageOrder.plist', () => []);
232
- bookInfoValue = await this.readPlistFromZip(zip, 'BookInfo.plist', () => ({}));
301
+ // Resolve plist locations once; shared by load/processTexts round-trip.
302
+ const files = zip.listFiles();
303
+ const pdEntry = this.resolveEntryName(files, 'PageData.plist', '');
304
+ const prefix = pdEntry ? pdEntry.slice(0, -'PageData.plist'.length) : '';
305
+ pageDataValue = await this.readPlistFromZip(zip, 'PageData.plist', pdEntry);
306
+ pageOrderValue = await this.readPlistFromZip(zip, 'PageOrder.plist', this.resolveEntryName(files, 'PageOrder.plist', prefix), () => []);
307
+ bookInfoValue = await this.readPlistFromZip(zip, 'BookInfo.plist', this.resolveEntryName(files, 'BookInfo.plist', prefix), () => ({}));
233
308
  }
234
309
  catch (err) {
235
310
  if (err instanceof validationTypes_1.ValidationFailureError)
@@ -268,6 +343,16 @@ class GotalkNowProcessor extends baseProcessor_1.BaseProcessor {
268
343
  }
269
344
  // Track available image files so we can resolve ButtonImages locations.
270
345
  const zipFiles = new Set(zip.listFiles());
346
+ // Basename (lowercase, non-__MACOSX) -> full entry name, for tolerant
347
+ // matching of bundled images in .gotalk-book share exports.
348
+ const zipBasenames = new Map();
349
+ for (const f of zipFiles) {
350
+ if (f.startsWith('__MACOSX/') || f.endsWith('/'))
351
+ continue;
352
+ const base = basenameLower(f);
353
+ if (base && !zipBasenames.has(base))
354
+ zipBasenames.set(base, f);
355
+ }
271
356
  // Build pages. Use the union of PageData keys and PageOrder ids.
272
357
  const allPageIds = new Set([...Object.keys(pageData), ...orderedPageIds]);
273
358
  // Deterministic ordering: ordered pages first (in order), then any extras sorted numerically.
@@ -317,7 +402,10 @@ class GotalkNowProcessor extends baseProcessor_1.BaseProcessor {
317
402
  const buttonsById = rawPage.Buttons || {};
318
403
  // Numeric sort of button indices ("0","1",…,"10",…)
319
404
  const buttonIndices = Object.keys(buttonsById).sort((a, b) => Number(a) - Number(b));
320
- const buttonCount = rawPage.ButtonCount ?? buttonIndices.length;
405
+ // Sparse button maps: the grid size is max(index)+1, not the number of
406
+ // keys (GoTalk snapshots and layouts are sized by the highest slot).
407
+ const buttonCount = rawPage.ButtonCount ??
408
+ (buttonIndices.length > 0 ? Math.max(...buttonIndices.map(Number)) + 1 : 0);
321
409
  const { rows, cols } = gridDimensionsFromButtonCount(buttonCount);
322
410
  const gridLayout = Array.from({ length: rows }, () => Array.from({ length: cols }, () => null));
323
411
  buttonIndices.forEach((btnIndex, i) => {
@@ -347,19 +435,47 @@ class GotalkNowProcessor extends baseProcessor_1.BaseProcessor {
347
435
  };
348
436
  }
349
437
  // Resolve image metadata.
350
- const images = rawButton.ButtonImages;
351
- if (Array.isArray(images) && images.length > 0) {
352
- const first = images[0];
353
- if (first?.Location) {
354
- button.image = first.Location;
355
- if (zipFiles.has(first.Location)) {
356
- button.resolvedImageEntry = first.Location;
438
+ // Heuristics (from real .gotalk-book share exports):
439
+ // - entries sourced from the app's bundled "GoTalk Image Library"
440
+ // are clip-art, not user content — skip them;
441
+ // - prefer the entry carrying QuickRecoveryPath; with multiple
442
+ // entries the second one is the user's image;
443
+ // - bundled files are matched by basename (Location/QRP are iOS
444
+ // container paths, never archive entry names);
445
+ // - last resort: the pre-rendered snapshot <page>-<btn>-<count>.png.
446
+ const imagesAll = rawButton.ButtonImages;
447
+ const images = Array.isArray(imagesAll)
448
+ ? imagesAll.filter((im) => (im?.SourceLibrary || '') !== 'GoTalk Image Library')
449
+ : [];
450
+ if (images.length > 0) {
451
+ const selected = images.find((im) => im.QuickRecoveryPath !== undefined) ??
452
+ (images.length >= 2 ? images[1] : images[0]);
453
+ if (selected) {
454
+ const sourceLibrary = (selected.SourceLibrary || '').toLowerCase();
455
+ const sourceImageName = selected.SourceImageName || '';
456
+ const isLibrarySymbol = /metacom|pcs|symbolstix|widgit/.test(sourceLibrary) && sourceImageName.length > 0;
457
+ if (selected.Location)
458
+ button.image = selected.Location;
459
+ if (selected.SourceLibrary)
460
+ button.symbolLibrary = selected.SourceLibrary;
461
+ if (sourceImageName)
462
+ button.symbolPath = sourceImageName;
463
+ if (!isLibrarySymbol) {
464
+ const bundled = findZipImageEntry(zipBasenames, [
465
+ selected.QuickRecoveryPath,
466
+ selected.Location,
467
+ selected.SourceImageName,
468
+ `${pageId}-${btnIndex}-${buttonCount}.png`,
469
+ ]);
470
+ if (bundled) {
471
+ button.image = bundled.entry;
472
+ button.resolvedImageEntry = bundled.entry;
473
+ }
474
+ }
475
+ else if (zipFiles.has(selected.Location || '')) {
476
+ button.resolvedImageEntry = selected.Location;
357
477
  }
358
478
  }
359
- if (first?.SourceLibrary)
360
- button.symbolLibrary = first.SourceLibrary;
361
- if (first?.SourceImageName)
362
- button.symbolPath = first.SourceImageName;
363
479
  }
364
480
  // Visibility: GoTalk buttons can be disabled via either field.
365
481
  if (rawButton.Enabled === false || rawButton.Disabled === true) {
@@ -413,8 +529,12 @@ class GotalkNowProcessor extends baseProcessor_1.BaseProcessor {
413
529
  async processTexts(filePathOrBuffer, translations, outputPath) {
414
530
  const { writeBinaryToPath } = this.options.fileAdapter;
415
531
  const originalZip = await this.openZip(filePathOrBuffer);
416
- // Parse PageData and apply translations in place.
417
- const pageDataBytes = await originalZip.readFile('PageData.plist');
532
+ // Parse PageData and apply translations in place. Locate the plist the
533
+ // same tolerant way loadIntoTree does (root or nested .gotalk-book folder)
534
+ // and write it back under its original entry name so the layout survives.
535
+ const files = originalZip.listFiles();
536
+ const pdEntry = this.resolveEntryName(files, 'PageData.plist', '') ?? 'PageData.plist';
537
+ const pageDataBytes = await originalZip.readFile(pdEntry);
418
538
  const pageDataText = typeof Buffer !== 'undefined' && Buffer.isBuffer(pageDataBytes)
419
539
  ? pageDataBytes.toString('utf8')
420
540
  : new TextDecoder().decode(pageDataBytes);
@@ -441,14 +561,14 @@ class GotalkNowProcessor extends baseProcessor_1.BaseProcessor {
441
561
  const rebuiltPageData = plist_1.default.build(pageData);
442
562
  // Repack: copy every original entry verbatim, replacing only PageData.plist.
443
563
  const outputZip = await this.options.zipAdapter(undefined, this.options.fileAdapter);
444
- const files = [];
564
+ const filesOut = [];
445
565
  for (const name of originalZip.listFiles()) {
446
- if (name === 'PageData.plist')
566
+ if (name === pdEntry)
447
567
  continue;
448
- files.push({ name, data: await originalZip.readFile(name) });
568
+ filesOut.push({ name, data: await originalZip.readFile(name) });
449
569
  }
450
- files.push({ name: 'PageData.plist', data: rebuiltPageData });
451
- const outputBuffer = await outputZip.writeFiles(files);
570
+ filesOut.push({ name: pdEntry, data: rebuiltPageData });
571
+ const outputBuffer = await outputZip.writeFiles(filesOut);
452
572
  await writeBinaryToPath(outputPath, outputBuffer);
453
573
  return outputBuffer;
454
574
  }
@@ -126,6 +126,7 @@ export interface AACPage {
126
126
  descriptionHtml?: string;
127
127
  images?: any[];
128
128
  sounds?: any[];
129
+ wordListItems?: AACWordListItem[];
129
130
  semantic_ids?: string[];
130
131
  clone_ids?: string[];
131
132
  scanningConfig?: ScanningConfig;
@@ -34,6 +34,7 @@
34
34
  * conflates linguistic/operational/strategic/social competence in AAC); it is
35
35
  * reported only as a distribution.
36
36
  */
37
+ import type { VocabularySummary } from './metrics/vocabularyDump';
37
38
  /** A single spoken utterance with a production timestamp (epoch ms). */
38
39
  export interface CompetenceUtterance {
39
40
  text: string;
@@ -117,6 +118,19 @@ export declare function morphologicalDiversity(words: WordStream, options?: Dive
117
118
  * Returns null (unavailable) if no dictionary is provided.
118
119
  */
119
120
  export declare function spellingValidity(words: WordStream, dictionary?: Set<string>): number | null;
121
+ export interface LexicalRichness {
122
+ /** Brunet's index W = N · V^(-0.165). Range ~10–30; LOWER = richer. */
123
+ brunetsW: number | null;
124
+ /** Honoré's statistic R = 100·ln N / (1 − V1/V). HIGHER = richer. */
125
+ honoresR: number | null;
126
+ /** Hapax legomena (words used exactly once) — count only. */
127
+ hapax: number;
128
+ /** Number of distinct words (V). */
129
+ types: number;
130
+ /** Number of tokens (N). */
131
+ tokens: number;
132
+ }
133
+ export declare function lexicalRichness(words: WordStream): LexicalRichness;
120
134
  export interface DistributionStats {
121
135
  median: number | null;
122
136
  mean: number | null;
@@ -150,6 +164,8 @@ export interface MonthBin {
150
164
  morphologicalDiversity: DiversityResult;
151
165
  /** Phonological — only when a dictionary is supplied. */
152
166
  spellingValidity: number | null;
167
+ /** Lexical richness indices (Brunet's W, Honoré's R) — always available. */
168
+ lexicalRichness: LexicalRichness;
153
169
  /** True when the month has too little data to trust the diversity figures. */
154
170
  suppressed: boolean;
155
171
  suppressReason: string | null;
@@ -185,6 +201,12 @@ export interface PagesetSummary {
185
201
  base: number | null;
186
202
  perLetter: number | null;
187
203
  };
204
+ /**
205
+ * Counts-only vocabulary inventory (buttons, wordlists, prediction
206
+ * dictionaries, smart-grammar word forms) — see dumpVocabulary().
207
+ * Null/undefined when the caller did not compute it. Contains counts only.
208
+ */
209
+ vocabulary?: VocabularySummary | null;
188
210
  error?: string;
189
211
  }
190
212
  export interface UserSettingsSummary {
@@ -42,6 +42,7 @@ exports.lexicalDiversity = lexicalDiversity;
42
42
  exports.syntacticDiversity = syntacticDiversity;
43
43
  exports.morphologicalDiversity = morphologicalDiversity;
44
44
  exports.spellingValidity = spellingValidity;
45
+ exports.lexicalRichness = lexicalRichness;
45
46
  exports.summarizeActivity = summarizeActivity;
46
47
  exports.analyzeTimeline = analyzeTimeline;
47
48
  /* ------------------------------------------------------------------ *
@@ -277,6 +278,21 @@ function spellingValidity(words, dictionary) {
277
278
  }
278
279
  return checked === 0 ? null : correct / checked;
279
280
  }
281
+ function lexicalRichness(words) {
282
+ const n = words.length;
283
+ const freq = new Map();
284
+ for (const w of words)
285
+ freq.set(w, (freq.get(w) ?? 0) + 1);
286
+ const v = freq.size;
287
+ let v1 = 0;
288
+ for (const c of freq.values())
289
+ if (c === 1)
290
+ v1++;
291
+ const brunetsW = n > 0 && v > 0 ? n * Math.pow(v, -0.165) : null;
292
+ // Honoré's R is undefined when every word is a hapax (V1 = V) or N <= 1.
293
+ const honoresR = n > 1 && v > 0 && v1 < v ? (100 * Math.log(n)) / (1 - v1 / v) : null;
294
+ return { brunetsW, honoresR, hapax: v1, types: v, tokens: n };
295
+ }
280
296
  function distribution(values) {
281
297
  return {
282
298
  median: median(values),
@@ -391,6 +407,9 @@ function analyzeTimeline(utterances, options = {}) {
391
407
  classifyInflection: resources.classifyInflection,
392
408
  });
393
409
  const spell = suppressed ? null : spellingValidity(stream, dictionary);
410
+ const rich = suppressed
411
+ ? { brunetsW: null, honoresR: null, hapax: 0, types: 0, tokens: 0 }
412
+ : lexicalRichness(stream);
394
413
  timeline.push({
395
414
  month: key,
396
415
  utterances: activity.utterances,
@@ -402,6 +421,7 @@ function analyzeTimeline(utterances, options = {}) {
402
421
  syntacticDiversity: syn,
403
422
  morphologicalDiversity: mor,
404
423
  spellingValidity: spell,
424
+ lexicalRichness: rich,
405
425
  suppressed,
406
426
  suppressReason,
407
427
  });
@@ -18,6 +18,8 @@ export * from './metrics/obl-types';
18
18
  export { OblUtil, OblAnonymizer } from './metrics/obl';
19
19
  export { MetricsCalculator } from './metrics/core';
20
20
  export { VocabularyAnalyzer } from './metrics/vocabulary';
21
+ export { dumpVocabulary } from './metrics/vocabularyDump';
22
+ export type { VocabularyDump, VocabularyDumpOptions, VocabularySourceCounts, VocabularySummary, } from './metrics/vocabularyDump';
21
23
  export { SentenceAnalyzer } from './metrics/sentence';
22
24
  export { ComparisonAnalyzer } from './metrics/comparison';
23
25
  export { ReferenceLoader } from './reference';
@@ -25,7 +25,7 @@ var __exportStar = (this && this.__exportStar) || function(m, exports) {
25
25
  for (var p in m) if (p !== "default" && !Object.prototype.hasOwnProperty.call(exports, p)) __createBinding(exports, m, p);
26
26
  };
27
27
  Object.defineProperty(exports, "__esModule", { value: true });
28
- exports.ReferenceLoader = exports.ComparisonAnalyzer = exports.SentenceAnalyzer = exports.VocabularyAnalyzer = exports.MetricsCalculator = exports.OblAnonymizer = exports.OblUtil = void 0;
28
+ exports.ReferenceLoader = exports.ComparisonAnalyzer = exports.SentenceAnalyzer = exports.dumpVocabulary = exports.VocabularyAnalyzer = exports.MetricsCalculator = exports.OblAnonymizer = exports.OblUtil = void 0;
29
29
  exports.getReferenceDataPath = getReferenceDataPath;
30
30
  exports.hasReferenceData = hasReferenceData;
31
31
  const io_1 = require("../../utils/io");
@@ -46,6 +46,8 @@ Object.defineProperty(exports, "MetricsCalculator", { enumerable: true, get: fun
46
46
  // Export vocabulary and comparison analyzers
47
47
  var vocabulary_1 = require("./metrics/vocabulary");
48
48
  Object.defineProperty(exports, "VocabularyAnalyzer", { enumerable: true, get: function () { return vocabulary_1.VocabularyAnalyzer; } });
49
+ var vocabularyDump_1 = require("./metrics/vocabularyDump");
50
+ Object.defineProperty(exports, "dumpVocabulary", { enumerable: true, get: function () { return vocabularyDump_1.dumpVocabulary; } });
49
51
  var sentence_1 = require("./metrics/sentence");
50
52
  Object.defineProperty(exports, "SentenceAnalyzer", { enumerable: true, get: function () { return sentence_1.SentenceAnalyzer; } });
51
53
  var comparison_1 = require("./metrics/comparison");
@@ -0,0 +1,88 @@
1
+ /**
2
+ * Vocabulary Dump (counts-only)
3
+ *
4
+ * Inventory of the vocabulary available in an AAC pageset, aggregated by
5
+ * source and part of speech. Only counts are emitted — never word lists —
6
+ * so the output is safe to embed in privacy-preserving reports.
7
+ *
8
+ * Sources counted:
9
+ * buttons — labelled buttons (the static on-board vocabulary)
10
+ * wordLists — page WordLists feeding dynamic AutoContent cells
11
+ * (Grid 3 `<WordList>`; absent in other formats)
12
+ * predictionDictionaries — prediction wordlists attached to prediction cells
13
+ * (Grid 3 `Prediction.PredictThis` dictionaries)
14
+ * wordForms — smart-grammar inflections generated by
15
+ * MetricsCalculator (only when a MetricsResult is
16
+ * supplied)
17
+ *
18
+ * Non-Grid 3 formats simply report zeros for the Grid 3-specific sources.
19
+ */
20
+ import type { AACTree } from '../../../types/aac';
21
+ import type { MetricsResult } from './types';
22
+ /** Counts for one vocabulary source. */
23
+ export interface VocabularySourceCounts {
24
+ /** Total entries found in this source (before deduplication). */
25
+ entries: number;
26
+ /** Unique single words after normalisation. */
27
+ uniqueWords: number;
28
+ /** Unique multi-word phrases after normalisation (e.g. "thank you"). */
29
+ uniquePhrases: number;
30
+ /** Unique entries per part-of-speech tag ('Unknown' when untagged). */
31
+ byPartOfSpeech: Record<string, number>;
32
+ }
33
+ /** Aggregated, counts-only vocabulary inventory. */
34
+ export interface VocabularySummary {
35
+ totalBoards: number;
36
+ totalButtons: number;
37
+ /** Vocabulary carried by labelled buttons. */
38
+ buttons: VocabularySourceCounts;
39
+ /** Page WordLists (dynamic content cells; Grid 3). */
40
+ wordLists: VocabularySourceCounts & {
41
+ lists: number;
42
+ };
43
+ /** Prediction dictionaries attached to prediction cells (Grid 3). */
44
+ predictionDictionaries: VocabularySourceCounts & {
45
+ buttonsWithDictionaries: number;
46
+ };
47
+ /** Smart-grammar inflections; null when no MetricsResult was supplied. */
48
+ wordForms: (VocabularySourceCounts & {
49
+ parentButtons: number;
50
+ }) | null;
51
+ /** Unique entries across all sources combined. */
52
+ combined: {
53
+ uniqueEntries: number;
54
+ uniqueWords: number;
55
+ uniquePhrases: number;
56
+ };
57
+ }
58
+ /** Full vocabulary dump document (schema-versioned). */
59
+ export interface VocabularyDump {
60
+ schema: 'aac-vocabulary-dump/v1';
61
+ generatedAt: string;
62
+ source: {
63
+ format?: string;
64
+ name?: string;
65
+ locale?: string;
66
+ };
67
+ summary: VocabularySummary;
68
+ }
69
+ export interface VocabularyDumpOptions {
70
+ /**
71
+ * Precomputed metrics (MetricsCalculator.analyze). Enables the
72
+ * smart-grammar word-form counts. Run analyze() BEFORE dumpVocabulary():
73
+ * analyze() expands morphological predictions on the tree.
74
+ */
75
+ metrics?: MetricsResult;
76
+ }
77
+ /**
78
+ * Count the vocabulary available in a pageset, by source and part of speech.
79
+ * Emits counts only — no word lists.
80
+ *
81
+ * @example
82
+ * const processor = new GridsetProcessor();
83
+ * const tree = await processor.loadIntoTree('my.gridset');
84
+ * const metrics = new MetricsCalculator().analyze(tree);
85
+ * const dump = dumpVocabulary(tree, { metrics });
86
+ * console.log(dump.summary.wordLists.lists, 'wordlists');
87
+ */
88
+ export declare function dumpVocabulary(tree: AACTree, options?: VocabularyDumpOptions): VocabularyDump;