@willwade/aac-processors 0.3.2 → 0.3.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,191 @@
1
+ "use strict";
2
+ /**
3
+ * Vocabulary Dump (counts-only)
4
+ *
5
+ * Inventory of the vocabulary available in an AAC pageset, aggregated by
6
+ * source and part of speech. Only counts are emitted — never word lists —
7
+ * so the output is safe to embed in privacy-preserving reports.
8
+ *
9
+ * Sources counted:
10
+ * buttons — labelled buttons (the static on-board vocabulary)
11
+ * wordLists — page WordLists feeding dynamic AutoContent cells
12
+ * (Grid 3 `<WordList>`; absent in other formats)
13
+ * predictionDictionaries — prediction wordlists attached to prediction cells
14
+ * (Grid 3 `Prediction.PredictThis` dictionaries)
15
+ * wordForms — smart-grammar inflections generated by
16
+ * MetricsCalculator (only when a MetricsResult is
17
+ * supplied)
18
+ *
19
+ * Non-Grid 3 formats simply report zeros for the Grid 3-specific sources.
20
+ */
21
+ Object.defineProperty(exports, "__esModule", { value: true });
22
+ exports.dumpVocabulary = dumpVocabulary;
23
+ /** Normalise an entry: trim, lowercase, collapse internal whitespace. */
24
+ function normalize(text) {
25
+ return text.trim().toLowerCase().replace(/\s+/g, ' ');
26
+ }
27
+ const UNTAGGED = 'Unknown';
28
+ /** Per-source tally accumulated while walking the tree. */
29
+ class SourceTally {
30
+ constructor() {
31
+ this.entries = 0;
32
+ this.words = new Set();
33
+ this.phrases = new Set();
34
+ this.byPos = new Map();
35
+ }
36
+ add(text, pos) {
37
+ const norm = normalize(text);
38
+ if (!norm)
39
+ return;
40
+ this.entries++;
41
+ if (/\s/.test(norm))
42
+ this.phrases.add(norm);
43
+ else
44
+ this.words.add(norm);
45
+ const tag = pos && pos.trim() ? pos.trim() : UNTAGGED;
46
+ let set = this.byPos.get(tag);
47
+ if (!set) {
48
+ set = new Set();
49
+ this.byPos.set(tag, set);
50
+ }
51
+ set.add(norm);
52
+ }
53
+ counts() {
54
+ const byPartOfSpeech = {};
55
+ for (const tag of Array.from(this.byPos.keys()).sort((a, b) => a.localeCompare(b))) {
56
+ byPartOfSpeech[tag] = this.byPos.get(tag).size;
57
+ }
58
+ return {
59
+ entries: this.entries,
60
+ uniqueWords: this.words.size,
61
+ uniquePhrases: this.phrases.size,
62
+ byPartOfSpeech,
63
+ };
64
+ }
65
+ get wordSet() {
66
+ return this.words;
67
+ }
68
+ get phraseSet() {
69
+ return this.phrases;
70
+ }
71
+ }
72
+ /**
73
+ * Count the vocabulary available in a pageset, by source and part of speech.
74
+ * Emits counts only — no word lists.
75
+ *
76
+ * @example
77
+ * const processor = new GridsetProcessor();
78
+ * const tree = await processor.loadIntoTree('my.gridset');
79
+ * const metrics = new MetricsCalculator().analyze(tree);
80
+ * const dump = dumpVocabulary(tree, { metrics });
81
+ * console.log(dump.summary.wordLists.lists, 'wordlists');
82
+ */
83
+ function dumpVocabulary(tree, options) {
84
+ const pages = Object.values(tree.pages);
85
+ const buttonTally = new SourceTally();
86
+ const wordListTally = new SourceTally();
87
+ const predictionTally = new SourceTally();
88
+ let wordListCount = 0;
89
+ let buttonsWithDictionaries = 0;
90
+ let totalButtons = 0;
91
+ for (const page of pages) {
92
+ totalButtons += page.buttons.length;
93
+ if (page.wordListItems && page.wordListItems.length > 0) {
94
+ wordListCount++;
95
+ for (const item of page.wordListItems) {
96
+ wordListTally.add(item.text, item.partOfSpeech);
97
+ }
98
+ }
99
+ for (const btn of page.buttons) {
100
+ const label = btn.label?.trim();
101
+ if (label) {
102
+ buttonTally.add(label, btn.pos);
103
+ }
104
+ // Prediction dictionaries: parameters.predictions keeps the original
105
+ // Prediction.PredictThis words even after analyze() expands
106
+ // btn.predictions with morphological forms.
107
+ const dict = btn.parameters?.predictions;
108
+ if (Array.isArray(dict) && dict.length > 0) {
109
+ buttonsWithDictionaries++;
110
+ for (const w of dict) {
111
+ if (typeof w === 'string') {
112
+ predictionTally.add(w, btn.pos);
113
+ }
114
+ }
115
+ }
116
+ }
117
+ }
118
+ // Smart-grammar word forms: inflected forms generated by MetricsCalculator
119
+ // (is_word_form buttons), excluding the original dictionary words they
120
+ // were generated from.
121
+ let wordForms = null;
122
+ let formTally = null;
123
+ if (options?.metrics) {
124
+ const originals = new Set();
125
+ for (const page of pages) {
126
+ for (const btn of page.buttons) {
127
+ const dict = btn.parameters?.predictions;
128
+ if (Array.isArray(dict)) {
129
+ for (const w of dict) {
130
+ if (typeof w === 'string') {
131
+ const norm = normalize(w);
132
+ if (norm)
133
+ originals.add(norm);
134
+ }
135
+ }
136
+ }
137
+ }
138
+ }
139
+ formTally = new SourceTally();
140
+ const parents = new Set();
141
+ const seen = new Set();
142
+ for (const b of options.metrics.buttons) {
143
+ if (!b.is_word_form)
144
+ continue;
145
+ const norm = normalize(b.label);
146
+ if (!norm || seen.has(norm) || originals.has(norm))
147
+ continue;
148
+ seen.add(norm);
149
+ formTally.add(b.label, b.pos);
150
+ if (b.parent_button_id)
151
+ parents.add(b.parent_button_id);
152
+ }
153
+ wordForms = { ...formTally.counts(), parentButtons: parents.size };
154
+ }
155
+ const combinedWords = new Set([
156
+ ...buttonTally.wordSet,
157
+ ...wordListTally.wordSet,
158
+ ...predictionTally.wordSet,
159
+ ...(formTally ? formTally.wordSet : []),
160
+ ]);
161
+ const combinedPhrases = new Set([
162
+ ...buttonTally.phraseSet,
163
+ ...wordListTally.phraseSet,
164
+ ...predictionTally.phraseSet,
165
+ ]);
166
+ return {
167
+ schema: 'aac-vocabulary-dump/v1',
168
+ generatedAt: new Date().toISOString(),
169
+ source: {
170
+ format: tree.metadata?.format,
171
+ name: tree.metadata?.name,
172
+ locale: tree.metadata?.locale,
173
+ },
174
+ summary: {
175
+ totalBoards: pages.length,
176
+ totalButtons,
177
+ buttons: buttonTally.counts(),
178
+ wordLists: { ...wordListTally.counts(), lists: wordListCount },
179
+ predictionDictionaries: {
180
+ ...predictionTally.counts(),
181
+ buttonsWithDictionaries,
182
+ },
183
+ wordForms,
184
+ combined: {
185
+ uniqueEntries: combinedWords.size + combinedPhrases.size,
186
+ uniqueWords: combinedWords.size,
187
+ uniquePhrases: combinedPhrases.size,
188
+ },
189
+ },
190
+ };
191
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@willwade/aac-processors",
3
- "version": "0.3.2",
3
+ "version": "0.3.4",
4
4
  "description": "A comprehensive TypeScript library for processing AAC (Augmentative and Alternative Communication) file formats with translation support",
5
5
  "main": "dist/index.js",
6
6
  "browser": "dist/browser/index.browser.js",