@capsiynau/intelligence 0.3.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,240 @@
1
+ /**
2
+ * @capsiynau/intelligence/dialect
3
+ *
4
+ * The Welsh regional LEXICON as data, not as a prompt.
5
+ *
6
+ * `welsh/normaliser.js` already exposes `getDialectSystemPrompt()`, which
7
+ * returns a register instruction ("avoid colloquialisms", "maintain formal
8
+ * third-person constructions"). Register genuinely is a prompt concern -
9
+ * there is no deterministic transform for "sound less formal".
10
+ *
11
+ * Word choice is not. "rŵan" versus "nawr" is a dictionary lookup, and a
12
+ * dictionary applied in code is right every time, costs nothing, and can be
13
+ * shown to a reviewer. Asked of a model in a system prompt it is applied
14
+ * unevenly and cannot be verified afterwards. So this module owns the lexical
15
+ * half and the prompt fragment keeps the register half. They compose; neither
16
+ * replaces the other.
17
+ *
18
+ * What this module DOES:
19
+ * - `checkDialect(text, region)` - report off-region terms, mutating nothing
20
+ * - `applyDialect(text, region)` - deterministic case-preserving substitution
21
+ * - `listDialectTerms(region)` - expose the table for review UI and export
22
+ *
23
+ * What this module does NOT do:
24
+ * - Decide register, tone or formality. That is the prompt fragment's job.
25
+ * - Match MUTATED surface forms. "taid" after a possessive is "daid", and
26
+ * recognising that needs syntactic context, exactly as documented in
27
+ * `welsh/mutations.js`. Callers normalise first or accept the miss.
28
+ * - Touch pronouns or copula forms. See EXCLUDED_TERMS below; the false
29
+ * positives are worse than the misses.
30
+ * - Infer a region from text. The caller declares it, per the house rule
31
+ * that source language and variety are never auto-detected.
32
+ *
33
+ * Pure JS, no external deps, no I/O. Runs in Node or the browser.
34
+ */
35
+
36
+ /**
37
+ * Provenance for the seed table, per the "evidence beats confidence" rule:
38
+ * state what this data IS rather than implying an authority it has not earned.
39
+ * These are widely attested north/south pairs, but they are a starting seed
40
+ * compiled alongside the existing dialect prompt fragments - NOT a reviewed
41
+ * lexicographic source. A native reviewer signs this off before any consumer
42
+ * treats `applyDialect` output as publishable without a human read.
43
+ */
44
+ export const DIALECT_TERMS_STATUS = {
45
+ reviewed: false,
46
+ reviewedBy: null,
47
+ seededOn: '2026-07-25',
48
+ note: 'Seed table. Needs native review before authoritative use.',
49
+ }
50
+
51
+ // Welsh letters, including the diacritics JS `\w` and `\b` do not cover. `\b`
52
+ // is unusable here: to `\b` the `ŵ` in `rŵan` is a NON-word character, so
53
+ // /\brŵan\b/ finds a boundary in the middle of the word and matches inside
54
+ // longer tokens. Lookarounds over this class are the fix.
55
+ const WELSH_LETTERS = "A-Za-zÂÊÎÔÛŴŶâêîôûŵŷÁÉÍÓÚáéíóúÀÈÌÒÙàèìòùÄËÏÖÜäëïöü'"
56
+
57
+ /**
58
+ * Seed lexicon. `neutral` is the form a broadcast or formal register would
59
+ * use; `null` means BOTH forms are regionally marked and there is no neutral
60
+ * standard, so a neutral target leaves the text alone rather than inventing
61
+ * one. Saying "no neutral form exists" is more useful than picking a side.
62
+ */
63
+ export const DIALECT_TERMS = [
64
+ { id: 'now', north: 'rŵan', south: 'nawr', neutral: 'nawr', gloss: 'now' },
65
+ { id: 'with', north: 'efo', south: 'gyda', neutral: 'gyda', gloss: 'with' },
66
+ { id: 'milk', north: 'llefrith', south: 'llaeth', neutral: 'llaeth', gloss: 'milk' },
67
+ { id: 'want', north: 'isio', south: 'moyn', neutral: 'eisiau', gloss: 'to want' },
68
+ { id: 'out', north: 'allan', south: 'mas', neutral: 'allan', gloss: 'out' },
69
+ { id: 'money', north: 'pres', south: 'arian', neutral: 'arian', gloss: 'money' },
70
+ { id: 'to-come', north: 'dŵad', south: 'dod', neutral: 'dod', gloss: 'to come' },
71
+ { id: 'boy', north: 'hogyn', south: 'crwt', neutral: 'bachgen', gloss: 'boy' },
72
+ { id: 'girl', north: 'hogan', south: 'croten', neutral: 'merch', gloss: 'girl' },
73
+ { id: 'cuppa', north: 'panad', south: 'dishgled', neutral: 'paned', gloss: 'a cuppa' },
74
+ // Both forms regionally marked, no neutral standard.
75
+ { id: 'grandfather', north: 'taid', south: 'tad-cu', neutral: null, gloss: 'grandfather' },
76
+ { id: 'grandmother', north: 'nain', south: 'mam-gu', neutral: null, gloss: 'grandmother' },
77
+ ]
78
+
79
+ /**
80
+ * Deliberately NOT in the table, with the reason. Kept in code because the
81
+ * next person to extend this will reach for exactly these, and the reason
82
+ * they are absent is not self-evident.
83
+ */
84
+ export const EXCLUDED_TERMS = [
85
+ {
86
+ id: 'pronoun-o-e',
87
+ reason:
88
+ 'North "mae o" / south "mae e". The north form collides with the ' +
89
+ 'preposition "o" (from): substituting in "mae o\'r ardal" (is from the ' +
90
+ 'area) produces nonsense. Needs part-of-speech context.',
91
+ },
92
+ {
93
+ id: 'up-lan',
94
+ reason:
95
+ 'South "lan" (up) collides with "glan" soft-mutated to "lan" (shore), ' +
96
+ 'so "ar lan y môr" would be rewritten. Needs mutation awareness.',
97
+ },
98
+ {
99
+ id: 'watching-gwatsiad',
100
+ reason:
101
+ 'Appears in the north prompt fragment but is an anglicism, not a ' +
102
+ 'north/south pair. Codifying it as regional Welsh would be wrong.',
103
+ },
104
+ ]
105
+
106
+ // `broadcast` and `formal` both want the unmarked standard form, so they map
107
+ // onto the same lexical target. They still differ in REGISTER, which is the
108
+ // prompt fragment's concern, not this module's.
109
+ const REGION_ALIASES = {
110
+ north: 'north',
111
+ south: 'south',
112
+ neutral: 'neutral',
113
+ broadcast: 'neutral',
114
+ formal: 'neutral',
115
+ }
116
+
117
+ /** @param {string} region @returns {'north'|'south'|'neutral'} */
118
+ function resolveRegion(region) {
119
+ return REGION_ALIASES[region] || 'neutral'
120
+ }
121
+
122
+ function escapeRegExp(s) {
123
+ return s.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')
124
+ }
125
+
126
+ /**
127
+ * Whole-term matcher. Multi-word entries ("tad-cu", and any future entry with
128
+ * a space) tolerate variable internal whitespace.
129
+ */
130
+ function matcherFor(term) {
131
+ const body = escapeRegExp(term).replace(/\s+/g, '\\s+')
132
+ return new RegExp(`(?<![${WELSH_LETTERS}])(${body})(?![${WELSH_LETTERS}])`, 'gi')
133
+ }
134
+
135
+ /**
136
+ * Carry the observed capitalisation onto the replacement, so "Rŵan" becomes
137
+ * "Nawr" and "RŴAN" becomes "NAWR". Mixed case falls through to the table's
138
+ * own casing rather than guessing.
139
+ */
140
+ function matchCase(observed, replacement) {
141
+ if (observed === observed.toUpperCase() && observed !== observed.toLowerCase()) {
142
+ return replacement.toUpperCase()
143
+ }
144
+ if (observed[0] === observed[0].toUpperCase()) {
145
+ return replacement[0].toUpperCase() + replacement.slice(1)
146
+ }
147
+ return replacement
148
+ }
149
+
150
+ /**
151
+ * The entries that carry a usable target for `region`, each with the forms
152
+ * that should be replaced by it. Entries with no neutral form are skipped for
153
+ * a neutral target rather than forced onto one side.
154
+ */
155
+ function activeEntries(region) {
156
+ const resolved = resolveRegion(region)
157
+ const out = []
158
+ for (const entry of DIALECT_TERMS) {
159
+ const preferred = entry[resolved]
160
+ if (!preferred) continue
161
+ // Dedupe by lowercase form. Two regions often share a form (south and
162
+ // neutral are both "gyda"), and without this the same occurrence is
163
+ // reported twice by checkDialect.
164
+ const seen = new Set()
165
+ const alternatives = []
166
+ for (const r of ['north', 'south', 'neutral']) {
167
+ const form = r === resolved ? null : entry[r]
168
+ if (!form) continue
169
+ const key = form.toLowerCase()
170
+ if (key === preferred.toLowerCase() || seen.has(key)) continue
171
+ seen.add(key)
172
+ alternatives.push(form)
173
+ }
174
+ if (alternatives.length) out.push({ entry, preferred, alternatives })
175
+ }
176
+ return out
177
+ }
178
+
179
+ /**
180
+ * The lexicon as a reviewable list. This is what a "your preferred terms"
181
+ * management screen renders, and what a workspace exports for sign-off.
182
+ *
183
+ * @param {'north'|'south'|'neutral'|'broadcast'|'formal'} region
184
+ * @returns {Array<{id:string, gloss:string, preferred:string, alternatives:string[]}>}
185
+ */
186
+ export function listDialectTerms(region) {
187
+ return activeEntries(region).map(({ entry, preferred, alternatives }) => ({
188
+ id: entry.id,
189
+ gloss: entry.gloss,
190
+ preferred,
191
+ alternatives,
192
+ }))
193
+ }
194
+
195
+ /**
196
+ * Report off-region terms WITHOUT changing the text. This is the validator
197
+ * half: it produces evidence a reviewer can act on, which is what makes the
198
+ * lexical layer auditable rather than merely applied.
199
+ *
200
+ * @param {string} text
201
+ * @param {'north'|'south'|'neutral'|'broadcast'|'formal'} region
202
+ * @returns {Array<{id:string, found:string, expected:string, index:number}>}
203
+ * Findings in document order. Empty array for empty input.
204
+ */
205
+ export function checkDialect(text, region) {
206
+ if (!text) return []
207
+ const findings = []
208
+ for (const { entry, preferred, alternatives } of activeEntries(region)) {
209
+ for (const alternative of alternatives) {
210
+ const matcher = matcherFor(alternative)
211
+ let hit
212
+ while ((hit = matcher.exec(text)) !== null) {
213
+ findings.push({ id: entry.id, found: hit[1], expected: preferred, index: hit.index })
214
+ }
215
+ }
216
+ }
217
+ return findings.sort((a, b) => a.index - b.index)
218
+ }
219
+
220
+ /**
221
+ * Rewrite off-region terms to the target region's form. Deterministic and
222
+ * case-preserving; returns the input unchanged when there is nothing to do.
223
+ *
224
+ * Substitution is whole-term only and mutation-blind, so it is safe to run on
225
+ * mixed Welsh/English text: no entry in the table is an English word.
226
+ *
227
+ * @param {string} text
228
+ * @param {'north'|'south'|'neutral'|'broadcast'|'formal'} region
229
+ * @returns {string}
230
+ */
231
+ export function applyDialect(text, region) {
232
+ if (!text) return ''
233
+ let out = text
234
+ for (const { preferred, alternatives } of activeEntries(region)) {
235
+ for (const alternative of alternatives) {
236
+ out = out.replace(matcherFor(alternative), (m) => matchCase(m, preferred))
237
+ }
238
+ }
239
+ return out
240
+ }
@@ -0,0 +1,204 @@
1
+ /**
2
+ * @capsiynau/intelligence/speakers
3
+ *
4
+ * Base-level speaker-handling primitives, shared by every app that consumes
5
+ * diarisation (Capsiynau engine, Nodiadau, future apps). Pure + dependency-
6
+ * free; runs in Node or the browser.
7
+ *
8
+ * This first export is the DIARISATION QUALITY GATE. Acoustic diarisation
9
+ * (AssemblyAI) is language-agnostic but not always reliable - especially on
10
+ * Welsh audio (unofficial support), short memos, or heavy cross-talk. The
11
+ * gate decides whether a diarisation result is trustworthy enough to SHOW
12
+ * speaker labels, or whether the consumer should suppress them. Showing
13
+ * "Speaker 1" on everything when four people spoke is worse than showing
14
+ * nothing - so when in doubt, suppress.
15
+ *
16
+ * Thresholds are documented constants, deliberately conservative. Tune them
17
+ * once real Welsh-meeting telemetry exists (schedule phase C2). The function
18
+ * never throws and never mutates its inputs.
19
+ */
20
+
21
+ export const DIARISATION_GATE_DEFAULTS = {
22
+ // A multi-minute recording that diarisation collapses to ONE speaker is the
23
+ // classic false-negative (it failed to separate voices). Below this span a
24
+ // genuine solo memo is plausible, so a sole speaker is allowed; above it a
25
+ // sole speaker is suspicious and we suppress.
26
+ soleSpeakerSuspectAboveMs: 90_000, // 90 seconds
27
+ // Fraction of the transcript span that labelled utterances must cover.
28
+ // Sparse coverage means the speaker timeline barely overlaps the words, so
29
+ // per-segment assignment would be mostly guesswork.
30
+ minCoverage: 0.6,
31
+ // This many distinct speakers (with adequate coverage) is trusted as a
32
+ // clear multi-party separation.
33
+ trustAtSpeakers: 2,
34
+ }
35
+
36
+ /**
37
+ * Total milliseconds the utterances cover WITHIN the transcript span
38
+ * `[spanStart, spanEnd]`, counted as a UNION so overlapping utterances don't
39
+ * double-count. Each utterance is first clipped to the span (speech entirely
40
+ * outside the transcript window contributes 0), then the clipped intervals are
41
+ * merged and their lengths summed. This is what makes coverage measure the
42
+ * fraction of the transcript actually overlapped by the speaker timeline,
43
+ * rather than total diarisation duration (which can sit outside the window
44
+ * when diarisation ran over the full audio but the transcript is a clip).
45
+ *
46
+ * @param {Array<{start?:number,end?:number}>} utts
47
+ * @param {number} spanStart inclusive lower bound (ms)
48
+ * @param {number} spanEnd inclusive upper bound (ms)
49
+ * @returns {number} unioned, clipped duration in ms (>= 0)
50
+ */
51
+ export function clampedUnionMs(utts, spanStart, spanEnd) {
52
+ const clipped = []
53
+ for (const u of utts) {
54
+ const start = Math.max(u.start ?? 0, spanStart)
55
+ const end = Math.min(u.end ?? 0, spanEnd)
56
+ if (end > start) clipped.push([start, end])
57
+ }
58
+ if (clipped.length === 0) return 0
59
+
60
+ clipped.sort((a, b) => a[0] - b[0])
61
+ let total = 0
62
+ let [curStart, curEnd] = clipped[0]
63
+ for (let i = 1; i < clipped.length; i++) {
64
+ const [s, e] = clipped[i]
65
+ if (s <= curEnd) {
66
+ // Overlapping or contiguous - extend the current merged interval.
67
+ if (e > curEnd) curEnd = e
68
+ } else {
69
+ total += curEnd - curStart
70
+ curStart = s
71
+ curEnd = e
72
+ }
73
+ }
74
+ total += curEnd - curStart
75
+ return total
76
+ }
77
+
78
+ /**
79
+ * Judge whether a diarisation timeline is reliable enough to label segments.
80
+ *
81
+ * @param {Array<{start:number,end:number,speaker:string}>} utterances
82
+ * diarisation timeline in milliseconds (AssemblyAI `utterances` shape).
83
+ * @param {Array<{start_ms:number,end_ms:number}>} segments
84
+ * transcript segments in milliseconds (used to measure coverage).
85
+ * @param {Partial<typeof DIARISATION_GATE_DEFAULTS>} [opts] threshold overrides.
86
+ * @returns {{ reliable: boolean, speakerCount: number, coverage: number,
87
+ * durationMs: number, reason: string }}
88
+ */
89
+ export function assessDiarisation(utterances, segments, opts = {}) {
90
+ const t = { ...DIARISATION_GATE_DEFAULTS, ...opts }
91
+ const utts = Array.isArray(utterances)
92
+ ? utterances.filter((u) => u && typeof u.speaker === 'string' && u.speaker)
93
+ : []
94
+
95
+ if (utts.length === 0) {
96
+ return { reliable: false, speakerCount: 0, coverage: 0, durationMs: 0, reason: 'no_utterances' }
97
+ }
98
+
99
+ const speakerCount = new Set(utts.map((u) => u.speaker)).size
100
+
101
+ // Total span: prefer the transcript's extent; fall back to the utterances'.
102
+ const segs = Array.isArray(segments) ? segments.filter(Boolean) : []
103
+ const starts = segs.length ? segs.map((s) => s.start_ms ?? 0) : utts.map((u) => u.start)
104
+ const ends = segs.length ? segs.map((s) => s.end_ms ?? 0) : utts.map((u) => u.end)
105
+ const spanStart = Math.min(...starts)
106
+ const spanEnd = Math.max(...ends)
107
+ const durationMs = Math.max(0, spanEnd - spanStart)
108
+ const totalSpan = Math.max(1, durationMs)
109
+
110
+ // Coverage = fraction of the transcript span actually overlapped by the
111
+ // speaker timeline. Clip each utterance to [spanStart, spanEnd] and union the
112
+ // clipped intervals: speech outside the transcript window (diarisation ran
113
+ // over full audio, transcript is a clip) must NOT inflate coverage, and
114
+ // overlapping utterances must NOT double-count past the real covered fraction.
115
+ const labelledMs = clampedUnionMs(utts, spanStart, spanEnd)
116
+ const coverage = Math.min(1, labelledMs / totalSpan)
117
+
118
+ if (coverage < t.minCoverage) {
119
+ return { reliable: false, speakerCount, coverage, durationMs, reason: 'low_coverage' }
120
+ }
121
+
122
+ // Single speaker: trust only on short audio (a real solo memo); on long
123
+ // audio a lone speaker almost always means diarisation failed to separate.
124
+ if (speakerCount < t.trustAtSpeakers) {
125
+ if (durationMs > t.soleSpeakerSuspectAboveMs) {
126
+ return { reliable: false, speakerCount, coverage, durationMs, reason: 'sole_speaker_long_audio' }
127
+ }
128
+ return { reliable: true, speakerCount, coverage, durationMs, reason: 'sole_speaker_short_audio' }
129
+ }
130
+
131
+ return { reliable: true, speakerCount, coverage, durationMs, reason: 'ok' }
132
+ }
133
+
134
+ export const SPEAKER_MAP_DEFAULTS = {
135
+ // The winning name must cover at least this fraction of the acoustic
136
+ // speaker's total speaking time. Below it the overlap is too weak to trust.
137
+ minOverlapFraction: 0.5,
138
+ // The winning name's overlap must beat the runner-up by at least this ratio,
139
+ // else the acoustic speaker straddles two names too ambiguously to name.
140
+ dominanceRatio: 1.5,
141
+ }
142
+
143
+ /**
144
+ * Map acoustic diarisation speakers ("Speaker 1/2/3") to REAL names from a
145
+ * labelled reference timeline (platform transcript / roster), by greatest
146
+ * total time-overlap. Upgrades anonymous acoustic labels to real names when a
147
+ * labelled source exists (speaker-ID ladder rung A over C). Pure.
148
+ *
149
+ * Only confident matches are returned: the winning name must cover
150
+ * >= minOverlapFraction of the acoustic speaker's speaking time AND beat the
151
+ * runner-up by >= dominanceRatio. Ambiguous / unmatched acoustic speakers are
152
+ * listed in `unmapped`, so the caller keeps their anonymous label.
153
+ *
154
+ * Note: two acoustic speakers can map to the same real name (acoustic over-
155
+ * split one person). That is acceptable - downstream can merge. 1:1 dedup is a
156
+ * future refinement.
157
+ *
158
+ * @param {Array<{start:number,end:number,speaker:string}>} acoustic utterances (ms)
159
+ * @param {Array<{start:number,end:number,name:string}>} labelled real-name turns (ms)
160
+ * @param {Partial<typeof SPEAKER_MAP_DEFAULTS>} [opts]
161
+ * @returns {{ map: Record<string,string>, unmapped: string[] }}
162
+ */
163
+ export function mapSpeakersToNames(acoustic, labelled, opts = {}) {
164
+ const t = { ...SPEAKER_MAP_DEFAULTS, ...opts }
165
+ const utts = Array.isArray(acoustic) ? acoustic.filter((u) => u && u.speaker) : []
166
+ const refs = Array.isArray(labelled) ? labelled.filter((r) => r && r.name) : []
167
+
168
+ const speakers = [...new Set(utts.map((u) => u.speaker))]
169
+ const map = {}
170
+ const unmapped = []
171
+ if (speakers.length === 0) return { map, unmapped }
172
+ if (refs.length === 0) return { map, unmapped: speakers }
173
+
174
+ const overlapMs = (a, b) => Math.max(0, Math.min(a.end, b.end) - Math.max(a.start, b.start))
175
+
176
+ for (const spk of speakers) {
177
+ const spkUtts = utts.filter((u) => u.speaker === spk)
178
+ const spkTotal = spkUtts.reduce((s, u) => s + Math.max(0, u.end - u.start), 0)
179
+
180
+ const byName = {}
181
+ for (const u of spkUtts) {
182
+ for (const r of refs) {
183
+ const ov = overlapMs(u, r)
184
+ if (ov > 0) byName[r.name] = (byName[r.name] || 0) + ov
185
+ }
186
+ }
187
+ const ranked = Object.entries(byName).sort((a, b) => b[1] - a[1])
188
+ if (ranked.length === 0) {
189
+ unmapped.push(spk)
190
+ continue
191
+ }
192
+ const [bestName, bestOv] = ranked[0]
193
+ const secondOv = ranked[1] ? ranked[1][1] : 0
194
+ const fraction = spkTotal > 0 ? bestOv / spkTotal : 0
195
+ const dominant = secondOv === 0 || bestOv >= secondOv * t.dominanceRatio
196
+
197
+ if (fraction >= t.minOverlapFraction && dominant) {
198
+ map[spk] = bestName
199
+ } else {
200
+ unmapped.push(spk)
201
+ }
202
+ }
203
+ return { map, unmapped }
204
+ }
@@ -20,7 +20,16 @@
20
20
  * and passed in. Caller manages caching.
21
21
  */
22
22
 
23
- import { possibleRoots } from '../welsh/mutations.js'
23
+ import { possibleRoots, applyMutation, MUTATION } from '../welsh/mutations.js'
24
+
25
+ // Edit-distance ceiling for the accept-list bias inside suggest().
26
+ // Raised to 3 (was 2 pre-audit 2026-05-20) so proper-noun typos one
27
+ // extra step beyond what nspell catches still surface the right
28
+ // glossary hit. Eg 'Gyfyrddin' -> 'Gaerfyrddin' is distance 3.
29
+ const ACCEPT_BIAS_MAX_DIST = 3
30
+
31
+ // Welsh mutations to try when forward-mutating accept-list entries.
32
+ const ACCEPT_MUTATION_KINDS = [MUTATION.soft, MUTATION.nasal, MUTATION.aspirate]
24
33
 
25
34
  /**
26
35
  * Check a single word against the supplied dictionary + accept-list.
@@ -41,6 +50,17 @@ export function checkWord(word, { nspell, accept, welsh }) {
41
50
  if (typeof word !== 'string' || word.length === 0) return true
42
51
  const cleaned = word.replace(/[^\p{L}\p{M}'’-]/gu, '')
43
52
  if (cleaned.length === 0) return true
53
+ // No letter characters at all (e.g. a bare hyphen) -- treat as OK.
54
+ if (!/\p{L}/u.test(cleaned)) return true
55
+ // Hyphenated compounds (e.g. 'cyd-destun', 'ar-ol', 'rhag-brawf'):
56
+ // Welsh dictionaries store roots, not every compound form. Split on
57
+ // hyphen and accept if every component passes individually.
58
+ if (cleaned.includes('-')) {
59
+ const parts = cleaned.split('-').filter(p => /\p{L}/u.test(p))
60
+ if (parts.length > 1) {
61
+ return parts.every(part => checkWord(part, { nspell, accept, welsh }))
62
+ }
63
+ }
44
64
  const lower = cleaned.toLowerCase()
45
65
  if (accept && accept.has(lower)) return true
46
66
  if (nspell.correct(cleaned)) return true
@@ -74,7 +94,12 @@ export function tokenise(text) {
74
94
  const re = /[\p{L}\p{M}'’-]+/gu
75
95
  let m
76
96
  while ((m = re.exec(text)) !== null) {
77
- tokens.push({ word: m[0], start: m.index, end: m.index + m[0].length })
97
+ // Guard: skip tokens with no letter at all. A bare '-' or apostrophe-only
98
+ // run is punctuation, not a word. Without this, 'word - word' produces a
99
+ // standalone '-' token that nspell.correct() rejects -> false red squiggle.
100
+ if (/\p{L}/u.test(m[0])) {
101
+ tokens.push({ word: m[0], start: m.index, end: m.index + m[0].length })
102
+ }
78
103
  }
79
104
  return tokens
80
105
  }
@@ -85,6 +110,9 @@ export function tokenise(text) {
85
110
  *
86
111
  * @param {string} text
87
112
  * @param {object} opts - same shape as checkWord opts
113
+ * @param {object} opts.nspell - An nspell instance.
114
+ * @param {Set<string>} [opts.accept] - Terms that must never be flagged.
115
+ * @param {boolean} [opts.welsh] - Try mutation reversal before flagging.
88
116
  * @returns {{ word: string, start: number, end: number }[]}
89
117
  */
90
118
  export function checkSentence(text, opts) {
@@ -100,22 +128,45 @@ export function checkSentence(text, opts) {
100
128
  * @param {object} opts
101
129
  * @param {object} opts.nspell
102
130
  * @param {Set<string>} [opts.accept]
131
+ * @param {boolean} [opts.welsh] - Bias suggestions for Welsh mutation forms.
103
132
  * @param {number} [opts.max] - default 6
104
133
  * @returns {string[]}
105
134
  */
106
- export function suggest(word, { nspell, accept, max = 6 }) {
135
+ export function suggest(word, { nspell, accept, welsh, max = 6 }) {
107
136
  if (typeof word !== 'string' || word.length === 0) return []
108
137
  const base = nspell.suggest(word) || []
109
138
  if (!accept || accept.size === 0) return base.slice(0, max)
110
- const acceptHits = []
111
139
  const lower = word.toLowerCase()
140
+
141
+ // Build candidate surface forms from each accept-list entry.
142
+ // Welsh: try the bare term plus each mutation, and pick whichever form
143
+ // is closest to the misspelling - so 'Caerfyrddin' in the glossary
144
+ // surfaces as 'Gaerfyrddin' when the typo was 'Gyfyrddin', preserving
145
+ // the grammatically-correct mutated surface form.
146
+ // English (welsh falsy): just the bare term, no mutation table.
147
+ const acceptHits = []
148
+ const seenHit = new Set()
112
149
  for (const term of accept) {
113
150
  if (term === lower) continue
114
- if (editDistance(lower, term, 2) <= 2) acceptHits.push(term)
151
+ const candidates = welsh ? mutatedCandidates(term) : [term]
152
+ let bestForm = null
153
+ let bestDist = ACCEPT_BIAS_MAX_DIST + 1
154
+ for (const cand of candidates) {
155
+ const d = editDistance(lower, cand.toLowerCase(), ACCEPT_BIAS_MAX_DIST)
156
+ if (d <= ACCEPT_BIAS_MAX_DIST && d < bestDist) {
157
+ bestDist = d
158
+ bestForm = cand
159
+ }
160
+ }
161
+ if (bestForm && !seenHit.has(bestForm.toLowerCase())) {
162
+ acceptHits.push(bestForm)
163
+ seenHit.add(bestForm.toLowerCase())
164
+ }
115
165
  }
166
+
116
167
  // De-dupe: glossary hits first, then nspell suggestions that aren't dupes.
117
- const seen = new Set(acceptHits.map(t => t.toLowerCase()))
118
168
  const out = [...acceptHits]
169
+ const seen = new Set(acceptHits.map(t => t.toLowerCase()))
119
170
  for (const s of base) {
120
171
  if (out.length >= max) break
121
172
  if (!seen.has(s.toLowerCase())) {
@@ -123,6 +174,22 @@ export function suggest(word, { nspell, accept, max = 6 }) {
123
174
  seen.add(s.toLowerCase())
124
175
  }
125
176
  }
177
+ return out.slice(0, max)
178
+ }
179
+
180
+ // Expand a Welsh accept-list term to its surface forms: bare + every
181
+ // mutation. applyMutation() returns the input unchanged when the initial
182
+ // consonant isn't subject to that mutation, so dedupe before returning.
183
+ function mutatedCandidates(term) {
184
+ const out = [term]
185
+ const seen = new Set([term.toLowerCase()])
186
+ for (const kind of ACCEPT_MUTATION_KINDS) {
187
+ const m = applyMutation(term, kind)
188
+ if (m && !seen.has(m.toLowerCase())) {
189
+ out.push(m)
190
+ seen.add(m.toLowerCase())
191
+ }
192
+ }
126
193
  return out
127
194
  }
128
195
 
@@ -20,6 +20,17 @@ const ZERO_WIDTH = /[\u200B\u200C\u200D\uFEFF]/g
20
20
  // Deterministic Welsh broken-contraction repairs (ASR artefacts)
21
21
  const CONTRACTION_REPAIRS = [
22
22
  [/\b(a|o|i|â) r\b/gi, "$1'r"],
23
+ // Verb + 'r: ASR drops the apostrophe before the definite article.
24
+ // e.g. "mae r athro" → "mae'r athro", "yw r plant" → "yw'r plant".
25
+ // Drift fix: these were absent from the drifted frontend copy (#449).
26
+ // VOWEL-ENDING verbs only ('r attaches after vowels).
27
+ [/\b(mae|yw|dyma|dyna) r\b/gi, "$1'r"],
28
+ // CONSONANT-ENDING verbs take the full article, not 'r: "oedd r ffordd"
29
+ // is the ASR mishearing "oedd y ffordd", and "oedd'r" is ungrammatical.
30
+ // Repair to yr before vowels/h, y otherwise (Codex on #1103; Aled's call
31
+ // 2026-06-12: restore the article rather than leave the artefact).
32
+ [/\b(oedd|fydd|bydd|sydd|roedd|caiff|gall) r (?=[aeiouwyhâêîôûŵŷáéíóúàèìòùäëïöü])/gi, '$1 yr '],
33
+ [/\b(oedd|fydd|bydd|sydd|roedd|caiff|gall) r\b/gi, '$1 y'],
23
34
  [/\bwedi (u|i|m|n|ch)\b/gi, "wedi'$1"],
24
35
  [/\bwedi eu\b/gi, "wedi'u"],
25
36
  [/\bwedi ei\b/gi, "wedi'i"],
@@ -0,0 +1,2 @@
1
+ export * from "./pure.js";
2
+ export * from "./tracker.js";
@@ -0,0 +1,21 @@
1
+ /**
2
+ * Compute the set of proper-noun corrections worth promoting.
3
+ *
4
+ * Diffs `before` against `after` on whitespace tokens and returns the
5
+ * corrected tokens that look like proper nouns. Structural edits (the
6
+ * token count changed) return `[]`: word-for-word swaps are the
7
+ * high-signal class and everything else is too noisy to learn from.
8
+ *
9
+ * In Capsiynau these terms become rows in `word_boost_approved` (per-org,
10
+ * tagged with the session's language) and the relay's loader treats those
11
+ * rows as language-agnostic on read - see the contract note on
12
+ * `looksLikeProperNoun` above for the full invariant and what must change
13
+ * if the heuristic ever expands beyond proper nouns. The function itself
14
+ * knows nothing about that table; a consumer on a different schema can
15
+ * do whatever it likes with the returned terms.
16
+ *
17
+ * @param {string} before Original text before the edit.
18
+ * @param {string} after Text after the edit.
19
+ * @returns {string[]} Deduplicated corrected tokens worth biasing toward.
20
+ */
21
+ export function extractProperNounCorrections(before: string, after: string): string[];
@@ -0,0 +1,36 @@
1
+ /**
2
+ * Upsert proper-noun corrections into word_boost_approved for the session's
3
+ * org + source language. Returns the list of terms persisted (excludes
4
+ * duplicates already in the table - upsert is silent on conflict).
5
+ *
6
+ * Never throws - failures are logged and swallowed so a segment-edit save
7
+ * is never blocked by a vocabulary-learning hiccup.
8
+ *
9
+ * CAPSIYNAU-SPECIFIC. Requires `live_sessions`, `live_segments` and
10
+ * `word_boost_approved`. If you only want the heuristic, import
11
+ * `extractProperNounCorrections` from
12
+ * `@capsiynau/intelligence/corrections/pure`.
13
+ *
14
+ * @param {string} before Original segment text before the edit.
15
+ * @param {string} after Text after the edit.
16
+ * @param {object} opts
17
+ * @param {string} opts.sessionId Live session UUID for org / fallback-lang lookup.
18
+ * @param {string} [opts.segmentId] Phase 1+ (notes-app#262): live_segments
19
+ * UUID for the edited segment. When provided,
20
+ * the segment's own `language` column is used
21
+ * for the new word_boost_approved row (drops
22
+ * silent rot in mixed-language sessions where
23
+ * Phase 3's per-chunk detected_language differs
24
+ * from session.source_language). Falls back to
25
+ * session.source_language if absent or the
26
+ * segment's language column is null.
27
+ * @param {object} opts.supabase Caller-provided Supabase client (DI).
28
+ * Must have `.from()` and `.rpc()` methods.
29
+ * RLS scopes writes by org.
30
+ * @returns {Promise<string[]>} Terms persisted (empty on any failure).
31
+ */
32
+ export function captureCorrections(before: string, after: string, { sessionId, segmentId, supabase }: {
33
+ sessionId: string;
34
+ segmentId?: string;
35
+ supabase: object;
36
+ }): Promise<string[]>;