lyric-align 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,19 @@
1
+ """lyric-align — place known lyrics on an audio timeline.
2
+
3
+ You already have the correct lyrics; you only need the *times*. lyric-align
4
+ anchors known lyric lines onto ASR word timings via character-level fuzzy
5
+ matching — built for space-less languages (Japanese, Chinese) and sung vocals
6
+ (including rap), where whitespace tokenization and forced aligners fall short.
7
+ """
8
+ from .anchor import align, auto_pairing, interpolate_gaps
9
+ from .model import AlignedLine, Segment, Word
10
+ from .normalize import normalize, similarity
11
+
12
+ __version__ = "0.3.0"
13
+
14
+ __all__ = [
15
+ "align", "auto_pairing", "interpolate_gaps",
16
+ "AlignedLine", "Segment", "Word",
17
+ "normalize", "similarity",
18
+ "__version__",
19
+ ]
lyric_align/anchor.py ADDED
@@ -0,0 +1,231 @@
1
+ """Anchor known lyric lines onto ASR segments (greedy forward fuzzy match).
2
+
3
+ This is the heart of lyric-align. Given the *correct* lyrics (from an official
4
+ source) and ASR segments with word timings, we place each lyric line/stanza on
5
+ the timeline by character-level fuzzy matching, moving forward monotonically so
6
+ repeated lines (choruses) consume segments in order.
7
+
8
+ The forward `window` is load-bearing, not just an optimisation: it keeps a line
9
+ from reaching a distant segment that happens to clear the threshold. Replacing
10
+ this scan with a globally optimal monotone assignment (maximise total similarity,
11
+ optionally minus a diagonal-drift penalty) was tried and measured worse — it
12
+ places every line, but repeated hooks carry no distinguishing similarity, so the
13
+ extra placements land on the wrong repetition: mean error 0.94 s -> 1.14-1.61 s,
14
+ worst 6.4 s -> 10.6 s on a track with a 4x-repeated hook. Locality beats global
15
+ optimality here because similarity alone cannot tell repetitions apart.
16
+
17
+ A timing prior — score candidates by `similarity - lam * |start - predicted|`,
18
+ predicting from the last placement plus the running median gap — was the obvious
19
+ next move and also measured worse (mean 0.94 s -> 1.67 s at lam=0.1, worst
20
+ 6.4 s -> 10.6 s, collapsing to 8/33 placements by lam=0.5). It predicts from
21
+ *our own* previous placements, so it cannot correct a bad one; it anchors on it
22
+ and drags the following lines along. Songs do not run at one pace either, so a
23
+ median gap mispredicts hardest across a section boundary — where repeated hooks
24
+ live. Restricted to breaking near-ties it stops hurting without clearly helping
25
+ (one line gained, below the ground truth's own resolution), so it is not here.
26
+ See the README for the full sweep.
27
+
28
+ Four further attempts to fix the remaining outlier — letting a candidate span
29
+ two consecutive segments, scoring candidates on how well they explain the unit's
30
+ *opening* rather than the whole unit, vetoing candidates that match only its
31
+ tail, and taking the start from the word the first character lands on — all
32
+ worked, and all cost more than they returned. The first three share a mechanism:
33
+ `idx` advances to just past whatever was chosen, so every neighbour is
34
+ downstream of every decision, and gains and losses arrive in adjacent pairs (the
35
+ best variant fixes 2.30 s -> 0.02 s at 2:20 on one track and breaks
36
+ 0.32 s -> 3.70 s at 2:26). The fourth fails differently: placements are already
37
+ late (signed mean +0.46 s / +0.21 s) because a sung phrase begins at its breath
38
+ and attack, earlier than the first word an ASR will timestamp, so refining into
39
+ the segment only adds lateness. See the README for both tables.
40
+
41
+ Letting the unit size vary — choosing, per step, how many lines this segment
42
+ should absorb instead of fixing it up front — was tried next and also measured
43
+ worse. It is a tempting move because `pairing` is a rounded average: at 76 lines
44
+ over 64 segments the true ratio is 1.19, so a fixed 1 leaves 27 lines unplaced
45
+ and a fixed 2 straddles boundaries. Scoring `(k, segment)` jointly does raise
46
+ placements (49/76 -> 66/76), and the first-line error barely moves. It is not
47
+ free: a unit chosen to maximise similarity will happily straddle a *section*
48
+ boundary, and such a unit needs only its tail to match. On the first track the
49
+ last verse line scores 0.000 against the hook segment alone and 0.286 once the
50
+ following hook line joins it — over the 0.25 threshold — so the breath split
51
+ drags that verse line 13.3 s forward into the hook, and displaces the correctly
52
+ placed hook line as collateral. Fixed pairing cannot do this at pairing=1: a
53
+ one-line unit has no tail to match with. Variable pairing manufactures the very
54
+ tail-match pathology that a dedicated veto (above) already failed to fix.
55
+
56
+ Note the shape of that failure, because it also refutes the reason for trying:
57
+ the earlier four attempts all changed *selection* (which segment a unit takes),
58
+ so the plan was to change *consumption* instead and avoid the coupling. It does
59
+ not avoid it. Consumption sets how fast the segment cursor advances relative to
60
+ the line cursor, so it moves `idx` too — just one step removed — and the same
61
+ gains-and-losses-in-adjacent-pairs behaviour returns. On the second track, where
62
+ the ASR merges two lines consistently, deviating downward orphans the remainder
63
+ onto the next segment and shifts every later unit's phase: within-0.5 s falls
64
+ 26/33 -> 18/33 and the 4x-repeated hook lands a repetition early. Restricting
65
+ deviation to *upward* only looks safe, but only because pairing=2 with k<=2
66
+ leaves it no room; allowing k<=3 breaks that track the same way (26/33 -> 13/33).
67
+
68
+ Design philosophy: when a line cannot be confidently matched, we mark it
69
+ unmatched rather than inventing a timestamp. Forced aligners always emit an
70
+ answer and thus fail *silently* (e.g. drifting into the intro); we prefer honest
71
+ gaps that a human — or a later interpolation pass — can fix. The same reasoning
72
+ rejects the global matcher above: a visible gap beats a confident wrong time.
73
+ The variable-pairing result is the sharpest case: its headline gain was 17 extra
74
+ placements, and on the only subset where those extra placements can be checked,
75
+ half were catastrophically wrong.
76
+ """
77
+ from __future__ import annotations
78
+
79
+ from .breath import split_words_by_breath
80
+ from .charmap import char_timings
81
+ from .model import AlignedLine, Segment
82
+ from .normalize import default_threshold, similarity
83
+
84
+
85
+ AUTO_PAIRING_MAX = 3
86
+
87
+
88
+ def auto_pairing(lyrics: list[str], segments: list[Segment]) -> int:
89
+ """How many lyric lines the ASR appears to have merged into one segment.
90
+
91
+ `pairing` is not a property of the song, it is a property of the *ASR's*
92
+ segmentation, and different models segment differently. faster-whisper
93
+ `medium` merges about two sung lines per segment on Japanese rap, which is
94
+ where the old fixed default of 2 came from — but `large-v3` splits much
95
+ finer (4.7 s average segment down to 2.8 s, 41 segments up to 64 on the same
96
+ four-minute track), so a two-line unit straddles a segment boundary and the
97
+ match degrades. Measured on that track: `large-v3` at the old default is
98
+ *worse* than `medium` (mean 0.50 s -> 1.13 s), and better than it once the
99
+ pairing follows (0.31 s, worst case 3.58 s -> 0.76 s). Upgrading the model
100
+ alone is a trap.
101
+
102
+ Lines per segment is exactly what pairing means, so it is also the estimate.
103
+ Capped at 3: beyond that the ASR has stopped producing line-like segments
104
+ (a full mix, where whole verses collapse into one), and the answer there is
105
+ to separate the vocal, not to widen the unit — which the CLI already says.
106
+ """
107
+ if not segments:
108
+ return 1
109
+ return max(1, min(AUTO_PAIRING_MAX, round(len(lyrics) / len(segments))))
110
+
111
+
112
+ def _stanzas(lines: list[str], pairing: int) -> list[list[str]]:
113
+ """Group lyric lines into stanza units of size `pairing` (1 = per line)."""
114
+ if pairing < 1:
115
+ pairing = 1
116
+ return [lines[i:i + pairing] for i in range(0, len(lines), pairing)]
117
+
118
+
119
+ def _containing(start: float, end: float, chars: list[dict] | None) -> tuple[float, float]:
120
+ """Widen a line span so it contains its own character timings.
121
+
122
+ A line takes its span from the segment, but the characters are interpolated
123
+ across the *word* timings, and faster-whisper does not guarantee that a
124
+ segment's end equals its last word's end. When it does not, the final
125
+ character runs past the line — which TTML forbids outright ("the timestamp
126
+ of a child element must be completely contained within the timestamp of its
127
+ parent") and which makes a strict player clamp or drop the tail.
128
+
129
+ The word timings are the precise signal here, so the line yields to them.
130
+ """
131
+ if not chars:
132
+ return start, end
133
+ return min(start, chars[0]["start"]), max(end, chars[-1]["end"])
134
+
135
+
136
+ def align(
137
+ segments: list[Segment],
138
+ lyrics: list[str],
139
+ *,
140
+ pairing: int | str = "auto",
141
+ threshold: float | None = None,
142
+ window: int = 4,
143
+ karaoke: bool = False,
144
+ ) -> list[AlignedLine]:
145
+ """Align known `lyrics` lines onto ASR `segments`.
146
+
147
+ Args:
148
+ pairing: lyric lines per stanza unit matched to one segment, or "auto"
149
+ (the default) to read it off the ASR's own segmentation — see
150
+ `auto_pairing`, which explains why a fixed value is tied to one
151
+ model. Pass an int to override.
152
+ threshold: minimum character similarity to accept a match. Defaults to a
153
+ script-aware value (see `normalize.default_threshold`).
154
+ window: how many segments ahead to search from the current position.
155
+ karaoke: if True, compute per-character timings (needs word timings).
156
+
157
+ Returns one AlignedLine per input lyric line (stanzas are expanded back to
158
+ lines via breath splitting).
159
+ """
160
+ if threshold is None:
161
+ threshold = default_threshold(lyrics)
162
+ if isinstance(pairing, str):
163
+ if pairing != "auto":
164
+ raise ValueError(f"pairing must be an int or 'auto', got {pairing!r}")
165
+ pairing = auto_pairing(lyrics, segments)
166
+
167
+ units = _stanzas(lyrics, pairing)
168
+ results: list[AlignedLine] = []
169
+ idx = 0
170
+
171
+ for unit in units:
172
+ joined = " ".join(unit)
173
+ best_score, best_i = 0.0, None
174
+ upper = min(idx + window, len(segments))
175
+ for i in range(idx, upper):
176
+ sc = similarity(joined, segments[i].text)
177
+ if sc > best_score:
178
+ best_score, best_i = sc, i
179
+
180
+ if best_i is not None and best_score > threshold:
181
+ seg = segments[best_i]
182
+ idx = best_i + 1
183
+ if len(unit) == 1 or not seg.words:
184
+ # Single line, or no word timings: use the segment span as-is.
185
+ for k, line in enumerate(unit):
186
+ chars = char_timings(line, seg.words) if (karaoke and k == 0) else None
187
+ start, end = _containing(seg.start, seg.end, chars)
188
+ results.append(AlignedLine(line, start, end,
189
+ best_score, True, chars))
190
+ else:
191
+ groups = split_words_by_breath(seg.words, unit)
192
+ for line, grp in zip(unit, groups):
193
+ if grp:
194
+ chars = char_timings(line, grp) if karaoke else None
195
+ start, end = _containing(grp[0].start, grp[-1].end, chars)
196
+ results.append(AlignedLine(line, start, end,
197
+ best_score, True, chars))
198
+ else:
199
+ results.append(AlignedLine(line, seg.start, seg.end,
200
+ best_score, True, None))
201
+ else:
202
+ for line in unit:
203
+ results.append(AlignedLine(line, None, None, best_score, False, None))
204
+
205
+ return results
206
+
207
+
208
+ def interpolate_gaps(aligned: list[AlignedLine]) -> list[AlignedLine]:
209
+ """Fill unmatched lines by linear interpolation between known neighbors.
210
+
211
+ Optional convenience for callers who want a fully-populated timeline. The
212
+ `matched` flag stays False so downstream code can still tell which lines
213
+ were guessed.
214
+ """
215
+ n = len(aligned)
216
+ for i, a in enumerate(aligned):
217
+ if a.matched or a.start is not None:
218
+ continue
219
+ prev = next((aligned[j] for j in range(i - 1, -1, -1)
220
+ if aligned[j].start is not None), None)
221
+ nxt = next((aligned[j] for j in range(i + 1, n)
222
+ if aligned[j].start is not None), None)
223
+ if prev and nxt:
224
+ span = (nxt.start - prev.end)
225
+ a.start = prev.end + span * 0.33
226
+ a.end = prev.end + span * 0.66
227
+ elif prev:
228
+ a.start, a.end = prev.end, prev.end + 2.0
229
+ elif nxt:
230
+ a.start, a.end = max(0.0, nxt.start - 2.0), nxt.start
231
+ return aligned
lyric_align/asr.py ADDED
@@ -0,0 +1,47 @@
1
+ """ASR backend: faster-whisper (optional dependency).
2
+
3
+ Kept behind a lazy import so the core aligner has zero heavy dependencies. If
4
+ you already have segments (from any Whisper flavor), skip this and feed the
5
+ aligner directly via `Segment.from_dict`.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ from pathlib import Path
10
+
11
+ from .model import Segment, Word
12
+
13
+
14
+ def transcribe(
15
+ audio_path: str | Path,
16
+ *,
17
+ language: str = "ja",
18
+ model_size: str = "medium",
19
+ device: str = "cpu",
20
+ compute_type: str = "int8",
21
+ vad: bool = True,
22
+ ) -> list[Segment]:
23
+ """Transcribe with faster-whisper, returning word-timed segments.
24
+
25
+ Requires the `[asr]` extra: pip install lyric-align[asr]
26
+ """
27
+ try:
28
+ from faster_whisper import WhisperModel
29
+ except ImportError as e: # pragma: no cover
30
+ raise ImportError(
31
+ "faster-whisper is required for transcription. "
32
+ "Install with: pip install lyric-align[asr]"
33
+ ) from e
34
+
35
+ model = WhisperModel(model_size, device=device, compute_type=compute_type)
36
+ kwargs = dict(language=language, word_timestamps=True)
37
+ if vad:
38
+ kwargs.update(vad_filter=True,
39
+ vad_parameters=dict(min_silence_duration_ms=500))
40
+ segments, _ = model.transcribe(str(audio_path), **kwargs)
41
+
42
+ out = []
43
+ for seg in segments:
44
+ words = [Word(float(w.start), float(w.end), w.word)
45
+ for w in (seg.words or [])]
46
+ out.append(Segment(float(seg.start), float(seg.end), seg.text.strip(), words))
47
+ return out
lyric_align/breath.py ADDED
@@ -0,0 +1,52 @@
1
+ """Split a merged ASR segment back into individual lyric lines.
2
+
3
+ ASR (Whisper etc.) tends to merge several sung lines into one segment. When we
4
+ know how many lines a segment should contain, we can recover per-line
5
+ boundaries by cutting at the largest inter-word time gap (a breath) near the
6
+ expected split point — estimated from the character-count ratio of the known
7
+ lines.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ from .model import Word
12
+ from .normalize import nchars
13
+
14
+
15
+ def split_words_by_breath(words: list[Word], lines: list[str],
16
+ search: int = 3) -> list[list[Word]]:
17
+ """Partition `words` into len(lines) groups.
18
+
19
+ For each split point, aim at the index implied by cumulative character
20
+ ratio of `lines`, then within ±`search` words pick the boundary with the
21
+ largest silence gap (the breath). Returns one word-list per line; groups
22
+ may be empty if there are fewer words than lines.
23
+ """
24
+ n = len(words)
25
+ if not lines:
26
+ return []
27
+ if len(lines) == 1 or n <= 1:
28
+ return [words] + [[] for _ in lines[1:]]
29
+
30
+ total_chars = sum(nchars(l) for l in lines) or 1
31
+ groups: list[list[Word]] = []
32
+ start = 0
33
+ cum_chars = 0
34
+ for li, line in enumerate(lines[:-1]):
35
+ cum_chars += nchars(line)
36
+ # words remaining must cover the remaining lines (>=1 each)
37
+ remaining_lines = len(lines) - li - 1
38
+ target = round(n * cum_chars / total_chars)
39
+ lo = max(start + 1, target - search)
40
+ hi = min(n - remaining_lines, target + search + 1)
41
+ if hi <= lo:
42
+ cut = min(max(lo, start + 1), n - remaining_lines)
43
+ else:
44
+ best_gap, cut = -1.0, lo
45
+ for i in range(lo, hi):
46
+ gap = words[i].start - words[i - 1].end
47
+ if gap > best_gap:
48
+ best_gap, cut = gap, i
49
+ groups.append(words[start:cut])
50
+ start = cut
51
+ groups.append(words[start:])
52
+ return groups
lyric_align/charmap.py ADDED
@@ -0,0 +1,117 @@
1
+ """Map known characters to sung times by proportional interpolation.
2
+
3
+ Given the ASR word boundaries for a line and the *known* character string, we
4
+ distribute the characters proportionally across the word-time span. This lets
5
+ karaoke output use the correct lyric text while borrowing only the timing from
6
+ the (possibly misspelled) ASR words.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ from .model import Word
11
+ from .normalize import cjk_ratio, nchars
12
+
13
+
14
+ MIN_CHAR_DUR = 0.01
15
+
16
+
17
+ def char_timings(text: str, words: list[Word]) -> list[dict]:
18
+ """Return [{"char", "start", "end"}] for each non-space character in text.
19
+
20
+ Spaces are skipped (no karaoke syllable). If `words` is empty, returns [].
21
+
22
+ Durations are kept strictly positive and non-overlapping: proportional
23
+ interpolation plus rounding to milliseconds can otherwise collapse a
24
+ character to zero length, which downstream means a `\\k0` sweep or a TTML span
25
+ with begin == end.
26
+ """
27
+ if not words:
28
+ return []
29
+ m = nchars(text)
30
+ if m == 0:
31
+ return []
32
+ bounds = [w.start for w in words] + [words[-1].end]
33
+ n = len(bounds) - 1
34
+
35
+ def t_at(j: float) -> float:
36
+ pos = j * n / m
37
+ i = min(int(pos), n - 1)
38
+ return bounds[i] + (bounds[i + 1] - bounds[i]) * (pos - i)
39
+
40
+ # A very short span cannot give every character MIN_CHAR_DUR; share it out.
41
+ span = max(bounds[-1] - bounds[0], 0.0)
42
+ min_dur = min(MIN_CHAR_DUR, span / m) if m else 0.0
43
+
44
+ out = []
45
+ j = 0
46
+ prev_end = bounds[0]
47
+ for ch in text:
48
+ if ch == ' ':
49
+ continue
50
+ start = max(t_at(j), prev_end)
51
+ end = max(t_at(j + 1), start + min_dur)
52
+ out.append({"char": ch, "start": round(start, 3), "end": round(end, 3)})
53
+ prev_end = end
54
+ j += 1
55
+
56
+ # Rounding can re-introduce a collision at millisecond resolution.
57
+ for a, b in zip(out, out[1:]):
58
+ if b["start"] < a["end"]:
59
+ b["start"] = a["end"]
60
+ if b["end"] <= b["start"]:
61
+ b["end"] = round(b["start"] + MIN_CHAR_DUR, 3)
62
+ if out and out[0]["end"] <= out[0]["start"]:
63
+ out[0]["end"] = round(out[0]["start"] + MIN_CHAR_DUR, 3)
64
+ return out
65
+
66
+
67
+ def syllable_timings(text: str, chars: list[dict]) -> list[dict]:
68
+ """Group per-character timings into the units a karaoke format highlights.
69
+
70
+ Returns ``[{"text", "start", "end", "space_after"}]``. The grouping differs
71
+ by script, because what counts as a "syllable" to highlight does:
72
+
73
+ - Alphabetic text is grouped on whitespace, so "Amazing grace" highlights two
74
+ words rather than twelve letters.
75
+ - CJK text keeps one unit per character, which is how per-character karaoke
76
+ formats (QQ/NetEase style) treat Chinese and Japanese. Japanese lyrics often
77
+ contain spaces as phrasing, so splitting on them would give useless chunks.
78
+
79
+ ``space_after`` records whether whitespace followed the unit *in the source
80
+ line*, so a formatter can rebuild the line exactly instead of guessing a
81
+ separator. The guess is what breaks CJK: a Japanese line is one unit per
82
+ character yet often carries a phrasing space, so any line-wide "is this
83
+ spaced text?" test either inserts a space between every character or drops
84
+ the phrasing space entirely.
85
+ """
86
+ if not chars:
87
+ return []
88
+ per_char = cjk_ratio(text) >= 0.2
89
+
90
+ units: list[list] = [] # [source_text, [timings], space_after]
91
+ buf_text, buf_times = "", []
92
+
93
+ def flush() -> None:
94
+ nonlocal buf_text, buf_times
95
+ if buf_times:
96
+ units.append([buf_text, buf_times, False])
97
+ buf_text, buf_times = "", []
98
+
99
+ it = iter(chars)
100
+ for ch in text:
101
+ if ch.isspace():
102
+ flush()
103
+ if units:
104
+ units[-1][2] = True
105
+ continue
106
+ try:
107
+ t = next(it)
108
+ except StopIteration:
109
+ break
110
+ buf_text += ch
111
+ buf_times.append(t)
112
+ if per_char:
113
+ flush()
114
+ flush()
115
+
116
+ return [{"text": t, "start": ts[0]["start"], "end": ts[-1]["end"],
117
+ "space_after": sp} for t, ts, sp in units]