lyric-align 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lyric_align/__init__.py +19 -0
- lyric_align/anchor.py +231 -0
- lyric_align/asr.py +47 -0
- lyric_align/breath.py +52 -0
- lyric_align/charmap.py +117 -0
- lyric_align/cli.py +326 -0
- lyric_align/formats.py +319 -0
- lyric_align/model.py +53 -0
- lyric_align/normalize.py +69 -0
- lyric_align/separate.py +84 -0
- lyric_align-0.3.0.dist-info/METADATA +627 -0
- lyric_align-0.3.0.dist-info/RECORD +16 -0
- lyric_align-0.3.0.dist-info/WHEEL +4 -0
- lyric_align-0.3.0.dist-info/entry_points.txt +2 -0
- lyric_align-0.3.0.dist-info/licenses/LICENSE +21 -0
- lyric_align-0.3.0.dist-info/licenses/NOTICE +15 -0
lyric_align/__init__.py
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"""lyric-align — place known lyrics on an audio timeline.
|
|
2
|
+
|
|
3
|
+
You already have the correct lyrics; you only need the *times*. lyric-align
|
|
4
|
+
anchors known lyric lines onto ASR word timings via character-level fuzzy
|
|
5
|
+
matching — built for space-less languages (Japanese, Chinese) and sung vocals
|
|
6
|
+
(including rap), where whitespace tokenization and forced aligners fall short.
|
|
7
|
+
"""
|
|
8
|
+
from .anchor import align, auto_pairing, interpolate_gaps
|
|
9
|
+
from .model import AlignedLine, Segment, Word
|
|
10
|
+
from .normalize import normalize, similarity
|
|
11
|
+
|
|
12
|
+
__version__ = "0.3.0"
|
|
13
|
+
|
|
14
|
+
__all__ = [
|
|
15
|
+
"align", "auto_pairing", "interpolate_gaps",
|
|
16
|
+
"AlignedLine", "Segment", "Word",
|
|
17
|
+
"normalize", "similarity",
|
|
18
|
+
"__version__",
|
|
19
|
+
]
|
lyric_align/anchor.py
ADDED
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
"""Anchor known lyric lines onto ASR segments (greedy forward fuzzy match).
|
|
2
|
+
|
|
3
|
+
This is the heart of lyric-align. Given the *correct* lyrics (from an official
|
|
4
|
+
source) and ASR segments with word timings, we place each lyric line/stanza on
|
|
5
|
+
the timeline by character-level fuzzy matching, moving forward monotonically so
|
|
6
|
+
repeated lines (choruses) consume segments in order.
|
|
7
|
+
|
|
8
|
+
The forward `window` is load-bearing, not just an optimisation: it keeps a line
|
|
9
|
+
from reaching a distant segment that happens to clear the threshold. Replacing
|
|
10
|
+
this scan with a globally optimal monotone assignment (maximise total similarity,
|
|
11
|
+
optionally minus a diagonal-drift penalty) was tried and measured worse — it
|
|
12
|
+
places every line, but repeated hooks carry no distinguishing similarity, so the
|
|
13
|
+
extra placements land on the wrong repetition: mean error 0.94 s -> 1.14-1.61 s,
|
|
14
|
+
worst 6.4 s -> 10.6 s on a track with a 4x-repeated hook. Locality beats global
|
|
15
|
+
optimality here because similarity alone cannot tell repetitions apart.
|
|
16
|
+
|
|
17
|
+
A timing prior — score candidates by `similarity - lam * |start - predicted|`,
|
|
18
|
+
predicting from the last placement plus the running median gap — was the obvious
|
|
19
|
+
next move and also measured worse (mean 0.94 s -> 1.67 s at lam=0.1, worst
|
|
20
|
+
6.4 s -> 10.6 s, collapsing to 8/33 placements by lam=0.5). It predicts from
|
|
21
|
+
*our own* previous placements, so it cannot correct a bad one; it anchors on it
|
|
22
|
+
and drags the following lines along. Songs do not run at one pace either, so a
|
|
23
|
+
median gap mispredicts hardest across a section boundary — where repeated hooks
|
|
24
|
+
live. Restricted to breaking near-ties it stops hurting without clearly helping
|
|
25
|
+
(one line gained, below the ground truth's own resolution), so it is not here.
|
|
26
|
+
See the README for the full sweep.
|
|
27
|
+
|
|
28
|
+
Four further attempts to fix the remaining outlier — letting a candidate span
|
|
29
|
+
two consecutive segments, scoring candidates on how well they explain the unit's
|
|
30
|
+
*opening* rather than the whole unit, vetoing candidates that match only its
|
|
31
|
+
tail, and taking the start from the word the first character lands on — all
|
|
32
|
+
worked, and all cost more than they returned. The first three share a mechanism:
|
|
33
|
+
`idx` advances to just past whatever was chosen, so every neighbour is
|
|
34
|
+
downstream of every decision, and gains and losses arrive in adjacent pairs (the
|
|
35
|
+
best variant fixes 2.30 s -> 0.02 s at 2:20 on one track and breaks
|
|
36
|
+
0.32 s -> 3.70 s at 2:26). The fourth fails differently: placements are already
|
|
37
|
+
late (signed mean +0.46 s / +0.21 s) because a sung phrase begins at its breath
|
|
38
|
+
and attack, earlier than the first word an ASR will timestamp, so refining into
|
|
39
|
+
the segment only adds lateness. See the README for both tables.
|
|
40
|
+
|
|
41
|
+
Letting the unit size vary — choosing, per step, how many lines this segment
|
|
42
|
+
should absorb instead of fixing it up front — was tried next and also measured
|
|
43
|
+
worse. It is a tempting move because `pairing` is a rounded average: at 76 lines
|
|
44
|
+
over 64 segments the true ratio is 1.19, so a fixed 1 leaves 27 lines unplaced
|
|
45
|
+
and a fixed 2 straddles boundaries. Scoring `(k, segment)` jointly does raise
|
|
46
|
+
placements (49/76 -> 66/76), and the first-line error barely moves. It is not
|
|
47
|
+
free: a unit chosen to maximise similarity will happily straddle a *section*
|
|
48
|
+
boundary, and such a unit needs only its tail to match. On the first track the
|
|
49
|
+
last verse line scores 0.000 against the hook segment alone and 0.286 once the
|
|
50
|
+
following hook line joins it — over the 0.25 threshold — so the breath split
|
|
51
|
+
drags that verse line 13.3 s forward into the hook, and displaces the correctly
|
|
52
|
+
placed hook line as collateral. Fixed pairing cannot do this at pairing=1: a
|
|
53
|
+
one-line unit has no tail to match with. Variable pairing manufactures the very
|
|
54
|
+
tail-match pathology that a dedicated veto (above) already failed to fix.
|
|
55
|
+
|
|
56
|
+
Note the shape of that failure, because it also refutes the reason for trying:
|
|
57
|
+
the earlier four attempts all changed *selection* (which segment a unit takes),
|
|
58
|
+
so the plan was to change *consumption* instead and avoid the coupling. It does
|
|
59
|
+
not avoid it. Consumption sets how fast the segment cursor advances relative to
|
|
60
|
+
the line cursor, so it moves `idx` too — just one step removed — and the same
|
|
61
|
+
gains-and-losses-in-adjacent-pairs behaviour returns. On the second track, where
|
|
62
|
+
the ASR merges two lines consistently, deviating downward orphans the remainder
|
|
63
|
+
onto the next segment and shifts every later unit's phase: within-0.5 s falls
|
|
64
|
+
26/33 -> 18/33 and the 4x-repeated hook lands a repetition early. Restricting
|
|
65
|
+
deviation to *upward* only looks safe, but only because pairing=2 with k<=2
|
|
66
|
+
leaves it no room; allowing k<=3 breaks that track the same way (26/33 -> 13/33).
|
|
67
|
+
|
|
68
|
+
Design philosophy: when a line cannot be confidently matched, we mark it
|
|
69
|
+
unmatched rather than inventing a timestamp. Forced aligners always emit an
|
|
70
|
+
answer and thus fail *silently* (e.g. drifting into the intro); we prefer honest
|
|
71
|
+
gaps that a human — or a later interpolation pass — can fix. The same reasoning
|
|
72
|
+
rejects the global matcher above: a visible gap beats a confident wrong time.
|
|
73
|
+
The variable-pairing result is the sharpest case: its headline gain was 17 extra
|
|
74
|
+
placements, and on the only subset where those extra placements can be checked,
|
|
75
|
+
half were catastrophically wrong.
|
|
76
|
+
"""
|
|
77
|
+
from __future__ import annotations
|
|
78
|
+
|
|
79
|
+
from .breath import split_words_by_breath
|
|
80
|
+
from .charmap import char_timings
|
|
81
|
+
from .model import AlignedLine, Segment
|
|
82
|
+
from .normalize import default_threshold, similarity
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
AUTO_PAIRING_MAX = 3
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def auto_pairing(lyrics: list[str], segments: list[Segment]) -> int:
|
|
89
|
+
"""How many lyric lines the ASR appears to have merged into one segment.
|
|
90
|
+
|
|
91
|
+
`pairing` is not a property of the song, it is a property of the *ASR's*
|
|
92
|
+
segmentation, and different models segment differently. faster-whisper
|
|
93
|
+
`medium` merges about two sung lines per segment on Japanese rap, which is
|
|
94
|
+
where the old fixed default of 2 came from — but `large-v3` splits much
|
|
95
|
+
finer (4.7 s average segment down to 2.8 s, 41 segments up to 64 on the same
|
|
96
|
+
four-minute track), so a two-line unit straddles a segment boundary and the
|
|
97
|
+
match degrades. Measured on that track: `large-v3` at the old default is
|
|
98
|
+
*worse* than `medium` (mean 0.50 s -> 1.13 s), and better than it once the
|
|
99
|
+
pairing follows (0.31 s, worst case 3.58 s -> 0.76 s). Upgrading the model
|
|
100
|
+
alone is a trap.
|
|
101
|
+
|
|
102
|
+
Lines per segment is exactly what pairing means, so it is also the estimate.
|
|
103
|
+
Capped at 3: beyond that the ASR has stopped producing line-like segments
|
|
104
|
+
(a full mix, where whole verses collapse into one), and the answer there is
|
|
105
|
+
to separate the vocal, not to widen the unit — which the CLI already says.
|
|
106
|
+
"""
|
|
107
|
+
if not segments:
|
|
108
|
+
return 1
|
|
109
|
+
return max(1, min(AUTO_PAIRING_MAX, round(len(lyrics) / len(segments))))
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def _stanzas(lines: list[str], pairing: int) -> list[list[str]]:
|
|
113
|
+
"""Group lyric lines into stanza units of size `pairing` (1 = per line)."""
|
|
114
|
+
if pairing < 1:
|
|
115
|
+
pairing = 1
|
|
116
|
+
return [lines[i:i + pairing] for i in range(0, len(lines), pairing)]
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _containing(start: float, end: float, chars: list[dict] | None) -> tuple[float, float]:
|
|
120
|
+
"""Widen a line span so it contains its own character timings.
|
|
121
|
+
|
|
122
|
+
A line takes its span from the segment, but the characters are interpolated
|
|
123
|
+
across the *word* timings, and faster-whisper does not guarantee that a
|
|
124
|
+
segment's end equals its last word's end. When it does not, the final
|
|
125
|
+
character runs past the line — which TTML forbids outright ("the timestamp
|
|
126
|
+
of a child element must be completely contained within the timestamp of its
|
|
127
|
+
parent") and which makes a strict player clamp or drop the tail.
|
|
128
|
+
|
|
129
|
+
The word timings are the precise signal here, so the line yields to them.
|
|
130
|
+
"""
|
|
131
|
+
if not chars:
|
|
132
|
+
return start, end
|
|
133
|
+
return min(start, chars[0]["start"]), max(end, chars[-1]["end"])
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def align(
|
|
137
|
+
segments: list[Segment],
|
|
138
|
+
lyrics: list[str],
|
|
139
|
+
*,
|
|
140
|
+
pairing: int | str = "auto",
|
|
141
|
+
threshold: float | None = None,
|
|
142
|
+
window: int = 4,
|
|
143
|
+
karaoke: bool = False,
|
|
144
|
+
) -> list[AlignedLine]:
|
|
145
|
+
"""Align known `lyrics` lines onto ASR `segments`.
|
|
146
|
+
|
|
147
|
+
Args:
|
|
148
|
+
pairing: lyric lines per stanza unit matched to one segment, or "auto"
|
|
149
|
+
(the default) to read it off the ASR's own segmentation — see
|
|
150
|
+
`auto_pairing`, which explains why a fixed value is tied to one
|
|
151
|
+
model. Pass an int to override.
|
|
152
|
+
threshold: minimum character similarity to accept a match. Defaults to a
|
|
153
|
+
script-aware value (see `normalize.default_threshold`).
|
|
154
|
+
window: how many segments ahead to search from the current position.
|
|
155
|
+
karaoke: if True, compute per-character timings (needs word timings).
|
|
156
|
+
|
|
157
|
+
Returns one AlignedLine per input lyric line (stanzas are expanded back to
|
|
158
|
+
lines via breath splitting).
|
|
159
|
+
"""
|
|
160
|
+
if threshold is None:
|
|
161
|
+
threshold = default_threshold(lyrics)
|
|
162
|
+
if isinstance(pairing, str):
|
|
163
|
+
if pairing != "auto":
|
|
164
|
+
raise ValueError(f"pairing must be an int or 'auto', got {pairing!r}")
|
|
165
|
+
pairing = auto_pairing(lyrics, segments)
|
|
166
|
+
|
|
167
|
+
units = _stanzas(lyrics, pairing)
|
|
168
|
+
results: list[AlignedLine] = []
|
|
169
|
+
idx = 0
|
|
170
|
+
|
|
171
|
+
for unit in units:
|
|
172
|
+
joined = " ".join(unit)
|
|
173
|
+
best_score, best_i = 0.0, None
|
|
174
|
+
upper = min(idx + window, len(segments))
|
|
175
|
+
for i in range(idx, upper):
|
|
176
|
+
sc = similarity(joined, segments[i].text)
|
|
177
|
+
if sc > best_score:
|
|
178
|
+
best_score, best_i = sc, i
|
|
179
|
+
|
|
180
|
+
if best_i is not None and best_score > threshold:
|
|
181
|
+
seg = segments[best_i]
|
|
182
|
+
idx = best_i + 1
|
|
183
|
+
if len(unit) == 1 or not seg.words:
|
|
184
|
+
# Single line, or no word timings: use the segment span as-is.
|
|
185
|
+
for k, line in enumerate(unit):
|
|
186
|
+
chars = char_timings(line, seg.words) if (karaoke and k == 0) else None
|
|
187
|
+
start, end = _containing(seg.start, seg.end, chars)
|
|
188
|
+
results.append(AlignedLine(line, start, end,
|
|
189
|
+
best_score, True, chars))
|
|
190
|
+
else:
|
|
191
|
+
groups = split_words_by_breath(seg.words, unit)
|
|
192
|
+
for line, grp in zip(unit, groups):
|
|
193
|
+
if grp:
|
|
194
|
+
chars = char_timings(line, grp) if karaoke else None
|
|
195
|
+
start, end = _containing(grp[0].start, grp[-1].end, chars)
|
|
196
|
+
results.append(AlignedLine(line, start, end,
|
|
197
|
+
best_score, True, chars))
|
|
198
|
+
else:
|
|
199
|
+
results.append(AlignedLine(line, seg.start, seg.end,
|
|
200
|
+
best_score, True, None))
|
|
201
|
+
else:
|
|
202
|
+
for line in unit:
|
|
203
|
+
results.append(AlignedLine(line, None, None, best_score, False, None))
|
|
204
|
+
|
|
205
|
+
return results
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def interpolate_gaps(aligned: list[AlignedLine]) -> list[AlignedLine]:
|
|
209
|
+
"""Fill unmatched lines by linear interpolation between known neighbors.
|
|
210
|
+
|
|
211
|
+
Optional convenience for callers who want a fully-populated timeline. The
|
|
212
|
+
`matched` flag stays False so downstream code can still tell which lines
|
|
213
|
+
were guessed.
|
|
214
|
+
"""
|
|
215
|
+
n = len(aligned)
|
|
216
|
+
for i, a in enumerate(aligned):
|
|
217
|
+
if a.matched or a.start is not None:
|
|
218
|
+
continue
|
|
219
|
+
prev = next((aligned[j] for j in range(i - 1, -1, -1)
|
|
220
|
+
if aligned[j].start is not None), None)
|
|
221
|
+
nxt = next((aligned[j] for j in range(i + 1, n)
|
|
222
|
+
if aligned[j].start is not None), None)
|
|
223
|
+
if prev and nxt:
|
|
224
|
+
span = (nxt.start - prev.end)
|
|
225
|
+
a.start = prev.end + span * 0.33
|
|
226
|
+
a.end = prev.end + span * 0.66
|
|
227
|
+
elif prev:
|
|
228
|
+
a.start, a.end = prev.end, prev.end + 2.0
|
|
229
|
+
elif nxt:
|
|
230
|
+
a.start, a.end = max(0.0, nxt.start - 2.0), nxt.start
|
|
231
|
+
return aligned
|
lyric_align/asr.py
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""ASR backend: faster-whisper (optional dependency).
|
|
2
|
+
|
|
3
|
+
Kept behind a lazy import so the core aligner has zero heavy dependencies. If
|
|
4
|
+
you already have segments (from any Whisper flavor), skip this and feed the
|
|
5
|
+
aligner directly via `Segment.from_dict`.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
from .model import Segment, Word
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def transcribe(
|
|
15
|
+
audio_path: str | Path,
|
|
16
|
+
*,
|
|
17
|
+
language: str = "ja",
|
|
18
|
+
model_size: str = "medium",
|
|
19
|
+
device: str = "cpu",
|
|
20
|
+
compute_type: str = "int8",
|
|
21
|
+
vad: bool = True,
|
|
22
|
+
) -> list[Segment]:
|
|
23
|
+
"""Transcribe with faster-whisper, returning word-timed segments.
|
|
24
|
+
|
|
25
|
+
Requires the `[asr]` extra: pip install lyric-align[asr]
|
|
26
|
+
"""
|
|
27
|
+
try:
|
|
28
|
+
from faster_whisper import WhisperModel
|
|
29
|
+
except ImportError as e: # pragma: no cover
|
|
30
|
+
raise ImportError(
|
|
31
|
+
"faster-whisper is required for transcription. "
|
|
32
|
+
"Install with: pip install lyric-align[asr]"
|
|
33
|
+
) from e
|
|
34
|
+
|
|
35
|
+
model = WhisperModel(model_size, device=device, compute_type=compute_type)
|
|
36
|
+
kwargs = dict(language=language, word_timestamps=True)
|
|
37
|
+
if vad:
|
|
38
|
+
kwargs.update(vad_filter=True,
|
|
39
|
+
vad_parameters=dict(min_silence_duration_ms=500))
|
|
40
|
+
segments, _ = model.transcribe(str(audio_path), **kwargs)
|
|
41
|
+
|
|
42
|
+
out = []
|
|
43
|
+
for seg in segments:
|
|
44
|
+
words = [Word(float(w.start), float(w.end), w.word)
|
|
45
|
+
for w in (seg.words or [])]
|
|
46
|
+
out.append(Segment(float(seg.start), float(seg.end), seg.text.strip(), words))
|
|
47
|
+
return out
|
lyric_align/breath.py
ADDED
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
"""Split a merged ASR segment back into individual lyric lines.
|
|
2
|
+
|
|
3
|
+
ASR (Whisper etc.) tends to merge several sung lines into one segment. When we
|
|
4
|
+
know how many lines a segment should contain, we can recover per-line
|
|
5
|
+
boundaries by cutting at the largest inter-word time gap (a breath) near the
|
|
6
|
+
expected split point — estimated from the character-count ratio of the known
|
|
7
|
+
lines.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from .model import Word
|
|
12
|
+
from .normalize import nchars
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def split_words_by_breath(words: list[Word], lines: list[str],
|
|
16
|
+
search: int = 3) -> list[list[Word]]:
|
|
17
|
+
"""Partition `words` into len(lines) groups.
|
|
18
|
+
|
|
19
|
+
For each split point, aim at the index implied by cumulative character
|
|
20
|
+
ratio of `lines`, then within ±`search` words pick the boundary with the
|
|
21
|
+
largest silence gap (the breath). Returns one word-list per line; groups
|
|
22
|
+
may be empty if there are fewer words than lines.
|
|
23
|
+
"""
|
|
24
|
+
n = len(words)
|
|
25
|
+
if not lines:
|
|
26
|
+
return []
|
|
27
|
+
if len(lines) == 1 or n <= 1:
|
|
28
|
+
return [words] + [[] for _ in lines[1:]]
|
|
29
|
+
|
|
30
|
+
total_chars = sum(nchars(l) for l in lines) or 1
|
|
31
|
+
groups: list[list[Word]] = []
|
|
32
|
+
start = 0
|
|
33
|
+
cum_chars = 0
|
|
34
|
+
for li, line in enumerate(lines[:-1]):
|
|
35
|
+
cum_chars += nchars(line)
|
|
36
|
+
# words remaining must cover the remaining lines (>=1 each)
|
|
37
|
+
remaining_lines = len(lines) - li - 1
|
|
38
|
+
target = round(n * cum_chars / total_chars)
|
|
39
|
+
lo = max(start + 1, target - search)
|
|
40
|
+
hi = min(n - remaining_lines, target + search + 1)
|
|
41
|
+
if hi <= lo:
|
|
42
|
+
cut = min(max(lo, start + 1), n - remaining_lines)
|
|
43
|
+
else:
|
|
44
|
+
best_gap, cut = -1.0, lo
|
|
45
|
+
for i in range(lo, hi):
|
|
46
|
+
gap = words[i].start - words[i - 1].end
|
|
47
|
+
if gap > best_gap:
|
|
48
|
+
best_gap, cut = gap, i
|
|
49
|
+
groups.append(words[start:cut])
|
|
50
|
+
start = cut
|
|
51
|
+
groups.append(words[start:])
|
|
52
|
+
return groups
|
lyric_align/charmap.py
ADDED
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
"""Map known characters to sung times by proportional interpolation.
|
|
2
|
+
|
|
3
|
+
Given the ASR word boundaries for a line and the *known* character string, we
|
|
4
|
+
distribute the characters proportionally across the word-time span. This lets
|
|
5
|
+
karaoke output use the correct lyric text while borrowing only the timing from
|
|
6
|
+
the (possibly misspelled) ASR words.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from .model import Word
|
|
11
|
+
from .normalize import cjk_ratio, nchars
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
MIN_CHAR_DUR = 0.01
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def char_timings(text: str, words: list[Word]) -> list[dict]:
|
|
18
|
+
"""Return [{"char", "start", "end"}] for each non-space character in text.
|
|
19
|
+
|
|
20
|
+
Spaces are skipped (no karaoke syllable). If `words` is empty, returns [].
|
|
21
|
+
|
|
22
|
+
Durations are kept strictly positive and non-overlapping: proportional
|
|
23
|
+
interpolation plus rounding to milliseconds can otherwise collapse a
|
|
24
|
+
character to zero length, which downstream means a `\\k0` sweep or a TTML span
|
|
25
|
+
with begin == end.
|
|
26
|
+
"""
|
|
27
|
+
if not words:
|
|
28
|
+
return []
|
|
29
|
+
m = nchars(text)
|
|
30
|
+
if m == 0:
|
|
31
|
+
return []
|
|
32
|
+
bounds = [w.start for w in words] + [words[-1].end]
|
|
33
|
+
n = len(bounds) - 1
|
|
34
|
+
|
|
35
|
+
def t_at(j: float) -> float:
|
|
36
|
+
pos = j * n / m
|
|
37
|
+
i = min(int(pos), n - 1)
|
|
38
|
+
return bounds[i] + (bounds[i + 1] - bounds[i]) * (pos - i)
|
|
39
|
+
|
|
40
|
+
# A very short span cannot give every character MIN_CHAR_DUR; share it out.
|
|
41
|
+
span = max(bounds[-1] - bounds[0], 0.0)
|
|
42
|
+
min_dur = min(MIN_CHAR_DUR, span / m) if m else 0.0
|
|
43
|
+
|
|
44
|
+
out = []
|
|
45
|
+
j = 0
|
|
46
|
+
prev_end = bounds[0]
|
|
47
|
+
for ch in text:
|
|
48
|
+
if ch == ' ':
|
|
49
|
+
continue
|
|
50
|
+
start = max(t_at(j), prev_end)
|
|
51
|
+
end = max(t_at(j + 1), start + min_dur)
|
|
52
|
+
out.append({"char": ch, "start": round(start, 3), "end": round(end, 3)})
|
|
53
|
+
prev_end = end
|
|
54
|
+
j += 1
|
|
55
|
+
|
|
56
|
+
# Rounding can re-introduce a collision at millisecond resolution.
|
|
57
|
+
for a, b in zip(out, out[1:]):
|
|
58
|
+
if b["start"] < a["end"]:
|
|
59
|
+
b["start"] = a["end"]
|
|
60
|
+
if b["end"] <= b["start"]:
|
|
61
|
+
b["end"] = round(b["start"] + MIN_CHAR_DUR, 3)
|
|
62
|
+
if out and out[0]["end"] <= out[0]["start"]:
|
|
63
|
+
out[0]["end"] = round(out[0]["start"] + MIN_CHAR_DUR, 3)
|
|
64
|
+
return out
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def syllable_timings(text: str, chars: list[dict]) -> list[dict]:
|
|
68
|
+
"""Group per-character timings into the units a karaoke format highlights.
|
|
69
|
+
|
|
70
|
+
Returns ``[{"text", "start", "end", "space_after"}]``. The grouping differs
|
|
71
|
+
by script, because what counts as a "syllable" to highlight does:
|
|
72
|
+
|
|
73
|
+
- Alphabetic text is grouped on whitespace, so "Amazing grace" highlights two
|
|
74
|
+
words rather than twelve letters.
|
|
75
|
+
- CJK text keeps one unit per character, which is how per-character karaoke
|
|
76
|
+
formats (QQ/NetEase style) treat Chinese and Japanese. Japanese lyrics often
|
|
77
|
+
contain spaces as phrasing, so splitting on them would give useless chunks.
|
|
78
|
+
|
|
79
|
+
``space_after`` records whether whitespace followed the unit *in the source
|
|
80
|
+
line*, so a formatter can rebuild the line exactly instead of guessing a
|
|
81
|
+
separator. The guess is what breaks CJK: a Japanese line is one unit per
|
|
82
|
+
character yet often carries a phrasing space, so any line-wide "is this
|
|
83
|
+
spaced text?" test either inserts a space between every character or drops
|
|
84
|
+
the phrasing space entirely.
|
|
85
|
+
"""
|
|
86
|
+
if not chars:
|
|
87
|
+
return []
|
|
88
|
+
per_char = cjk_ratio(text) >= 0.2
|
|
89
|
+
|
|
90
|
+
units: list[list] = [] # [source_text, [timings], space_after]
|
|
91
|
+
buf_text, buf_times = "", []
|
|
92
|
+
|
|
93
|
+
def flush() -> None:
|
|
94
|
+
nonlocal buf_text, buf_times
|
|
95
|
+
if buf_times:
|
|
96
|
+
units.append([buf_text, buf_times, False])
|
|
97
|
+
buf_text, buf_times = "", []
|
|
98
|
+
|
|
99
|
+
it = iter(chars)
|
|
100
|
+
for ch in text:
|
|
101
|
+
if ch.isspace():
|
|
102
|
+
flush()
|
|
103
|
+
if units:
|
|
104
|
+
units[-1][2] = True
|
|
105
|
+
continue
|
|
106
|
+
try:
|
|
107
|
+
t = next(it)
|
|
108
|
+
except StopIteration:
|
|
109
|
+
break
|
|
110
|
+
buf_text += ch
|
|
111
|
+
buf_times.append(t)
|
|
112
|
+
if per_char:
|
|
113
|
+
flush()
|
|
114
|
+
flush()
|
|
115
|
+
|
|
116
|
+
return [{"text": t, "start": ts[0]["start"], "end": ts[-1]["end"],
|
|
117
|
+
"space_after": sp} for t, ts, sp in units]
|