echoact 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- echoact/__init__.py +3 -0
- echoact/__main__.py +117 -0
- echoact/app.py +315 -0
- echoact/audio/__init__.py +0 -0
- echoact/audio/devices.py +192 -0
- echoact/audio/player.py +611 -0
- echoact/audio/wav.py +854 -0
- echoact/config/__init__.py +0 -0
- echoact/config/budget.py +370 -0
- echoact/config/settings.py +1244 -0
- echoact/db/__init__.py +0 -0
- echoact/db/backup.py +2429 -0
- echoact/db/migrations.py +434 -0
- echoact/db/schema.sql +214 -0
- echoact/db/store.py +2062 -0
- echoact/diagnostics.py +902 -0
- echoact/domain.py +487 -0
- echoact/engine/__init__.py +0 -0
- echoact/engine/container.py +843 -0
- echoact/engine/protocol.py +241 -0
- echoact/engine/runtime.py +324 -0
- echoact/engine/supervisor.py +961 -0
- echoact/engine/worker.py +659 -0
- echoact/errors.py +281 -0
- echoact/instance.py +172 -0
- echoact/jobs/__init__.py +0 -0
- echoact/jobs/engine.py +776 -0
- echoact/jobs/request.py +300 -0
- echoact/mcp/__init__.py +0 -0
- echoact/mcp/__main__.py +50 -0
- echoact/mcp/client.py +202 -0
- echoact/mcp/config.py +112 -0
- echoact/mcp/server.py +340 -0
- echoact/models/__init__.py +0 -0
- echoact/models/catalog.py +273 -0
- echoact/models/manifest.py +278 -0
- echoact/models/registry.py +1551 -0
- echoact/paths.py +93 -0
- echoact/policy.py +189 -0
- echoact/security/__init__.py +0 -0
- echoact/security/credentials.py +930 -0
- echoact/security/ratelimit.py +534 -0
- echoact/service/__init__.py +20 -0
- echoact/service/app.py +182 -0
- echoact/service/deps.py +563 -0
- echoact/service/errors.py +241 -0
- echoact/service/routes.py +1125 -0
- echoact/service/schemas.py +509 -0
- echoact/service/server.py +270 -0
- echoact/text/__init__.py +0 -0
- echoact/text/language.py +44 -0
- echoact/text/loader.py +577 -0
- echoact/text/normalize.py +924 -0
- echoact/text/segment.py +499 -0
- echoact/text/sniff.py +1202 -0
- echoact/ui/__init__.py +0 -0
- echoact/ui/bridge.py +50 -0
- echoact/ui/controls.py +360 -0
- echoact/ui/credential_dialog.py +131 -0
- echoact/ui/fonts.py +94 -0
- echoact/ui/i18n.py +260 -0
- echoact/ui/icons.py +440 -0
- echoact/ui/library.py +1642 -0
- echoact/ui/licence.py +162 -0
- echoact/ui/main_window.py +1202 -0
- echoact/ui/mcp_setup.py +494 -0
- echoact/ui/models_view.py +1142 -0
- echoact/ui/notifications.py +202 -0
- echoact/ui/reading.py +494 -0
- echoact/ui/settings_view.py +2258 -0
- echoact/ui/status_view.py +1193 -0
- echoact/ui/theme.py +579 -0
- echoact/util/__init__.py +0 -0
- echoact/util/ids.py +62 -0
- echoact/util/logging.py +127 -0
- echoact-0.1.0.dist-info/METADATA +162 -0
- echoact-0.1.0.dist-info/RECORD +80 -0
- echoact-0.1.0.dist-info/WHEEL +4 -0
- echoact-0.1.0.dist-info/entry_points.txt +3 -0
- echoact-0.1.0.dist-info/licenses/LICENSE +21 -0
echoact/text/segment.py
ADDED
|
@@ -0,0 +1,499 @@
|
|
|
1
|
+
"""F-81's segmentation: sentences first, then clauses, deterministically.
|
|
2
|
+
|
|
3
|
+
Every segment carries a source range in Unicode code points into the text the
|
|
4
|
+
user entered, and those ranges tile that text exactly -- no gaps, no overlaps,
|
|
5
|
+
first starting at 0 and last ending at ``len(text)``. F-27 makes this a
|
|
6
|
+
correctness requirement rather than a convenience: the highlight, the segment
|
|
7
|
+
time table, and every external projection of a job are built on it, and a
|
|
8
|
+
range that is off by one is a defect no interface can hide (A.2, step 2).
|
|
9
|
+
|
|
10
|
+
Three properties are worth stating because the rest of the product relies on
|
|
11
|
+
them and none of them is obvious from the code:
|
|
12
|
+
|
|
13
|
+
* The function is pure. Same text and same settings, same segmentation --
|
|
14
|
+
F-81 requires it of every entry path, and it is also what lets a REST
|
|
15
|
+
caller and the GUI agree on segment indices without sharing state.
|
|
16
|
+
* Cuts land only on boundaries of the F-27 alignment, which makes "never
|
|
17
|
+
inside a word, a number, or a grapheme cluster" structural: a number and
|
|
18
|
+
its expansion are one indivisible piece, so there is no offset inside one
|
|
19
|
+
to cut at.
|
|
20
|
+
* Unspoken spans -- emoji, decorative symbols, a run of whitespace -- are
|
|
21
|
+
never segments of their own where a neighbour exists. F-27 attaches them
|
|
22
|
+
to a neighbouring segment so the highlight still travels across them, and
|
|
23
|
+
the preceding segment is that neighbour, matching the way F-82's silence
|
|
24
|
+
is attributed backwards.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import re
|
|
30
|
+
from bisect import bisect_right
|
|
31
|
+
from dataclasses import dataclass
|
|
32
|
+
from typing import Final
|
|
33
|
+
|
|
34
|
+
from ..domain import Segment, SpeakingStyle, TextRange, VoiceSettings, clamp
|
|
35
|
+
from ..policy import (
|
|
36
|
+
ESTIMATE_CODEPOINTS_PER_SECOND_EN,
|
|
37
|
+
ESTIMATE_CODEPOINTS_PER_SECOND_KO,
|
|
38
|
+
FIRST_SEGMENT_MAX_SECONDS,
|
|
39
|
+
PARAGRAPH_EXTRA_GAP_MS,
|
|
40
|
+
SEGMENT_MAX_CODEPOINTS,
|
|
41
|
+
SEGMENT_MAX_SECONDS,
|
|
42
|
+
SEGMENT_MIN_CODEPOINTS,
|
|
43
|
+
STYLE_SEGMENT_GAP_MS,
|
|
44
|
+
STYLE_TEMPO_MULTIPLIER,
|
|
45
|
+
TEMPO_MAX,
|
|
46
|
+
TEMPO_MIN,
|
|
47
|
+
)
|
|
48
|
+
from .language import KO, resolve_language
|
|
49
|
+
from .normalize import (
|
|
50
|
+
AMBIGUOUS_ABBREVIATIONS,
|
|
51
|
+
NON_TERMINAL_ABBREVIATIONS,
|
|
52
|
+
Normalized,
|
|
53
|
+
is_grapheme_boundary,
|
|
54
|
+
normalize,
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
# ======================================================================
|
|
58
|
+
# Sentence boundaries
|
|
59
|
+
# ======================================================================
|
|
60
|
+
|
|
61
|
+
#: Fullwidth forms included: a Korean or Japanese keyboard produces them and
|
|
62
|
+
#: they are unambiguously terminal, unlike the ASCII period.
|
|
63
|
+
_TERMINATORS: Final = frozenset(".!?。!?.…‥⋯؟।")
|
|
64
|
+
_STRONG_TERMINATORS: Final = frozenset("!?。!?…‥⋯؟।")
|
|
65
|
+
#: A quote or bracket that closes after the terminator belongs to the
|
|
66
|
+
#: sentence it closes -- '"Stop!" she said' must not break before `she`.
|
|
67
|
+
_CLOSERS: Final = frozenset("\"'”’))]}»›〉》」』〞")
|
|
68
|
+
_LINE_BREAKS: Final = frozenset("\n\r
\f\v")
|
|
69
|
+
|
|
70
|
+
#: Verb endings that make a Korean sentence's final period unmistakable, so
|
|
71
|
+
#: that a missing space after it ("...합니다.다음") still splits. English
|
|
72
|
+
#: cannot use the same trick: its abbreviations end in letters too.
|
|
73
|
+
_KO_SENTENCE_ENDINGS: Final = frozenset("다요죠까네오음함임슴")
|
|
74
|
+
|
|
75
|
+
_WS_RUN: Final = re.compile(r"\s+")
|
|
76
|
+
_WORD: Final = re.compile(r"\S+")
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _absorb_whitespace(text: str, index: int) -> int:
|
|
80
|
+
match = _WS_RUN.match(text, index)
|
|
81
|
+
return match.end() if match else index
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _abbreviation_before(text: str, dot: int) -> str:
|
|
85
|
+
"""The token ending at ``text[dot]``, e.g. ``Dr`` or ``e.g``."""
|
|
86
|
+
start = dot
|
|
87
|
+
while start > 0:
|
|
88
|
+
previous = text[start - 1]
|
|
89
|
+
if not ((previous.isalpha() and previous.isascii()) or previous == "."):
|
|
90
|
+
break
|
|
91
|
+
start -= 1
|
|
92
|
+
return text[start:dot].rstrip(".")
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _terminates(text: str, first: int, last: int, after_closers: int) -> bool:
|
|
96
|
+
"""Whether the terminator run ``text[first:last]`` ends a sentence."""
|
|
97
|
+
run = text[first:last]
|
|
98
|
+
if any(ch in _STRONG_TERMINATORS for ch in run):
|
|
99
|
+
return True
|
|
100
|
+
if last - first >= 3:
|
|
101
|
+
return True # "..." is an ellipsis, not three abbreviations
|
|
102
|
+
before = text[first - 1] if first else ""
|
|
103
|
+
following = text[after_closers] if after_closers < len(text) else ""
|
|
104
|
+
if before.isdigit() and following.isdigit():
|
|
105
|
+
return False
|
|
106
|
+
if before in _KO_SENTENCE_ENDINGS:
|
|
107
|
+
return True
|
|
108
|
+
token = _abbreviation_before(text, first)
|
|
109
|
+
if token in NON_TERMINAL_ABBREVIATIONS:
|
|
110
|
+
return False
|
|
111
|
+
if len(token) == 1 and token.isupper():
|
|
112
|
+
return False # an initial: "J. R. R. Tolkien"
|
|
113
|
+
if token in AMBIGUOUS_ABBREVIATIONS:
|
|
114
|
+
# "etc." ends a sentence only when what follows starts one.
|
|
115
|
+
nxt = _absorb_whitespace(text, after_closers)
|
|
116
|
+
return nxt >= len(text) or text[nxt].isupper() or not text[nxt].isascii()
|
|
117
|
+
# A period glued to more text is inside something -- a host name, a
|
|
118
|
+
# version, a file name -- not at the end of a sentence.
|
|
119
|
+
return following == "" or following.isspace()
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def split_sentences(text: str) -> list[TextRange]:
|
|
123
|
+
"""Sentence ranges that tile ``text``.
|
|
124
|
+
|
|
125
|
+
Trailing whitespace, including the line break, belongs to the sentence it
|
|
126
|
+
follows: F-27 attributes the pause after a sentence to that sentence, and
|
|
127
|
+
keeping the two together is what makes a paragraph break detectable later
|
|
128
|
+
without a second pass over the source.
|
|
129
|
+
"""
|
|
130
|
+
spans: list[TextRange] = []
|
|
131
|
+
start = 0
|
|
132
|
+
index = 0
|
|
133
|
+
length = len(text)
|
|
134
|
+
while index < length:
|
|
135
|
+
char = text[index]
|
|
136
|
+
if char in _TERMINATORS:
|
|
137
|
+
run_end = index
|
|
138
|
+
while run_end < length and text[run_end] in _TERMINATORS:
|
|
139
|
+
run_end += 1
|
|
140
|
+
closed = run_end
|
|
141
|
+
while closed < length and text[closed] in _CLOSERS:
|
|
142
|
+
closed += 1
|
|
143
|
+
if _terminates(text, index, run_end, closed):
|
|
144
|
+
end = _absorb_whitespace(text, closed)
|
|
145
|
+
spans.append(TextRange(start, end))
|
|
146
|
+
start = index = end
|
|
147
|
+
else:
|
|
148
|
+
index = run_end
|
|
149
|
+
continue
|
|
150
|
+
if char in _LINE_BREAKS:
|
|
151
|
+
# A line with no terminator is still a unit: a heading, a list
|
|
152
|
+
# item, a line of dialogue. Reading it into the next line would
|
|
153
|
+
# be wrong however the punctuation falls.
|
|
154
|
+
end = _absorb_whitespace(text, index)
|
|
155
|
+
spans.append(TextRange(start, end))
|
|
156
|
+
start = index = end
|
|
157
|
+
continue
|
|
158
|
+
index += 1
|
|
159
|
+
if start < length:
|
|
160
|
+
spans.append(TextRange(start, length))
|
|
161
|
+
return spans
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
# ======================================================================
|
|
165
|
+
# Clause boundaries
|
|
166
|
+
# ======================================================================
|
|
167
|
+
|
|
168
|
+
_CLAUSE_PUNCTUATION: Final = frozenset(",;:,;:、")
|
|
169
|
+
_EN_CONJUNCTIONS: Final = frozenset({"and", "but", "or", "so"})
|
|
170
|
+
_KO_CONJUNCTIONS: Final = ("그리고", "하지만", "그래서", "또는", "그러나", "그런데")
|
|
171
|
+
#: Korean connective endings. Two-syllable forms are unambiguous; the
|
|
172
|
+
#: single-syllable ones can also end a noun (사고, 참고), which costs a pause
|
|
173
|
+
#: in the wrong place at worst -- never a cut inside a word.
|
|
174
|
+
_KO_CONNECTIVE_LONG: Final = ("지만", "는데", "거나", "면서", "아서", "어서", "해서")
|
|
175
|
+
_KO_CONNECTIVE_SHORT: Final = frozenset("고며면서")
|
|
176
|
+
|
|
177
|
+
_RANK_PUNCTUATION: Final = 0
|
|
178
|
+
_RANK_CONJUNCTION: Final = 1
|
|
179
|
+
_RANK_CONNECTIVE: Final = 2
|
|
180
|
+
_RANK_WORD: Final = 3
|
|
181
|
+
#: Ranks F-81 calls clause boundaries. A bare word boundary is not one, and
|
|
182
|
+
#: is only used where a sentence offers nothing better (see ``_cut``).
|
|
183
|
+
_CLAUSE_RANKS: Final = (_RANK_PUNCTUATION, _RANK_CONJUNCTION, _RANK_CONNECTIVE)
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def _clause_candidates(text: str) -> dict[int, int]:
|
|
187
|
+
"""Offsets where ``text`` may be cut, each with its rank (lower better)."""
|
|
188
|
+
found: dict[int, int] = {}
|
|
189
|
+
|
|
190
|
+
def offer(position: int, rank: int) -> None:
|
|
191
|
+
if 0 < position < len(text) and rank < found.get(position, _RANK_WORD + 1):
|
|
192
|
+
found[position] = rank
|
|
193
|
+
|
|
194
|
+
for index, char in enumerate(text):
|
|
195
|
+
if char not in _CLAUSE_PUNCTUATION:
|
|
196
|
+
continue
|
|
197
|
+
before = text[index - 1] if index else ""
|
|
198
|
+
after = text[index + 1] if index + 1 < len(text) else ""
|
|
199
|
+
if before.isdigit() and after.isdigit():
|
|
200
|
+
continue # a thousands separator or a decimal comma
|
|
201
|
+
offer(_absorb_whitespace(text, index + 1), _RANK_PUNCTUATION)
|
|
202
|
+
|
|
203
|
+
for match in _WS_RUN.finditer(text):
|
|
204
|
+
word_start = match.end()
|
|
205
|
+
offer(word_start, _RANK_WORD)
|
|
206
|
+
word_end = match.start()
|
|
207
|
+
if word_end >= 2 and (
|
|
208
|
+
text[word_end - 2 : word_end] in _KO_CONNECTIVE_LONG
|
|
209
|
+
or (text[word_end - 1] in _KO_CONNECTIVE_SHORT and _is_hangul(text[word_end - 2]))
|
|
210
|
+
):
|
|
211
|
+
offer(word_start, _RANK_CONNECTIVE)
|
|
212
|
+
following = _WORD.match(text, word_start)
|
|
213
|
+
if following is not None:
|
|
214
|
+
word = following.group()
|
|
215
|
+
if word.strip(",.!?").lower() in _EN_CONJUNCTIONS or word.startswith(_KO_CONJUNCTIONS):
|
|
216
|
+
offer(word_start, _RANK_CONJUNCTION)
|
|
217
|
+
return found
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def _is_hangul(char: str) -> bool:
|
|
221
|
+
return "가" <= char <= "힣" or "ᄀ" <= char <= "ᇿ"
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
# ======================================================================
|
|
225
|
+
# Duration estimate
|
|
226
|
+
# ======================================================================
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def codepoints_per_second(lang: str) -> float:
|
|
230
|
+
return ESTIMATE_CODEPOINTS_PER_SECOND_KO if lang == KO else ESTIMATE_CODEPOINTS_PER_SECOND_EN
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def estimate_seconds(spoken: str, lang: str, tempo: float) -> float:
|
|
234
|
+
"""Audio seconds for text already normalised, per policy's measurements.
|
|
235
|
+
|
|
236
|
+
The estimate runs on the *spoken* text, not the source: "1,234" is five
|
|
237
|
+
code points and "one thousand two hundred thirty-four" is thirty-six, and
|
|
238
|
+
it is the second that the engine has to say.
|
|
239
|
+
"""
|
|
240
|
+
return _seconds(len(spoken), lang, tempo)
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def _seconds(codepoints: int, lang: str, tempo: float) -> float:
|
|
244
|
+
return codepoints / codepoints_per_second(lang) / tempo
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def effective_tempo(settings: VoiceSettings) -> float:
|
|
248
|
+
"""F-08's style multiplier over F-07's tempo, clamped as policy requires."""
|
|
249
|
+
multiplier = STYLE_TEMPO_MULTIPLIER[settings.style.value]
|
|
250
|
+
return clamp(settings.tempo * multiplier, TEMPO_MIN, TEMPO_MAX)
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def trailing_silence_ms(style: SpeakingStyle, *, paragraph_break: bool) -> int:
|
|
254
|
+
"""The pause after a segment: F-08's style preset, plus F-27's paragraph.
|
|
255
|
+
|
|
256
|
+
The app inserts this itself. A.5 measured the engine's own silence
|
|
257
|
+
parameter doing nothing once the text is chunked first, so F-82 makes the
|
|
258
|
+
gap the application's and this is where its length is decided.
|
|
259
|
+
"""
|
|
260
|
+
gap = STYLE_SEGMENT_GAP_MS[style.value]
|
|
261
|
+
return gap + PARAGRAPH_EXTRA_GAP_MS if paragraph_break else gap
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
_PARAGRAPH_TAIL: Final = re.compile(r"\s*$")
|
|
265
|
+
_LINE_BREAK_RUN: Final = re.compile(r"\r\n|[\n\r
\f\v]")
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
def ends_paragraph(source_slice: str) -> bool:
|
|
269
|
+
"""Whether this span's own trailing whitespace contains a blank line."""
|
|
270
|
+
tail = _PARAGRAPH_TAIL.search(source_slice)
|
|
271
|
+
return tail is not None and len(_LINE_BREAK_RUN.findall(tail.group())) >= 2
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
# ======================================================================
|
|
275
|
+
# Splitting one sentence
|
|
276
|
+
# ======================================================================
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
@dataclass(frozen=True, slots=True)
|
|
280
|
+
class _Chunk:
|
|
281
|
+
start: int
|
|
282
|
+
end: int
|
|
283
|
+
spoken: str
|
|
284
|
+
lang: str
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
def _cut(
|
|
288
|
+
text: str,
|
|
289
|
+
norm: Normalized,
|
|
290
|
+
candidates: dict[int, int],
|
|
291
|
+
sorted_offsets: list[int],
|
|
292
|
+
start: int,
|
|
293
|
+
lang: str,
|
|
294
|
+
tempo: float,
|
|
295
|
+
max_seconds: float,
|
|
296
|
+
) -> int:
|
|
297
|
+
"""Where the segment starting at ``start`` should end.
|
|
298
|
+
|
|
299
|
+
Returns ``len(text)`` when the rest fits, or when the sentence offers no
|
|
300
|
+
boundary that may legally be cut -- F-81 forbids cutting inside a word,
|
|
301
|
+
so an unbreakable run stays whole and over-long rather than being
|
|
302
|
+
chopped. Latency is a quality of service; a word cut in half is not.
|
|
303
|
+
"""
|
|
304
|
+
length = len(text)
|
|
305
|
+
produced_start = norm.produced_offset(start)
|
|
306
|
+
remaining = len(norm.text) - produced_start
|
|
307
|
+
if (
|
|
308
|
+
length - start <= SEGMENT_MAX_CODEPOINTS
|
|
309
|
+
and _seconds(remaining, lang, tempo) <= max_seconds
|
|
310
|
+
):
|
|
311
|
+
return length
|
|
312
|
+
|
|
313
|
+
budget = int(max_seconds * codepoints_per_second(lang) * tempo)
|
|
314
|
+
seconds_end = norm.source_offset(produced_start + budget)
|
|
315
|
+
hard_end = min(length, start + SEGMENT_MAX_CODEPOINTS, max(seconds_end, start + 1))
|
|
316
|
+
hard_end = max(hard_end, min(length, start + SEGMENT_MIN_CODEPOINTS))
|
|
317
|
+
|
|
318
|
+
# Order of preference: a clause boundary that also leaves a long enough
|
|
319
|
+
# tail, then a clause boundary that does not, then -- only if the
|
|
320
|
+
# sentence offers no clause boundary at all -- the gap between two words.
|
|
321
|
+
# F-81 asks for both the clause rule and the 20-code-point minimum; where
|
|
322
|
+
# they conflict, a short tail is a worse pause and a mid-clause cut is a
|
|
323
|
+
# worse sentence.
|
|
324
|
+
for ranks in (_CLAUSE_RANKS, (_RANK_WORD,)):
|
|
325
|
+
for require_tail in (True, False):
|
|
326
|
+
chosen = _best_offset(
|
|
327
|
+
candidates, sorted_offsets, start, hard_end, length, ranks, require_tail
|
|
328
|
+
)
|
|
329
|
+
if chosen is not None:
|
|
330
|
+
return chosen
|
|
331
|
+
return length
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
def _best_offset(
|
|
335
|
+
candidates: dict[int, int],
|
|
336
|
+
sorted_offsets: list[int],
|
|
337
|
+
start: int,
|
|
338
|
+
hard_end: int,
|
|
339
|
+
length: int,
|
|
340
|
+
ranks: tuple[int, ...],
|
|
341
|
+
require_tail: bool,
|
|
342
|
+
) -> int | None:
|
|
343
|
+
"""The latest acceptable cut at or before ``hard_end``.
|
|
344
|
+
|
|
345
|
+
Latest, not best-ranked: a comma five characters in is a clause boundary
|
|
346
|
+
too, and preferring it would shred the sentence into segments far shorter
|
|
347
|
+
than F-81 allows.
|
|
348
|
+
"""
|
|
349
|
+
upper = bisect_right(sorted_offsets, hard_end)
|
|
350
|
+
for index in range(upper - 1, -1, -1):
|
|
351
|
+
offset = sorted_offsets[index]
|
|
352
|
+
if offset - start < SEGMENT_MIN_CODEPOINTS:
|
|
353
|
+
break
|
|
354
|
+
if candidates[offset] not in ranks:
|
|
355
|
+
continue
|
|
356
|
+
if require_tail and 0 < length - offset < SEGMENT_MIN_CODEPOINTS:
|
|
357
|
+
continue
|
|
358
|
+
return offset
|
|
359
|
+
return None
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
def _split_sentence(
|
|
363
|
+
text: str, norm: Normalized, lang: str, tempo: float, first_max_seconds: float
|
|
364
|
+
) -> list[_Chunk]:
|
|
365
|
+
cuttable = set(norm.boundaries)
|
|
366
|
+
candidates = {
|
|
367
|
+
offset: rank
|
|
368
|
+
for offset, rank in _clause_candidates(text).items()
|
|
369
|
+
if offset in cuttable and is_grapheme_boundary(text, offset)
|
|
370
|
+
}
|
|
371
|
+
sorted_offsets = sorted(candidates)
|
|
372
|
+
chunks: list[_Chunk] = []
|
|
373
|
+
start = 0
|
|
374
|
+
max_seconds = first_max_seconds
|
|
375
|
+
while start < len(text):
|
|
376
|
+
end = _cut(text, norm, candidates, sorted_offsets, start, lang, tempo, max_seconds)
|
|
377
|
+
chunks.append(_Chunk(start, end, norm.spoken_between(start, end), lang))
|
|
378
|
+
start = end
|
|
379
|
+
max_seconds = SEGMENT_MAX_SECONDS
|
|
380
|
+
return chunks
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
# ======================================================================
|
|
384
|
+
# The public entry point
|
|
385
|
+
# ======================================================================
|
|
386
|
+
|
|
387
|
+
|
|
388
|
+
def segment_text(source: str, settings: VoiceSettings) -> list[Segment]:
|
|
389
|
+
"""Split ``source`` into the segments one job will generate (F-81, F-27).
|
|
390
|
+
|
|
391
|
+
``settings`` supplies F-05's language selection, F-07's tempo, and F-08's
|
|
392
|
+
style, all three of which change where the cuts fall: the language picks
|
|
393
|
+
the reading and the estimate, and tempo and style set how much audio a
|
|
394
|
+
given number of code points becomes.
|
|
395
|
+
"""
|
|
396
|
+
if not source:
|
|
397
|
+
return []
|
|
398
|
+
tempo = effective_tempo(settings)
|
|
399
|
+
chunks: list[_Chunk] = []
|
|
400
|
+
# The 8-second cap belongs to the first segment that actually produces
|
|
401
|
+
# audio. Leading blank lines or an emoji-only first line are attached to
|
|
402
|
+
# a neighbour later, so counting them would spend the cap on silence.
|
|
403
|
+
spoken_seen = False
|
|
404
|
+
for span in split_sentences(source):
|
|
405
|
+
sentence = span.slice(source)
|
|
406
|
+
lang = resolve_language(sentence, settings.language)
|
|
407
|
+
norm = normalize(sentence, lang)
|
|
408
|
+
first_cap = SEGMENT_MAX_SECONDS if spoken_seen else FIRST_SEGMENT_MAX_SECONDS
|
|
409
|
+
for chunk in _split_sentence(sentence, norm, lang, tempo, first_cap):
|
|
410
|
+
spoken_seen = spoken_seen or bool(chunk.spoken.strip())
|
|
411
|
+
chunks.append(
|
|
412
|
+
_Chunk(span.start + chunk.start, span.start + chunk.end, chunk.spoken, chunk.lang)
|
|
413
|
+
)
|
|
414
|
+
chunks = _attach_unspoken(chunks)
|
|
415
|
+
segments = _build(chunks, source, settings.style)
|
|
416
|
+
_assert_tiles(segments, len(source))
|
|
417
|
+
return segments
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
def _attach_unspoken(chunks: list[_Chunk]) -> list[_Chunk]:
|
|
421
|
+
"""F-27: a span that yields no audio joins a neighbour, keeping its range.
|
|
422
|
+
|
|
423
|
+
Backwards by preference, so that the emoji or the blank line after a
|
|
424
|
+
sentence is highlighted while that sentence's own trailing silence plays.
|
|
425
|
+
A document with nothing to say keeps one segment covering all of it --
|
|
426
|
+
the range still has to exist, or the source text stops being complete.
|
|
427
|
+
"""
|
|
428
|
+
merged: list[_Chunk] = []
|
|
429
|
+
for chunk in chunks:
|
|
430
|
+
if chunk.spoken.strip() or not merged:
|
|
431
|
+
merged.append(chunk)
|
|
432
|
+
continue
|
|
433
|
+
previous = merged[-1]
|
|
434
|
+
merged[-1] = _Chunk(
|
|
435
|
+
previous.start, chunk.end, previous.spoken + chunk.spoken, previous.lang
|
|
436
|
+
)
|
|
437
|
+
if len(merged) > 1 and not merged[0].spoken.strip():
|
|
438
|
+
head, second = merged[0], merged[1]
|
|
439
|
+
merged[1] = _Chunk(head.start, second.end, head.spoken + second.spoken, second.lang)
|
|
440
|
+
del merged[0]
|
|
441
|
+
return merged
|
|
442
|
+
|
|
443
|
+
|
|
444
|
+
def _build(chunks: list[_Chunk], source: str, style: SpeakingStyle) -> list[Segment]:
|
|
445
|
+
segments: list[Segment] = []
|
|
446
|
+
for index, chunk in enumerate(chunks):
|
|
447
|
+
spoken = chunk.spoken.strip()
|
|
448
|
+
last = index == len(chunks) - 1
|
|
449
|
+
silence = (
|
|
450
|
+
0
|
|
451
|
+
if last
|
|
452
|
+
else trailing_silence_ms(
|
|
453
|
+
style, paragraph_break=ends_paragraph(source[chunk.start : chunk.end])
|
|
454
|
+
)
|
|
455
|
+
)
|
|
456
|
+
segments.append(
|
|
457
|
+
Segment(
|
|
458
|
+
index=index,
|
|
459
|
+
source=TextRange(chunk.start, chunk.end),
|
|
460
|
+
spoken_text=spoken,
|
|
461
|
+
language=chunk.lang,
|
|
462
|
+
trailing_silence_ms=silence,
|
|
463
|
+
)
|
|
464
|
+
)
|
|
465
|
+
return segments
|
|
466
|
+
|
|
467
|
+
|
|
468
|
+
def _assert_tiles(segments: list[Segment], length: int) -> None:
|
|
469
|
+
"""F-27's invariant, checked rather than trusted."""
|
|
470
|
+
cursor = 0
|
|
471
|
+
for segment in segments:
|
|
472
|
+
if segment.source.start != cursor:
|
|
473
|
+
raise AssertionError(
|
|
474
|
+
f"segment {segment.index} starts at {segment.source.start}, expected {cursor}"
|
|
475
|
+
)
|
|
476
|
+
cursor = segment.source.end
|
|
477
|
+
if cursor != length:
|
|
478
|
+
raise AssertionError(f"segments cover {cursor} of {length} code points")
|
|
479
|
+
|
|
480
|
+
|
|
481
|
+
def estimate_job_seconds(segments: list[Segment], tempo: float) -> float:
|
|
482
|
+
"""Total audio a job will produce, silence included (F-88's input)."""
|
|
483
|
+
total = 0.0
|
|
484
|
+
for segment in segments:
|
|
485
|
+
total += estimate_seconds(segment.spoken_text, segment.language, tempo)
|
|
486
|
+
total += segment.trailing_silence_ms / 1000
|
|
487
|
+
return total
|
|
488
|
+
|
|
489
|
+
|
|
490
|
+
__all__ = [
|
|
491
|
+
"codepoints_per_second",
|
|
492
|
+
"effective_tempo",
|
|
493
|
+
"ends_paragraph",
|
|
494
|
+
"estimate_job_seconds",
|
|
495
|
+
"estimate_seconds",
|
|
496
|
+
"segment_text",
|
|
497
|
+
"split_sentences",
|
|
498
|
+
"trailing_silence_ms",
|
|
499
|
+
]
|