ssmd 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ssmd/__init__.py +143 -0
- ssmd/_version.py +34 -0
- ssmd/annotations/__init__.py +44 -0
- ssmd/annotations/audio.py +52 -0
- ssmd/annotations/base.py +61 -0
- ssmd/annotations/extension.py +87 -0
- ssmd/annotations/language.py +67 -0
- ssmd/annotations/phoneme.py +100 -0
- ssmd/annotations/prosody.py +131 -0
- ssmd/annotations/say_as.py +58 -0
- ssmd/annotations/substitution.py +41 -0
- ssmd/annotations/xsampa_to_ipa.txt +174 -0
- ssmd/capabilities.py +277 -0
- ssmd/converter.py +117 -0
- ssmd/document.py +845 -0
- ssmd/processors/__init__.py +37 -0
- ssmd/processors/annotation.py +130 -0
- ssmd/processors/base.py +125 -0
- ssmd/processors/break_processor.py +82 -0
- ssmd/processors/emphasis.py +54 -0
- ssmd/processors/heading.py +85 -0
- ssmd/processors/mark.py +50 -0
- ssmd/processors/paragraph.py +80 -0
- ssmd/processors/prosody.py +104 -0
- ssmd/processors/sentence.py +79 -0
- ssmd/py.typed +0 -0
- ssmd/ssml_parser.py +485 -0
- ssmd/utils.py +81 -0
- ssmd-0.1.0.dist-info/METADATA +691 -0
- ssmd-0.1.0.dist-info/RECORD +33 -0
- ssmd-0.1.0.dist-info/WHEEL +5 -0
- ssmd-0.1.0.dist-info/licenses/LICENSE +21 -0
- ssmd-0.1.0.dist-info/top_level.txt +1 -0
ssmd/__init__.py
ADDED
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
"""SSMD - Speech Synthesis Markdown to SSML converter.
|
|
2
|
+
|
|
3
|
+
SSMD provides a lightweight markdown-like syntax for creating SSML
|
|
4
|
+
(Speech Synthesis Markup Language) documents. It's designed to be
|
|
5
|
+
more human-friendly than raw SSML while maintaining full compatibility.
|
|
6
|
+
|
|
7
|
+
Example:
|
|
8
|
+
Basic usage::
|
|
9
|
+
|
|
10
|
+
import ssmd
|
|
11
|
+
|
|
12
|
+
# Create and build a document
|
|
13
|
+
doc = ssmd.Document()
|
|
14
|
+
doc.add_sentence("Hello *world*!")
|
|
15
|
+
doc.add_sentence("This is SSMD.")
|
|
16
|
+
|
|
17
|
+
# Export to different formats
|
|
18
|
+
ssml = doc.to_ssml()
|
|
19
|
+
text = doc.to_text()
|
|
20
|
+
|
|
21
|
+
# Or use convenience functions for one-off conversions
|
|
22
|
+
ssml = ssmd.to_ssml("Hello *world*!")
|
|
23
|
+
|
|
24
|
+
Advanced usage with streaming::
|
|
25
|
+
|
|
26
|
+
# Create parser with custom config
|
|
27
|
+
doc = ssmd.Document(
|
|
28
|
+
capabilities='pyttsx3',
|
|
29
|
+
config={'auto_sentence_tags': True}
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
# Build document incrementally
|
|
33
|
+
doc.add_paragraph("# Welcome")
|
|
34
|
+
doc.add_sentence("Hello and *welcome* to SSMD!")
|
|
35
|
+
|
|
36
|
+
# Stream to TTS
|
|
37
|
+
for sentence in doc.sentences():
|
|
38
|
+
tts_engine.speak(sentence)
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
from typing import Any
|
|
42
|
+
|
|
43
|
+
from ssmd.document import Document
|
|
44
|
+
from ssmd.ssml_parser import SSMLParser
|
|
45
|
+
from ssmd.capabilities import (
|
|
46
|
+
TTSCapabilities,
|
|
47
|
+
get_preset,
|
|
48
|
+
ESPEAK_CAPABILITIES,
|
|
49
|
+
PYTTSX3_CAPABILITIES,
|
|
50
|
+
GOOGLE_TTS_CAPABILITIES,
|
|
51
|
+
AMAZON_POLLY_CAPABILITIES,
|
|
52
|
+
AZURE_TTS_CAPABILITIES,
|
|
53
|
+
MINIMAL_CAPABILITIES,
|
|
54
|
+
FULL_CAPABILITIES,
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
try:
|
|
58
|
+
from ssmd._version import version as __version__
|
|
59
|
+
except ImportError:
|
|
60
|
+
__version__ = "unknown"
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
# ═══════════════════════════════════════════════════════════
|
|
64
|
+
# CONVENIENCE FUNCTIONS
|
|
65
|
+
# ═══════════════════════════════════════════════════════════
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def to_ssml(ssmd_text: str, **config: Any) -> str:
|
|
69
|
+
"""Convert SSMD to SSML (convenience function).
|
|
70
|
+
|
|
71
|
+
Creates a temporary Document and converts to SSML.
|
|
72
|
+
For repeated conversions with the same config, create a Document instance.
|
|
73
|
+
|
|
74
|
+
Args:
|
|
75
|
+
ssmd_text: SSMD markdown text
|
|
76
|
+
**config: Optional configuration parameters
|
|
77
|
+
|
|
78
|
+
Returns:
|
|
79
|
+
SSML string
|
|
80
|
+
|
|
81
|
+
Example:
|
|
82
|
+
>>> ssmd.to_ssml("Hello *world*!")
|
|
83
|
+
'<speak>Hello <emphasis>world</emphasis>!</speak>'
|
|
84
|
+
"""
|
|
85
|
+
return Document(ssmd_text, config).to_ssml()
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def to_text(ssmd_text: str, **config: Any) -> str:
|
|
89
|
+
"""Convert SSMD to plain text (convenience function).
|
|
90
|
+
|
|
91
|
+
Strips all SSMD markup, returning plain text.
|
|
92
|
+
|
|
93
|
+
Args:
|
|
94
|
+
ssmd_text: SSMD markdown text
|
|
95
|
+
**config: Optional configuration parameters
|
|
96
|
+
|
|
97
|
+
Returns:
|
|
98
|
+
Plain text with markup removed
|
|
99
|
+
|
|
100
|
+
Example:
|
|
101
|
+
>>> ssmd.to_text("Hello *world* @marker!")
|
|
102
|
+
'Hello world!'
|
|
103
|
+
"""
|
|
104
|
+
return Document(ssmd_text, config).to_text()
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def from_ssml(ssml_text: str, **config: Any) -> str:
|
|
108
|
+
"""Convert SSML to SSMD format (convenience function).
|
|
109
|
+
|
|
110
|
+
Args:
|
|
111
|
+
ssml_text: SSML XML string
|
|
112
|
+
**config: Optional configuration parameters
|
|
113
|
+
|
|
114
|
+
Returns:
|
|
115
|
+
SSMD markdown string
|
|
116
|
+
|
|
117
|
+
Example:
|
|
118
|
+
>>> ssml = '<speak><emphasis>Hello</emphasis> world</speak>'
|
|
119
|
+
>>> ssmd.from_ssml(ssml)
|
|
120
|
+
'*Hello* world'
|
|
121
|
+
"""
|
|
122
|
+
parser = SSMLParser(config)
|
|
123
|
+
return parser.to_ssmd(ssml_text)
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
__all__ = [
|
|
127
|
+
"Document",
|
|
128
|
+
"to_ssml",
|
|
129
|
+
"to_text",
|
|
130
|
+
"from_ssml",
|
|
131
|
+
"SSMLParser",
|
|
132
|
+
"TTSCapabilities",
|
|
133
|
+
"get_preset",
|
|
134
|
+
# Capability presets
|
|
135
|
+
"ESPEAK_CAPABILITIES",
|
|
136
|
+
"PYTTSX3_CAPABILITIES",
|
|
137
|
+
"GOOGLE_TTS_CAPABILITIES",
|
|
138
|
+
"AMAZON_POLLY_CAPABILITIES",
|
|
139
|
+
"AZURE_TTS_CAPABILITIES",
|
|
140
|
+
"MINIMAL_CAPABILITIES",
|
|
141
|
+
"FULL_CAPABILITIES",
|
|
142
|
+
"__version__",
|
|
143
|
+
]
|
ssmd/_version.py
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
# file generated by setuptools-scm
|
|
2
|
+
# don't change, don't track in version control
|
|
3
|
+
|
|
4
|
+
__all__ = [
|
|
5
|
+
"__version__",
|
|
6
|
+
"__version_tuple__",
|
|
7
|
+
"version",
|
|
8
|
+
"version_tuple",
|
|
9
|
+
"__commit_id__",
|
|
10
|
+
"commit_id",
|
|
11
|
+
]
|
|
12
|
+
|
|
13
|
+
TYPE_CHECKING = False
|
|
14
|
+
if TYPE_CHECKING:
|
|
15
|
+
from typing import Tuple
|
|
16
|
+
from typing import Union
|
|
17
|
+
|
|
18
|
+
VERSION_TUPLE = Tuple[Union[int, str], ...]
|
|
19
|
+
COMMIT_ID = Union[str, None]
|
|
20
|
+
else:
|
|
21
|
+
VERSION_TUPLE = object
|
|
22
|
+
COMMIT_ID = object
|
|
23
|
+
|
|
24
|
+
version: str
|
|
25
|
+
__version__: str
|
|
26
|
+
__version_tuple__: VERSION_TUPLE
|
|
27
|
+
version_tuple: VERSION_TUPLE
|
|
28
|
+
commit_id: COMMIT_ID
|
|
29
|
+
__commit_id__: COMMIT_ID
|
|
30
|
+
|
|
31
|
+
__version__ = version = '0.1.0'
|
|
32
|
+
__version_tuple__ = version_tuple = (0, 1, 0)
|
|
33
|
+
|
|
34
|
+
__commit_id__ = commit_id = None
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
"""SSMD annotations for extended syntax in [text](annotations) format."""
|
|
2
|
+
|
|
3
|
+
from typing import Optional
|
|
4
|
+
|
|
5
|
+
from ssmd.annotations.base import BaseAnnotation
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def get_annotation(annotation_str: str) -> BaseAnnotation | None:
|
|
9
|
+
"""Try to create an annotation from a string.
|
|
10
|
+
|
|
11
|
+
Args:
|
|
12
|
+
annotation_str: Annotation string (e.g., "en", "ph: dIC", "vrp: 555")
|
|
13
|
+
|
|
14
|
+
Returns:
|
|
15
|
+
Annotation instance or None if no match
|
|
16
|
+
"""
|
|
17
|
+
from ssmd.annotations.audio import AudioAnnotation
|
|
18
|
+
from ssmd.annotations.extension import ExtensionAnnotation
|
|
19
|
+
from ssmd.annotations.language import LanguageAnnotation
|
|
20
|
+
from ssmd.annotations.phoneme import PhonemeAnnotation
|
|
21
|
+
from ssmd.annotations.prosody import ProsodyAnnotation
|
|
22
|
+
from ssmd.annotations.say_as import SayAsAnnotation
|
|
23
|
+
from ssmd.annotations.substitution import SubstitutionAnnotation
|
|
24
|
+
|
|
25
|
+
# Try each annotation type in order
|
|
26
|
+
annotation_types: list[type[BaseAnnotation]] = [
|
|
27
|
+
AudioAnnotation, # Try audio first (has URL pattern)
|
|
28
|
+
ExtensionAnnotation, # Extensions (ext: name)
|
|
29
|
+
SayAsAnnotation, # Say-as (as: type)
|
|
30
|
+
PhonemeAnnotation, # Phonemes (ph: ..., ipa: ...)
|
|
31
|
+
ProsodyAnnotation, # Prosody (vrp: 555, v: 5, etc.)
|
|
32
|
+
SubstitutionAnnotation, # Substitution (sub: alias)
|
|
33
|
+
LanguageAnnotation, # Language (en, en-US, etc.)
|
|
34
|
+
]
|
|
35
|
+
|
|
36
|
+
for annotation_type in annotation_types:
|
|
37
|
+
annotation = annotation_type.try_create(annotation_str)
|
|
38
|
+
if annotation:
|
|
39
|
+
return annotation
|
|
40
|
+
|
|
41
|
+
return None
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
__all__ = ["get_annotation"]
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
"""Audio annotation: [desc](url.mp3 alt) → <audio>desc</audio>"""
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
|
|
5
|
+
from ssmd.annotations.base import BaseAnnotation
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class AudioAnnotation(BaseAnnotation):
|
|
9
|
+
"""Process audio file annotations.
|
|
10
|
+
|
|
11
|
+
Examples:
|
|
12
|
+
[boing](https://example.com/sounds/boing.mp3) →
|
|
13
|
+
<audio src="..."><desc>boing</desc></audio>
|
|
14
|
+
[purr](cat.ogg Sound didn't load) →
|
|
15
|
+
<audio src="cat.ogg"><desc>purr</desc>Sound didn't load</audio>
|
|
16
|
+
[](miaou.mp3) → <audio src="miaou.mp3"></audio>
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
def __init__(self, match: re.Match):
|
|
20
|
+
"""Initialize with URL and optional alt text.
|
|
21
|
+
|
|
22
|
+
Args:
|
|
23
|
+
match: Regex match containing URL and alt text
|
|
24
|
+
"""
|
|
25
|
+
self.url = match.group(1).strip()
|
|
26
|
+
self.alt_text = match.group(2).strip() if match.group(2) else ""
|
|
27
|
+
|
|
28
|
+
@classmethod
|
|
29
|
+
def regex(cls) -> re.Pattern:
|
|
30
|
+
"""Match audio file URLs.
|
|
31
|
+
|
|
32
|
+
Returns:
|
|
33
|
+
Pattern matching audio file extensions with optional alt text
|
|
34
|
+
"""
|
|
35
|
+
# Match URL (anything starting with http/https or ending in audio extension)
|
|
36
|
+
# Followed by optional alt text
|
|
37
|
+
return re.compile(
|
|
38
|
+
r"^((?:https?://)?[^\s]+\.(?:mp3|ogg|wav|m4a|aac|flac))(?:\s+(.+))?$",
|
|
39
|
+
re.IGNORECASE,
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
def wrap(self, text: str) -> str:
|
|
43
|
+
"""Wrap in audio tag with description and alt text.
|
|
44
|
+
|
|
45
|
+
Args:
|
|
46
|
+
text: Description text (used in <desc>)
|
|
47
|
+
|
|
48
|
+
Returns:
|
|
49
|
+
SSML <audio> element
|
|
50
|
+
"""
|
|
51
|
+
desc = f"<desc>{text}</desc>" if text else ""
|
|
52
|
+
return f'<audio src="{self.url}">{desc}{self.alt_text}</audio>'
|
ssmd/annotations/base.py
ADDED
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""Base annotation class for SSMD conversion."""
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from abc import ABC, abstractmethod
|
|
5
|
+
from typing import Optional
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class BaseAnnotation(ABC):
|
|
9
|
+
"""Abstract base class for annotations.
|
|
10
|
+
|
|
11
|
+
Annotations are specified in the format [text](annotation)
|
|
12
|
+
where annotation can be language codes, phonemes, prosody, etc.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
@classmethod
|
|
16
|
+
@abstractmethod
|
|
17
|
+
def regex(cls) -> re.Pattern:
|
|
18
|
+
"""Pattern to match this annotation type.
|
|
19
|
+
|
|
20
|
+
Returns:
|
|
21
|
+
Compiled regex pattern
|
|
22
|
+
"""
|
|
23
|
+
pass
|
|
24
|
+
|
|
25
|
+
@classmethod
|
|
26
|
+
def try_create(cls, annotation_str: str) -> Optional["BaseAnnotation"]:
|
|
27
|
+
"""Try to create annotation from string.
|
|
28
|
+
|
|
29
|
+
Args:
|
|
30
|
+
annotation_str: Annotation string to parse
|
|
31
|
+
|
|
32
|
+
Returns:
|
|
33
|
+
Annotation instance or None if no match
|
|
34
|
+
"""
|
|
35
|
+
match = cls.regex().match(annotation_str.strip())
|
|
36
|
+
if match:
|
|
37
|
+
return cls(match) # type: ignore[call-arg]
|
|
38
|
+
return None
|
|
39
|
+
|
|
40
|
+
@abstractmethod
|
|
41
|
+
def wrap(self, text: str) -> str:
|
|
42
|
+
"""Wrap text in SSML element.
|
|
43
|
+
|
|
44
|
+
Args:
|
|
45
|
+
text: Content to wrap
|
|
46
|
+
|
|
47
|
+
Returns:
|
|
48
|
+
SSML string with text wrapped
|
|
49
|
+
"""
|
|
50
|
+
pass
|
|
51
|
+
|
|
52
|
+
def combine(self, other: "BaseAnnotation") -> None: # noqa: B027
|
|
53
|
+
"""Combine with duplicate annotation.
|
|
54
|
+
|
|
55
|
+
Default behavior: first annotation wins, ignore duplicates.
|
|
56
|
+
Override this method to implement custom combining logic.
|
|
57
|
+
|
|
58
|
+
Args:
|
|
59
|
+
other: Another annotation of the same type
|
|
60
|
+
"""
|
|
61
|
+
pass # Default: ignore duplicate
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
"""Extension annotation: [text](ext: whisper) → platform-specific SSML"""
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from collections.abc import Callable
|
|
5
|
+
from typing import Optional
|
|
6
|
+
|
|
7
|
+
from ssmd.annotations.base import BaseAnnotation
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class ExtensionAnnotation(BaseAnnotation):
|
|
11
|
+
"""Process platform-specific extension annotations.
|
|
12
|
+
|
|
13
|
+
Examples:
|
|
14
|
+
[whispers](ext: whisper) →
|
|
15
|
+
<amazon:effect name="whispered">whispers</amazon:effect>
|
|
16
|
+
[url](ext: audio) → <audio src="url"/>
|
|
17
|
+
|
|
18
|
+
Extensions can be registered via config['extensions'].
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
# Built-in extensions
|
|
22
|
+
DEFAULT_EXTENSIONS: dict[str, Callable[[str], str]] = {
|
|
23
|
+
"whisper": lambda text: (
|
|
24
|
+
f'<amazon:effect name="whispered">{text}</amazon:effect>'
|
|
25
|
+
),
|
|
26
|
+
"audio": lambda text: f'<audio src="{text}"/>',
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
def __init__(
|
|
30
|
+
self,
|
|
31
|
+
match: re.Match,
|
|
32
|
+
custom_extensions: dict[str, Callable[[str], str]] | None = None,
|
|
33
|
+
):
|
|
34
|
+
"""Initialize with extension name.
|
|
35
|
+
|
|
36
|
+
Args:
|
|
37
|
+
match: Regex match containing extension name
|
|
38
|
+
custom_extensions: Optional custom extension handlers
|
|
39
|
+
"""
|
|
40
|
+
self.extension_name = match.group(1).strip()
|
|
41
|
+
self.extensions: dict[str, Callable[[str], str]] = {
|
|
42
|
+
**self.DEFAULT_EXTENSIONS,
|
|
43
|
+
**(custom_extensions or {}),
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
@classmethod
|
|
47
|
+
def regex(cls) -> re.Pattern:
|
|
48
|
+
"""Match extension annotations.
|
|
49
|
+
|
|
50
|
+
Returns:
|
|
51
|
+
Pattern matching ext: annotations
|
|
52
|
+
"""
|
|
53
|
+
return re.compile(r"^ext:\s*(\w+)$")
|
|
54
|
+
|
|
55
|
+
@classmethod
|
|
56
|
+
def try_create(cls, annotation_str: str) -> Optional["ExtensionAnnotation"]:
|
|
57
|
+
"""Try to create annotation with custom extensions from config.
|
|
58
|
+
|
|
59
|
+
Note: This override is needed to pass custom extensions from config.
|
|
60
|
+
In practice, the AnnotationProcessor should handle this.
|
|
61
|
+
|
|
62
|
+
Args:
|
|
63
|
+
annotation_str: Annotation string
|
|
64
|
+
|
|
65
|
+
Returns:
|
|
66
|
+
ExtensionAnnotation instance or None
|
|
67
|
+
"""
|
|
68
|
+
match = cls.regex().match(annotation_str.strip())
|
|
69
|
+
if match:
|
|
70
|
+
return cls(match)
|
|
71
|
+
return None
|
|
72
|
+
|
|
73
|
+
def wrap(self, text: str) -> str:
|
|
74
|
+
"""Wrap text using extension handler.
|
|
75
|
+
|
|
76
|
+
Args:
|
|
77
|
+
text: Content to wrap
|
|
78
|
+
|
|
79
|
+
Returns:
|
|
80
|
+
Platform-specific SSML
|
|
81
|
+
"""
|
|
82
|
+
handler = self.extensions.get(self.extension_name)
|
|
83
|
+
if handler is not None:
|
|
84
|
+
return handler(text)
|
|
85
|
+
|
|
86
|
+
# Unknown extension, return unchanged
|
|
87
|
+
return text
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
"""Language annotation: [text](en) → <lang xml:lang="en-US">text</lang>"""
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
|
|
5
|
+
from ssmd.annotations.base import BaseAnnotation
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class LanguageAnnotation(BaseAnnotation):
|
|
9
|
+
"""Process language code annotations.
|
|
10
|
+
|
|
11
|
+
Examples:
|
|
12
|
+
[text](en) → <lang xml:lang="en-US">text</lang>
|
|
13
|
+
[text](en-GB) → <lang xml:lang="en-GB">text</lang>
|
|
14
|
+
[text](de) → <lang xml:lang="de-DE">text</lang>
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
# Language code defaults (2-letter code → full locale)
|
|
18
|
+
DEFAULTS = {
|
|
19
|
+
"en": "en-US",
|
|
20
|
+
"de": "de-DE",
|
|
21
|
+
"fr": "fr-FR",
|
|
22
|
+
"es": "es-ES",
|
|
23
|
+
"it": "it-IT",
|
|
24
|
+
"pt": "pt-PT",
|
|
25
|
+
"ru": "ru-RU",
|
|
26
|
+
"zh": "zh-CN",
|
|
27
|
+
"ja": "ja-JP",
|
|
28
|
+
"ko": "ko-KR",
|
|
29
|
+
"ar": "ar-SA",
|
|
30
|
+
"hi": "hi-IN",
|
|
31
|
+
"nl": "nl-NL",
|
|
32
|
+
"pl": "pl-PL",
|
|
33
|
+
"sv": "sv-SE",
|
|
34
|
+
"da": "da-DK",
|
|
35
|
+
"no": "no-NO",
|
|
36
|
+
"fi": "fi-FI",
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
def __init__(self, match: re.Match):
|
|
40
|
+
"""Initialize with language code.
|
|
41
|
+
|
|
42
|
+
Args:
|
|
43
|
+
match: Regex match containing language code
|
|
44
|
+
"""
|
|
45
|
+
lang_code = match.group(1)
|
|
46
|
+
# Auto-complete if only 2-letter code
|
|
47
|
+
self.lang = self.DEFAULTS.get(lang_code, lang_code)
|
|
48
|
+
|
|
49
|
+
@classmethod
|
|
50
|
+
def regex(cls) -> re.Pattern:
|
|
51
|
+
"""Match language codes.
|
|
52
|
+
|
|
53
|
+
Returns:
|
|
54
|
+
Pattern matching 2-letter or full locale codes
|
|
55
|
+
"""
|
|
56
|
+
return re.compile(r"^([a-z]{2}(?:-[A-Z]{2})?)$")
|
|
57
|
+
|
|
58
|
+
def wrap(self, text: str) -> str:
|
|
59
|
+
"""Wrap text in lang tag.
|
|
60
|
+
|
|
61
|
+
Args:
|
|
62
|
+
text: Content to wrap
|
|
63
|
+
|
|
64
|
+
Returns:
|
|
65
|
+
SSML <lang> element
|
|
66
|
+
"""
|
|
67
|
+
return f'<lang xml:lang="{self.lang}">{text}</lang>'
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
"""Phoneme annotation: [text](ph: dIC) → <phoneme>text</phoneme>"""
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
from ssmd.annotations.base import BaseAnnotation
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class PhonemeAnnotation(BaseAnnotation):
|
|
10
|
+
"""Process phoneme annotations.
|
|
11
|
+
|
|
12
|
+
Supports both X-SAMPA and IPA notation:
|
|
13
|
+
[text](ph: dIC) → <phoneme alphabet="ipa" ph="dɪç">text</phoneme>
|
|
14
|
+
[text](ipa: dɪç) → <phoneme alphabet="ipa" ph="dɪç">text</phoneme>
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
# Lazy-loaded X-SAMPA to IPA conversion table
|
|
18
|
+
_XSAMPA_TABLE: dict[str, str] | None = None
|
|
19
|
+
|
|
20
|
+
def __init__(self, match: re.Match):
|
|
21
|
+
"""Initialize with phoneme data.
|
|
22
|
+
|
|
23
|
+
Args:
|
|
24
|
+
match: Regex match containing alphabet type and phonemes
|
|
25
|
+
"""
|
|
26
|
+
alphabet = match.group(1) # 'ph' or 'ipa'
|
|
27
|
+
phonemes = match.group(2).strip()
|
|
28
|
+
|
|
29
|
+
# Convert X-SAMPA to IPA if needed
|
|
30
|
+
if alphabet == "ph":
|
|
31
|
+
self.phonemes = self._xsampa_to_ipa(phonemes)
|
|
32
|
+
else:
|
|
33
|
+
self.phonemes = phonemes
|
|
34
|
+
|
|
35
|
+
@classmethod
|
|
36
|
+
def regex(cls) -> re.Pattern:
|
|
37
|
+
"""Match phoneme annotations.
|
|
38
|
+
|
|
39
|
+
Returns:
|
|
40
|
+
Pattern matching ph: or ipa: annotations
|
|
41
|
+
"""
|
|
42
|
+
return re.compile(r"^(ph|ipa):\s*(.+)$")
|
|
43
|
+
|
|
44
|
+
def wrap(self, text: str) -> str:
|
|
45
|
+
"""Wrap text in phoneme tag.
|
|
46
|
+
|
|
47
|
+
Args:
|
|
48
|
+
text: Content to wrap
|
|
49
|
+
|
|
50
|
+
Returns:
|
|
51
|
+
SSML <phoneme> element
|
|
52
|
+
"""
|
|
53
|
+
return f'<phoneme alphabet="ipa" ph="{self.phonemes}">{text}</phoneme>'
|
|
54
|
+
|
|
55
|
+
@classmethod
|
|
56
|
+
def _load_xsampa_table(cls) -> dict[str, str]:
|
|
57
|
+
"""Load X-SAMPA to IPA conversion table.
|
|
58
|
+
|
|
59
|
+
Returns:
|
|
60
|
+
Dictionary mapping X-SAMPA to IPA
|
|
61
|
+
"""
|
|
62
|
+
if cls._XSAMPA_TABLE is not None:
|
|
63
|
+
return cls._XSAMPA_TABLE
|
|
64
|
+
|
|
65
|
+
table = {}
|
|
66
|
+
table_file = Path(__file__).parent / "xsampa_to_ipa.txt"
|
|
67
|
+
|
|
68
|
+
if table_file.exists():
|
|
69
|
+
with open(table_file, encoding="utf-8") as f:
|
|
70
|
+
for line in f:
|
|
71
|
+
line = line.strip()
|
|
72
|
+
if line and not line.startswith("#"):
|
|
73
|
+
parts = line.split(maxsplit=1)
|
|
74
|
+
if len(parts) == 2:
|
|
75
|
+
xsampa, ipa = parts
|
|
76
|
+
table[xsampa] = ipa
|
|
77
|
+
|
|
78
|
+
cls._XSAMPA_TABLE = table
|
|
79
|
+
return table
|
|
80
|
+
|
|
81
|
+
@classmethod
|
|
82
|
+
def _xsampa_to_ipa(cls, xsampa: str) -> str:
|
|
83
|
+
"""Convert X-SAMPA notation to IPA.
|
|
84
|
+
|
|
85
|
+
Args:
|
|
86
|
+
xsampa: X-SAMPA phoneme string
|
|
87
|
+
|
|
88
|
+
Returns:
|
|
89
|
+
IPA phoneme string
|
|
90
|
+
"""
|
|
91
|
+
table = cls._load_xsampa_table()
|
|
92
|
+
|
|
93
|
+
# Sort by length (longest first) for proper replacement
|
|
94
|
+
sorted_keys = sorted(table.keys(), key=len, reverse=True)
|
|
95
|
+
|
|
96
|
+
result = xsampa
|
|
97
|
+
for x in sorted_keys:
|
|
98
|
+
result = result.replace(x, table[x])
|
|
99
|
+
|
|
100
|
+
return result
|