speechmatics-agent-stt 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,191 @@
1
+ Metadata-Version: 2.4
2
+ Name: speechmatics-agent-stt
3
+ Version: 0.1.0
4
+ Summary: Speechmatics Agent STT Python client for agent transcription
5
+ Author-email: Speechmatics <support@speechmatics.com>
6
+ License-Expression: MIT
7
+ Project-URL: homepage, https://github.com/speechmatics/speechmatics-python-sdk
8
+ Project-URL: documentation, https://docs.speechmatics.com/
9
+ Project-URL: repository, https://github.com/speechmatics/speechmatics-python-sdk
10
+ Project-URL: issues, https://github.com/speechmatics/speechmatics-python-sdk/issues
11
+ Keywords: speechmatics,speech-to-text,conversational-ai,voice,agents,real-time,websocket,pipecat,livekit
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.9
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Operating System :: OS Independent
20
+ Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
21
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
22
+ Requires-Python: >=3.9
23
+ Description-Content-Type: text/markdown
24
+ Requires-Dist: speechmatics-rt>=1.1.1
25
+ Provides-Extra: jwt
26
+ Requires-Dist: aiohttp; extra == "jwt"
27
+ Provides-Extra: dev
28
+ Requires-Dist: black; extra == "dev"
29
+ Requires-Dist: ruff; extra == "dev"
30
+ Requires-Dist: mypy; extra == "dev"
31
+ Requires-Dist: pre-commit; extra == "dev"
32
+ Requires-Dist: pytest; extra == "dev"
33
+ Requires-Dist: pytest-asyncio; extra == "dev"
34
+ Requires-Dist: pytest-cov; extra == "dev"
35
+ Requires-Dist: pytest-mock; extra == "dev"
36
+ Requires-Dist: build; extra == "dev"
37
+
38
+ # Speechmatics Agent STT SDK
39
+
40
+ Python client for the Speechmatics **Agent STT** service, built on
41
+ [`speechmatics-rt`](https://pypi.org/project/speechmatics-rt/).
42
+
43
+ The Agent STT service works in **segments** rather than word groups, and reports the speech and
44
+ turn events a voice agent needs. This SDK runs **no VAD and no turn detection of its own** -
45
+ either the service's VAD closes turns, or your application's does.
46
+
47
+ ```bash
48
+ pip install speechmatics-agent-stt
49
+ ```
50
+
51
+ ## Quick start
52
+
53
+ ```python
54
+ import asyncio
55
+ from speechmatics.agent_stt import AgentSttAsyncClient, ServerMessageType, TranscriptionConfig
56
+
57
+ async def main():
58
+ # Uses SPEECHMATICS_API_KEY from the environment
59
+ client = AgentSttAsyncClient(
60
+ transcription_config=TranscriptionConfig(language="en", enable_partials=True)
61
+ )
62
+
63
+ # Register handlers before opening the session, so no message can arrive unhandled
64
+ @client.on(ServerMessageType.ADD_SEGMENT)
65
+ def handle_segment(message):
66
+ print(message["segment"]["transcript"])
67
+
68
+ async with client:
69
+ while chunk := next_audio_chunk():
70
+ await client.send_audio(chunk)
71
+
72
+ print(client.transcript)
73
+
74
+ asyncio.run(main())
75
+ ```
76
+
77
+ ## Who closes the turn
78
+
79
+ The service needs a boundary to close a segment on. Pick where it comes from:
80
+
81
+ ```python
82
+ from speechmatics.agent_stt import TurnConfig, TurnDetectionMode
83
+
84
+ # The service's VAD (default). It emits SpeechStarted/SpeechEnded and StartOfTurn/EndOfTurn.
85
+ turn_config = TurnConfig(turn_detection_mode=TurnDetectionMode.VAD)
86
+
87
+ # Your endpointing - Pipecat, LiveKit, or your own. The service's VAD stays off.
88
+ turn_config = TurnConfig(turn_detection_mode=TurnDetectionMode.EXTERNAL)
89
+
90
+ client = AgentSttAsyncClient(
91
+ transcription_config=TranscriptionConfig(language="en"), turn_config=turn_config
92
+ )
93
+ ```
94
+
95
+ With `TurnDetectionMode.EXTERNAL`, close each turn when your side decides speech has ended:
96
+
97
+ ```python
98
+ client.finalize() # from a sync callback
99
+ await client.force_end_of_utterance() # from async code
100
+ ```
101
+
102
+ Either sends `ForceEndOfUtterance` stamped with the audio position at the moment of the call, so
103
+ the service cuts the turn where you heard the end of speech rather than wherever the send lands.
104
+ The flushed segment comes back as a normal `AddSegment`.
105
+
106
+ What decides that is entirely yours - a VAD, an ML turn model, or a push-to-talk button. The SDK
107
+ only cares that something calls `finalize()`.
108
+
109
+ ## Session output
110
+
111
+ Every server message is dispatched to your handlers and also kept on the client:
112
+
113
+ ```python
114
+ client.transcript # final segments joined by the language's word delimiter
115
+ client.segments # list[Segment] - transcript, timing, speaker, is_final
116
+ client.partial_segment # the segment currently in flight, or None
117
+ client.timeline # list[TimedEvent] - the speech and turn events, in order
118
+ client.events # every raw message, including ones this SDK does not model
119
+ client.session_info # session id and the language pack the service reported
120
+
121
+ client.transcript_text(speaker_labels=True, include_partial=False)
122
+ ```
123
+
124
+ Pass `record_events=False` to `AgentSttAsyncClient` for long-running sessions where the raw log is not
125
+ wanted.
126
+
127
+ ## Messages
128
+
129
+ Emitted by the service:
130
+
131
+ | Message | Payload |
132
+ | --- | --- |
133
+ | `AddSegment` | `segment.transcript`, optional `segment.speaker`, `metadata.start_time`, `metadata.end_time` |
134
+ | `AddPartialSegment` | interim preview of the segment being built |
135
+ | `SpeechStarted` / `SpeechEnded` | `metadata.start_time` / `metadata.end_time` (service VAD) |
136
+ | `StartOfTurn` / `EndOfTurn` | `metadata.start_time` / `metadata.end_time` (service turn detection) |
137
+
138
+ Passed through from the RT engine: `RecognitionStarted`, `AudioAdded`, `EndOfTranscript`,
139
+ `SpeakersResult`, `Info`, `Warning`, `Error`.
140
+
141
+ Anything else the engine sends - the word-level `AddTranscript`/`AddPartialTranscript`, audio
142
+ events - is not modelled here, but still reaches `client.events` and any handler registered
143
+ under its name.
144
+
145
+ ## Configuration
146
+
147
+ `TranscriptionConfig` is the RT transcription config with the service's own model names.
148
+ Turn taking is configured separately, and is fixed for the life of the session:
149
+
150
+ | Config | Field | Meaning |
151
+ | --- | --- | --- |
152
+ | `TurnConfig` | `turn_detection_mode` | `TurnDetectionMode.VAD` (default) or `TurnDetectionMode.EXTERNAL` |
153
+
154
+ `model` takes an Agent STT `Model` and defaults to `DEFAULT_MODEL` (`Model.LINDEN_1`):
155
+
156
+ ```python
157
+ from speechmatics.agent_stt import Model, TranscriptionConfig
158
+
159
+ transcription_config = TranscriptionConfig(model=Model.LINDEN_1)
160
+ ```
161
+
162
+ The proxy in front of the service resolves the Agent STT model name onto the engine's operating
163
+ point, so the transcriber never sees a name it has no notion of. The RT models (`enhanced`,
164
+ `standard`) are not Agent STT models and are not accepted here; the deprecated `operating_point`
165
+ still passes through, and suppresses the `model` default so the two never arrive together.
166
+
167
+ Engine silence-based end of utterance is not offered here. A turn ends either because the
168
+ service's VAD said so, or because you called `finalize()`.
169
+
170
+ ## Endpoint
171
+
172
+ The Agent STT endpoint is the RT endpoint plus `/agent`:
173
+
174
+ ```python
175
+ AgentSttAsyncClient(url="wss://eu2.rt.speechmatics.com/v2") # -> /v2/agent
176
+ AgentSttAsyncClient(url="ws://localhost:8000/v2") # -> /v2/agent
177
+ AgentSttAsyncClient(app="pipecat/1.0") # reported as sm-app
178
+ ```
179
+
180
+ Resolution order: the `url` argument, `SPEECHMATICS_RT_URL`, then the EU endpoint. The `/agent`
181
+ segment is appended when it is missing.
182
+
183
+ ## Audio
184
+
185
+ The service requires **16 kHz raw PCM**, `pcm_s16le` or `pcm_f32le`, which is what the client
186
+ defaults to. Audio sent before the session is ready, or after it closes, is dropped rather than
187
+ raising, so an audio callback does not have to track session state.
188
+
189
+ ## Examples
190
+
191
+ See [examples/agent_stt](../../examples/agent_stt).
@@ -0,0 +1,154 @@
1
+ # Speechmatics Agent STT SDK
2
+
3
+ Python client for the Speechmatics **Agent STT** service, built on
4
+ [`speechmatics-rt`](https://pypi.org/project/speechmatics-rt/).
5
+
6
+ The Agent STT service works in **segments** rather than word groups, and reports the speech and
7
+ turn events a voice agent needs. This SDK runs **no VAD and no turn detection of its own** -
8
+ either the service's VAD closes turns, or your application's does.
9
+
10
+ ```bash
11
+ pip install speechmatics-agent-stt
12
+ ```
13
+
14
+ ## Quick start
15
+
16
+ ```python
17
+ import asyncio
18
+ from speechmatics.agent_stt import AgentSttAsyncClient, ServerMessageType, TranscriptionConfig
19
+
20
+ async def main():
21
+ # Uses SPEECHMATICS_API_KEY from the environment
22
+ client = AgentSttAsyncClient(
23
+ transcription_config=TranscriptionConfig(language="en", enable_partials=True)
24
+ )
25
+
26
+ # Register handlers before opening the session, so no message can arrive unhandled
27
+ @client.on(ServerMessageType.ADD_SEGMENT)
28
+ def handle_segment(message):
29
+ print(message["segment"]["transcript"])
30
+
31
+ async with client:
32
+ while chunk := next_audio_chunk():
33
+ await client.send_audio(chunk)
34
+
35
+ print(client.transcript)
36
+
37
+ asyncio.run(main())
38
+ ```
39
+
40
+ ## Who closes the turn
41
+
42
+ The service needs a boundary to close a segment on. Pick where it comes from:
43
+
44
+ ```python
45
+ from speechmatics.agent_stt import TurnConfig, TurnDetectionMode
46
+
47
+ # The service's VAD (default). It emits SpeechStarted/SpeechEnded and StartOfTurn/EndOfTurn.
48
+ turn_config = TurnConfig(turn_detection_mode=TurnDetectionMode.VAD)
49
+
50
+ # Your endpointing - Pipecat, LiveKit, or your own. The service's VAD stays off.
51
+ turn_config = TurnConfig(turn_detection_mode=TurnDetectionMode.EXTERNAL)
52
+
53
+ client = AgentSttAsyncClient(
54
+ transcription_config=TranscriptionConfig(language="en"), turn_config=turn_config
55
+ )
56
+ ```
57
+
58
+ With `TurnDetectionMode.EXTERNAL`, close each turn when your side decides speech has ended:
59
+
60
+ ```python
61
+ client.finalize() # from a sync callback
62
+ await client.force_end_of_utterance() # from async code
63
+ ```
64
+
65
+ Either sends `ForceEndOfUtterance` stamped with the audio position at the moment of the call, so
66
+ the service cuts the turn where you heard the end of speech rather than wherever the send lands.
67
+ The flushed segment comes back as a normal `AddSegment`.
68
+
69
+ What decides that is entirely yours - a VAD, an ML turn model, or a push-to-talk button. The SDK
70
+ only cares that something calls `finalize()`.
71
+
72
+ ## Session output
73
+
74
+ Every server message is dispatched to your handlers and also kept on the client:
75
+
76
+ ```python
77
+ client.transcript # final segments joined by the language's word delimiter
78
+ client.segments # list[Segment] - transcript, timing, speaker, is_final
79
+ client.partial_segment # the segment currently in flight, or None
80
+ client.timeline # list[TimedEvent] - the speech and turn events, in order
81
+ client.events # every raw message, including ones this SDK does not model
82
+ client.session_info # session id and the language pack the service reported
83
+
84
+ client.transcript_text(speaker_labels=True, include_partial=False)
85
+ ```
86
+
87
+ Pass `record_events=False` to `AgentSttAsyncClient` for long-running sessions where the raw log is not
88
+ wanted.
89
+
90
+ ## Messages
91
+
92
+ Emitted by the service:
93
+
94
+ | Message | Payload |
95
+ | --- | --- |
96
+ | `AddSegment` | `segment.transcript`, optional `segment.speaker`, `metadata.start_time`, `metadata.end_time` |
97
+ | `AddPartialSegment` | interim preview of the segment being built |
98
+ | `SpeechStarted` / `SpeechEnded` | `metadata.start_time` / `metadata.end_time` (service VAD) |
99
+ | `StartOfTurn` / `EndOfTurn` | `metadata.start_time` / `metadata.end_time` (service turn detection) |
100
+
101
+ Passed through from the RT engine: `RecognitionStarted`, `AudioAdded`, `EndOfTranscript`,
102
+ `SpeakersResult`, `Info`, `Warning`, `Error`.
103
+
104
+ Anything else the engine sends - the word-level `AddTranscript`/`AddPartialTranscript`, audio
105
+ events - is not modelled here, but still reaches `client.events` and any handler registered
106
+ under its name.
107
+
108
+ ## Configuration
109
+
110
+ `TranscriptionConfig` is the RT transcription config with the service's own model names.
111
+ Turn taking is configured separately, and is fixed for the life of the session:
112
+
113
+ | Config | Field | Meaning |
114
+ | --- | --- | --- |
115
+ | `TurnConfig` | `turn_detection_mode` | `TurnDetectionMode.VAD` (default) or `TurnDetectionMode.EXTERNAL` |
116
+
117
+ `model` takes an Agent STT `Model` and defaults to `DEFAULT_MODEL` (`Model.LINDEN_1`):
118
+
119
+ ```python
120
+ from speechmatics.agent_stt import Model, TranscriptionConfig
121
+
122
+ transcription_config = TranscriptionConfig(model=Model.LINDEN_1)
123
+ ```
124
+
125
+ The proxy in front of the service resolves the Agent STT model name onto the engine's operating
126
+ point, so the transcriber never sees a name it has no notion of. The RT models (`enhanced`,
127
+ `standard`) are not Agent STT models and are not accepted here; the deprecated `operating_point`
128
+ still passes through, and suppresses the `model` default so the two never arrive together.
129
+
130
+ Engine silence-based end of utterance is not offered here. A turn ends either because the
131
+ service's VAD said so, or because you called `finalize()`.
132
+
133
+ ## Endpoint
134
+
135
+ The Agent STT endpoint is the RT endpoint plus `/agent`:
136
+
137
+ ```python
138
+ AgentSttAsyncClient(url="wss://eu2.rt.speechmatics.com/v2") # -> /v2/agent
139
+ AgentSttAsyncClient(url="ws://localhost:8000/v2") # -> /v2/agent
140
+ AgentSttAsyncClient(app="pipecat/1.0") # reported as sm-app
141
+ ```
142
+
143
+ Resolution order: the `url` argument, `SPEECHMATICS_RT_URL`, then the EU endpoint. The `/agent`
144
+ segment is appended when it is missing.
145
+
146
+ ## Audio
147
+
148
+ The service requires **16 kHz raw PCM**, `pcm_s16le` or `pcm_f32le`, which is what the client
149
+ defaults to. Audio sent before the session is ready, or after it closes, is dropped rather than
150
+ raising, so an audio callback does not have to track session state.
151
+
152
+ ## Examples
153
+
154
+ See [examples/agent_stt](../../examples/agent_stt).
@@ -0,0 +1,69 @@
1
+ [build-system]
2
+ requires = ["setuptools>=61.0.0"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "speechmatics-agent-stt"
7
+ dynamic = ["version"]
8
+ description = "Speechmatics Agent STT Python client for agent transcription"
9
+ readme = "README.md"
10
+ authors = [{ name = "Speechmatics", email = "support@speechmatics.com" }]
11
+ license = "MIT"
12
+ requires-python = ">=3.9"
13
+ dependencies = ["speechmatics-rt>=1.1.1"]
14
+ classifiers = [
15
+ "Development Status :: 4 - Beta",
16
+ "Intended Audience :: Developers",
17
+ "Programming Language :: Python :: 3",
18
+ "Programming Language :: Python :: 3.9",
19
+ "Programming Language :: Python :: 3.10",
20
+ "Programming Language :: Python :: 3.11",
21
+ "Programming Language :: Python :: 3.12",
22
+ "Operating System :: OS Independent",
23
+ "Topic :: Multimedia :: Sound/Audio :: Speech",
24
+ "Topic :: Software Development :: Libraries :: Python Modules",
25
+ ]
26
+ keywords = [
27
+ "speechmatics",
28
+ "speech-to-text",
29
+ "conversational-ai",
30
+ "voice",
31
+ "agents",
32
+ "real-time",
33
+ "websocket",
34
+ "pipecat",
35
+ "livekit",
36
+ ]
37
+
38
+ [project.optional-dependencies]
39
+ jwt = ["aiohttp"]
40
+ dev = [
41
+ "black",
42
+ "ruff",
43
+ "mypy",
44
+ "pre-commit",
45
+ "pytest",
46
+ "pytest-asyncio",
47
+ "pytest-cov",
48
+ "pytest-mock",
49
+ "build",
50
+ ]
51
+
52
+ [project.urls]
53
+ homepage = "https://github.com/speechmatics/speechmatics-python-sdk"
54
+ documentation = "https://docs.speechmatics.com/"
55
+ repository = "https://github.com/speechmatics/speechmatics-python-sdk"
56
+ issues = "https://github.com/speechmatics/speechmatics-python-sdk/issues"
57
+
58
+ [tool.setuptools.dynamic]
59
+ version = { attr = "speechmatics.agent_stt.__version__" }
60
+
61
+ [tool.setuptools.package-data]
62
+ "speechmatics.agent_stt" = ["py.typed"]
63
+
64
+ [tool.setuptools.packages.find]
65
+ where = ["."]
66
+
67
+ [[tool.mypy.overrides]]
68
+ module = ["speechmatics.rt.*"]
69
+ ignore_missing_imports = true
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1 @@
1
+ __path__ = __import__("pkgutil").extend_path(__path__, __name__)
@@ -0,0 +1,103 @@
1
+ #
2
+ # Copyright (c) 2026, Speechmatics / Cantab Research Ltd
3
+ #
4
+
5
+ """Speechmatics Agent STT SDK.
6
+
7
+ A client for the Speechmatics Agent STT service, built on the Speechmatics Python Real-Time
8
+ SDK. The service works in segments rather than word groups and reports speech and turn events.
9
+
10
+ This SDK runs no VAD and no turn detection of its own: either the service's VAD closes turns,
11
+ or the application's does (Pipecat, LiveKit, ...) by calling `finalize()`.
12
+ """
13
+
14
+ __version__ = "0.1.0"
15
+
16
+ from speechmatics.rt import AudioEncoding
17
+ from speechmatics.rt import AudioError
18
+ from speechmatics.rt import AudioFormat
19
+ from speechmatics.rt import AuthBase
20
+ from speechmatics.rt import AuthenticationError
21
+ from speechmatics.rt import ConfigurationError
22
+ from speechmatics.rt import ConnectionConfig
23
+ from speechmatics.rt import ConnectionError
24
+ from speechmatics.rt import EventEmitter
25
+ from speechmatics.rt import JWTAuth
26
+ from speechmatics.rt import Microphone
27
+ from speechmatics.rt import SessionError
28
+ from speechmatics.rt import SpeakerDiarizationConfig
29
+ from speechmatics.rt import SpeakerIdentifier
30
+ from speechmatics.rt import StaticKeyAuth
31
+ from speechmatics.rt import TimeoutError
32
+ from speechmatics.rt import TranscriptionError
33
+ from speechmatics.rt import TransportError
34
+
35
+ from ._client import AgentSttAsyncClient
36
+ from ._models import DEFAULT_CHUNK_SIZE
37
+ from ._models import DEFAULT_MODEL
38
+ from ._models import DEFAULT_SAMPLE_RATE
39
+ from ._models import DEFAULT_WORD_DELIMITER
40
+ from ._models import SEGMENT_MESSAGES
41
+ from ._models import TIMED_MESSAGES
42
+ from ._models import AdditionalVocabEntry
43
+ from ._models import ClientMessageType
44
+ from ._models import LanguagePackInfo
45
+ from ._models import Model
46
+ from ._models import Segment
47
+ from ._models import ServerMessageType
48
+ from ._models import SessionInfo
49
+ from ._models import TimedEvent
50
+ from ._models import TranscriptionConfig
51
+ from ._models import TurnConfig
52
+ from ._models import TurnDetectionMode
53
+ from ._transcript import Transcript
54
+ from ._url import resolve_url
55
+
56
+ __all__ = [
57
+ "DEFAULT_CHUNK_SIZE",
58
+ "DEFAULT_MODEL",
59
+ "DEFAULT_SAMPLE_RATE",
60
+ "DEFAULT_WORD_DELIMITER",
61
+ "SEGMENT_MESSAGES",
62
+ "TIMED_MESSAGES",
63
+ "__version__",
64
+ # Client
65
+ "AgentSttAsyncClient",
66
+ # Config
67
+ "AdditionalVocabEntry",
68
+ "AudioEncoding",
69
+ "AudioFormat",
70
+ "ConnectionConfig",
71
+ "SpeakerDiarizationConfig",
72
+ "SpeakerIdentifier",
73
+ "TranscriptionConfig",
74
+ "TurnConfig",
75
+ "TurnDetectionMode",
76
+ "Model",
77
+ # Auth
78
+ "AuthBase",
79
+ "JWTAuth",
80
+ "StaticKeyAuth",
81
+ # Messages
82
+ "ClientMessageType",
83
+ "ServerMessageType",
84
+ "Segment",
85
+ "TimedEvent",
86
+ # Session
87
+ "LanguagePackInfo",
88
+ "SessionInfo",
89
+ "Transcript",
90
+ "resolve_url",
91
+ # Utilities
92
+ "EventEmitter",
93
+ "Microphone",
94
+ # Exceptions
95
+ "AudioError",
96
+ "AuthenticationError",
97
+ "ConfigurationError",
98
+ "ConnectionError",
99
+ "SessionError",
100
+ "TimeoutError",
101
+ "TranscriptionError",
102
+ "TransportError",
103
+ ]