@ondewo/s2t-client-typescript 7.4.0 → 7.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -52,13 +52,18 @@ npm
52
52
  │ ├── google
53
53
  │ │ └── protobuf
54
54
  │ │ ├── empty_pb.d.ts
55
- │ │ └── empty_pb.js
55
+ │ │ ├── empty_pb.js
56
+ │ │ ├── struct_pb.d.ts
57
+ │ │ └── struct_pb.js
56
58
  │ └── ondewo
57
59
  │ └── s2t
58
60
  │ ├── speech-to-text_grpc_web_pb.d.ts
59
61
  │ ├── speech-to-text_grpc_web_pb.js
60
62
  │ ├── speech-to-text_pb.d.ts
61
63
  │ └── speech-to-text_pb.js
64
+ ├── auth
65
+ │ ├── offlineTokenProvider.d.ts
66
+ │ └── offlineTokenProvider.js
62
67
  ├── LICENSE
63
68
  ├── package.json
64
69
  ├── public-api.d.ts
@@ -66,3 +71,38 @@ npm
66
71
  └── README.md
67
72
  ```
68
73
 
74
+ `api/` and `public-api.*` are generated from `ondewo/s2t/speech-to-text.proto` by the
75
+ [ONDEWO PROTO COMPILER](https://github.com/ondewo/ondewo-proto-compiler); `auth/` is hand-written and compiled into
76
+ the package by `make create_npm_package`.
77
+
78
+ ## Authentication
79
+
80
+ The service expects a Keycloak bearer token in the `authorization` gRPC metadata header. `auth/offlineTokenProvider`
81
+ performs the headless (2FA-exempt) ROPC + `offline_access` login against the public SDK client and keeps the
82
+ short-lived access token fresh in the background until `tokenExpirationInS` elapses:
83
+
84
+ ```typescript
85
+ import { login, OfflineTokenProvider } from '@ondewo/s2t-client-typescript/auth/offlineTokenProvider';
86
+ import { Speech2TextPromiseClient } from '@ondewo/s2t-client-typescript/api/ondewo/s2t/speech-to-text_grpc_web_pb';
87
+ import { ListS2tPipelinesRequest } from '@ondewo/s2t-client-typescript/api/ondewo/s2t/speech-to-text_pb';
88
+
89
+ const provider: OfflineTokenProvider = await login({
90
+ keycloakUrl: 'https://auth.ondewo.com/auth',
91
+ realm: 'ondewo-ccai-platform',
92
+ clientId: 'ondewo-nlu-cai-sdk-public',
93
+ username: process.env.KEYCLOAK_USER_NAME ?? '',
94
+ password: process.env.KEYCLOAK_PASSWORD ?? ''
95
+ // keycloakVerifySsl: false // ONLY for a self-signed local Envoy; Node-only, ignored in a browser
96
+ });
97
+
98
+ const client = new Speech2TextPromiseClient('http://localhost:8080');
99
+ const request = new ListS2tPipelinesRequest();
100
+ request.setLanguagesList(['en-US']);
101
+ const response = await client.listS2tPipelines(request, { Authorization: provider.getAuthorizationHeader() });
102
+
103
+ provider.stop(); // stops the background refresh loop so the process can exit
104
+ ```
105
+
106
+ A runnable version of exactly this flow, configured from `examples/environment.env`, lives in
107
+ `examples/ts-client.ts`.
108
+
@@ -2091,6 +2091,22 @@ export class VoiceActivityDetection extends jspb.Message {
2091
2091
  hasPyannote(): boolean;
2092
2092
  clearPyannote(): VoiceActivityDetection;
2093
2093
 
2094
+ getSilero(): Silero | undefined;
2095
+ setSilero(value?: Silero): VoiceActivityDetection;
2096
+ hasSilero(): boolean;
2097
+ clearSilero(): VoiceActivityDetection;
2098
+
2099
+ getWespeakerTsd(): WespeakerTsd | undefined;
2100
+ setWespeakerTsd(value?: WespeakerTsd): VoiceActivityDetection;
2101
+ hasWespeakerTsd(): boolean;
2102
+ clearWespeakerTsd(): VoiceActivityDetection;
2103
+
2104
+ getVadMethod(): VadMethod;
2105
+ setVadMethod(value: VadMethod): VoiceActivityDetection;
2106
+
2107
+ getTsdMethod(): TsdMethod;
2108
+ setTsdMethod(value: TsdMethod): VoiceActivityDetection;
2109
+
2094
2110
  serializeBinary(): Uint8Array;
2095
2111
  toObject(includeInstance?: boolean): VoiceActivityDetection.AsObject;
2096
2112
  static toObject(includeInstance: boolean, msg: VoiceActivityDetection): VoiceActivityDetection.AsObject;
@@ -2104,6 +2120,10 @@ export namespace VoiceActivityDetection {
2104
2120
  active: string,
2105
2121
  samplingRate: number,
2106
2122
  pyannote?: Pyannote.AsObject,
2123
+ silero?: Silero.AsObject,
2124
+ wespeakerTsd?: WespeakerTsd.AsObject,
2125
+ vadMethod: VadMethod,
2126
+ tsdMethod: TsdMethod,
2107
2127
  }
2108
2128
  }
2109
2129
 
@@ -2145,6 +2165,136 @@ export namespace Pyannote {
2145
2165
  }
2146
2166
  }
2147
2167
 
2168
+ export class Silero extends jspb.Message {
2169
+ getModelName(): string;
2170
+ setModelName(value: string): Silero;
2171
+
2172
+ getMinAudioSize(): number;
2173
+ setMinAudioSize(value: number): Silero;
2174
+
2175
+ getThreshold(): number;
2176
+ setThreshold(value: number): Silero;
2177
+ hasThreshold(): boolean;
2178
+ clearThreshold(): Silero;
2179
+
2180
+ getMinSpeechDurationMs(): number;
2181
+ setMinSpeechDurationMs(value: number): Silero;
2182
+ hasMinSpeechDurationMs(): boolean;
2183
+ clearMinSpeechDurationMs(): Silero;
2184
+
2185
+ getMinSilenceDurationMs(): number;
2186
+ setMinSilenceDurationMs(value: number): Silero;
2187
+ hasMinSilenceDurationMs(): boolean;
2188
+ clearMinSilenceDurationMs(): Silero;
2189
+
2190
+ getSpeechPadMs(): number;
2191
+ setSpeechPadMs(value: number): Silero;
2192
+ hasSpeechPadMs(): boolean;
2193
+ clearSpeechPadMs(): Silero;
2194
+
2195
+ getTritonServerHost(): string;
2196
+ setTritonServerHost(value: string): Silero;
2197
+
2198
+ getTritonServerPort(): number;
2199
+ setTritonServerPort(value: number): Silero;
2200
+
2201
+ serializeBinary(): Uint8Array;
2202
+ toObject(includeInstance?: boolean): Silero.AsObject;
2203
+ static toObject(includeInstance: boolean, msg: Silero): Silero.AsObject;
2204
+ static serializeBinaryToWriter(message: Silero, writer: jspb.BinaryWriter): void;
2205
+ static deserializeBinary(bytes: Uint8Array): Silero;
2206
+ static deserializeBinaryFromReader(message: Silero, reader: jspb.BinaryReader): Silero;
2207
+ }
2208
+
2209
+ export namespace Silero {
2210
+ export type AsObject = {
2211
+ modelName: string,
2212
+ minAudioSize: number,
2213
+ threshold?: number,
2214
+ minSpeechDurationMs?: number,
2215
+ minSilenceDurationMs?: number,
2216
+ speechPadMs?: number,
2217
+ tritonServerHost: string,
2218
+ tritonServerPort: number,
2219
+ }
2220
+
2221
+ export enum ThresholdCase {
2222
+ _THRESHOLD_NOT_SET = 0,
2223
+ THRESHOLD = 3,
2224
+ }
2225
+
2226
+ export enum MinSpeechDurationMsCase {
2227
+ _MIN_SPEECH_DURATION_MS_NOT_SET = 0,
2228
+ MIN_SPEECH_DURATION_MS = 4,
2229
+ }
2230
+
2231
+ export enum MinSilenceDurationMsCase {
2232
+ _MIN_SILENCE_DURATION_MS_NOT_SET = 0,
2233
+ MIN_SILENCE_DURATION_MS = 5,
2234
+ }
2235
+
2236
+ export enum SpeechPadMsCase {
2237
+ _SPEECH_PAD_MS_NOT_SET = 0,
2238
+ SPEECH_PAD_MS = 6,
2239
+ }
2240
+ }
2241
+
2242
+ export class WespeakerTsd extends jspb.Message {
2243
+ getActive(): boolean;
2244
+ setActive(value: boolean): WespeakerTsd;
2245
+
2246
+ getModelName(): string;
2247
+ setModelName(value: string): WespeakerTsd;
2248
+
2249
+ getTritonServerHost(): string;
2250
+ setTritonServerHost(value: string): WespeakerTsd;
2251
+
2252
+ getTritonServerPort(): number;
2253
+ setTritonServerPort(value: number): WespeakerTsd;
2254
+
2255
+ getSimilarityThreshold(): number;
2256
+ setSimilarityThreshold(value: number): WespeakerTsd;
2257
+ hasSimilarityThreshold(): boolean;
2258
+ clearSimilarityThreshold(): WespeakerTsd;
2259
+
2260
+ getMinAudioLength(): number;
2261
+ setMinAudioLength(value: number): WespeakerTsd;
2262
+ hasMinAudioLength(): boolean;
2263
+ clearMinAudioLength(): WespeakerTsd;
2264
+
2265
+ getReferenceMaxLength(): number;
2266
+ setReferenceMaxLength(value: number): WespeakerTsd;
2267
+
2268
+ serializeBinary(): Uint8Array;
2269
+ toObject(includeInstance?: boolean): WespeakerTsd.AsObject;
2270
+ static toObject(includeInstance: boolean, msg: WespeakerTsd): WespeakerTsd.AsObject;
2271
+ static serializeBinaryToWriter(message: WespeakerTsd, writer: jspb.BinaryWriter): void;
2272
+ static deserializeBinary(bytes: Uint8Array): WespeakerTsd;
2273
+ static deserializeBinaryFromReader(message: WespeakerTsd, reader: jspb.BinaryReader): WespeakerTsd;
2274
+ }
2275
+
2276
+ export namespace WespeakerTsd {
2277
+ export type AsObject = {
2278
+ active: boolean,
2279
+ modelName: string,
2280
+ tritonServerHost: string,
2281
+ tritonServerPort: number,
2282
+ similarityThreshold?: number,
2283
+ minAudioLength?: number,
2284
+ referenceMaxLength: number,
2285
+ }
2286
+
2287
+ export enum SimilarityThresholdCase {
2288
+ _SIMILARITY_THRESHOLD_NOT_SET = 0,
2289
+ SIMILARITY_THRESHOLD = 5,
2290
+ }
2291
+
2292
+ export enum MinAudioLengthCase {
2293
+ _MIN_AUDIO_LENGTH_NOT_SET = 0,
2294
+ MIN_AUDIO_LENGTH = 6,
2295
+ }
2296
+ }
2297
+
2148
2298
  export class PostProcessing extends jspb.Message {
2149
2299
  getPipelineList(): Array<string>;
2150
2300
  setPipelineList(value: Array<string>): PostProcessing;
@@ -2964,3 +3114,14 @@ export enum ReasoningEffort {
2964
3114
  REASONING_EFFORT_MEDIUM = 3,
2965
3115
  REASONING_EFFORT_HIGH = 4,
2966
3116
  }
3117
+ export enum VadMethod {
3118
+ VAD_METHOD_UNSPECIFIED = 0,
3119
+ VAD_METHOD_PYANNOTE = 1,
3120
+ VAD_METHOD_SILERO = 2,
3121
+ }
3122
+ export enum TsdMethod {
3123
+ TSD_METHOD_UNSPECIFIED = 0,
3124
+ TSD_METHOD_NONE = 1,
3125
+ TSD_METHOD_PYANNOTE = 2,
3126
+ TSD_METHOD_WESPEAKER = 3,
3127
+ }