@pie-players/tts-server-core 0.3.61 → 0.3.62
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cache.js +0 -1
- package/dist/index.js +0 -1
- package/dist/provider.js +0 -1
- package/dist/speech-marks.js +0 -1
- package/dist/types.js +0 -1
- package/package.json +2 -2
- package/dist/cache.js.map +0 -1
- package/dist/index.js.map +0 -1
- package/dist/provider.js.map +0 -1
- package/dist/speech-marks.js.map +0 -1
- package/dist/types.js.map +0 -1
package/dist/cache.js
CHANGED
package/dist/index.js
CHANGED
|
@@ -7,4 +7,3 @@ export { BaseTTSProvider } from "./provider.js";
|
|
|
7
7
|
// Export speech marks utilities
|
|
8
8
|
export { adjustSpeechMarksForRate, estimateSpeechMarks, filterSpeechMarksByType, getSpeechMarkAtTime, getSpeechMarksStats, mergeSpeechMarks, validateSpeechMarks, } from "./speech-marks.js";
|
|
9
9
|
export { TTSError, TTSErrorCode } from "./types.js";
|
|
10
|
-
//# sourceMappingURL=index.js.map
|
package/dist/provider.js
CHANGED
package/dist/speech-marks.js
CHANGED
package/dist/types.js
CHANGED
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@pie-players/tts-server-core",
|
|
3
|
-
"version": "0.3.
|
|
3
|
+
"version": "0.3.62",
|
|
4
4
|
"author": "PIE Framework",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
"main": "./dist/index.js",
|
|
14
14
|
"devDependencies": {
|
|
15
15
|
"typescript": "^5.9.3",
|
|
16
|
-
"vitest": "^4.1.
|
|
16
|
+
"vitest": "^4.1.10"
|
|
17
17
|
},
|
|
18
18
|
"exports": {
|
|
19
19
|
".": {
|
package/dist/cache.js.map
DELETED
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"cache.js","sourceRoot":"","sources":["../src/cache.ts"],"names":[],"mappings":"AAAA;;;GAGG;AA8FH;;;;;GAKG;AACH,MAAM,UAAU,gBAAgB,CAAC,UAA8B;IAC9D,MAAM,EACL,UAAU,EACV,IAAI,EACJ,KAAK,EACL,QAAQ,GAAG,EAAE,EACb,IAAI,GAAG,GAAG,EACV,MAAM,GAAG,KAAK,GACd,GAAG,UAAU,CAAC;IAEf,2CAA2C;IAC3C,MAAM,QAAQ,GAAG;QAChB,KAAK;QACL,UAAU;QACV,KAAK;QACL,QAAQ;QACR,IAAI,CAAC,OAAO,CAAC,CAAC,CAAC;QACf,MAAM;QACN,IAAI;KACJ,CAAC;IAEF,0CAA0C;IAC1C,iEAAiE;IACjE,OAAO,QAAQ,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC;AAC3B,CAAC;AAED;;;;;;GAMG;AACH,MAAM,CAAC,KAAK,UAAU,QAAQ,CAAC,IAAY;IAC1C,gEAAgE;IAChE,MAAM,OAAO,GAAG,IAAI,WAAW,EAAE,CAAC;IAClC,MAAM,IAAI,GAAG,OAAO,CAAC,MAAM,CAAC,IAAI,CAAC,CAAC;IAClC,MAAM,UAAU,GAAG,MAAM,MAAM,CAAC,MAAM,CAAC,MAAM,CAAC,SAAS,EAAE,IAAI,CAAC,CAAC;IAC/D,MAAM,SAAS,GAAG,KAAK,CAAC,IAAI,CAAC,IAAI,UAAU,CAAC,UAAU,CAAC,CAAC,CAAC;IACzD,OAAO,SAAS,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,QAAQ,CAAC,EAAE,CAAC,CAAC,QAAQ,CAAC,CAAC,EAAE,GAAG,CAAC,CAAC,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;AACvE,CAAC;AAED;;;;;GAKG;AACH,MAAM,CAAC,KAAK,UAAU,sBAAsB,CAC3C,UAA8B;IAE9B,MAAM,EACL,UAAU,EACV,IAAI,EACJ,KAAK,EACL,QAAQ,GAAG,EAAE,EACb,IAAI,GAAG,GAAG,EACV,MAAM,GAAG,KAAK,GACd,GAAG,UAAU,CAAC;IAEf,8CAA8C;IAC9C,MAAM,QAAQ,GAAG,MAAM,QAAQ,CAAC,IAAI,CAAC,CAAC;IAEtC,MAAM,QAAQ,GAAG;QAChB,KAAK;QACL,UAAU;QACV,KAAK;QACL,QAAQ;QACR,IAAI,CAAC,OAAO,CAAC,CAAC,CAAC;QACf,MAAM;QACN,QAAQ;KACR,CAAC;IAEF,OAAO,QAAQ,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC;AAC3B,CAAC;AAED;;;GAGG;AACH,MAAM,OAAO,WAAW;IACf,KAAK,GAAG,IAAI,GAAG,EAGpB,CAAC;IACI,IAAI,GAAG,CAAC,CAAC;IACT,MAAM,GAAG,CAAC,CAAC;IACX,OAAO,CAAS;IAExB,YAAY,OAAO,GAAG,GAAG;QACxB,IAAI,CAAC,OAAO,GAAG,OAAO,CAAC;IACxB,CAAC;IAED,KAAK,CAAC,GAAG,CAAC,GAAW;QACpB,MAAM,KAAK,GAAG,IAAI,CAAC,KAAK,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC;QAElC,IAAI,CAAC,KAAK,EAAE,CAAC;YACZ,IAAI,CAAC,MAAM,EAAE,CAAC;YACd,OAAO,IAAI,CAAC;QACb,CAAC;QAED,mBAAmB;QACnB,IAAI,IAAI,CAAC,GAAG,EAAE,GAAG,KAAK,CAAC,OAAO,EAAE,CAAC;YAChC,IAAI,CAAC,KAAK,CAAC,MAAM,CAAC,GAAG,CAAC,CAAC;YACvB,IAAI,CAAC,MAAM,EAAE,CAAC;YACd,OAAO,IAAI,CAAC;QACb,CAAC;QAED,IAAI,CAAC,IAAI,EAAE,CAAC;QAEZ,+CAA+C;QAC/C,MAAM,MAAM,GAAG,EAAE,GAAG,KAAK,CAAC,KAAK,EAAE,CAAC;QAClC,MAAM,CAAC,QAAQ,GAAG,EAAE,GAAG,MAAM,CAAC,QAAQ,EAAE,MAAM,EAAE,IAAI,EAAE,CAAC;QAEvD,OAAO,MAAM,CAAC;IACf,CAAC;IAED,KAAK,CAAC,GAAG,CACR,GAAW,EACX,KAAyB,EACzB,GAAG,GAAG,KAAK;QAEX,gCAAgC;QAChC,IAAI,IAAI,CAAC,KAAK,CAAC,IAAI,IAAI,IAAI,CAAC,OAAO,EAAE,CAAC;YACrC,kCAAkC;YAClC,MAAM,QAAQ,GAAG,IAAI,CAAC,KAAK,CAAC,IAAI,EAAE,CAAC,IAAI,EAAE,CAAC,KAAK,CAAC;YAChD,IAAI,QAAQ,EAAE,CAAC;gBACd,IAAI,CAAC,KAAK,CAAC,MAAM,CAAC,QAAQ,CAAC,CAAC;YAC7B,CAAC;QACF,CAAC;QAED,IAAI,CAAC,KAAK,CAAC,GAAG,CAAC,GAAG,EAAE;YACnB,KAAK;YACL,OAAO,EAAE,IAAI,CAAC,GAAG,EAAE,GAAG,GAAG,GAAG,IAAI;SAChC,CAAC,CAAC;IACJ,CAAC;IAED,KAAK,CAAC,GAAG,CAAC,GAAW;QACpB,MAAM,KAAK,GAAG,IAAI,CAAC,KAAK,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC;QAClC,IAAI,CAAC,KAAK;YAAE,OAAO,KAAK,CAAC;QAEzB,mBAAmB;QACnB,IAAI,IAAI,CAAC,GAAG,EAAE,GAAG,KAAK,CAAC,OAAO,EAAE,CAAC;YAChC,IAAI,CAAC,KAAK,CAAC,MAAM,CAAC,GAAG,CAAC,CAAC;YACvB,OAAO,KAAK,CAAC;QACd,CAAC;QAED,OAAO,IAAI,CAAC;IACb,CAAC;IAED,KAAK,CAAC,MAAM,CAAC,GAAW;QACvB,IAAI,CAAC,KAAK,CAAC,MAAM,CAAC,GAAG,CAAC,CAAC;IACxB,CAAC;IAED,KAAK,CAAC,KAAK;QACV,IAAI,CAAC,KAAK,CAAC,KAAK,EAAE,CAAC;QACnB,IAAI,CAAC,IAAI,GAAG,CAAC,CAAC;QACd,IAAI,CAAC,MAAM,GAAG,CAAC,CAAC;IACjB,CAAC;IAED,KAAK,CAAC,QAAQ;QACb,MAAM,KAAK,GAAG,IAAI,CAAC,IAAI,GAAG,IAAI,CAAC,MAAM,CAAC;QACtC,OAAO;YACN,IAAI,EAAE,IAAI,CAAC,IAAI;YACf,MAAM,EAAE,IAAI,CAAC,MAAM;YACnB,OAAO,EAAE,KAAK,GAAG,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,IAAI,GAAG,KAAK,CAAC,CAAC,CAAC,CAAC;YAC1C,QAAQ,EAAE,IAAI,CAAC,KAAK,CAAC,IAAI;SACzB,CAAC;IACH,CAAC;CACD","sourcesContent":["/**\n * Caching interface for TTS results\n * @module @pie-players/tts-server-core\n */\n\nimport type { SynthesizeResponse } from \"./types.js\";\n\n/**\n * Cache key components for TTS synthesis\n */\nexport interface CacheKeyComponents {\n\t/** Provider identifier */\n\tproviderId: string;\n\n\t/** Text to synthesize */\n\ttext: string;\n\n\t/** Voice ID */\n\tvoice: string;\n\n\t/** Language code */\n\tlanguage?: string;\n\n\t/** Speech rate */\n\trate?: number;\n\n\t/** Audio format */\n\tformat?: string;\n}\n\n/**\n * Cache interface for TTS providers\n */\nexport interface ITTSCache {\n\t/**\n\t * Get cached synthesis result\n\t *\n\t * @param key - Cache key\n\t * @returns Cached result or null if not found\n\t */\n\tget(key: string): Promise<SynthesizeResponse | null>;\n\n\t/**\n\t * Store synthesis result in cache\n\t *\n\t * @param key - Cache key\n\t * @param value - Synthesis response to cache\n\t * @param ttl - Time to live in seconds (optional)\n\t */\n\tset(key: string, value: SynthesizeResponse, ttl?: number): Promise<void>;\n\n\t/**\n\t * Check if key exists in cache\n\t *\n\t * @param key - Cache key\n\t * @returns True if key exists\n\t */\n\thas(key: string): Promise<boolean>;\n\n\t/**\n\t * Delete cached result\n\t *\n\t * @param key - Cache key\n\t */\n\tdelete(key: string): Promise<void>;\n\n\t/**\n\t * Clear all cached results\n\t */\n\tclear(): Promise<void>;\n\n\t/**\n\t * Get cache statistics\n\t */\n\tgetStats?(): Promise<CacheStats>;\n}\n\n/**\n * Cache statistics\n */\nexport interface CacheStats {\n\t/** Total cache hits */\n\thits: number;\n\n\t/** Total cache misses */\n\tmisses: number;\n\n\t/** Hit rate (0.0 to 1.0) */\n\thitRate: number;\n\n\t/** Number of keys in cache */\n\tkeyCount: number;\n\n\t/** Total size in bytes (if available) */\n\tsizeBytes?: number;\n}\n\n/**\n * Generate cache key from components\n *\n * @param components - Cache key components\n * @returns Cache key string\n */\nexport function generateCacheKey(components: CacheKeyComponents): string {\n\tconst {\n\t\tproviderId,\n\t\ttext,\n\t\tvoice,\n\t\tlanguage = \"\",\n\t\trate = 1.0,\n\t\tformat = \"mp3\",\n\t} = components;\n\n\t// Create deterministic key from components\n\tconst keyParts = [\n\t\t\"tts\",\n\t\tproviderId,\n\t\tvoice,\n\t\tlanguage,\n\t\trate.toFixed(2),\n\t\tformat,\n\t\ttext,\n\t];\n\n\t// Use simple concatenation with delimiter\n\t// In production, consider using a hash function for shorter keys\n\treturn keyParts.join(\":\");\n}\n\n/**\n * Generate SHA-256 hash for cache key\n * Useful for creating shorter keys from long text\n *\n * @param text - Text to hash\n * @returns Hex string hash\n */\nexport async function hashText(text: string): Promise<string> {\n\t// Use Web Crypto API (available in modern Node.js and browsers)\n\tconst encoder = new TextEncoder();\n\tconst data = encoder.encode(text);\n\tconst hashBuffer = await crypto.subtle.digest(\"SHA-256\", data);\n\tconst hashArray = Array.from(new Uint8Array(hashBuffer));\n\treturn hashArray.map((b) => b.toString(16).padStart(2, \"0\")).join(\"\");\n}\n\n/**\n * Generate short cache key using hash\n *\n * @param components - Cache key components\n * @returns Promise resolving to cache key\n */\nexport async function generateHashedCacheKey(\n\tcomponents: CacheKeyComponents,\n): Promise<string> {\n\tconst {\n\t\tproviderId,\n\t\ttext,\n\t\tvoice,\n\t\tlanguage = \"\",\n\t\trate = 1.0,\n\t\tformat = \"mp3\",\n\t} = components;\n\n\t// Hash the text to keep key length reasonable\n\tconst textHash = await hashText(text);\n\n\tconst keyParts = [\n\t\t\"tts\",\n\t\tproviderId,\n\t\tvoice,\n\t\tlanguage,\n\t\trate.toFixed(2),\n\t\tformat,\n\t\ttextHash,\n\t];\n\n\treturn keyParts.join(\":\");\n}\n\n/**\n * In-memory cache implementation\n * Simple LRU cache for development/testing\n */\nexport class MemoryCache implements ITTSCache {\n\tprivate cache = new Map<\n\t\tstring,\n\t\t{ value: SynthesizeResponse; expires: number }\n\t>();\n\tprivate hits = 0;\n\tprivate misses = 0;\n\tprivate maxSize: number;\n\n\tconstructor(maxSize = 100) {\n\t\tthis.maxSize = maxSize;\n\t}\n\n\tasync get(key: string): Promise<SynthesizeResponse | null> {\n\t\tconst entry = this.cache.get(key);\n\n\t\tif (!entry) {\n\t\t\tthis.misses++;\n\t\t\treturn null;\n\t\t}\n\n\t\t// Check expiration\n\t\tif (Date.now() > entry.expires) {\n\t\t\tthis.cache.delete(key);\n\t\t\tthis.misses++;\n\t\t\treturn null;\n\t\t}\n\n\t\tthis.hits++;\n\n\t\t// Update metadata to mark as served from cache\n\t\tconst result = { ...entry.value };\n\t\tresult.metadata = { ...result.metadata, cached: true };\n\n\t\treturn result;\n\t}\n\n\tasync set(\n\t\tkey: string,\n\t\tvalue: SynthesizeResponse,\n\t\tttl = 86400,\n\t): Promise<void> {\n\t\t// Enforce max size (simple LRU)\n\t\tif (this.cache.size >= this.maxSize) {\n\t\t\t// Delete oldest entry (first key)\n\t\t\tconst firstKey = this.cache.keys().next().value;\n\t\t\tif (firstKey) {\n\t\t\t\tthis.cache.delete(firstKey);\n\t\t\t}\n\t\t}\n\n\t\tthis.cache.set(key, {\n\t\t\tvalue,\n\t\t\texpires: Date.now() + ttl * 1000,\n\t\t});\n\t}\n\n\tasync has(key: string): Promise<boolean> {\n\t\tconst entry = this.cache.get(key);\n\t\tif (!entry) return false;\n\n\t\t// Check expiration\n\t\tif (Date.now() > entry.expires) {\n\t\t\tthis.cache.delete(key);\n\t\t\treturn false;\n\t\t}\n\n\t\treturn true;\n\t}\n\n\tasync delete(key: string): Promise<void> {\n\t\tthis.cache.delete(key);\n\t}\n\n\tasync clear(): Promise<void> {\n\t\tthis.cache.clear();\n\t\tthis.hits = 0;\n\t\tthis.misses = 0;\n\t}\n\n\tasync getStats(): Promise<CacheStats> {\n\t\tconst total = this.hits + this.misses;\n\t\treturn {\n\t\t\thits: this.hits,\n\t\t\tmisses: this.misses,\n\t\t\thitRate: total > 0 ? this.hits / total : 0,\n\t\t\tkeyCount: this.cache.size,\n\t\t};\n\t}\n}\n"]}
|
package/dist/index.js.map
DELETED
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"index.js","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAQH,OAAO,EACN,gBAAgB,EAChB,sBAAsB,EACtB,QAAQ,EACR,WAAW,GACX,MAAM,YAAY,CAAC;AAQpB,OAAO,EAAE,eAAe,EAAE,MAAM,eAAe,CAAC;AAEhD,gCAAgC;AAChC,OAAO,EACN,wBAAwB,EACxB,mBAAmB,EACnB,uBAAuB,EACvB,mBAAmB,EACnB,mBAAmB,EACnB,gBAAgB,EAChB,mBAAmB,GACnB,MAAM,mBAAmB,CAAC;AAc3B,OAAO,EAAE,QAAQ,EAAE,YAAY,EAAE,MAAM,YAAY,CAAC","sourcesContent":["/**\n * Core types and interfaces for server-side TTS providers\n * @module @pie-players/tts-server-core\n */\n\n// Export cache interfaces\nexport type {\n\tCacheKeyComponents,\n\tCacheStats,\n\tITTSCache,\n} from \"./cache.js\";\nexport {\n\tgenerateCacheKey,\n\tgenerateHashedCacheKey,\n\thashText,\n\tMemoryCache,\n} from \"./cache.js\";\n\n// Export provider interfaces\nexport type {\n\tITTSServerProvider,\n\tTTSServerConfig,\n} from \"./provider.js\";\n\nexport { BaseTTSProvider } from \"./provider.js\";\n\n// Export speech marks utilities\nexport {\n\tadjustSpeechMarksForRate,\n\testimateSpeechMarks,\n\tfilterSpeechMarksByType,\n\tgetSpeechMarkAtTime,\n\tgetSpeechMarksStats,\n\tmergeSpeechMarks,\n\tvalidateSpeechMarks,\n} from \"./speech-marks.js\";\n// Export types\nexport type {\n\tGetVoicesOptions,\n\tServerProviderCapabilities,\n\tSpeechMark,\n\tStandardTTSParameters,\n\tSynthesizeMetadata,\n\tSynthesizeRequest,\n\tSynthesizeResponse,\n\tTTSProviderExtensions,\n\tVoice,\n\tVoiceFeatures,\n} from \"./types.js\";\nexport { TTSError, TTSErrorCode } from \"./types.js\";\n"]}
|
package/dist/provider.js.map
DELETED
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"provider.js","sourceRoot":"","sources":["../src/provider.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAoHH;;;GAGG;AACH,MAAM,OAAgB,eAAe;IAK1B,MAAM,GAAoB,EAAE,CAAC;IAC7B,WAAW,GAAG,KAAK,CAAC;IAO9B,KAAK,CAAC,OAAO;QACZ,IAAI,CAAC,WAAW,GAAG,KAAK,CAAC;QACzB,IAAI,CAAC,MAAM,GAAG,EAAE,CAAC;IAClB,CAAC;IAED;;;OAGG;IACO,iBAAiB;QAC1B,IAAI,CAAC,IAAI,CAAC,WAAW,EAAE,CAAC;YACvB,MAAM,IAAI,KAAK,CAAC,YAAY,IAAI,CAAC,UAAU,kBAAkB,CAAC,CAAC;QAChE,CAAC;IACF,CAAC;IAED;;;OAGG;IACO,eAAe,CACxB,OAA0B,EAC1B,YAAwC;QAExC,IAAI,CAAC,OAAO,CAAC,IAAI,IAAI,OAAO,CAAC,IAAI,CAAC,IAAI,EAAE,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;YACvD,MAAM,IAAI,KAAK,CAAC,sCAAsC,CAAC,CAAC;QACzD,CAAC;QAED,IAAI,OAAO,CAAC,IAAI,CAAC,MAAM,GAAG,YAAY,CAAC,QAAQ,CAAC,aAAa,EAAE,CAAC;YAC/D,MAAM,IAAI,KAAK,CACd,gBAAgB,OAAO,CAAC,IAAI,CAAC,MAAM,sBAAsB,YAAY,CAAC,QAAQ,CAAC,aAAa,GAAG,CAC/F,CAAC;QACH,CAAC;QAED,IACC,OAAO,CAAC,MAAM;YACd,CAAC,YAAY,CAAC,UAAU,CAAC,gBAAgB,CAAC,QAAQ,CAAC,OAAO,CAAC,MAAM,CAAC,EACjE,CAAC;YACF,MAAM,IAAI,KAAK,CACd,WAAW,OAAO,CAAC,MAAM,uCAAuC,YAAY,CAAC,UAAU,CAAC,gBAAgB,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CACrH,CAAC;QACH,CAAC;QAED,IACC,OAAO,CAAC,IAAI,KAAK,SAAS;YAC1B,CAAC,OAAO,CAAC,IAAI,GAAG,IAAI,IAAI,OAAO,CAAC,IAAI,GAAG,GAAG,CAAC,EAC1C,CAAC;YACF,MAAM,IAAI,KAAK,CAAC,mCAAmC,CAAC,CAAC;QACtD,CAAC;QAED,IACC,OAAO,CAAC,KAAK,KAAK,SAAS;YAC3B,CAAC,OAAO,CAAC,KAAK,GAAG,CAAC,EAAE,IAAI,OAAO,CAAC,KAAK,GAAG,EAAE,CAAC,EAC1C,CAAC;YACF,MAAM,IAAI,KAAK,CAAC,kCAAkC,CAAC,CAAC;QACrD,CAAC;QAED,IACC,OAAO,CAAC,MAAM,KAAK,SAAS;YAC5B,CAAC,OAAO,CAAC,MAAM,GAAG,CAAC,IAAI,OAAO,CAAC,MAAM,GAAG,CAAC,CAAC,EACzC,CAAC;YACF,MAAM,IAAI,KAAK,CAAC,gCAAgC,CAAC,CAAC;QACnD,CAAC;IACF,CAAC;CACD","sourcesContent":["/**\n * Server-side TTS Provider interface\n * @module @pie-players/tts-server-core\n */\n\nimport type {\n\tGetVoicesOptions,\n\tServerProviderCapabilities,\n\tSynthesizeRequest,\n\tSynthesizeResponse,\n\tVoice,\n} from \"./types.js\";\n\n/**\n * Base configuration for TTS providers\n */\nexport interface TTSServerConfig {\n\t/** Provider-specific configuration */\n\t[key: string]: unknown;\n}\n\n/**\n * Server-side TTS Provider interface\n *\n * All server-side TTS providers must implement this interface.\n * Providers handle synthesis requests and return audio with speech marks.\n *\n * ## Initialization Performance\n *\n * The `initialize()` method MUST be fast and lightweight:\n * - Should only validate config and create API clients\n * - MUST NOT fetch voices or make expensive API calls\n * - MUST NOT perform test synthesis requests\n *\n * Use `getVoices()` explicitly when voice discovery is needed (e.g., in demo/admin UIs).\n * Runtime synthesis should work with hardcoded voice IDs without querying available voices.\n *\n * @example Fast initialization (runtime)\n * ```typescript\n * const provider = new PollyServerProvider();\n * await provider.initialize({ region: 'us-east-1', defaultVoice: 'Joanna' });\n * // Ready to synthesize immediately - no voices query\n * await provider.synthesize({ text: 'Hello', voice: 'Joanna' });\n * ```\n *\n * @example Explicit voice discovery (admin/demo UIs)\n * ```typescript\n * const provider = new PollyServerProvider();\n * await provider.initialize({ region: 'us-east-1' });\n * const voices = await provider.getVoices(); // Explicit, separate call\n * ```\n */\nexport interface ITTSServerProvider {\n\t/**\n\t * Unique provider identifier (e.g., 'aws-polly', 'google-cloud-tts')\n\t */\n\treadonly providerId: string;\n\n\t/**\n\t * Human-readable provider name\n\t */\n\treadonly providerName: string;\n\n\t/**\n\t * Provider version\n\t */\n\treadonly version: string;\n\n\t/**\n\t * Initialize the provider with configuration.\n\t *\n\t * MUST be fast and lightweight - only validates config and creates clients.\n\t * MUST NOT fetch voices or make expensive API calls during initialization.\n\t *\n\t * @param config - Provider-specific configuration\n\t * @throws {TTSError} If initialization fails\n\t * @performance Should complete in <100ms\n\t */\n\tinitialize(config: TTSServerConfig): Promise<void>;\n\n\t/**\n\t * Synthesize speech from text\n\t *\n\t * @param request - Synthesis request parameters\n\t * @returns Audio data and speech marks\n\t * @throws {TTSError} If synthesis fails\n\t */\n\tsynthesize(request: SynthesizeRequest): Promise<SynthesizeResponse>;\n\n\t/**\n\t * Get available voices (explicit, secondary query).\n\t *\n\t * This is an EXPLICIT operation for voice discovery in demo/admin UIs.\n\t * NOT called during initialization - call separately when needed.\n\t *\n\t * @param options - Optional filters for voices\n\t * @returns List of available voices\n\t * @throws {TTSError} If voice listing fails\n\t * @note May take 200-500ms depending on provider\n\t */\n\tgetVoices(options?: GetVoicesOptions): Promise<Voice[]>;\n\n\t/**\n\t * Get provider capabilities (synchronous, fast).\n\t *\n\t * Returns static capability information without API calls.\n\t *\n\t * @returns Provider feature support\n\t * @performance Should complete in <1ms (synchronous)\n\t */\n\tgetCapabilities(): ServerProviderCapabilities;\n\n\t/**\n\t * Clean up provider resources\n\t * Called when provider is no longer needed\n\t */\n\tdestroy(): Promise<void>;\n}\n\n/**\n * Abstract base class for TTS providers\n * Provides common functionality and helpers\n */\nexport abstract class BaseTTSProvider implements ITTSServerProvider {\n\tabstract readonly providerId: string;\n\tabstract readonly providerName: string;\n\tabstract readonly version: string;\n\n\tprotected config: TTSServerConfig = {};\n\tprotected initialized = false;\n\n\tabstract initialize(config: TTSServerConfig): Promise<void>;\n\tabstract synthesize(request: SynthesizeRequest): Promise<SynthesizeResponse>;\n\tabstract getVoices(options?: GetVoicesOptions): Promise<Voice[]>;\n\tabstract getCapabilities(): ServerProviderCapabilities;\n\n\tasync destroy(): Promise<void> {\n\t\tthis.initialized = false;\n\t\tthis.config = {};\n\t}\n\n\t/**\n\t * Ensure provider is initialized before operations\n\t * @throws {TTSError} If provider not initialized\n\t */\n\tprotected ensureInitialized(): void {\n\t\tif (!this.initialized) {\n\t\t\tthrow new Error(`Provider ${this.providerId} not initialized`);\n\t\t}\n\t}\n\n\t/**\n\t * Validate synthesis request\n\t * @throws {TTSError} If request is invalid\n\t */\n\tprotected validateRequest(\n\t\trequest: SynthesizeRequest,\n\t\tcapabilities: ServerProviderCapabilities,\n\t): void {\n\t\tif (!request.text || request.text.trim().length === 0) {\n\t\t\tthrow new Error(\"Text is required and cannot be empty\");\n\t\t}\n\n\t\tif (request.text.length > capabilities.standard.maxTextLength) {\n\t\t\tthrow new Error(\n\t\t\t\t`Text length (${request.text.length}) exceeds maximum (${capabilities.standard.maxTextLength})`,\n\t\t\t);\n\t\t}\n\n\t\tif (\n\t\t\trequest.format &&\n\t\t\t!capabilities.extensions.supportedFormats.includes(request.format)\n\t\t) {\n\t\t\tthrow new Error(\n\t\t\t\t`Format '${request.format}' not supported. Supported formats: ${capabilities.extensions.supportedFormats.join(\", \")}`,\n\t\t\t);\n\t\t}\n\n\t\tif (\n\t\t\trequest.rate !== undefined &&\n\t\t\t(request.rate < 0.25 || request.rate > 4.0)\n\t\t) {\n\t\t\tthrow new Error(\"Rate must be between 0.25 and 4.0\");\n\t\t}\n\n\t\tif (\n\t\t\trequest.pitch !== undefined &&\n\t\t\t(request.pitch < -20 || request.pitch > 20)\n\t\t) {\n\t\t\tthrow new Error(\"Pitch must be between -20 and 20\");\n\t\t}\n\n\t\tif (\n\t\t\trequest.volume !== undefined &&\n\t\t\t(request.volume < 0 || request.volume > 1)\n\t\t) {\n\t\t\tthrow new Error(\"Volume must be between 0 and 1\");\n\t\t}\n\t}\n}\n"]}
|
package/dist/speech-marks.js.map
DELETED
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"speech-marks.js","sourceRoot":"","sources":["../src/speech-marks.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAIH;;;;;;;;;GASG;AACH,MAAM,UAAU,mBAAmB,CAClC,IAAY,EACZ,iBAAiB,GAAG,GAAG;IAEvB,MAAM,KAAK,GAAG,IAAI,CAAC,KAAK,CAAC,KAAK,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC;IAC5D,MAAM,SAAS,GAAG,CAAC,EAAE,GAAG,IAAI,CAAC,GAAG,iBAAiB,CAAC;IAElD,MAAM,KAAK,GAAiB,EAAE,CAAC;IAC/B,IAAI,SAAS,GAAG,CAAC,CAAC;IAElB,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,KAAK,CAAC,MAAM,EAAE,CAAC,EAAE,EAAE,CAAC;QACvC,MAAM,IAAI,GAAG,KAAK,CAAC,CAAC,CAAC,CAAC;QAEtB,0DAA0D;QAC1D,MAAM,SAAS,GAAG,IAAI,CAAC,OAAO,CAAC,IAAI,EAAE,SAAS,CAAC,CAAC;QAChD,IAAI,SAAS,KAAK,CAAC,CAAC,EAAE,CAAC;YACtB,0CAA0C;YAC1C,SAAS,IAAI,IAAI,CAAC,MAAM,GAAG,CAAC,CAAC;YAC7B,SAAS;QACV,CAAC;QAED,KAAK,CAAC,IAAI,CAAC;YACV,IAAI,EAAE,IAAI,CAAC,KAAK,CAAC,CAAC,GAAG,SAAS,CAAC;YAC/B,IAAI,EAAE,MAAM;YACZ,KAAK,EAAE,SAAS;YAChB,GAAG,EAAE,SAAS,GAAG,IAAI,CAAC,MAAM;YAC5B,KAAK,EAAE,IAAI;SACX,CAAC,CAAC;QAEH,SAAS,GAAG,SAAS,GAAG,IAAI,CAAC,MAAM,CAAC;IACrC,CAAC;IAED,OAAO,KAAK,CAAC;AACd,CAAC;AAED;;;;;;GAMG;AACH,MAAM,UAAU,wBAAwB,CACvC,KAAmB,EACnB,IAAY;IAEZ,IAAI,IAAI,KAAK,GAAG,EAAE,CAAC;QAClB,OAAO,KAAK,CAAC,CAAC,uBAAuB;IACtC,CAAC;IAED,OAAO,KAAK,CAAC,GAAG,CAAC,CAAC,IAAI,EAAE,EAAE,CAAC,CAAC;QAC3B,GAAG,IAAI;QACP,IAAI,EAAE,IAAI,CAAC,KAAK,CAAC,IAAI,CAAC,IAAI,GAAG,IAAI,CAAC;KAClC,CAAC,CAAC,CAAC;AACL,CAAC;AAED;;;;;;GAMG;AACH,MAAM,UAAU,mBAAmB,CAAC,KAAmB;IACtD,MAAM,MAAM,GAAa,EAAE,CAAC;IAE5B,IAAI,CAAC,KAAK,IAAI,KAAK,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QAClC,OAAO,MAAM,CAAC,CAAC,iBAAiB;IACjC,CAAC;IAED,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,KAAK,CAAC,MAAM,EAAE,CAAC,EAAE,EAAE,CAAC;QACvC,MAAM,IAAI,GAAG,KAAK,CAAC,CAAC,CAAC,CAAC;QAEtB,wBAAwB;QACxB,IAAI,OAAO,IAAI,CAAC,IAAI,KAAK,QAAQ,IAAI,IAAI,CAAC,IAAI,GAAG,CAAC,EAAE,CAAC;YACpD,MAAM,CAAC,IAAI,CAAC,QAAQ,CAAC,mBAAmB,IAAI,CAAC,IAAI,GAAG,CAAC,CAAC;QACvD,CAAC;QAED,IAAI,OAAO,IAAI,CAAC,KAAK,KAAK,QAAQ,IAAI,IAAI,CAAC,KAAK,GAAG,CAAC,EAAE,CAAC;YACtD,MAAM,CAAC,IAAI,CAAC,QAAQ,CAAC,oBAAoB,IAAI,CAAC,KAAK,GAAG,CAAC,CAAC;QACzD,CAAC;QAED,IAAI,OAAO,IAAI,CAAC,GAAG,KAAK,QAAQ,IAAI,IAAI,CAAC,GAAG,IAAI,IAAI,CAAC,KAAK,EAAE,CAAC;YAC5D,MAAM,CAAC,IAAI,CAAC,QAAQ,CAAC,kBAAkB,IAAI,CAAC,GAAG,YAAY,IAAI,CAAC,KAAK,GAAG,CAAC,CAAC;QAC3E,CAAC;QAED,IAAI,CAAC,IAAI,CAAC,KAAK,IAAI,OAAO,IAAI,CAAC,KAAK,KAAK,QAAQ,EAAE,CAAC;YACnD,MAAM,CAAC,IAAI,CAAC,QAAQ,CAAC,iBAAiB,CAAC,CAAC;QACzC,CAAC;QAED,2DAA2D;QAC3D,IAAI,CAAC,GAAG,CAAC,IAAI,IAAI,CAAC,IAAI,GAAG,KAAK,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,IAAI,EAAE,CAAC;YAC5C,MAAM,CAAC,IAAI,CACV,QAAQ,CAAC,WAAW,IAAI,CAAC,IAAI,iCAAiC,KAAK,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,IAAI,GAAG,CAClF,CAAC;QACH,CAAC;IACF,CAAC;IAED,OAAO,MAAM,CAAC;AACf,CAAC;AAED;;;;;;GAMG;AACH,MAAM,UAAU,gBAAgB,CAAC,KAAmB;IACnD,IAAI,KAAK,CAAC,MAAM,IAAI,CAAC,EAAE,CAAC;QACvB,OAAO,KAAK,CAAC;IACd,CAAC;IAED,yBAAyB;IACzB,MAAM,MAAM,GAAG,CAAC,GAAG,KAAK,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,KAAK,GAAG,CAAC,CAAC,KAAK,CAAC,CAAC;IAC5D,MAAM,MAAM,GAAiB,CAAC,MAAM,CAAC,CAAC,CAAC,CAAC,CAAC;IAEzC,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,MAAM,CAAC,MAAM,EAAE,CAAC,EAAE,EAAE,CAAC;QACxC,MAAM,OAAO,GAAG,MAAM,CAAC,CAAC,CAAC,CAAC;QAC1B,MAAM,QAAQ,GAAG,MAAM,CAAC,MAAM,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC;QAE3C,yCAAyC;QACzC,IAAI,OAAO,CAAC,KAAK,IAAI,QAAQ,CAAC,GAAG,EAAE,CAAC;YACnC,2BAA2B;YAC3B,QAAQ,CAAC,GAAG,GAAG,IAAI,CAAC,GAAG,CAAC,QAAQ,CAAC,GAAG,EAAE,OAAO,CAAC,GAAG,CAAC,CAAC;YACnD,QAAQ,CAAC,KAAK,GAAG,QAAQ,CAAC,KAAK,GAAG,GAAG,GAAG,OAAO,CAAC,KAAK,CAAC;YACtD,QAAQ,CAAC,IAAI,GAAG,IAAI,CAAC,GAAG,CAAC,QAAQ,CAAC,IAAI,EAAE,OAAO,CAAC,IAAI,CAAC,CAAC,CAAC,mBAAmB;QAC3E,CAAC;aAAM,CAAC;YACP,8BAA8B;YAC9B,MAAM,CAAC,IAAI,CAAC,OAAO,CAAC,CAAC;QACtB,CAAC;IACF,CAAC;IAED,OAAO,MAAM,CAAC;AACf,CAAC;AAED;;;;;;GAMG;AACH,MAAM,UAAU,uBAAuB,CACtC,KAAmB,EACnB,IAAkC;IAElC,OAAO,KAAK,CAAC,MAAM,CAAC,CAAC,IAAI,EAAE,EAAE,CAAC,IAAI,CAAC,IAAI,KAAK,IAAI,CAAC,CAAC;AACnD,CAAC;AAED;;;;;;GAMG;AACH,MAAM,UAAU,mBAAmB,CAClC,KAAmB,EACnB,IAAY;IAEZ,+BAA+B;IAC/B,IAAI,IAAI,GAAG,CAAC,CAAC;IACb,IAAI,KAAK,GAAG,KAAK,CAAC,MAAM,GAAG,CAAC,CAAC;IAC7B,IAAI,OAAO,GAAsB,IAAI,CAAC;IAEtC,OAAO,IAAI,IAAI,KAAK,EAAE,CAAC;QACtB,MAAM,GAAG,GAAG,IAAI,CAAC,KAAK,CAAC,CAAC,IAAI,GAAG,KAAK,CAAC,GAAG,CAAC,CAAC,CAAC;QAC3C,MAAM,IAAI,GAAG,KAAK,CAAC,GAAG,CAAC,CAAC;QAExB,IAAI,IAAI,CAAC,IAAI,KAAK,IAAI,EAAE,CAAC;YACxB,OAAO,IAAI,CAAC;QACb,CAAC;QAED,qBAAqB;QACrB,IACC,CAAC,OAAO;YACR,IAAI,CAAC,GAAG,CAAC,IAAI,CAAC,IAAI,GAAG,IAAI,CAAC,GAAG,IAAI,CAAC,GAAG,CAAC,OAAO,CAAC,IAAI,GAAG,IAAI,CAAC,EACzD,CAAC;YACF,OAAO,GAAG,IAAI,CAAC;QAChB,CAAC;QAED,IAAI,IAAI,CAAC,IAAI,GAAG,IAAI,EAAE,CAAC;YACtB,IAAI,GAAG,GAAG,GAAG,CAAC,CAAC;QAChB,CAAC;aAAM,CAAC;YACP,KAAK,GAAG,GAAG,GAAG,CAAC,CAAC;QACjB,CAAC;IACF,CAAC;IAED,6DAA6D;IAC7D,IAAI,OAAO,IAAI,IAAI,CAAC,GAAG,CAAC,OAAO,CAAC,IAAI,GAAG,IAAI,CAAC,IAAI,GAAG,EAAE,CAAC;QACrD,OAAO,OAAO,CAAC;IAChB,CAAC;IAED,OAAO,IAAI,CAAC;AACb,CAAC;AAED;;;;;GAKG;AACH,MAAM,UAAU,mBAAmB,CAAC,KAAmB;IACtD,IAAI,KAAK,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QACxB,OAAO;YACN,KAAK,EAAE,CAAC;YACR,aAAa,EAAE,CAAC;YAChB,eAAe,EAAE,CAAC;YAClB,cAAc,EAAE,CAAC;SACjB,CAAC;IACH,CAAC;IAED,MAAM,SAAS,GAAG,uBAAuB,CAAC,KAAK,EAAE,MAAM,CAAC,CAAC;IACzD,MAAM,aAAa,GAAG,KAAK,CAAC,KAAK,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC,IAAI,CAAC;IACnD,MAAM,eAAe,GAAG,aAAa,GAAG,SAAS,CAAC,MAAM,CAAC;IACzD,MAAM,cAAc,GAAG,CAAC,SAAS,CAAC,MAAM,GAAG,aAAa,CAAC,GAAG,EAAE,GAAG,IAAI,CAAC;IAEtE,OAAO;QACN,KAAK,EAAE,KAAK,CAAC,MAAM;QACnB,SAAS,EAAE,SAAS,CAAC,MAAM;QAC3B,aAAa;QACb,eAAe;QACf,cAAc;KACd,CAAC;AACH,CAAC","sourcesContent":["/**\n * Speech marks utilities\n * @module @pie-players/tts-server-core\n */\n\nimport type { SpeechMark } from \"./types.js\";\n\n/**\n * Estimate speech marks for text when provider doesn't support them\n *\n * Uses average speaking rate to estimate word timing.\n * Not as accurate as provider-generated marks, but better than nothing.\n *\n * @param text - Text to generate marks for\n * @param avgWordsPerMinute - Average speaking rate (default 150)\n * @returns Estimated speech marks\n */\nexport function estimateSpeechMarks(\n\ttext: string,\n\tavgWordsPerMinute = 150,\n): SpeechMark[] {\n\tconst words = text.split(/\\s+/).filter((w) => w.length > 0);\n\tconst msPerWord = (60 * 1000) / avgWordsPerMinute;\n\n\tconst marks: SpeechMark[] = [];\n\tlet charIndex = 0;\n\n\tfor (let i = 0; i < words.length; i++) {\n\t\tconst word = words[i];\n\n\t\t// Find word position in original text (preserves spacing)\n\t\tconst wordStart = text.indexOf(word, charIndex);\n\t\tif (wordStart === -1) {\n\t\t\t// Word not found (shouldn't happen), skip\n\t\t\tcharIndex += word.length + 1;\n\t\t\tcontinue;\n\t\t}\n\n\t\tmarks.push({\n\t\t\ttime: Math.round(i * msPerWord),\n\t\t\ttype: \"word\",\n\t\t\tstart: wordStart,\n\t\t\tend: wordStart + word.length,\n\t\t\tvalue: word,\n\t\t});\n\n\t\tcharIndex = wordStart + word.length;\n\t}\n\n\treturn marks;\n}\n\n/**\n * Adjust speech marks timing for different speaking rates\n *\n * @param marks - Original speech marks\n * @param rate - Speech rate multiplier (0.25 to 4.0)\n * @returns Adjusted speech marks\n */\nexport function adjustSpeechMarksForRate(\n\tmarks: SpeechMark[],\n\trate: number,\n): SpeechMark[] {\n\tif (rate === 1.0) {\n\t\treturn marks; // No adjustment needed\n\t}\n\n\treturn marks.map((mark) => ({\n\t\t...mark,\n\t\ttime: Math.round(mark.time / rate),\n\t}));\n}\n\n/**\n * Validate speech marks\n * Ensures marks are properly ordered and have valid data\n *\n * @param marks - Speech marks to validate\n * @returns Validation errors (empty array if valid)\n */\nexport function validateSpeechMarks(marks: SpeechMark[]): string[] {\n\tconst errors: string[] = [];\n\n\tif (!marks || marks.length === 0) {\n\t\treturn errors; // Empty is valid\n\t}\n\n\tfor (let i = 0; i < marks.length; i++) {\n\t\tconst mark = marks[i];\n\n\t\t// Check required fields\n\t\tif (typeof mark.time !== \"number\" || mark.time < 0) {\n\t\t\terrors.push(`Mark ${i}: invalid time (${mark.time})`);\n\t\t}\n\n\t\tif (typeof mark.start !== \"number\" || mark.start < 0) {\n\t\t\terrors.push(`Mark ${i}: invalid start (${mark.start})`);\n\t\t}\n\n\t\tif (typeof mark.end !== \"number\" || mark.end <= mark.start) {\n\t\t\terrors.push(`Mark ${i}: invalid end (${mark.end}, start: ${mark.start})`);\n\t\t}\n\n\t\tif (!mark.value || typeof mark.value !== \"string\") {\n\t\t\terrors.push(`Mark ${i}: invalid value`);\n\t\t}\n\n\t\t// Check ordering (time should be monotonically increasing)\n\t\tif (i > 0 && mark.time < marks[i - 1].time) {\n\t\t\terrors.push(\n\t\t\t\t`Mark ${i}: time (${mark.time}) is less than previous mark (${marks[i - 1].time})`,\n\t\t\t);\n\t\t}\n\t}\n\n\treturn errors;\n}\n\n/**\n * Merge overlapping or adjacent speech marks\n * Useful when combining marks from multiple sources\n *\n * @param marks - Speech marks to merge\n * @returns Merged speech marks\n */\nexport function mergeSpeechMarks(marks: SpeechMark[]): SpeechMark[] {\n\tif (marks.length <= 1) {\n\t\treturn marks;\n\t}\n\n\t// Sort by start position\n\tconst sorted = [...marks].sort((a, b) => a.start - b.start);\n\tconst merged: SpeechMark[] = [sorted[0]];\n\n\tfor (let i = 1; i < sorted.length; i++) {\n\t\tconst current = sorted[i];\n\t\tconst previous = merged[merged.length - 1];\n\n\t\t// Check if marks overlap or are adjacent\n\t\tif (current.start <= previous.end) {\n\t\t\t// Merge with previous mark\n\t\t\tprevious.end = Math.max(previous.end, current.end);\n\t\t\tprevious.value = previous.value + \" \" + current.value;\n\t\t\tprevious.time = Math.min(previous.time, current.time); // Use earlier time\n\t\t} else {\n\t\t\t// No overlap, add as new mark\n\t\t\tmerged.push(current);\n\t\t}\n\t}\n\n\treturn merged;\n}\n\n/**\n * Filter speech marks by type\n *\n * @param marks - Speech marks to filter\n * @param type - Type to filter by\n * @returns Filtered speech marks\n */\nexport function filterSpeechMarksByType(\n\tmarks: SpeechMark[],\n\ttype: \"word\" | \"sentence\" | \"ssml\",\n): SpeechMark[] {\n\treturn marks.filter((mark) => mark.type === type);\n}\n\n/**\n * Get speech mark at specific time\n *\n * @param marks - Speech marks\n * @param time - Time in milliseconds\n * @returns Speech mark at time, or null if none found\n */\nexport function getSpeechMarkAtTime(\n\tmarks: SpeechMark[],\n\ttime: number,\n): SpeechMark | null {\n\t// Binary search for efficiency\n\tlet left = 0;\n\tlet right = marks.length - 1;\n\tlet closest: SpeechMark | null = null;\n\n\twhile (left <= right) {\n\t\tconst mid = Math.floor((left + right) / 2);\n\t\tconst mark = marks[mid];\n\n\t\tif (mark.time === time) {\n\t\t\treturn mark;\n\t\t}\n\n\t\t// Track closest mark\n\t\tif (\n\t\t\t!closest ||\n\t\t\tMath.abs(mark.time - time) < Math.abs(closest.time - time)\n\t\t) {\n\t\t\tclosest = mark;\n\t\t}\n\n\t\tif (mark.time < time) {\n\t\t\tleft = mid + 1;\n\t\t} else {\n\t\t\tright = mid - 1;\n\t\t}\n\t}\n\n\t// Return closest mark if within reasonable threshold (500ms)\n\tif (closest && Math.abs(closest.time - time) <= 500) {\n\t\treturn closest;\n\t}\n\n\treturn null;\n}\n\n/**\n * Calculate statistics for speech marks\n *\n * @param marks - Speech marks\n * @returns Statistics about the marks\n */\nexport function getSpeechMarksStats(marks: SpeechMark[]) {\n\tif (marks.length === 0) {\n\t\treturn {\n\t\t\tcount: 0,\n\t\t\ttotalDuration: 0,\n\t\t\tavgWordDuration: 0,\n\t\t\twordsPerMinute: 0,\n\t\t};\n\t}\n\n\tconst wordMarks = filterSpeechMarksByType(marks, \"word\");\n\tconst totalDuration = marks[marks.length - 1].time;\n\tconst avgWordDuration = totalDuration / wordMarks.length;\n\tconst wordsPerMinute = (wordMarks.length / totalDuration) * 60 * 1000;\n\n\treturn {\n\t\tcount: marks.length,\n\t\twordCount: wordMarks.length,\n\t\ttotalDuration,\n\t\tavgWordDuration,\n\t\twordsPerMinute,\n\t};\n}\n"]}
|
package/dist/types.js.map
DELETED
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"types.js","sourceRoot":"","sources":["../src/types.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAyXH;;GAEG;AACH,MAAM,CAAN,IAAY,YAUX;AAVD,WAAY,YAAY;IACvB,mDAAmC,CAAA;IACnC,+CAA+B,CAAA;IAC/B,qDAAqC,CAAA;IACrC,+CAA+B,CAAA;IAC/B,iDAAiC,CAAA;IACjC,+CAA+B,CAAA;IAC/B,6DAA6C,CAAA;IAC7C,2DAA2C,CAAA;IAC3C,6DAA6C,CAAA;AAC9C,CAAC,EAVW,YAAY,KAAZ,YAAY,QAUvB;AAED;;GAEG;AACH,MAAM,OAAO,QAAS,SAAQ,KAAK;IAE1B;IAEA;IACA;IAJR,YACQ,IAAkB,EACzB,OAAe,EACR,OAAiC,EACjC,UAAmB;QAE1B,KAAK,CAAC,OAAO,CAAC,CAAC;QALR,SAAI,GAAJ,IAAI,CAAc;QAElB,YAAO,GAAP,OAAO,CAA0B;QACjC,eAAU,GAAV,UAAU,CAAS;QAG1B,IAAI,CAAC,IAAI,GAAG,UAAU,CAAC;QAEvB,oEAAoE;QACpE,IAAI,KAAK,CAAC,iBAAiB,EAAE,CAAC;YAC7B,KAAK,CAAC,iBAAiB,CAAC,IAAI,EAAE,QAAQ,CAAC,CAAC;QACzC,CAAC;IACF,CAAC;IAED,MAAM;QACL,OAAO;YACN,KAAK,EAAE;gBACN,IAAI,EAAE,IAAI,CAAC,IAAI;gBACf,OAAO,EAAE,IAAI,CAAC,OAAO;gBACrB,OAAO,EAAE,IAAI,CAAC,OAAO;gBACrB,QAAQ,EAAE,IAAI,CAAC,UAAU;aACzB;SACD,CAAC;IACH,CAAC;CACD","sourcesContent":["/**\n * Core types for server-side TTS providers\n * @module @pie-players/tts-server-core\n */\n\n/**\n * Speech mark representing a timing event in synthesized speech\n * Unified format across all TTS providers\n */\nexport interface SpeechMark {\n\t/** Milliseconds from start of audio */\n\ttime: number;\n\n\t/** Type of speech mark */\n\ttype: \"word\" | \"sentence\" | \"ssml\";\n\n\t/** Character index in original text (inclusive) */\n\tstart: number;\n\n\t/** Character index in original text (exclusive) */\n\tend: number;\n\n\t/** The actual word or text */\n\tvalue: string;\n}\n\n/**\n * Standard TTS parameters based on W3C Web Speech API and SSML specifications.\n *\n * These parameters are widely supported across TTS providers (browsers, cloud services)\n * and follow established standards:\n * - W3C Web Speech API (SpeechSynthesisUtterance)\n * - W3C SSML 1.1 specification\n * - BCP47 language tags (RFC 5646)\n *\n * @see https://w3c.github.io/speech-api/\n * @see https://www.w3.org/TR/speech-synthesis/\n */\nexport interface StandardTTSParameters {\n\t/**\n\t * Text to synthesize (plain text or SSML markup)\n\t *\n\t * @standard W3C Web Speech API\n\t */\n\ttext: string;\n\n\t/**\n\t * Voice identifier (provider-specific voice names)\n\t * Examples: \"Joanna\" (Polly), \"en-US-Standard-A\" (Google), browser voice names\n\t *\n\t * @standard W3C Web Speech API (concept)\n\t * @note Voice names are provider-specific but the concept is standard\n\t */\n\tvoice?: string;\n\n\t/**\n\t * Language code using BCP47 format (e.g., 'en-US', 'es-ES', 'fr-FR')\n\t *\n\t * @standard BCP47 (RFC 5646), W3C Web Speech API\n\t * @see https://tools.ietf.org/html/rfc5646\n\t */\n\tlanguage?: string;\n\n\t/**\n\t * Speech rate (speed multiplier)\n\t * - Range: 0.25 to 4.0\n\t * - Default: 1.0 (normal speed)\n\t * - 0.5 = half speed, 2.0 = double speed\n\t *\n\t * @standard W3C Web Speech API, SSML <prosody rate>\n\t */\n\trate?: number;\n\n\t/**\n\t * Pitch adjustment\n\t * - Range: -20 to +20 semitones (or 0 to 2 as multiplier depending on provider)\n\t * - Default: 0 (or 1.0 as multiplier)\n\t * - Negative values = lower pitch, positive = higher pitch\n\t *\n\t * @standard W3C Web Speech API, SSML <prosody pitch>\n\t * @note Some providers use semitones (-20 to +20), others use multipliers (0 to 2)\n\t */\n\tpitch?: number;\n\n\t/**\n\t * Volume level\n\t * - Range: 0.0 to 1.0\n\t * - Default: 1.0 (full volume)\n\t * - 0.0 = silent, 0.5 = half volume\n\t *\n\t * @standard W3C Web Speech API, SSML <prosody volume>\n\t */\n\tvolume?: number;\n}\n\n/**\n * Provider-specific extensions for advanced TTS control.\n *\n * These parameters are NOT part of W3C standards and have varying support\n * across providers. Use with caution for portability.\n *\n * Common extensions include:\n * - Audio format selection (mp3, wav, ogg)\n * - Sample rate control\n * - Engine selection (neural vs standard)\n * - Regional endpoints\n * - Speech marks / word timing\n *\n * @note Providers may ignore unsupported extensions silently or throw errors\n */\nexport interface TTSProviderExtensions {\n\t/**\n\t * Audio format for output\n\t *\n\t * @extension Common across providers but values vary\n\t * @support AWS Polly (mp3, ogg, pcm), Google Cloud TTS (mp3, wav, ogg), Azure (mp3, wav, ogg)\n\t */\n\tformat?: \"mp3\" | \"wav\" | \"ogg\" | \"pcm\";\n\n\t/**\n\t * Sample rate in Hz (e.g., 8000, 16000, 22050, 24000)\n\t *\n\t * @extension Common audio parameter\n\t * @note Higher sample rates = better quality but larger file sizes\n\t */\n\tsampleRate?: number;\n\n\t/**\n\t * Request word-level timing data (speech marks)\n\t *\n\t * @extension Provider-specific but common pattern\n\t * @support AWS Polly (SpeechMarks), Google Cloud TTS (timepoints), Azure (word boundaries)\n\t * @default true\n\t */\n\tincludeSpeechMarks?: boolean;\n\n\t/**\n\t * Provider-specific options (extensibility point)\n\t *\n\t * Examples:\n\t * - AWS Polly: { engine: 'neural' | 'standard', lexiconNames: string[] }\n\t * - Google Cloud TTS: { audioEncoding: string, effectsProfileId: string[] }\n\t * - Azure: { voiceType: string, stylesList: string[] }\n\t *\n\t * @extension Arbitrary provider-specific data\n\t */\n\tproviderOptions?: Record<string, unknown>;\n}\n\n/**\n * Complete synthesis request combining standard parameters and extensions.\n *\n * This interface provides the full set of options for text-to-speech synthesis,\n * clearly separating W3C-standard parameters from provider-specific extensions.\n *\n * @example Basic usage (portable across providers)\n * ```typescript\n * const request: SynthesizeRequest = {\n * text: \"Hello world\",\n * voice: \"Joanna\",\n * rate: 1.0,\n * language: \"en-US\"\n * };\n * ```\n *\n * @example Advanced usage with extensions (provider-specific)\n * ```typescript\n * const request: SynthesizeRequest = {\n * text: \"Hello world\",\n * voice: \"Joanna\",\n * rate: 1.0,\n * // Extensions - may not be portable\n * format: 'mp3',\n * sampleRate: 24000,\n * includeSpeechMarks: true,\n * providerOptions: {\n * engine: 'neural' // AWS Polly specific\n * }\n * };\n * ```\n */\nexport interface SynthesizeRequest\n\textends StandardTTSParameters,\n\t\tTTSProviderExtensions {}\n\n/**\n * Response from speech synthesis\n */\nexport interface SynthesizeResponse {\n\t/** Audio data (Buffer for server, base64 string for client) */\n\taudio: Buffer | string;\n\n\t/** MIME type of audio (e.g., 'audio/mpeg') */\n\tcontentType: string;\n\n\t/** Speech marks for word-level timing */\n\tspeechMarks: SpeechMark[];\n\n\t/** Metadata about the synthesis */\n\tmetadata: SynthesizeMetadata;\n}\n\n/**\n * Metadata about synthesized speech\n */\nexport interface SynthesizeMetadata {\n\t/** Provider that generated the audio */\n\tproviderId: string;\n\n\t/** Voice ID used */\n\tvoice: string;\n\n\t/** Audio duration in seconds */\n\tduration: number;\n\n\t/** Character count of input text */\n\tcharCount: number;\n\n\t/** Whether response was served from cache */\n\tcached: boolean;\n\n\t/** ISO timestamp of synthesis */\n\ttimestamp?: string;\n}\n\n/**\n * Voice definition\n */\nexport interface Voice {\n\t/** Unique voice identifier */\n\tid: string;\n\n\t/** Human-readable name */\n\tname: string;\n\n\t/** Language name (e.g., \"English\", \"Spanish\") */\n\tlanguage: string;\n\n\t/** Language code (e.g., \"en-US\", \"es-ES\") */\n\tlanguageCode: string;\n\n\t/** Gender of voice */\n\tgender?: \"male\" | \"female\" | \"neutral\";\n\n\t/** Voice quality level */\n\tquality: \"standard\" | \"premium\" | \"neural\";\n\n\t/** Supported features */\n\tsupportedFeatures: VoiceFeatures;\n\n\t/** Provider-specific metadata */\n\tproviderMetadata?: Record<string, unknown>;\n}\n\n/**\n * Voice feature flags\n */\nexport interface VoiceFeatures {\n\t/** Supports SSML markup */\n\tssml: boolean;\n\n\t/** Supports emotional expression */\n\temotions: boolean;\n\n\t/** Supports speaking styles */\n\tstyles: boolean;\n}\n\n/**\n * Options for listing voices\n */\nexport interface GetVoicesOptions {\n\t/** Filter by language code */\n\tlanguage?: string;\n\n\t/** Filter by quality level */\n\tquality?: \"standard\" | \"premium\" | \"neural\";\n\n\t/** Filter by gender */\n\tgender?: \"male\" | \"female\" | \"neutral\";\n}\n\n/**\n * Provider capabilities split into standard features and extensions.\n *\n * This interface helps consumers understand what features are universally\n * supported (W3C standards) vs provider-specific extensions.\n */\nexport interface ServerProviderCapabilities {\n\t/**\n\t * Standard W3C features that should be widely supported\n\t */\n\tstandard: {\n\t\t/**\n\t\t * Supports SSML markup (W3C SSML 1.1)\n\t\t *\n\t\t * @standard W3C SSML 1.1\n\t\t * @support Most cloud TTS providers, limited browser support\n\t\t */\n\t\tsupportsSSML: boolean;\n\n\t\t/**\n\t\t * Supports pitch control via rate parameter or SSML <prosody>\n\t\t *\n\t\t * @standard W3C Web Speech API, SSML <prosody pitch>\n\t\t * @note May be via API parameter or SSML only\n\t\t */\n\t\tsupportsPitch: boolean;\n\n\t\t/**\n\t\t * Supports rate (speed) control via rate parameter or SSML <prosody>\n\t\t *\n\t\t * @standard W3C Web Speech API, SSML <prosody rate>\n\t\t */\n\t\tsupportsRate: boolean;\n\n\t\t/**\n\t\t * Supports volume control via volume parameter or SSML <prosody>\n\t\t *\n\t\t * @standard W3C Web Speech API, SSML <prosody volume>\n\t\t * @note Often better handled client-side for server TTS\n\t\t */\n\t\tsupportsVolume: boolean;\n\n\t\t/**\n\t\t * Supports multiple voices (voice selection)\n\t\t *\n\t\t * @standard W3C Web Speech API (concept)\n\t\t */\n\t\tsupportsMultipleVoices: boolean;\n\n\t\t/**\n\t\t * Maximum text length in characters\n\t\t *\n\t\t * @note Varies by provider: Polly=3000, Google=5000, browser=~32k\n\t\t */\n\t\tmaxTextLength: number;\n\t};\n\n\t/**\n\t * Provider-specific extensions\n\t */\n\textensions: {\n\t\t/**\n\t\t * Supports word-level timing data (speech marks)\n\t\t *\n\t\t * @extension Provider-specific but common\n\t\t * @support AWS Polly ✅, Google Cloud TTS ✅, Azure TTS ✅, Browser ⚠️\n\t\t * @note Format and precision vary by provider\n\t\t */\n\t\tsupportsSpeechMarks: boolean;\n\n\t\t/**\n\t\t * Supported audio output formats\n\t\t *\n\t\t * @extension Common but not standardized\n\t\t */\n\t\tsupportedFormats: (\"mp3\" | \"wav\" | \"ogg\" | \"pcm\")[];\n\n\t\t/**\n\t\t * Supports sample rate configuration\n\t\t *\n\t\t * @extension Common audio parameter\n\t\t */\n\t\tsupportsSampleRate: boolean;\n\n\t\t/**\n\t\t * Provider-specific features (extensibility point)\n\t\t *\n\t\t * Examples:\n\t\t * - AWS Polly: { engines: ['neural', 'standard'], lexicons: true }\n\t\t * - Google Cloud TTS: { audioProfiles: true, voiceEffects: true }\n\t\t * - Azure: { styles: true, emotions: true }\n\t\t *\n\t\t * @extension Arbitrary provider capabilities\n\t\t */\n\t\tproviderSpecific?: Record<string, unknown>;\n\t};\n}\n\n/**\n * TTS error codes\n */\nexport enum TTSErrorCode {\n\tINVALID_REQUEST = \"INVALID_REQUEST\",\n\tINVALID_VOICE = \"INVALID_VOICE\",\n\tINVALID_PROVIDER = \"INVALID_PROVIDER\",\n\tTEXT_TOO_LONG = \"TEXT_TOO_LONG\",\n\tPROVIDER_ERROR = \"PROVIDER_ERROR\",\n\tNETWORK_ERROR = \"NETWORK_ERROR\",\n\tAUTHENTICATION_ERROR = \"AUTHENTICATION_ERROR\",\n\tRATE_LIMIT_EXCEEDED = \"RATE_LIMIT_EXCEEDED\",\n\tINITIALIZATION_ERROR = \"INITIALIZATION_ERROR\",\n}\n\n/**\n * TTS error with structured information\n */\nexport class TTSError extends Error {\n\tconstructor(\n\t\tpublic code: TTSErrorCode,\n\t\tmessage: string,\n\t\tpublic details?: Record<string, unknown>,\n\t\tpublic providerId?: string,\n\t) {\n\t\tsuper(message);\n\t\tthis.name = \"TTSError\";\n\n\t\t// Maintains proper stack trace for where error was thrown (V8 only)\n\t\tif (Error.captureStackTrace) {\n\t\t\tError.captureStackTrace(this, TTSError);\n\t\t}\n\t}\n\n\ttoJSON() {\n\t\treturn {\n\t\t\terror: {\n\t\t\t\tcode: this.code,\n\t\t\t\tmessage: this.message,\n\t\t\t\tdetails: this.details,\n\t\t\t\tprovider: this.providerId,\n\t\t\t},\n\t\t};\n\t}\n}\n"]}
|