@ossy/media-tasks 3.17.0 → 3.19.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,5 @@
1
+ export const metadata = {
2
+ id: '@ossy/media-tasks/actions/text-to-voice',
3
+ access: 'workspace',
4
+ label: 'automation.task.text-to-voice.title',
5
+ }
@@ -0,0 +1,160 @@
1
+ /**
2
+ * Text-to-voice via OpenRouter TTS (#898).
3
+ *
4
+ * Provider: shared `openrouter` integration (`OPENROUTER_API_KEY`) — OpenRouter
5
+ * exposes OpenAI-compatible `/audio/speech` (see
6
+ * https://openrouter.ai/docs/guides/overview/multimodal/tts). Default model is
7
+ * `mistralai/voxtral-mini-tts-2603` (listed in OpenRouter speech catalog).
8
+ * Output is MP3 (`audio/mpeg`).
9
+ */
10
+
11
+ import { MARKDOWN_DOCUMENT_TYPE } from './run-prompt.js'
12
+
13
+ /** @see https://openrouter.ai/docs/guides/overview/multimodal/tts */
14
+ export const DEFAULT_TEXT_TO_VOICE_MODEL = 'mistralai/voxtral-mini-tts-2603'
15
+ export const DEFAULT_TEXT_TO_VOICE_VOICE = 'en_paul_neutral'
16
+ export const DEFAULT_TEXT_TO_VOICE_FORMAT = 'mp3'
17
+ export const TEXT_TO_VOICE_CONTENT_TYPE = 'audio/mpeg'
18
+
19
+ /** Soft character ceiling for spoken input. */
20
+ export const MAX_TEXT_TO_VOICE_CHARS = 4096
21
+
22
+ /** Legacy direct-OpenAI short model names → OpenRouter default. */
23
+ const LEGACY_MODEL_ALIASES = Object.freeze({
24
+ 'tts-1': DEFAULT_TEXT_TO_VOICE_MODEL,
25
+ 'tts-1-hd': DEFAULT_TEXT_TO_VOICE_MODEL,
26
+ })
27
+
28
+ /** Legacy OpenAI voice names → Voxtral defaults (provider voices differ). */
29
+ const LEGACY_VOICE_ALIASES = Object.freeze({
30
+ alloy: 'en_paul_neutral',
31
+ echo: 'en_paul_neutral',
32
+ fable: 'en_paul_cheerful',
33
+ onyx: 'en_paul_confident',
34
+ nova: 'en_paul_happy',
35
+ shimmer: 'en_paul_excited',
36
+ })
37
+
38
+ /**
39
+ * Voices are provider-specific — accept any non-empty id; map legacy OpenAI names.
40
+ *
41
+ * @param {unknown} value
42
+ * @returns {string}
43
+ */
44
+ export function normalizeVoice (value) {
45
+ const raw = typeof value === 'string' ? value.trim() : ''
46
+ if (!raw) return DEFAULT_TEXT_TO_VOICE_VOICE
47
+ const lower = raw.toLowerCase()
48
+ if (LEGACY_VOICE_ALIASES[lower]) return LEGACY_VOICE_ALIASES[lower]
49
+ return raw
50
+ }
51
+
52
+ /**
53
+ * Accept OpenRouter model slugs; map legacy `tts-1` / `tts-1-hd` aliases.
54
+ *
55
+ * @param {unknown} value
56
+ * @returns {string}
57
+ */
58
+ export function normalizeModel (value) {
59
+ const raw = typeof value === 'string' ? value.trim() : ''
60
+ if (!raw) return DEFAULT_TEXT_TO_VOICE_MODEL
61
+ const lower = raw.toLowerCase()
62
+ if (LEGACY_MODEL_ALIASES[lower]) return LEGACY_MODEL_ALIASES[lower]
63
+ // OpenRouter slugs are provider/model (e.g. mistralai/voxtral-mini-tts-2603).
64
+ if (raw.includes('/')) return raw
65
+ return DEFAULT_TEXT_TO_VOICE_MODEL
66
+ }
67
+
68
+ /**
69
+ * @param {unknown} value
70
+ * @returns {string}
71
+ */
72
+ export function normalizeSpeechText (value) {
73
+ if (typeof value !== 'string') return ''
74
+ return value.trim()
75
+ }
76
+
77
+ /**
78
+ * Truncate to the TTS character ceiling.
79
+ *
80
+ * @param {string} text
81
+ * @returns {{ text: string, truncated: boolean }}
82
+ */
83
+ export function clampSpeechText (text) {
84
+ const normalized = normalizeSpeechText(text)
85
+ if (normalized.length <= MAX_TEXT_TO_VOICE_CHARS) {
86
+ return { text: normalized, truncated: false }
87
+ }
88
+ return {
89
+ text: normalized.slice(0, MAX_TEXT_TO_VOICE_CHARS),
90
+ truncated: true,
91
+ }
92
+ }
93
+
94
+ /**
95
+ * Prefer explicit input text; otherwise use Markdown `content.body` or file bytes.
96
+ *
97
+ * @param {{
98
+ * inputText?: unknown,
99
+ * resource?: object | null,
100
+ * fileText?: string | null,
101
+ * }} opts
102
+ * @returns {string}
103
+ */
104
+ export function resolveSpeechSourceText ({ inputText, resource, fileText } = {}) {
105
+ const fromInput = normalizeSpeechText(inputText)
106
+ if (fromInput) return fromInput
107
+
108
+ if (resource?.type === MARKDOWN_DOCUMENT_TYPE) {
109
+ const body = resource?.content?.body
110
+ if (typeof body === 'string' && body.trim()) return body.trim()
111
+ }
112
+
113
+ if (typeof fileText === 'string' && fileText.trim()) return fileText.trim()
114
+ return ''
115
+ }
116
+
117
+ /**
118
+ * Call OpenRouter TTS (OpenAI-compatible audio.speech) and return MP3 bytes.
119
+ *
120
+ * @param {{ audio: { speech: { create: Function } } }} client
121
+ * @param {{ text: string, voice?: string, model?: string }} opts
122
+ * @returns {Promise<{
123
+ * buffer: Buffer,
124
+ * voice: string,
125
+ * model: string,
126
+ * format: string,
127
+ * contentType: string,
128
+ * size: number,
129
+ * truncated: boolean,
130
+ * characterCount: number,
131
+ * }>}
132
+ */
133
+ export async function synthesizeSpeech (client, opts = {}) {
134
+ const voice = normalizeVoice(opts.voice)
135
+ const model = normalizeModel(opts.model)
136
+ const { text, truncated } = clampSpeechText(opts.text)
137
+ if (!text) {
138
+ throw new Error('[media-tasks/text-to-voice] text is required')
139
+ }
140
+
141
+ const response = await client.audio.speech.create({
142
+ model,
143
+ voice,
144
+ input: text,
145
+ response_format: DEFAULT_TEXT_TO_VOICE_FORMAT,
146
+ })
147
+
148
+ const arrayBuffer = await response.arrayBuffer()
149
+ const buffer = Buffer.from(arrayBuffer)
150
+ return {
151
+ buffer,
152
+ voice,
153
+ model,
154
+ format: DEFAULT_TEXT_TO_VOICE_FORMAT,
155
+ contentType: TEXT_TO_VOICE_CONTENT_TYPE,
156
+ size: buffer.length,
157
+ truncated,
158
+ characterCount: text.length,
159
+ }
160
+ }
@@ -0,0 +1,51 @@
1
+ import { TASK_OUTPUT_PROVENANCE_FIELDS } from '@ossy/schema'
2
+
3
+ export default {
4
+ name: 'Text to voice',
5
+ id: '@ossy/media-tasks/schema/text-to-voice',
6
+ categoryName: 'Media processing',
7
+ icon: 'mic',
8
+ fields: [
9
+ {
10
+ name: 'voice',
11
+ type: 'text',
12
+ description: 'TTS voice id (provider-dependent)',
13
+ },
14
+ {
15
+ name: 'model',
16
+ type: 'text',
17
+ description: 'OpenRouter TTS model slug',
18
+ },
19
+ {
20
+ name: 'format',
21
+ type: 'text',
22
+ description: 'Audio container format (mp3)',
23
+ },
24
+ {
25
+ name: 'contentType',
26
+ type: 'text',
27
+ description: 'MIME type of the audio bytes',
28
+ },
29
+ {
30
+ name: 'size',
31
+ type: 'number',
32
+ description: 'Audio byte length',
33
+ },
34
+ {
35
+ name: 'characterCount',
36
+ type: 'number',
37
+ description: 'Characters sent to TTS (after clamp)',
38
+ },
39
+ {
40
+ name: 'truncated',
41
+ type: 'boolean',
42
+ description: 'True when input was truncated to the TTS character limit',
43
+ },
44
+ {
45
+ name: 'audioKey',
46
+ type: 'text',
47
+ description: 'Storage key for the MP3 binary artifact',
48
+ },
49
+ ...TASK_OUTPUT_PROVENANCE_FIELDS,
50
+ ],
51
+ }
@@ -0,0 +1,102 @@
1
+ import {
2
+ DEFAULT_TEXT_TO_VOICE_MODEL,
3
+ DEFAULT_TEXT_TO_VOICE_VOICE,
4
+ MAX_TEXT_TO_VOICE_CHARS,
5
+ TEXT_TO_VOICE_CONTENT_TYPE,
6
+ clampSpeechText,
7
+ normalizeModel,
8
+ normalizeVoice,
9
+ resolveSpeechSourceText,
10
+ synthesizeSpeech,
11
+ } from './text-to-voice.js'
12
+ import { MARKDOWN_DOCUMENT_TYPE } from './run-prompt.js'
13
+
14
+ describe('text-to-voice helpers', () => {
15
+ it('normalizes voice and OpenRouter model slugs', () => {
16
+ expect(normalizeVoice('en_paul_happy')).toBe('en_paul_happy')
17
+ expect(normalizeVoice('NOVA')).toBe('en_paul_happy')
18
+ expect(normalizeVoice('')).toBe(DEFAULT_TEXT_TO_VOICE_VOICE)
19
+ expect(normalizeModel('mistralai/voxtral-mini-tts-2603')).toBe('mistralai/voxtral-mini-tts-2603')
20
+ expect(normalizeModel('google/gemini-3.8-flash-lite-tts')).toBe('google/gemini-3.8-flash-lite-tts')
21
+ expect(normalizeModel('TTS-1-HD')).toBe(DEFAULT_TEXT_TO_VOICE_MODEL)
22
+ expect(normalizeModel('tts-1')).toBe(DEFAULT_TEXT_TO_VOICE_MODEL)
23
+ expect(normalizeModel('openai/gpt-4o-mini-tts-2025-12-15')).toBe('openai/gpt-4o-mini-tts-2025-12-15')
24
+ expect(normalizeModel('whisper')).toBe(DEFAULT_TEXT_TO_VOICE_MODEL)
25
+ })
26
+
27
+ it('clamps speech text to the character ceiling', () => {
28
+ const long = 'a'.repeat(MAX_TEXT_TO_VOICE_CHARS + 10)
29
+ const clamped = clampSpeechText(long)
30
+ expect(clamped.truncated).toBe(true)
31
+ expect(clamped.text).toHaveLength(MAX_TEXT_TO_VOICE_CHARS)
32
+
33
+ expect(clampSpeechText(' hello ')).toEqual({ text: 'hello', truncated: false })
34
+ })
35
+
36
+ it('resolves speech source from input, markdown, or file text', () => {
37
+ expect(resolveSpeechSourceText({ inputText: ' from input ' })).toBe('from input')
38
+ expect(resolveSpeechSourceText({
39
+ resource: { type: MARKDOWN_DOCUMENT_TYPE, content: { body: ' md body ' } },
40
+ })).toBe('md body')
41
+ expect(resolveSpeechSourceText({ fileText: ' file bytes ' })).toBe('file bytes')
42
+ expect(resolveSpeechSourceText({})).toBe('')
43
+ })
44
+ })
45
+
46
+ describe('synthesizeSpeech', () => {
47
+ it('calls OpenRouter-compatible audio.speech and returns an MP3 buffer', async () => {
48
+ const fakeMp3 = Buffer.from([0xff, 0xfb, 0x90, 0x00])
49
+ const calls = []
50
+ const client = {
51
+ audio: {
52
+ speech: {
53
+ create: async (args) => {
54
+ calls.push(args)
55
+ return {
56
+ arrayBuffer: async () => fakeMp3.buffer.slice(
57
+ fakeMp3.byteOffset,
58
+ fakeMp3.byteOffset + fakeMp3.byteLength,
59
+ ),
60
+ }
61
+ },
62
+ },
63
+ },
64
+ }
65
+
66
+ const result = await synthesizeSpeech(client, {
67
+ text: 'Hello Ossy',
68
+ voice: 'en_paul_happy',
69
+ model: 'mistralai/voxtral-mini-tts-2603',
70
+ })
71
+
72
+ expect(calls[0]).toEqual({
73
+ model: 'mistralai/voxtral-mini-tts-2603',
74
+ voice: 'en_paul_happy',
75
+ input: 'Hello Ossy',
76
+ response_format: 'mp3',
77
+ })
78
+ expect(result.contentType).toBe(TEXT_TO_VOICE_CONTENT_TYPE)
79
+ expect(result.format).toBe('mp3')
80
+ expect(result.voice).toBe('en_paul_happy')
81
+ expect(result.model).toBe('mistralai/voxtral-mini-tts-2603')
82
+ expect(result.characterCount).toBe(10)
83
+ expect(result.truncated).toBe(false)
84
+ expect(Buffer.compare(result.buffer, fakeMp3)).toBe(0)
85
+ })
86
+
87
+ it('rejects empty text', async () => {
88
+ let called = false
89
+ const client = {
90
+ audio: {
91
+ speech: {
92
+ create: async () => {
93
+ called = true
94
+ return { arrayBuffer: async () => new ArrayBuffer(0) }
95
+ },
96
+ },
97
+ },
98
+ }
99
+ await expect(synthesizeSpeech(client, { text: ' ' })).rejects.toThrow(/text is required/)
100
+ expect(called).toBe(false)
101
+ })
102
+ })
@@ -0,0 +1,191 @@
1
+ import { createLogger } from '@ossy/observability'
2
+ import { StorageClient } from '@ossy/platform'
3
+ import { originalObjectKey, taskArtifactObjectKey } from '@ossy/platform/storage-keys'
4
+ import { GetResource, placeTaskArtifacts } from '@ossy/resources'
5
+ import { loadSourceBuffer } from './load-source-buffer.js'
6
+ import { resolveConfiguredPlacements } from './resolve-configured-placement.js'
7
+ import {
8
+ MARKDOWN_DOCUMENT_TYPE,
9
+ isTextContentType,
10
+ markdownDocumentBody,
11
+ } from './run-prompt.js'
12
+ import {
13
+ DEFAULT_TEXT_TO_VOICE_MODEL,
14
+ DEFAULT_TEXT_TO_VOICE_VOICE,
15
+ resolveSpeechSourceText,
16
+ synthesizeSpeech,
17
+ } from './text-to-voice.js'
18
+
19
+ const log = createLogger('media-tasks')
20
+
21
+ const TASK_ID = '@ossy/media-tasks/tasks/text-to-voice'
22
+ const SCHEMA_ID = '@ossy/media-tasks/schema/text-to-voice'
23
+ const CONCEPT = 'text-to-voice'
24
+ const AUDIO_ARTIFACT = 'audio'
25
+ const OUTPUT = {
26
+ name: 'speech',
27
+ kind: 'location',
28
+ schemaId: SCHEMA_ID,
29
+ contentType: 'application/json',
30
+ }
31
+
32
+ export const metadata = {
33
+ id: TASK_ID,
34
+ triggers: [
35
+ {
36
+ kind: 'on_action',
37
+ action: '@ossy/media-tasks/actions/text-to-voice',
38
+ },
39
+ {
40
+ type: MARKDOWN_DOCUMENT_TYPE,
41
+ event: 'Created',
42
+ },
43
+ {
44
+ type: '@ossy/platform/schema/file',
45
+ event: 'Created',
46
+ match: { 'content.ContentType': 'text/*' },
47
+ },
48
+ ],
49
+ outputs: [OUTPUT],
50
+ inputs: [
51
+ {
52
+ name: 'text',
53
+ type: 'text',
54
+ multiline: true,
55
+ sources: ['payload', 'config'],
56
+ description: 'Text to speak. When empty, uses the Markdown body or text file contents.',
57
+ },
58
+ {
59
+ name: 'voice',
60
+ type: 'text',
61
+ default: DEFAULT_TEXT_TO_VOICE_VOICE,
62
+ sources: ['payload', 'config'],
63
+ description: 'OpenRouter TTS voice id (provider-specific; default en_paul_neutral for Voxtral)',
64
+ },
65
+ {
66
+ name: 'model',
67
+ type: 'text',
68
+ default: DEFAULT_TEXT_TO_VOICE_MODEL,
69
+ sources: ['payload', 'config'],
70
+ description: 'OpenRouter speech model slug (default mistralai/voxtral-mini-tts-2603)',
71
+ },
72
+ ],
73
+ configurable: true,
74
+ }
75
+
76
+ export async function run ({ event, payload, sdk, integrations, inputs }) {
77
+ const openrouter = integrations?.get?.('openrouter')
78
+ if (!openrouter) {
79
+ log.warn('[media-tasks/tasks/text-to-voice] openrouter integration not connected')
80
+ throw Object.assign(new Error('OpenRouter integration not available'), { status: 503 })
81
+ }
82
+
83
+ const resourceId = event?.resourceId || payload?.resourceId || null
84
+ log.info(`[media-tasks/tasks/text-to-voice] starting${resourceId ? ` for resource ${resourceId}` : ''}`)
85
+
86
+ let resource = null
87
+ let fileText = null
88
+ if (resourceId) {
89
+ resource = await sdk.invoke(GetResource, { resourceId })
90
+ fileText = await loadFileText(resource)
91
+ }
92
+
93
+ const sourceText = resolveSpeechSourceText({
94
+ inputText: inputs?.text,
95
+ resource,
96
+ fileText,
97
+ })
98
+ if (!sourceText) {
99
+ throw Object.assign(
100
+ new Error('[media-tasks/tasks/text-to-voice] text is required (input, Markdown body, or text file)'),
101
+ { status: 400 },
102
+ )
103
+ }
104
+
105
+ const speech = await synthesizeSpeech(openrouter, {
106
+ text: sourceText,
107
+ voice: inputs?.voice,
108
+ model: inputs?.model,
109
+ })
110
+
111
+ if (!resource?.id) {
112
+ return {
113
+ output: OUTPUT.name,
114
+ voice: speech.voice,
115
+ model: speech.model,
116
+ format: speech.format,
117
+ contentType: speech.contentType,
118
+ size: speech.size,
119
+ characterCount: speech.characterCount,
120
+ truncated: speech.truncated,
121
+ placedResourceIds: [],
122
+ }
123
+ }
124
+
125
+ const audioKey = taskArtifactObjectKey(resource.id, CONCEPT, AUDIO_ARTIFACT)
126
+ await StorageClient.save(audioKey, speech.buffer)
127
+ log.info(`[media-tasks/tasks/text-to-voice] saved audio ${audioKey} (${speech.size} bytes)`)
128
+
129
+ const body = {
130
+ voice: speech.voice,
131
+ model: speech.model,
132
+ format: speech.format,
133
+ contentType: speech.contentType,
134
+ size: speech.size,
135
+ characterCount: speech.characterCount,
136
+ truncated: speech.truncated,
137
+ audioKey,
138
+ }
139
+
140
+ const metaKey = taskArtifactObjectKey(resource.id, CONCEPT, OUTPUT.name)
141
+ await StorageClient.save(metaKey, Buffer.from(JSON.stringify(body), 'utf8'))
142
+ log.info(`[media-tasks/tasks/text-to-voice] saved ${metaKey}`)
143
+
144
+ const placementLocations = await resolveConfiguredPlacements(sdk, TASK_ID, OUTPUT.name)
145
+ const placed = await placeTaskArtifacts({
146
+ sdk,
147
+ sourceResource: resource,
148
+ taskId: TASK_ID,
149
+ output: OUTPUT,
150
+ body,
151
+ artifactKey: metaKey,
152
+ placementLocations,
153
+ })
154
+ if (placed.length) {
155
+ log.info(`[media-tasks/tasks/text-to-voice] placed ${placed.length} resource(s)`)
156
+ } else {
157
+ log.info('[media-tasks/tasks/text-to-voice] artifact only (no placement location configured)')
158
+ }
159
+
160
+ return {
161
+ output: OUTPUT.name,
162
+ key: metaKey,
163
+ audioKey,
164
+ voice: speech.voice,
165
+ model: speech.model,
166
+ format: speech.format,
167
+ contentType: speech.contentType,
168
+ size: speech.size,
169
+ characterCount: speech.characterCount,
170
+ truncated: speech.truncated,
171
+ placedResourceIds: placed.map((r) => r.id),
172
+ placedResourceId: placed[0]?.id ?? null,
173
+ }
174
+ }
175
+
176
+ /**
177
+ * @param {object | null | undefined} resource
178
+ * @returns {Promise<string | null>}
179
+ */
180
+ async function loadFileText (resource) {
181
+ if (!resource) return null
182
+ if (resource.type === MARKDOWN_DOCUMENT_TYPE) {
183
+ return markdownDocumentBody(resource) || null
184
+ }
185
+ const contentType = resource?.content?.ContentType || ''
186
+ if (!isTextContentType(contentType)) return null
187
+ // content.Key is stripped by the API media layer; derive from resource id.
188
+ const key = originalObjectKey(resource.id)
189
+ const buffer = await loadSourceBuffer(key)
190
+ return buffer ? buffer.toString('utf8') : null
191
+ }
@@ -0,0 +1,5 @@
1
+ export const metadata = {
2
+ id: '@ossy/media-tasks/actions/visual-content-descriptors',
3
+ access: 'workspace',
4
+ label: 'automation.task.visual-content-descriptors.title',
5
+ }
@@ -36,6 +36,10 @@ export const metadata = {
36
36
  event: 'Created',
37
37
  match: { 'content.ContentType': 'video/*' },
38
38
  },
39
+ {
40
+ kind: 'on_action',
41
+ action: '@ossy/media-tasks/actions/visual-content-descriptors',
42
+ },
39
43
  ],
40
44
  outputs: [OUTPUT],
41
45
  inputs: [
@@ -51,7 +55,10 @@ export const metadata = {
51
55
  }
52
56
 
53
57
  export async function run ({ event, payload, req, sdk, integrations, inputs }) {
54
- const resourceId = event.resourceId
58
+ const resourceId = event?.resourceId || payload?.resourceId
59
+ if (!resourceId) {
60
+ throw new Error('[media-tasks/tasks/visual-content-descriptors] resourceId is required')
61
+ }
55
62
 
56
63
  log.info(`[media-tasks/tasks/visual-content-descriptors] starting for resource ${resourceId}`)
57
64