@ossy/media-tasks 3.18.0 → 3.19.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +29 -3
- package/package.json +16 -6
- package/src/Definition.js +7 -0
- package/src/convert-common-audio.action.js +5 -0
- package/src/convert-common-audio.js +243 -0
- package/src/convert-common-audio.schema.js +41 -0
- package/src/convert-common-audio.spec.js +188 -0
- package/src/convert-common-audio.task.js +142 -0
- package/src/convert-common-image.action.js +5 -0
- package/src/convert-common-image.js +143 -0
- package/src/convert-common-image.schema.js +46 -0
- package/src/convert-common-image.spec.js +101 -0
- package/src/convert-common-image.task.js +155 -0
- package/src/embed-resource.action.js +5 -0
- package/src/embed-resource.js +111 -0
- package/src/embed-resource.schema.js +46 -0
- package/src/embed-resource.spec.js +179 -0
- package/src/embed-resource.task.js +199 -0
- package/src/en.translations.json +4 -0
- package/src/extract-colors.action.js +5 -0
- package/src/extract-colors.task.js +13 -7
- package/src/openrouter.integration.js +2 -1
- package/src/resize-common-web.action.js +5 -0
- package/src/resize-common-web.task.js +13 -7
- package/src/run-prompt.task.js +6 -0
- package/src/sv.translations.json +4 -0
- package/src/text-to-voice.action.js +5 -0
- package/src/text-to-voice.js +160 -0
- package/src/text-to-voice.schema.js +51 -0
- package/src/text-to-voice.spec.js +102 -0
- package/src/text-to-voice.task.js +191 -0
- package/src/visual-content-descriptors.action.js +5 -0
- package/src/visual-content-descriptors.task.js +8 -1
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Text-to-voice via OpenRouter TTS (#898).
|
|
3
|
+
*
|
|
4
|
+
* Provider: shared `openrouter` integration (`OPENROUTER_API_KEY`) — OpenRouter
|
|
5
|
+
* exposes OpenAI-compatible `/audio/speech` (see
|
|
6
|
+
* https://openrouter.ai/docs/guides/overview/multimodal/tts). Default model is
|
|
7
|
+
* `mistralai/voxtral-mini-tts-2603` (listed in OpenRouter speech catalog).
|
|
8
|
+
* Output is MP3 (`audio/mpeg`).
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
import { MARKDOWN_DOCUMENT_TYPE } from './run-prompt.js'
|
|
12
|
+
|
|
13
|
+
/** @see https://openrouter.ai/docs/guides/overview/multimodal/tts */
|
|
14
|
+
export const DEFAULT_TEXT_TO_VOICE_MODEL = 'mistralai/voxtral-mini-tts-2603'
|
|
15
|
+
export const DEFAULT_TEXT_TO_VOICE_VOICE = 'en_paul_neutral'
|
|
16
|
+
export const DEFAULT_TEXT_TO_VOICE_FORMAT = 'mp3'
|
|
17
|
+
export const TEXT_TO_VOICE_CONTENT_TYPE = 'audio/mpeg'
|
|
18
|
+
|
|
19
|
+
/** Soft character ceiling for spoken input. */
|
|
20
|
+
export const MAX_TEXT_TO_VOICE_CHARS = 4096
|
|
21
|
+
|
|
22
|
+
/** Legacy direct-OpenAI short model names → OpenRouter default. */
|
|
23
|
+
const LEGACY_MODEL_ALIASES = Object.freeze({
|
|
24
|
+
'tts-1': DEFAULT_TEXT_TO_VOICE_MODEL,
|
|
25
|
+
'tts-1-hd': DEFAULT_TEXT_TO_VOICE_MODEL,
|
|
26
|
+
})
|
|
27
|
+
|
|
28
|
+
/** Legacy OpenAI voice names → Voxtral defaults (provider voices differ). */
|
|
29
|
+
const LEGACY_VOICE_ALIASES = Object.freeze({
|
|
30
|
+
alloy: 'en_paul_neutral',
|
|
31
|
+
echo: 'en_paul_neutral',
|
|
32
|
+
fable: 'en_paul_cheerful',
|
|
33
|
+
onyx: 'en_paul_confident',
|
|
34
|
+
nova: 'en_paul_happy',
|
|
35
|
+
shimmer: 'en_paul_excited',
|
|
36
|
+
})
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* Voices are provider-specific — accept any non-empty id; map legacy OpenAI names.
|
|
40
|
+
*
|
|
41
|
+
* @param {unknown} value
|
|
42
|
+
* @returns {string}
|
|
43
|
+
*/
|
|
44
|
+
export function normalizeVoice (value) {
|
|
45
|
+
const raw = typeof value === 'string' ? value.trim() : ''
|
|
46
|
+
if (!raw) return DEFAULT_TEXT_TO_VOICE_VOICE
|
|
47
|
+
const lower = raw.toLowerCase()
|
|
48
|
+
if (LEGACY_VOICE_ALIASES[lower]) return LEGACY_VOICE_ALIASES[lower]
|
|
49
|
+
return raw
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* Accept OpenRouter model slugs; map legacy `tts-1` / `tts-1-hd` aliases.
|
|
54
|
+
*
|
|
55
|
+
* @param {unknown} value
|
|
56
|
+
* @returns {string}
|
|
57
|
+
*/
|
|
58
|
+
export function normalizeModel (value) {
|
|
59
|
+
const raw = typeof value === 'string' ? value.trim() : ''
|
|
60
|
+
if (!raw) return DEFAULT_TEXT_TO_VOICE_MODEL
|
|
61
|
+
const lower = raw.toLowerCase()
|
|
62
|
+
if (LEGACY_MODEL_ALIASES[lower]) return LEGACY_MODEL_ALIASES[lower]
|
|
63
|
+
// OpenRouter slugs are provider/model (e.g. mistralai/voxtral-mini-tts-2603).
|
|
64
|
+
if (raw.includes('/')) return raw
|
|
65
|
+
return DEFAULT_TEXT_TO_VOICE_MODEL
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* @param {unknown} value
|
|
70
|
+
* @returns {string}
|
|
71
|
+
*/
|
|
72
|
+
export function normalizeSpeechText (value) {
|
|
73
|
+
if (typeof value !== 'string') return ''
|
|
74
|
+
return value.trim()
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Truncate to the TTS character ceiling.
|
|
79
|
+
*
|
|
80
|
+
* @param {string} text
|
|
81
|
+
* @returns {{ text: string, truncated: boolean }}
|
|
82
|
+
*/
|
|
83
|
+
export function clampSpeechText (text) {
|
|
84
|
+
const normalized = normalizeSpeechText(text)
|
|
85
|
+
if (normalized.length <= MAX_TEXT_TO_VOICE_CHARS) {
|
|
86
|
+
return { text: normalized, truncated: false }
|
|
87
|
+
}
|
|
88
|
+
return {
|
|
89
|
+
text: normalized.slice(0, MAX_TEXT_TO_VOICE_CHARS),
|
|
90
|
+
truncated: true,
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* Prefer explicit input text; otherwise use Markdown `content.body` or file bytes.
|
|
96
|
+
*
|
|
97
|
+
* @param {{
|
|
98
|
+
* inputText?: unknown,
|
|
99
|
+
* resource?: object | null,
|
|
100
|
+
* fileText?: string | null,
|
|
101
|
+
* }} opts
|
|
102
|
+
* @returns {string}
|
|
103
|
+
*/
|
|
104
|
+
export function resolveSpeechSourceText ({ inputText, resource, fileText } = {}) {
|
|
105
|
+
const fromInput = normalizeSpeechText(inputText)
|
|
106
|
+
if (fromInput) return fromInput
|
|
107
|
+
|
|
108
|
+
if (resource?.type === MARKDOWN_DOCUMENT_TYPE) {
|
|
109
|
+
const body = resource?.content?.body
|
|
110
|
+
if (typeof body === 'string' && body.trim()) return body.trim()
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
if (typeof fileText === 'string' && fileText.trim()) return fileText.trim()
|
|
114
|
+
return ''
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
/**
|
|
118
|
+
* Call OpenRouter TTS (OpenAI-compatible audio.speech) and return MP3 bytes.
|
|
119
|
+
*
|
|
120
|
+
* @param {{ audio: { speech: { create: Function } } }} client
|
|
121
|
+
* @param {{ text: string, voice?: string, model?: string }} opts
|
|
122
|
+
* @returns {Promise<{
|
|
123
|
+
* buffer: Buffer,
|
|
124
|
+
* voice: string,
|
|
125
|
+
* model: string,
|
|
126
|
+
* format: string,
|
|
127
|
+
* contentType: string,
|
|
128
|
+
* size: number,
|
|
129
|
+
* truncated: boolean,
|
|
130
|
+
* characterCount: number,
|
|
131
|
+
* }>}
|
|
132
|
+
*/
|
|
133
|
+
export async function synthesizeSpeech (client, opts = {}) {
|
|
134
|
+
const voice = normalizeVoice(opts.voice)
|
|
135
|
+
const model = normalizeModel(opts.model)
|
|
136
|
+
const { text, truncated } = clampSpeechText(opts.text)
|
|
137
|
+
if (!text) {
|
|
138
|
+
throw new Error('[media-tasks/text-to-voice] text is required')
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
const response = await client.audio.speech.create({
|
|
142
|
+
model,
|
|
143
|
+
voice,
|
|
144
|
+
input: text,
|
|
145
|
+
response_format: DEFAULT_TEXT_TO_VOICE_FORMAT,
|
|
146
|
+
})
|
|
147
|
+
|
|
148
|
+
const arrayBuffer = await response.arrayBuffer()
|
|
149
|
+
const buffer = Buffer.from(arrayBuffer)
|
|
150
|
+
return {
|
|
151
|
+
buffer,
|
|
152
|
+
voice,
|
|
153
|
+
model,
|
|
154
|
+
format: DEFAULT_TEXT_TO_VOICE_FORMAT,
|
|
155
|
+
contentType: TEXT_TO_VOICE_CONTENT_TYPE,
|
|
156
|
+
size: buffer.length,
|
|
157
|
+
truncated,
|
|
158
|
+
characterCount: text.length,
|
|
159
|
+
}
|
|
160
|
+
}
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
import { TASK_OUTPUT_PROVENANCE_FIELDS } from '@ossy/schema'
|
|
2
|
+
|
|
3
|
+
export default {
|
|
4
|
+
name: 'Text to voice',
|
|
5
|
+
id: '@ossy/media-tasks/schema/text-to-voice',
|
|
6
|
+
categoryName: 'Media processing',
|
|
7
|
+
icon: 'mic',
|
|
8
|
+
fields: [
|
|
9
|
+
{
|
|
10
|
+
name: 'voice',
|
|
11
|
+
type: 'text',
|
|
12
|
+
description: 'TTS voice id (provider-dependent)',
|
|
13
|
+
},
|
|
14
|
+
{
|
|
15
|
+
name: 'model',
|
|
16
|
+
type: 'text',
|
|
17
|
+
description: 'OpenRouter TTS model slug',
|
|
18
|
+
},
|
|
19
|
+
{
|
|
20
|
+
name: 'format',
|
|
21
|
+
type: 'text',
|
|
22
|
+
description: 'Audio container format (mp3)',
|
|
23
|
+
},
|
|
24
|
+
{
|
|
25
|
+
name: 'contentType',
|
|
26
|
+
type: 'text',
|
|
27
|
+
description: 'MIME type of the audio bytes',
|
|
28
|
+
},
|
|
29
|
+
{
|
|
30
|
+
name: 'size',
|
|
31
|
+
type: 'number',
|
|
32
|
+
description: 'Audio byte length',
|
|
33
|
+
},
|
|
34
|
+
{
|
|
35
|
+
name: 'characterCount',
|
|
36
|
+
type: 'number',
|
|
37
|
+
description: 'Characters sent to TTS (after clamp)',
|
|
38
|
+
},
|
|
39
|
+
{
|
|
40
|
+
name: 'truncated',
|
|
41
|
+
type: 'boolean',
|
|
42
|
+
description: 'True when input was truncated to the TTS character limit',
|
|
43
|
+
},
|
|
44
|
+
{
|
|
45
|
+
name: 'audioKey',
|
|
46
|
+
type: 'text',
|
|
47
|
+
description: 'Storage key for the MP3 binary artifact',
|
|
48
|
+
},
|
|
49
|
+
...TASK_OUTPUT_PROVENANCE_FIELDS,
|
|
50
|
+
],
|
|
51
|
+
}
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
import {
|
|
2
|
+
DEFAULT_TEXT_TO_VOICE_MODEL,
|
|
3
|
+
DEFAULT_TEXT_TO_VOICE_VOICE,
|
|
4
|
+
MAX_TEXT_TO_VOICE_CHARS,
|
|
5
|
+
TEXT_TO_VOICE_CONTENT_TYPE,
|
|
6
|
+
clampSpeechText,
|
|
7
|
+
normalizeModel,
|
|
8
|
+
normalizeVoice,
|
|
9
|
+
resolveSpeechSourceText,
|
|
10
|
+
synthesizeSpeech,
|
|
11
|
+
} from './text-to-voice.js'
|
|
12
|
+
import { MARKDOWN_DOCUMENT_TYPE } from './run-prompt.js'
|
|
13
|
+
|
|
14
|
+
describe('text-to-voice helpers', () => {
|
|
15
|
+
it('normalizes voice and OpenRouter model slugs', () => {
|
|
16
|
+
expect(normalizeVoice('en_paul_happy')).toBe('en_paul_happy')
|
|
17
|
+
expect(normalizeVoice('NOVA')).toBe('en_paul_happy')
|
|
18
|
+
expect(normalizeVoice('')).toBe(DEFAULT_TEXT_TO_VOICE_VOICE)
|
|
19
|
+
expect(normalizeModel('mistralai/voxtral-mini-tts-2603')).toBe('mistralai/voxtral-mini-tts-2603')
|
|
20
|
+
expect(normalizeModel('google/gemini-3.8-flash-lite-tts')).toBe('google/gemini-3.8-flash-lite-tts')
|
|
21
|
+
expect(normalizeModel('TTS-1-HD')).toBe(DEFAULT_TEXT_TO_VOICE_MODEL)
|
|
22
|
+
expect(normalizeModel('tts-1')).toBe(DEFAULT_TEXT_TO_VOICE_MODEL)
|
|
23
|
+
expect(normalizeModel('openai/gpt-4o-mini-tts-2025-12-15')).toBe('openai/gpt-4o-mini-tts-2025-12-15')
|
|
24
|
+
expect(normalizeModel('whisper')).toBe(DEFAULT_TEXT_TO_VOICE_MODEL)
|
|
25
|
+
})
|
|
26
|
+
|
|
27
|
+
it('clamps speech text to the character ceiling', () => {
|
|
28
|
+
const long = 'a'.repeat(MAX_TEXT_TO_VOICE_CHARS + 10)
|
|
29
|
+
const clamped = clampSpeechText(long)
|
|
30
|
+
expect(clamped.truncated).toBe(true)
|
|
31
|
+
expect(clamped.text).toHaveLength(MAX_TEXT_TO_VOICE_CHARS)
|
|
32
|
+
|
|
33
|
+
expect(clampSpeechText(' hello ')).toEqual({ text: 'hello', truncated: false })
|
|
34
|
+
})
|
|
35
|
+
|
|
36
|
+
it('resolves speech source from input, markdown, or file text', () => {
|
|
37
|
+
expect(resolveSpeechSourceText({ inputText: ' from input ' })).toBe('from input')
|
|
38
|
+
expect(resolveSpeechSourceText({
|
|
39
|
+
resource: { type: MARKDOWN_DOCUMENT_TYPE, content: { body: ' md body ' } },
|
|
40
|
+
})).toBe('md body')
|
|
41
|
+
expect(resolveSpeechSourceText({ fileText: ' file bytes ' })).toBe('file bytes')
|
|
42
|
+
expect(resolveSpeechSourceText({})).toBe('')
|
|
43
|
+
})
|
|
44
|
+
})
|
|
45
|
+
|
|
46
|
+
describe('synthesizeSpeech', () => {
|
|
47
|
+
it('calls OpenRouter-compatible audio.speech and returns an MP3 buffer', async () => {
|
|
48
|
+
const fakeMp3 = Buffer.from([0xff, 0xfb, 0x90, 0x00])
|
|
49
|
+
const calls = []
|
|
50
|
+
const client = {
|
|
51
|
+
audio: {
|
|
52
|
+
speech: {
|
|
53
|
+
create: async (args) => {
|
|
54
|
+
calls.push(args)
|
|
55
|
+
return {
|
|
56
|
+
arrayBuffer: async () => fakeMp3.buffer.slice(
|
|
57
|
+
fakeMp3.byteOffset,
|
|
58
|
+
fakeMp3.byteOffset + fakeMp3.byteLength,
|
|
59
|
+
),
|
|
60
|
+
}
|
|
61
|
+
},
|
|
62
|
+
},
|
|
63
|
+
},
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
const result = await synthesizeSpeech(client, {
|
|
67
|
+
text: 'Hello Ossy',
|
|
68
|
+
voice: 'en_paul_happy',
|
|
69
|
+
model: 'mistralai/voxtral-mini-tts-2603',
|
|
70
|
+
})
|
|
71
|
+
|
|
72
|
+
expect(calls[0]).toEqual({
|
|
73
|
+
model: 'mistralai/voxtral-mini-tts-2603',
|
|
74
|
+
voice: 'en_paul_happy',
|
|
75
|
+
input: 'Hello Ossy',
|
|
76
|
+
response_format: 'mp3',
|
|
77
|
+
})
|
|
78
|
+
expect(result.contentType).toBe(TEXT_TO_VOICE_CONTENT_TYPE)
|
|
79
|
+
expect(result.format).toBe('mp3')
|
|
80
|
+
expect(result.voice).toBe('en_paul_happy')
|
|
81
|
+
expect(result.model).toBe('mistralai/voxtral-mini-tts-2603')
|
|
82
|
+
expect(result.characterCount).toBe(10)
|
|
83
|
+
expect(result.truncated).toBe(false)
|
|
84
|
+
expect(Buffer.compare(result.buffer, fakeMp3)).toBe(0)
|
|
85
|
+
})
|
|
86
|
+
|
|
87
|
+
it('rejects empty text', async () => {
|
|
88
|
+
let called = false
|
|
89
|
+
const client = {
|
|
90
|
+
audio: {
|
|
91
|
+
speech: {
|
|
92
|
+
create: async () => {
|
|
93
|
+
called = true
|
|
94
|
+
return { arrayBuffer: async () => new ArrayBuffer(0) }
|
|
95
|
+
},
|
|
96
|
+
},
|
|
97
|
+
},
|
|
98
|
+
}
|
|
99
|
+
await expect(synthesizeSpeech(client, { text: ' ' })).rejects.toThrow(/text is required/)
|
|
100
|
+
expect(called).toBe(false)
|
|
101
|
+
})
|
|
102
|
+
})
|
|
@@ -0,0 +1,191 @@
|
|
|
1
|
+
import { createLogger } from '@ossy/observability'
|
|
2
|
+
import { StorageClient } from '@ossy/platform'
|
|
3
|
+
import { originalObjectKey, taskArtifactObjectKey } from '@ossy/platform/storage-keys'
|
|
4
|
+
import { GetResource, placeTaskArtifacts } from '@ossy/resources'
|
|
5
|
+
import { loadSourceBuffer } from './load-source-buffer.js'
|
|
6
|
+
import { resolveConfiguredPlacements } from './resolve-configured-placement.js'
|
|
7
|
+
import {
|
|
8
|
+
MARKDOWN_DOCUMENT_TYPE,
|
|
9
|
+
isTextContentType,
|
|
10
|
+
markdownDocumentBody,
|
|
11
|
+
} from './run-prompt.js'
|
|
12
|
+
import {
|
|
13
|
+
DEFAULT_TEXT_TO_VOICE_MODEL,
|
|
14
|
+
DEFAULT_TEXT_TO_VOICE_VOICE,
|
|
15
|
+
resolveSpeechSourceText,
|
|
16
|
+
synthesizeSpeech,
|
|
17
|
+
} from './text-to-voice.js'
|
|
18
|
+
|
|
19
|
+
const log = createLogger('media-tasks')
|
|
20
|
+
|
|
21
|
+
const TASK_ID = '@ossy/media-tasks/tasks/text-to-voice'
|
|
22
|
+
const SCHEMA_ID = '@ossy/media-tasks/schema/text-to-voice'
|
|
23
|
+
const CONCEPT = 'text-to-voice'
|
|
24
|
+
const AUDIO_ARTIFACT = 'audio'
|
|
25
|
+
const OUTPUT = {
|
|
26
|
+
name: 'speech',
|
|
27
|
+
kind: 'location',
|
|
28
|
+
schemaId: SCHEMA_ID,
|
|
29
|
+
contentType: 'application/json',
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
export const metadata = {
|
|
33
|
+
id: TASK_ID,
|
|
34
|
+
triggers: [
|
|
35
|
+
{
|
|
36
|
+
kind: 'on_action',
|
|
37
|
+
action: '@ossy/media-tasks/actions/text-to-voice',
|
|
38
|
+
},
|
|
39
|
+
{
|
|
40
|
+
type: MARKDOWN_DOCUMENT_TYPE,
|
|
41
|
+
event: 'Created',
|
|
42
|
+
},
|
|
43
|
+
{
|
|
44
|
+
type: '@ossy/platform/schema/file',
|
|
45
|
+
event: 'Created',
|
|
46
|
+
match: { 'content.ContentType': 'text/*' },
|
|
47
|
+
},
|
|
48
|
+
],
|
|
49
|
+
outputs: [OUTPUT],
|
|
50
|
+
inputs: [
|
|
51
|
+
{
|
|
52
|
+
name: 'text',
|
|
53
|
+
type: 'text',
|
|
54
|
+
multiline: true,
|
|
55
|
+
sources: ['payload', 'config'],
|
|
56
|
+
description: 'Text to speak. When empty, uses the Markdown body or text file contents.',
|
|
57
|
+
},
|
|
58
|
+
{
|
|
59
|
+
name: 'voice',
|
|
60
|
+
type: 'text',
|
|
61
|
+
default: DEFAULT_TEXT_TO_VOICE_VOICE,
|
|
62
|
+
sources: ['payload', 'config'],
|
|
63
|
+
description: 'OpenRouter TTS voice id (provider-specific; default en_paul_neutral for Voxtral)',
|
|
64
|
+
},
|
|
65
|
+
{
|
|
66
|
+
name: 'model',
|
|
67
|
+
type: 'text',
|
|
68
|
+
default: DEFAULT_TEXT_TO_VOICE_MODEL,
|
|
69
|
+
sources: ['payload', 'config'],
|
|
70
|
+
description: 'OpenRouter speech model slug (default mistralai/voxtral-mini-tts-2603)',
|
|
71
|
+
},
|
|
72
|
+
],
|
|
73
|
+
configurable: true,
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
export async function run ({ event, payload, sdk, integrations, inputs }) {
|
|
77
|
+
const openrouter = integrations?.get?.('openrouter')
|
|
78
|
+
if (!openrouter) {
|
|
79
|
+
log.warn('[media-tasks/tasks/text-to-voice] openrouter integration not connected')
|
|
80
|
+
throw Object.assign(new Error('OpenRouter integration not available'), { status: 503 })
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
const resourceId = event?.resourceId || payload?.resourceId || null
|
|
84
|
+
log.info(`[media-tasks/tasks/text-to-voice] starting${resourceId ? ` for resource ${resourceId}` : ''}`)
|
|
85
|
+
|
|
86
|
+
let resource = null
|
|
87
|
+
let fileText = null
|
|
88
|
+
if (resourceId) {
|
|
89
|
+
resource = await sdk.invoke(GetResource, { resourceId })
|
|
90
|
+
fileText = await loadFileText(resource)
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
const sourceText = resolveSpeechSourceText({
|
|
94
|
+
inputText: inputs?.text,
|
|
95
|
+
resource,
|
|
96
|
+
fileText,
|
|
97
|
+
})
|
|
98
|
+
if (!sourceText) {
|
|
99
|
+
throw Object.assign(
|
|
100
|
+
new Error('[media-tasks/tasks/text-to-voice] text is required (input, Markdown body, or text file)'),
|
|
101
|
+
{ status: 400 },
|
|
102
|
+
)
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
const speech = await synthesizeSpeech(openrouter, {
|
|
106
|
+
text: sourceText,
|
|
107
|
+
voice: inputs?.voice,
|
|
108
|
+
model: inputs?.model,
|
|
109
|
+
})
|
|
110
|
+
|
|
111
|
+
if (!resource?.id) {
|
|
112
|
+
return {
|
|
113
|
+
output: OUTPUT.name,
|
|
114
|
+
voice: speech.voice,
|
|
115
|
+
model: speech.model,
|
|
116
|
+
format: speech.format,
|
|
117
|
+
contentType: speech.contentType,
|
|
118
|
+
size: speech.size,
|
|
119
|
+
characterCount: speech.characterCount,
|
|
120
|
+
truncated: speech.truncated,
|
|
121
|
+
placedResourceIds: [],
|
|
122
|
+
}
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
const audioKey = taskArtifactObjectKey(resource.id, CONCEPT, AUDIO_ARTIFACT)
|
|
126
|
+
await StorageClient.save(audioKey, speech.buffer)
|
|
127
|
+
log.info(`[media-tasks/tasks/text-to-voice] saved audio ${audioKey} (${speech.size} bytes)`)
|
|
128
|
+
|
|
129
|
+
const body = {
|
|
130
|
+
voice: speech.voice,
|
|
131
|
+
model: speech.model,
|
|
132
|
+
format: speech.format,
|
|
133
|
+
contentType: speech.contentType,
|
|
134
|
+
size: speech.size,
|
|
135
|
+
characterCount: speech.characterCount,
|
|
136
|
+
truncated: speech.truncated,
|
|
137
|
+
audioKey,
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
const metaKey = taskArtifactObjectKey(resource.id, CONCEPT, OUTPUT.name)
|
|
141
|
+
await StorageClient.save(metaKey, Buffer.from(JSON.stringify(body), 'utf8'))
|
|
142
|
+
log.info(`[media-tasks/tasks/text-to-voice] saved ${metaKey}`)
|
|
143
|
+
|
|
144
|
+
const placementLocations = await resolveConfiguredPlacements(sdk, TASK_ID, OUTPUT.name)
|
|
145
|
+
const placed = await placeTaskArtifacts({
|
|
146
|
+
sdk,
|
|
147
|
+
sourceResource: resource,
|
|
148
|
+
taskId: TASK_ID,
|
|
149
|
+
output: OUTPUT,
|
|
150
|
+
body,
|
|
151
|
+
artifactKey: metaKey,
|
|
152
|
+
placementLocations,
|
|
153
|
+
})
|
|
154
|
+
if (placed.length) {
|
|
155
|
+
log.info(`[media-tasks/tasks/text-to-voice] placed ${placed.length} resource(s)`)
|
|
156
|
+
} else {
|
|
157
|
+
log.info('[media-tasks/tasks/text-to-voice] artifact only (no placement location configured)')
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
return {
|
|
161
|
+
output: OUTPUT.name,
|
|
162
|
+
key: metaKey,
|
|
163
|
+
audioKey,
|
|
164
|
+
voice: speech.voice,
|
|
165
|
+
model: speech.model,
|
|
166
|
+
format: speech.format,
|
|
167
|
+
contentType: speech.contentType,
|
|
168
|
+
size: speech.size,
|
|
169
|
+
characterCount: speech.characterCount,
|
|
170
|
+
truncated: speech.truncated,
|
|
171
|
+
placedResourceIds: placed.map((r) => r.id),
|
|
172
|
+
placedResourceId: placed[0]?.id ?? null,
|
|
173
|
+
}
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
/**
|
|
177
|
+
* @param {object | null | undefined} resource
|
|
178
|
+
* @returns {Promise<string | null>}
|
|
179
|
+
*/
|
|
180
|
+
async function loadFileText (resource) {
|
|
181
|
+
if (!resource) return null
|
|
182
|
+
if (resource.type === MARKDOWN_DOCUMENT_TYPE) {
|
|
183
|
+
return markdownDocumentBody(resource) || null
|
|
184
|
+
}
|
|
185
|
+
const contentType = resource?.content?.ContentType || ''
|
|
186
|
+
if (!isTextContentType(contentType)) return null
|
|
187
|
+
// content.Key is stripped by the API media layer; derive from resource id.
|
|
188
|
+
const key = originalObjectKey(resource.id)
|
|
189
|
+
const buffer = await loadSourceBuffer(key)
|
|
190
|
+
return buffer ? buffer.toString('utf8') : null
|
|
191
|
+
}
|
|
@@ -36,6 +36,10 @@ export const metadata = {
|
|
|
36
36
|
event: 'Created',
|
|
37
37
|
match: { 'content.ContentType': 'video/*' },
|
|
38
38
|
},
|
|
39
|
+
{
|
|
40
|
+
kind: 'on_action',
|
|
41
|
+
action: '@ossy/media-tasks/actions/visual-content-descriptors',
|
|
42
|
+
},
|
|
39
43
|
],
|
|
40
44
|
outputs: [OUTPUT],
|
|
41
45
|
inputs: [
|
|
@@ -51,7 +55,10 @@ export const metadata = {
|
|
|
51
55
|
}
|
|
52
56
|
|
|
53
57
|
export async function run ({ event, payload, req, sdk, integrations, inputs }) {
|
|
54
|
-
const resourceId = event
|
|
58
|
+
const resourceId = event?.resourceId || payload?.resourceId
|
|
59
|
+
if (!resourceId) {
|
|
60
|
+
throw new Error('[media-tasks/tasks/visual-content-descriptors] resourceId is required')
|
|
61
|
+
}
|
|
55
62
|
|
|
56
63
|
log.info(`[media-tasks/tasks/visual-content-descriptors] starting for resource ${resourceId}`)
|
|
57
64
|
|