@ossy/media-tasks 3.15.0 → 3.17.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,44 @@
1
+ /**
2
+ * Resolve workspace-configured placement folders for a task output.
3
+ * Uses `@ossy/automation/actions/get-task-instance` (string id — no package import).
4
+ *
5
+ * @param {{ invoke: Function }} sdk
6
+ * @param {string} taskId
7
+ * @param {string} outputName
8
+ * @returns {Promise<string[]>}
9
+ */
10
+ export async function resolveConfiguredPlacements (sdk, taskId, outputName) {
11
+ if (!sdk?.invoke || !taskId || !outputName) return []
12
+ try {
13
+ const instance = await sdk.invoke('@ossy/automation/actions/get-task-instance', { taskId })
14
+ const config = instance?.outputs?.[outputName]
15
+ const fromList = Array.isArray(config?.placementLocations)
16
+ ? config.placementLocations
17
+ : null
18
+ const raw = fromList ?? config?.placementLocation
19
+ const list = Array.isArray(raw) ? raw : (typeof raw === 'string' ? [raw] : [])
20
+ const seen = new Set()
21
+ const result = []
22
+ for (const entry of list) {
23
+ const trimmed = typeof entry === 'string' ? entry.trim() : ''
24
+ if (!trimmed || seen.has(trimmed)) continue
25
+ seen.add(trimmed)
26
+ result.push(trimmed)
27
+ }
28
+ return result
29
+ } catch {
30
+ return []
31
+ }
32
+ }
33
+
34
+ /**
35
+ * @deprecated Use {@link resolveConfiguredPlacements}
36
+ * @param {{ invoke: Function }} sdk
37
+ * @param {string} taskId
38
+ * @param {string} outputName
39
+ * @returns {Promise<string | null>}
40
+ */
41
+ export async function resolveConfiguredPlacement (sdk, taskId, outputName) {
42
+ const locations = await resolveConfiguredPlacements(sdk, taskId, outputName)
43
+ return locations[0] ?? null
44
+ }
@@ -0,0 +1,58 @@
1
+ import {
2
+ resolveConfiguredPlacement,
3
+ resolveConfiguredPlacements,
4
+ } from './resolve-configured-placement.js'
5
+
6
+ describe('resolveConfiguredPlacements', () => {
7
+ it('returns empty without sdk or ids', async () => {
8
+ expect(await resolveConfiguredPlacements(null, 't', 'o')).toEqual([])
9
+ expect(await resolveConfiguredPlacements({ invoke: async () => ({}) }, '', 'o')).toEqual([])
10
+ })
11
+
12
+ it('reads placementLocations from get-task-instance', async () => {
13
+ const calls = []
14
+ const sdk = {
15
+ invoke: async (action, payload) => {
16
+ calls.push([action, payload])
17
+ return {
18
+ outputs: {
19
+ palette: { placementLocations: ['/derived/colors/', ' /also/ ', '/derived/colors/'] },
20
+ },
21
+ }
22
+ },
23
+ }
24
+ await expect(resolveConfiguredPlacements(sdk, 'task-1', 'palette')).resolves.toEqual([
25
+ '/derived/colors/',
26
+ '/also/',
27
+ ])
28
+ expect(calls).toEqual([
29
+ ['@ossy/automation/actions/get-task-instance', { taskId: 'task-1' }],
30
+ ])
31
+ })
32
+
33
+ it('falls back to singular placementLocation', async () => {
34
+ await expect(resolveConfiguredPlacements({
35
+ invoke: async () => ({ outputs: { palette: { placementLocation: '/one/' } } }),
36
+ }, 't', 'palette')).resolves.toEqual(['/one/'])
37
+ })
38
+
39
+ it('returns empty when invoke fails or locations empty', async () => {
40
+ await expect(resolveConfiguredPlacements({
41
+ invoke: async () => { throw new Error('missing') },
42
+ }, 't', 'palette')).resolves.toEqual([])
43
+
44
+ await expect(resolveConfiguredPlacements({
45
+ invoke: async () => ({ outputs: { palette: { placementLocations: [' '] } } }),
46
+ }, 't', 'palette')).resolves.toEqual([])
47
+ })
48
+ })
49
+
50
+ describe('resolveConfiguredPlacement', () => {
51
+ it('returns the first location', async () => {
52
+ await expect(resolveConfiguredPlacement({
53
+ invoke: async () => ({
54
+ outputs: { palette: { placementLocations: ['/a/', '/b/'] } },
55
+ }),
56
+ }, 't', 'palette')).resolves.toBe('/a/')
57
+ })
58
+ })
@@ -0,0 +1,5 @@
1
+ export const metadata = {
2
+ id: '@ossy/media-tasks/actions/run-prompt',
3
+ access: 'workspace',
4
+ label: 'automation.task.run-prompt.title',
5
+ }
@@ -0,0 +1,126 @@
1
+ import { usageFromResponse } from './llm-usage.js'
2
+
3
+ /** Default OpenRouter model for freeform completions. */
4
+ export const DEFAULT_RUN_PROMPT_MODEL = 'openai/gpt-4o-mini'
5
+
6
+ /** Document schema whose `content.body` is the prompt context. */
7
+ export const MARKDOWN_DOCUMENT_TYPE = '@ossy/resources/schema/markdown'
8
+
9
+ /** Cap resource text included beside the prompt so one file cannot fill the context. */
10
+ export const MAX_PROMPT_CONTEXT_CHARS = 12_000
11
+
12
+ /**
13
+ * @param {string} [contentType]
14
+ * @returns {boolean}
15
+ */
16
+ export function isTextContentType (contentType) {
17
+ const mime = String(contentType || '').split(';')[0].trim().toLowerCase()
18
+ if (!mime) return false
19
+ return mime.startsWith('text/')
20
+ || mime === 'application/json'
21
+ || mime === 'application/xml'
22
+ || mime.endsWith('+json')
23
+ || mime.endsWith('+xml')
24
+ }
25
+
26
+ /**
27
+ * @param {unknown} temperature
28
+ * @returns {number | undefined}
29
+ */
30
+ export function clampTemperature (temperature) {
31
+ if (temperature == null || temperature === '') return undefined
32
+ const n = typeof temperature === 'number' ? temperature : Number(temperature)
33
+ if (!Number.isFinite(n)) return undefined
34
+ return Math.min(2, Math.max(0, n))
35
+ }
36
+
37
+ /**
38
+ * True for a Markdown document (text in `content.body`, not a stored file).
39
+ *
40
+ * @param {object | null | undefined} resource
41
+ * @returns {boolean}
42
+ */
43
+ export function isMarkdownDocument (resource) {
44
+ return resource?.type === MARKDOWN_DOCUMENT_TYPE
45
+ }
46
+
47
+ /**
48
+ * @param {object | null | undefined} resource
49
+ * @returns {string}
50
+ */
51
+ export function markdownDocumentBody (resource) {
52
+ const body = resource?.content?.body
53
+ return typeof body === 'string' ? body : ''
54
+ }
55
+
56
+ /**
57
+ * Short context block prepended under the workspace prompt.
58
+ *
59
+ * @param {{ name?: string, contentType?: string, text?: string }} resource
60
+ * @returns {string}
61
+ */
62
+ export function resourceContextText ({ name, contentType, text } = {}) {
63
+ const lines = []
64
+ if (name) lines.push(`Name: ${name}`)
65
+ if (contentType) lines.push(`Content type: ${contentType}`)
66
+ const body = typeof text === 'string' ? text.trim() : ''
67
+ if (body) {
68
+ const clipped = body.length > MAX_PROMPT_CONTEXT_CHARS
69
+ ? `${body.slice(0, MAX_PROMPT_CONTEXT_CHARS)}…`
70
+ : body
71
+ lines.push('', clipped)
72
+ }
73
+ return lines.join('\n').trim()
74
+ }
75
+
76
+ /**
77
+ * @param {import('openai').OpenAI} client
78
+ * @param {{ prompt: string, model?: string, temperature?: number, contextText?: string, imageUrl?: string }} opts
79
+ * @returns {Promise<{
80
+ * text: string,
81
+ * model: string,
82
+ * usage: { model: string, promptTokens: number | null, completionTokens: number | null, totalTokens: number | null },
83
+ * }>}
84
+ */
85
+ export async function runOpenRouterPrompt (client, opts = {}) {
86
+ if (!client?.chat?.completions?.create) {
87
+ throw new Error('openrouter client is required')
88
+ }
89
+ const prompt = typeof opts.prompt === 'string' ? opts.prompt.trim() : ''
90
+ if (!prompt) {
91
+ throw new Error('prompt is required')
92
+ }
93
+
94
+ const model = opts.model || DEFAULT_RUN_PROMPT_MODEL
95
+ const temperature = clampTemperature(opts.temperature)
96
+ const contextText = typeof opts.contextText === 'string' ? opts.contextText.trim() : ''
97
+ const text = contextText ? `${prompt}\n\n---\n${contextText}` : prompt
98
+ const imageUrl = typeof opts.imageUrl === 'string' && opts.imageUrl ? opts.imageUrl : ''
99
+
100
+ /** @type {string | Array<{ type: string, text?: string, image_url?: { url: string } }>} */
101
+ const content = imageUrl
102
+ ? [
103
+ { type: 'text', text },
104
+ { type: 'image_url', image_url: { url: imageUrl } },
105
+ ]
106
+ : text
107
+
108
+ const response = await client.chat.completions.create({
109
+ model,
110
+ ...(temperature != null ? { temperature } : {}),
111
+ messages: [{ role: 'user', content }],
112
+ })
113
+
114
+ const completion = response?.choices?.[0]?.message?.content
115
+ if (typeof completion !== 'string' || !completion.trim()) {
116
+ throw new Error('run-prompt: model returned an empty completion')
117
+ }
118
+
119
+ const usage = usageFromResponse(response, model)
120
+
121
+ return {
122
+ text: completion,
123
+ model: usage.model,
124
+ usage,
125
+ }
126
+ }
@@ -0,0 +1,144 @@
1
+ import {
2
+ DEFAULT_RUN_PROMPT_MODEL,
3
+ MARKDOWN_DOCUMENT_TYPE,
4
+ clampTemperature,
5
+ isMarkdownDocument,
6
+ isTextContentType,
7
+ markdownDocumentBody,
8
+ resourceContextText,
9
+ runOpenRouterPrompt,
10
+ } from './run-prompt.js'
11
+
12
+ describe('isTextContentType', () => {
13
+ it('accepts text, json, and xml', () => {
14
+ expect(isTextContentType('text/plain')).toBe(true)
15
+ expect(isTextContentType('text/markdown; charset=utf-8')).toBe(true)
16
+ expect(isTextContentType('application/json')).toBe(true)
17
+ expect(isTextContentType('application/ld+json')).toBe(true)
18
+ expect(isTextContentType('image/png')).toBe(false)
19
+ })
20
+ })
21
+
22
+ describe('isMarkdownDocument', () => {
23
+ it('recognises the markdown schema and reads its body', () => {
24
+ const resource = {
25
+ type: MARKDOWN_DOCUMENT_TYPE,
26
+ name: 'notes',
27
+ content: { body: '# Hello' },
28
+ }
29
+ expect(isMarkdownDocument(resource)).toBe(true)
30
+ expect(markdownDocumentBody(resource)).toBe('# Hello')
31
+ expect(isMarkdownDocument({
32
+ type: '@ossy/platform/schema/file',
33
+ content: { ContentType: 'text/plain', Key: 'abc' },
34
+ })).toBe(false)
35
+ })
36
+ })
37
+
38
+ describe('resourceContextText', () => {
39
+ it('includes name, type, and clipped body', () => {
40
+ const text = resourceContextText({
41
+ name: 'notes.md',
42
+ contentType: 'text/markdown',
43
+ text: 'hello',
44
+ })
45
+ expect(text).toContain('Name: notes.md')
46
+ expect(text).toContain('hello')
47
+ })
48
+ })
49
+
50
+ describe('clampTemperature', () => {
51
+ it('keeps 0 and clamps the range', () => {
52
+ expect(clampTemperature(0)).toBe(0)
53
+ expect(clampTemperature(3)).toBe(2)
54
+ expect(clampTemperature(-1)).toBe(0)
55
+ expect(clampTemperature(undefined)).toBeUndefined()
56
+ })
57
+ })
58
+
59
+ describe('runOpenRouterPrompt', () => {
60
+ it('sends the prompt and optional temperature, and returns the completion', async () => {
61
+ const calls = []
62
+ const client = {
63
+ chat: {
64
+ completions: {
65
+ create: async (args) => {
66
+ calls.push(args)
67
+ return {
68
+ model: 'openai/gpt-4o-mini',
69
+ choices: [{ message: { content: 'A short title' } }],
70
+ }
71
+ },
72
+ },
73
+ },
74
+ }
75
+
76
+ const result = await runOpenRouterPrompt(client, {
77
+ prompt: 'Name this file',
78
+ temperature: 0.2,
79
+ contextText: 'Name: a.png',
80
+ })
81
+
82
+ expect(result).toEqual({
83
+ text: 'A short title',
84
+ model: 'openai/gpt-4o-mini',
85
+ usage: { model: 'openai/gpt-4o-mini', promptTokens: null, completionTokens: null, totalTokens: null },
86
+ })
87
+ expect(calls[0].model).toBe(DEFAULT_RUN_PROMPT_MODEL)
88
+ expect(calls[0].temperature).toBe(0.2)
89
+ expect(calls[0].messages[0].content).toContain('Name this file')
90
+ expect(calls[0].messages[0].content).toContain('Name: a.png')
91
+ })
92
+
93
+ it('captures token usage from the OpenRouter response', async () => {
94
+ const client = {
95
+ chat: {
96
+ completions: {
97
+ create: async () => ({
98
+ model: 'openai/gpt-4o-mini-2024-07-18',
99
+ usage: { prompt_tokens: 42, completion_tokens: 8, total_tokens: 50 },
100
+ choices: [{ message: { content: 'ok' } }],
101
+ }),
102
+ },
103
+ },
104
+ }
105
+
106
+ const result = await runOpenRouterPrompt(client, { prompt: 'Name this file' })
107
+
108
+ expect(result.usage).toEqual({
109
+ model: 'openai/gpt-4o-mini-2024-07-18',
110
+ promptTokens: 42,
111
+ completionTokens: 8,
112
+ totalTokens: 50,
113
+ })
114
+ })
115
+
116
+ it('attaches an image as a content part', async () => {
117
+ const calls = []
118
+ const client = {
119
+ chat: {
120
+ completions: {
121
+ create: async (args) => {
122
+ calls.push(args)
123
+ return { choices: [{ message: { content: 'ok' } }] }
124
+ },
125
+ },
126
+ },
127
+ }
128
+
129
+ await runOpenRouterPrompt(client, {
130
+ prompt: 'Describe',
131
+ model: 'google/gemini-2.5-flash-lite',
132
+ imageUrl: 'data:image/png;base64,aaaa',
133
+ })
134
+
135
+ expect(calls[0].model).toBe('google/gemini-2.5-flash-lite')
136
+ expect(calls[0].temperature).toBeUndefined()
137
+ expect(calls[0].messages[0].content[1].image_url.url).toBe('data:image/png;base64,aaaa')
138
+ })
139
+
140
+ it('rejects an empty prompt', async () => {
141
+ const client = { chat: { completions: { create: async () => ({}) } } }
142
+ await expect(runOpenRouterPrompt(client, { prompt: ' ' })).rejects.toThrow(/prompt/)
143
+ })
144
+ })
@@ -0,0 +1,197 @@
1
+ import { createLogger } from '@ossy/observability'
2
+ import { StorageClient } from '@ossy/platform'
3
+ import { taskArtifactObjectKey } from '@ossy/platform/storage-keys'
4
+ import { assertLlmUsageBudget, resolveWorkspaceIdForBudget } from '@ossy/platform/metering'
5
+ import { GetResource, placeTaskArtifacts } from '@ossy/resources'
6
+ import { loadSourceBuffer } from './load-source-buffer.js'
7
+ import { toImageDataUrl } from './visual-content-descriptors.js'
8
+ import {
9
+ DEFAULT_RUN_PROMPT_MODEL,
10
+ MARKDOWN_DOCUMENT_TYPE,
11
+ isMarkdownDocument,
12
+ isTextContentType,
13
+ markdownDocumentBody,
14
+ resourceContextText,
15
+ runOpenRouterPrompt,
16
+ } from './run-prompt.js'
17
+ import { resolveConfiguredPlacements } from './resolve-configured-placement.js'
18
+
19
+ const log = createLogger('media-tasks')
20
+
21
+ const TASK_ID = '@ossy/media-tasks/tasks/run-prompt'
22
+ // Freeform text/md completion → platform schema (see docs/concepts/TaskOutput.md),
23
+ // not a feature-owned one. Structured JSON outputs (e.g. visual-content-descriptors)
24
+ // still declare their own schema.
25
+ const SCHEMA_ID = '@ossy/platform/schema/text-completion'
26
+ const CONCEPT = 'run-prompt'
27
+ const OUTPUT = {
28
+ name: 'completion',
29
+ kind: 'location',
30
+ schemaId: SCHEMA_ID,
31
+ contentType: 'application/json',
32
+ }
33
+
34
+ export const metadata = {
35
+ id: TASK_ID,
36
+ triggers: [
37
+ {
38
+ type: '@ossy/platform/schema/file',
39
+ event: 'Created',
40
+ match: { 'content.ContentType': 'text/*' },
41
+ },
42
+ {
43
+ type: '@ossy/platform/schema/file',
44
+ event: 'Created',
45
+ match: { 'content.ContentType': 'image/*' },
46
+ },
47
+ {
48
+ type: '@ossy/platform/schema/file',
49
+ event: 'Created',
50
+ match: { 'content.ContentType': 'application/json' },
51
+ },
52
+ {
53
+ type: MARKDOWN_DOCUMENT_TYPE,
54
+ event: 'Created',
55
+ },
56
+ ],
57
+ outputs: [OUTPUT],
58
+ inputs: [
59
+ {
60
+ name: 'prompt',
61
+ type: 'text',
62
+ multiline: true,
63
+ sources: ['payload', 'config'],
64
+ required: true,
65
+ description: 'Instructions for the model',
66
+ },
67
+ {
68
+ name: 'model',
69
+ type: 'text',
70
+ default: DEFAULT_RUN_PROMPT_MODEL,
71
+ sources: ['payload', 'config'],
72
+ description: 'OpenRouter model slug',
73
+ },
74
+ {
75
+ name: 'temperature',
76
+ type: 'number',
77
+ sources: ['payload', 'config'],
78
+ description: 'Sampling temperature from 0 to 2. Leave empty for the provider default.',
79
+ },
80
+ ],
81
+ configurable: true,
82
+ }
83
+
84
+ export async function run ({ event, payload, req, sdk, integrations, inputs }) {
85
+ // Fail closed before spending anything — ADR 0017's first QuotaPolicy.
86
+ await assertLlmUsageBudget({
87
+ workspaceId: resolveWorkspaceIdForBudget({ event, payload, req }),
88
+ })
89
+
90
+ const openrouter = integrations?.get?.('openrouter')
91
+ if (!openrouter) {
92
+ log.warn('[media-tasks/tasks/run-prompt] openrouter integration not connected')
93
+ throw Object.assign(new Error('OpenRouter integration not available'), { status: 503 })
94
+ }
95
+
96
+ const prompt = typeof inputs?.prompt === 'string' ? inputs.prompt : ''
97
+ const resourceId = event?.resourceId || payload?.resourceId || null
98
+ log.info(`[media-tasks/tasks/run-prompt] starting${resourceId ? ` for resource ${resourceId}` : ''}`)
99
+
100
+ let resource = null
101
+ let contextText = ''
102
+ let imageUrl
103
+ if (resourceId) {
104
+ resource = await sdk.invoke(GetResource, { resourceId })
105
+ const loaded = await loadResourceContext(resource)
106
+ contextText = loaded.contextText
107
+ imageUrl = loaded.imageUrl
108
+ }
109
+
110
+ const completion = await runOpenRouterPrompt(openrouter, {
111
+ prompt,
112
+ model: inputs?.model,
113
+ temperature: inputs?.temperature,
114
+ contextText,
115
+ imageUrl,
116
+ })
117
+ const body = {
118
+ text: completion.text,
119
+ model: completion.model,
120
+ prompt,
121
+ }
122
+
123
+ if (!resource?.id) {
124
+ return { output: OUTPUT.name, completion: body, placedResourceIds: [], usage: completion.usage }
125
+ }
126
+
127
+ const objectKey = taskArtifactObjectKey(resource.id, CONCEPT, OUTPUT.name)
128
+ await StorageClient.save(objectKey, Buffer.from(JSON.stringify(body), 'utf8'))
129
+ log.info(`[media-tasks/tasks/run-prompt] saved ${objectKey}`)
130
+
131
+ const placementLocations = await resolveConfiguredPlacements(sdk, TASK_ID, OUTPUT.name)
132
+ const placed = await placeTaskArtifacts({
133
+ sdk,
134
+ sourceResource: resource,
135
+ taskId: TASK_ID,
136
+ output: OUTPUT,
137
+ body,
138
+ artifactKey: objectKey,
139
+ placementLocations,
140
+ })
141
+ if (placed.length) {
142
+ log.info(`[media-tasks/tasks/run-prompt] placed ${placed.length} resource(s)`)
143
+ }
144
+
145
+ return {
146
+ output: OUTPUT.name,
147
+ key: objectKey,
148
+ completion: body,
149
+ placedResourceIds: placed.map((item) => item.id),
150
+ placedResourceId: placed[0]?.id ?? null,
151
+ usage: completion.usage,
152
+ }
153
+ }
154
+
155
+ /**
156
+ * @param {object | null | undefined} resource
157
+ * @returns {Promise<{ contextText: string, imageUrl?: string }>}
158
+ */
159
+ async function loadResourceContext (resource) {
160
+ const name = resource?.name
161
+ if (isMarkdownDocument(resource)) {
162
+ return {
163
+ contextText: resourceContextText({
164
+ name,
165
+ contentType: MARKDOWN_DOCUMENT_TYPE,
166
+ text: markdownDocumentBody(resource),
167
+ }),
168
+ }
169
+ }
170
+
171
+ const contentType = resource?.content?.ContentType || ''
172
+ const key = resource?.content?.Key
173
+ if (!key) {
174
+ return { contextText: resourceContextText({ name, contentType }) }
175
+ }
176
+
177
+ if (String(contentType).startsWith('image/')) {
178
+ const buffer = await loadSourceBuffer(key)
179
+ return {
180
+ contextText: resourceContextText({ name, contentType }),
181
+ ...(buffer ? { imageUrl: toImageDataUrl(buffer, contentType) } : {}),
182
+ }
183
+ }
184
+
185
+ if (isTextContentType(contentType)) {
186
+ const buffer = await loadSourceBuffer(key)
187
+ return {
188
+ contextText: resourceContextText({
189
+ name,
190
+ contentType,
191
+ text: buffer ? buffer.toString('utf8') : '',
192
+ }),
193
+ }
194
+ }
195
+
196
+ return { contextText: resourceContextText({ name, contentType }) }
197
+ }
@@ -0,0 +1,79 @@
1
+ import { usageFromResponse } from './llm-usage.js'
2
+
3
+ /** Default OpenRouter model slug (vision-capable). */
4
+ export const DEFAULT_VCD_MODEL = 'openai/gpt-4o-mini'
5
+
6
+ const SYSTEM_PROMPT = `You will be provided with images, and your task is to create a json object including the following properties:
7
+ - title: Short attention grabbing title that describes the image
8
+ - description: A short text describing the image, suitable for SEO purposes
9
+ - tags: an array of tags describing the image that will be used for categorization and SEO purposes
10
+ - alt: A short text that describes the image, suitable for alt text`
11
+
12
+ /**
13
+ * Build a data URL so the model provider can read the image without fetching
14
+ * our storage URL (localhost / auth-gated URLs fail with OpenRouter 400).
15
+ *
16
+ * @param {Buffer} buffer
17
+ * @param {string} [contentType]
18
+ * @returns {string}
19
+ */
20
+ export function toImageDataUrl (buffer, contentType = 'application/octet-stream') {
21
+ if (!Buffer.isBuffer(buffer) || buffer.length === 0) {
22
+ throw new Error('image buffer is required')
23
+ }
24
+ const mime = String(contentType || 'application/octet-stream').split(';')[0].trim() || 'application/octet-stream'
25
+ return `data:${mime};base64,${buffer.toString('base64')}`
26
+ }
27
+
28
+ /**
29
+ * Ask an OpenRouter (OpenAI-compatible) client for SEO descriptors of an image.
30
+ *
31
+ * @param {import('openai').OpenAI} client
32
+ * @param {string} imageUrl Absolute http(s) URL or data: URL of the image
33
+ * @param {{ model?: string }} [opts]
34
+ * @returns {Promise<{
35
+ * title: string | null,
36
+ * description: string | null,
37
+ * tags: string[],
38
+ * alt: string | null,
39
+ * usage: { model: string, promptTokens: number | null, completionTokens: number | null, totalTokens: number | null },
40
+ * }>}
41
+ */
42
+ export async function getVisualContentDescriptors (client, imageUrl, opts = {}) {
43
+ if (!client?.chat?.completions?.create) {
44
+ throw new Error('openrouter client is required')
45
+ }
46
+ if (!imageUrl) {
47
+ throw new Error('imageUrl is required')
48
+ }
49
+
50
+ const model = opts.model || DEFAULT_VCD_MODEL
51
+ const response = await client.chat.completions.create({
52
+ model,
53
+ max_tokens: 4000,
54
+ response_format: { type: 'json_object' },
55
+ messages: [
56
+ { role: 'system', content: SYSTEM_PROMPT },
57
+ {
58
+ role: 'user',
59
+ content: [{ type: 'image_url', image_url: { url: imageUrl } }],
60
+ },
61
+ ],
62
+ })
63
+
64
+ const raw = response?.choices?.[0]?.message?.content ?? '{}'
65
+ let parsed
66
+ try {
67
+ parsed = JSON.parse(raw)
68
+ } catch {
69
+ throw new Error('visual-content-descriptors: model returned non-JSON content')
70
+ }
71
+
72
+ return {
73
+ title: typeof parsed.title === 'string' ? parsed.title : null,
74
+ description: typeof parsed.description === 'string' ? parsed.description : null,
75
+ tags: Array.isArray(parsed.tags) ? parsed.tags.filter((t) => typeof t === 'string') : [],
76
+ alt: typeof parsed.alt === 'string' ? parsed.alt : null,
77
+ usage: usageFromResponse(response, model),
78
+ }
79
+ }
@@ -1,20 +1,31 @@
1
+ import { TASK_OUTPUT_PROVENANCE_FIELDS } from '@ossy/schema'
2
+
1
3
  export default {
2
4
  name: 'Visual Content Descriptors',
3
5
  id: '@ossy/media-tasks/schema/visual-content-descriptors',
4
6
  categoryName: 'Media processing',
5
- icon: 'attribution',
7
+ icon: 'details-more',
6
8
  fields: [
7
9
  {
8
- name: 'resourceId',
10
+ name: 'title',
9
11
  type: 'text',
12
+ description: 'Suggested resource name',
10
13
  },
11
14
  {
12
- name: 'status',
13
- type: 'text',
15
+ name: 'description',
16
+ type: 'textarea',
17
+ description: 'Suggested SEO description',
14
18
  },
15
19
  {
16
- name: 'result',
20
+ name: 'tags',
17
21
  type: 'textarea',
22
+ description: 'Suggested tags',
23
+ },
24
+ {
25
+ name: 'alt',
26
+ type: 'text',
27
+ description: 'Suggested alt text',
18
28
  },
29
+ ...TASK_OUTPUT_PROVENANCE_FIELDS,
19
30
  ],
20
31
  }