@ossy/media-tasks 3.15.0 → 3.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +43 -24
- package/package.json +6 -4
- package/src/extract-colors.js +71 -0
- package/src/extract-colors.schema.js +21 -0
- package/src/extract-colors.spec.js +44 -0
- package/src/extract-colors.task.js +79 -0
- package/src/llm-usage.js +21 -0
- package/src/llm-usage.spec.js +41 -0
- package/src/load-source-buffer.js +17 -0
- package/src/openrouter.integration.js +30 -0
- package/src/openrouter.integration.spec.js +37 -0
- package/src/resize-common-web.schema.js +16 -5
- package/src/resize-common-web.task.js +59 -31
- package/src/resolve-configured-placement.js +44 -0
- package/src/resolve-configured-placement.spec.js +58 -0
- package/src/run-prompt.action.js +5 -0
- package/src/run-prompt.js +126 -0
- package/src/run-prompt.spec.js +144 -0
- package/src/run-prompt.task.js +197 -0
- package/src/visual-content-descriptors.js +79 -0
- package/src/visual-content-descriptors.schema.js +16 -5
- package/src/visual-content-descriptors.spec.js +130 -0
- package/src/visual-content-descriptors.task.js +99 -13
- package/src/openai.integration.js +0 -74
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
import {
|
|
2
|
+
DEFAULT_VCD_MODEL,
|
|
3
|
+
getVisualContentDescriptors,
|
|
4
|
+
toImageDataUrl,
|
|
5
|
+
} from './visual-content-descriptors.js'
|
|
6
|
+
|
|
7
|
+
describe('toImageDataUrl', () => {
|
|
8
|
+
it('encodes buffer as a data URL with mime type', () => {
|
|
9
|
+
const buf = Buffer.from([0x89, 0x50, 0x4e, 0x47])
|
|
10
|
+
expect(toImageDataUrl(buf, 'image/png')).toBe(
|
|
11
|
+
`data:image/png;base64,${buf.toString('base64')}`,
|
|
12
|
+
)
|
|
13
|
+
})
|
|
14
|
+
|
|
15
|
+
it('rejects empty buffers', () => {
|
|
16
|
+
expect(() => toImageDataUrl(Buffer.alloc(0))).toThrow(/image buffer/)
|
|
17
|
+
})
|
|
18
|
+
})
|
|
19
|
+
|
|
20
|
+
describe('getVisualContentDescriptors', () => {
|
|
21
|
+
it('calls chat.completions once with json_object and OpenRouter model slug', async () => {
|
|
22
|
+
const calls = []
|
|
23
|
+
const client = {
|
|
24
|
+
chat: {
|
|
25
|
+
completions: {
|
|
26
|
+
create: async (args) => {
|
|
27
|
+
calls.push(args)
|
|
28
|
+
return {
|
|
29
|
+
choices: [{
|
|
30
|
+
message: {
|
|
31
|
+
content: JSON.stringify({
|
|
32
|
+
title: 'Sunset',
|
|
33
|
+
description: 'Orange sky over water',
|
|
34
|
+
tags: ['sunset', 'nature'],
|
|
35
|
+
alt: 'Sunset over water',
|
|
36
|
+
}),
|
|
37
|
+
},
|
|
38
|
+
}],
|
|
39
|
+
}
|
|
40
|
+
},
|
|
41
|
+
},
|
|
42
|
+
},
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
const dataUrl = 'data:image/jpeg;base64,abc'
|
|
46
|
+
await expect(getVisualContentDescriptors(client, dataUrl))
|
|
47
|
+
.resolves.toEqual({
|
|
48
|
+
title: 'Sunset',
|
|
49
|
+
description: 'Orange sky over water',
|
|
50
|
+
tags: ['sunset', 'nature'],
|
|
51
|
+
alt: 'Sunset over water',
|
|
52
|
+
usage: { model: DEFAULT_VCD_MODEL, promptTokens: null, completionTokens: null, totalTokens: null },
|
|
53
|
+
})
|
|
54
|
+
|
|
55
|
+
expect(calls).toHaveLength(1)
|
|
56
|
+
expect(calls[0]).toMatchObject({
|
|
57
|
+
model: DEFAULT_VCD_MODEL,
|
|
58
|
+
response_format: { type: 'json_object' },
|
|
59
|
+
messages: [
|
|
60
|
+
{ role: 'system', content: expect.any(String) },
|
|
61
|
+
{
|
|
62
|
+
role: 'user',
|
|
63
|
+
content: [{ type: 'image_url', image_url: { url: dataUrl } }],
|
|
64
|
+
},
|
|
65
|
+
],
|
|
66
|
+
})
|
|
67
|
+
})
|
|
68
|
+
|
|
69
|
+
it('normalizes missing fields and non-string tags', async () => {
|
|
70
|
+
const client = {
|
|
71
|
+
chat: {
|
|
72
|
+
completions: {
|
|
73
|
+
create: async () => ({
|
|
74
|
+
choices: [{ message: { content: '{"title":1,"tags":["ok",2]}' } }],
|
|
75
|
+
}),
|
|
76
|
+
},
|
|
77
|
+
},
|
|
78
|
+
}
|
|
79
|
+
await expect(getVisualContentDescriptors(client, 'data:image/png;base64,x'))
|
|
80
|
+
.resolves.toEqual({
|
|
81
|
+
title: null,
|
|
82
|
+
description: null,
|
|
83
|
+
tags: ['ok'],
|
|
84
|
+
alt: null,
|
|
85
|
+
usage: { model: DEFAULT_VCD_MODEL, promptTokens: null, completionTokens: null, totalTokens: null },
|
|
86
|
+
})
|
|
87
|
+
})
|
|
88
|
+
|
|
89
|
+
it('captures token usage and the model that actually answered', async () => {
|
|
90
|
+
const client = {
|
|
91
|
+
chat: {
|
|
92
|
+
completions: {
|
|
93
|
+
create: async () => ({
|
|
94
|
+
model: 'openai/gpt-4o-mini-2024-07-18',
|
|
95
|
+
usage: { prompt_tokens: 120, completion_tokens: 40, total_tokens: 160 },
|
|
96
|
+
choices: [{ message: { content: '{"title":"Sunset"}' } }],
|
|
97
|
+
}),
|
|
98
|
+
},
|
|
99
|
+
},
|
|
100
|
+
}
|
|
101
|
+
await expect(getVisualContentDescriptors(client, 'data:image/png;base64,x'))
|
|
102
|
+
.resolves.toMatchObject({
|
|
103
|
+
usage: {
|
|
104
|
+
model: 'openai/gpt-4o-mini-2024-07-18',
|
|
105
|
+
promptTokens: 120,
|
|
106
|
+
completionTokens: 40,
|
|
107
|
+
totalTokens: 160,
|
|
108
|
+
},
|
|
109
|
+
})
|
|
110
|
+
})
|
|
111
|
+
|
|
112
|
+
it('rejects missing client or imageUrl', async () => {
|
|
113
|
+
await expect(getVisualContentDescriptors(null, 'data:image/png;base64,x'))
|
|
114
|
+
.rejects.toThrow(/openrouter client/)
|
|
115
|
+
await expect(getVisualContentDescriptors({ chat: { completions: { create: async () => ({}) } } }, ''))
|
|
116
|
+
.rejects.toThrow(/imageUrl/)
|
|
117
|
+
})
|
|
118
|
+
|
|
119
|
+
it('rejects non-JSON model content', async () => {
|
|
120
|
+
const client = {
|
|
121
|
+
chat: {
|
|
122
|
+
completions: {
|
|
123
|
+
create: async () => ({ choices: [{ message: { content: 'not-json' } }] }),
|
|
124
|
+
},
|
|
125
|
+
},
|
|
126
|
+
}
|
|
127
|
+
await expect(getVisualContentDescriptors(client, 'data:image/png;base64,x'))
|
|
128
|
+
.rejects.toThrow(/non-JSON/)
|
|
129
|
+
})
|
|
130
|
+
})
|
|
@@ -1,8 +1,30 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import {
|
|
1
|
+
import { createLogger } from '@ossy/observability'
|
|
2
|
+
import { StorageClient } from '@ossy/platform'
|
|
3
|
+
import { taskArtifactObjectKey } from '@ossy/platform/storage-keys'
|
|
4
|
+
import { assertLlmUsageBudget, resolveWorkspaceIdForBudget } from '@ossy/platform/metering'
|
|
5
|
+
import { GetResource, placeTaskArtifacts } from '@ossy/resources'
|
|
6
|
+
import { loadSourceBuffer } from './load-source-buffer.js'
|
|
7
|
+
import {
|
|
8
|
+
DEFAULT_VCD_MODEL,
|
|
9
|
+
getVisualContentDescriptors,
|
|
10
|
+
toImageDataUrl,
|
|
11
|
+
} from './visual-content-descriptors.js'
|
|
12
|
+
import { resolveConfiguredPlacements } from './resolve-configured-placement.js'
|
|
13
|
+
|
|
14
|
+
const log = createLogger('media-tasks')
|
|
15
|
+
|
|
16
|
+
const TASK_ID = '@ossy/media-tasks/tasks/visual-content-descriptors'
|
|
17
|
+
const SCHEMA_ID = '@ossy/media-tasks/schema/visual-content-descriptors'
|
|
18
|
+
const CONCEPT = 'visual-content-descriptors'
|
|
19
|
+
const OUTPUT = {
|
|
20
|
+
name: 'descriptors',
|
|
21
|
+
kind: 'location',
|
|
22
|
+
schemaId: SCHEMA_ID,
|
|
23
|
+
contentType: 'application/json',
|
|
24
|
+
}
|
|
3
25
|
|
|
4
26
|
export const metadata = {
|
|
5
|
-
id:
|
|
27
|
+
id: TASK_ID,
|
|
6
28
|
triggers: [
|
|
7
29
|
{
|
|
8
30
|
type: '@ossy/platform/schema/file',
|
|
@@ -15,20 +37,84 @@ export const metadata = {
|
|
|
15
37
|
match: { 'content.ContentType': 'video/*' },
|
|
16
38
|
},
|
|
17
39
|
],
|
|
40
|
+
outputs: [OUTPUT],
|
|
41
|
+
inputs: [
|
|
42
|
+
{
|
|
43
|
+
name: 'model',
|
|
44
|
+
type: 'text',
|
|
45
|
+
default: DEFAULT_VCD_MODEL,
|
|
46
|
+
sources: ['payload', 'config'],
|
|
47
|
+
description: 'OpenRouter model slug',
|
|
48
|
+
},
|
|
49
|
+
],
|
|
50
|
+
configurable: true,
|
|
18
51
|
}
|
|
19
52
|
|
|
20
|
-
export async function run ({ event, sdk }) {
|
|
53
|
+
export async function run ({ event, payload, req, sdk, integrations, inputs }) {
|
|
21
54
|
const resourceId = event.resourceId
|
|
22
55
|
|
|
56
|
+
log.info(`[media-tasks/tasks/visual-content-descriptors] starting for resource ${resourceId}`)
|
|
57
|
+
|
|
58
|
+
// Fail closed before spending anything — ADR 0017's first QuotaPolicy.
|
|
59
|
+
await assertLlmUsageBudget({
|
|
60
|
+
workspaceId: resolveWorkspaceIdForBudget({ event, payload, req }),
|
|
61
|
+
})
|
|
62
|
+
|
|
63
|
+
const openrouter = integrations?.get?.('openrouter')
|
|
64
|
+
if (!openrouter) {
|
|
65
|
+
log.warn('[media-tasks/tasks/visual-content-descriptors] openrouter integration not connected')
|
|
66
|
+
throw Object.assign(new Error('OpenRouter integration not available'), { status: 503 })
|
|
67
|
+
}
|
|
68
|
+
|
|
23
69
|
const resource = await sdk.invoke(GetResource, { resourceId })
|
|
24
|
-
const
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
70
|
+
const key = resource?.content?.Key
|
|
71
|
+
if (!key) {
|
|
72
|
+
throw new Error(`[media-tasks/tasks/visual-content-descriptors] missing storage key for ${resourceId}`)
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
const contentType = resource?.content?.ContentType || 'application/octet-stream'
|
|
76
|
+
if (!String(contentType).startsWith('image/')) {
|
|
77
|
+
throw new Error(
|
|
78
|
+
`[media-tasks/tasks/visual-content-descriptors] unsupported ContentType ${contentType} for ${resourceId}`,
|
|
79
|
+
)
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
const sourceBuffer = await loadSourceBuffer(key)
|
|
83
|
+
if (!sourceBuffer) {
|
|
84
|
+
throw new Error(`[media-tasks/tasks/visual-content-descriptors] missing file bytes for ${resourceId}`)
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
// Inline bytes — OpenRouter cannot fetch localhost / auth-gated storage URLs.
|
|
88
|
+
const imageUrl = toImageDataUrl(sourceBuffer, contentType)
|
|
89
|
+
// `usage` is metering-only — never persisted on the artifact/placed document.
|
|
90
|
+
const { usage, ...body } = await getVisualContentDescriptors(openrouter, imageUrl, {
|
|
91
|
+
model: inputs?.model,
|
|
92
|
+
})
|
|
93
|
+
const objectKey = taskArtifactObjectKey(resourceId, CONCEPT, OUTPUT.name)
|
|
94
|
+
await StorageClient.save(objectKey, Buffer.from(JSON.stringify(body), 'utf8'))
|
|
95
|
+
log.info(`[media-tasks/tasks/visual-content-descriptors] saved ${objectKey}`)
|
|
96
|
+
|
|
97
|
+
const placementLocations = await resolveConfiguredPlacements(sdk, TASK_ID, OUTPUT.name)
|
|
98
|
+
const placed = await placeTaskArtifacts({
|
|
99
|
+
sdk,
|
|
100
|
+
sourceResource: resource,
|
|
101
|
+
taskId: TASK_ID,
|
|
102
|
+
output: OUTPUT,
|
|
103
|
+
body,
|
|
104
|
+
artifactKey: objectKey,
|
|
105
|
+
placementLocations,
|
|
33
106
|
})
|
|
107
|
+
if (placed.length) {
|
|
108
|
+
log.info(`[media-tasks/tasks/visual-content-descriptors] placed ${placed.length} resource(s)`)
|
|
109
|
+
} else {
|
|
110
|
+
log.info(`[media-tasks/tasks/visual-content-descriptors] artifact only (no placement location configured)`)
|
|
111
|
+
}
|
|
112
|
+
return {
|
|
113
|
+
output: OUTPUT.name,
|
|
114
|
+
key: objectKey,
|
|
115
|
+
descriptors: body,
|
|
116
|
+
placedResourceIds: placed.map((r) => r.id),
|
|
117
|
+
placedResourceId: placed[0]?.id ?? null,
|
|
118
|
+
usage,
|
|
119
|
+
}
|
|
34
120
|
}
|
|
@@ -1,74 +0,0 @@
|
|
|
1
|
-
import OpenAiSDK from 'openai'
|
|
2
|
-
|
|
3
|
-
export const id = 'openai'
|
|
4
|
-
|
|
5
|
-
export const credentials = ['OPENAI_API_KEY']
|
|
6
|
-
|
|
7
|
-
/**
|
|
8
|
-
* Returns an OpenAI client configured from env vars.
|
|
9
|
-
* Tasks receive this via `integrations.get('openai')`.
|
|
10
|
-
*
|
|
11
|
-
* @param {{ env: NodeJS.ProcessEnv }} opts
|
|
12
|
-
* @returns {import('openai').OpenAI}
|
|
13
|
-
*/
|
|
14
|
-
export async function connect({ env }) {
|
|
15
|
-
return new OpenAiSDK({
|
|
16
|
-
apiKey: env.OPENAI_API_KEY,
|
|
17
|
-
organization: env.OPENAI_ORGANIZATION,
|
|
18
|
-
})
|
|
19
|
-
}
|
|
20
|
-
|
|
21
|
-
let _openai = null
|
|
22
|
-
function getOpenAi() {
|
|
23
|
-
if (!_openai) _openai = new OpenAiSDK({
|
|
24
|
-
apiKey: process.env.OPENAI_API_KEY,
|
|
25
|
-
organization: process.env.OPENAI_ORGANIZATION,
|
|
26
|
-
})
|
|
27
|
-
return _openai
|
|
28
|
-
}
|
|
29
|
-
|
|
30
|
-
export class OpenAi {
|
|
31
|
-
static async getVisualContentDescriptors(imageSrc) {
|
|
32
|
-
const systemPrompt = `
|
|
33
|
-
You will be provided with images, and your task is to create a json object including the following properties:
|
|
34
|
-
-title: Short attention grabbing title that describes the image
|
|
35
|
-
-description: A short text describing the image, suitable for SEO purposes
|
|
36
|
-
-tags: an array of tags describing the image that will be used for categorization and SEO purposes
|
|
37
|
-
-alt: A short text that describes the image, suitable for alt text
|
|
38
|
-
`
|
|
39
|
-
|
|
40
|
-
const response = await getOpenAi().chat.completions.create({
|
|
41
|
-
model: 'gpt-4o-mini',
|
|
42
|
-
max_tokens: 4000,
|
|
43
|
-
messages: [
|
|
44
|
-
{ role: 'system', content: systemPrompt },
|
|
45
|
-
{ role: 'user', content: [{ type: 'image_url', image_url: { url: imageSrc } }] },
|
|
46
|
-
],
|
|
47
|
-
})
|
|
48
|
-
|
|
49
|
-
const parsedResponse = await OpenAi.ensureJSONResponse(response)
|
|
50
|
-
const visualDescriptors = JSON.parse(
|
|
51
|
-
parsedResponse?.choices?.[0]?.message?.content ?? '{}'
|
|
52
|
-
)
|
|
53
|
-
return visualDescriptors
|
|
54
|
-
}
|
|
55
|
-
|
|
56
|
-
static ensureJSONResponse(response) {
|
|
57
|
-
const systemPrompt = `
|
|
58
|
-
You will be recieving a text string that contains a json object.
|
|
59
|
-
Your task is to parse the text string and convert the response into a json object.
|
|
60
|
-
`
|
|
61
|
-
|
|
62
|
-
const textString = response?.choices?.[0]?.message?.content
|
|
63
|
-
|
|
64
|
-
return getOpenAi().chat.completions.create({
|
|
65
|
-
model: 'gpt-4-turbo-preview',
|
|
66
|
-
max_tokens: 4000,
|
|
67
|
-
response_format: { type: 'json_object' },
|
|
68
|
-
messages: [
|
|
69
|
-
{ role: 'system', content: systemPrompt },
|
|
70
|
-
{ role: 'user', content: textString ?? '' },
|
|
71
|
-
],
|
|
72
|
-
})
|
|
73
|
-
}
|
|
74
|
-
}
|