mixdog 0.9.69 → 0.9.71
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +4 -1
- package/src/runtime/agent/orchestrator/session/manager/pending-messages.mjs +18 -4
- package/src/runtime/agent/orchestrator/session/manager/session-lifecycle.mjs +45 -12
- package/src/runtime/agent/orchestrator/session/store/listing.mjs +1 -0
- package/src/runtime/agent/orchestrator/session/store/paths-heartbeat.mjs +51 -18
- package/src/runtime/agent/orchestrator/session/store.mjs +3 -0
- package/src/runtime/channels/lib/owned-runtime.mjs +8 -1
- package/src/runtime/media/adapters/codex-image.mjs +109 -0
- package/src/runtime/media/adapters/gemini-image.mjs +68 -0
- package/src/runtime/media/adapters/gemini-video.mjs +119 -0
- package/src/runtime/media/adapters/xai-media.mjs +116 -0
- package/src/runtime/media/auth.mjs +51 -0
- package/src/runtime/media/index.mjs +18 -0
- package/src/runtime/media/jobs.mjs +174 -0
- package/src/runtime/media/lanes.mjs +230 -0
- package/src/runtime/media/renditions.mjs +203 -0
- package/src/runtime/media/store.mjs +356 -0
- package/src/runtime/media/store.test.mjs +103 -0
- package/src/runtime/media/upstream-error.mjs +39 -0
- package/src/session-runtime/channel-config-api.mjs +12 -1
- package/src/session-runtime/lifecycle-api.mjs +13 -4
- package/src/session-runtime/media-api.mjs +50 -0
- package/src/session-runtime/model-route-api.mjs +4 -1
- package/src/session-runtime/runtime-core.mjs +17 -1
- package/src/session-runtime/session-text.mjs +0 -1
- package/src/session-runtime/tool-catalog.mjs +32 -0
- package/src/session-runtime/warmup-schedulers.mjs +14 -1
- package/src/session-runtime/workflow-agents-api.mjs +29 -4
- package/src/tui/dist/index.mjs +21 -2
- package/src/tui/engine/session-api-ext.mjs +15 -0
- package/src/tui/engine.mjs +39 -1
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* xAI Imagine adapter (images + video) for the `grok-oauth` and `xai` lanes.
|
|
3
|
+
*
|
|
4
|
+
* Both lanes speak the same api.x.ai contract; only the bearer differs. Video
|
|
5
|
+
* is a submit + poll job: POST /videos/generations returns a request id and the
|
|
6
|
+
* poll route answers 202 while pending, 200 with a signed URL when done.
|
|
7
|
+
*/
|
|
8
|
+
import { resolveXaiAuth } from '../auth.mjs';
|
|
9
|
+
import { mediaError } from '../lanes.mjs';
|
|
10
|
+
import { upstreamError } from '../upstream-error.mjs';
|
|
11
|
+
|
|
12
|
+
const POLL_INTERVAL_MS = 4_000;
|
|
13
|
+
const START_TIMEOUT_MS = 60_000;
|
|
14
|
+
const TOTAL_TIMEOUT_MS = 900_000;
|
|
15
|
+
|
|
16
|
+
async function readError(res) {
|
|
17
|
+
const text = await res.text().catch(() => '');
|
|
18
|
+
return text.slice(0, 400);
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
function sizeParams({ aspectRatio, resolution }) {
|
|
22
|
+
const params = {};
|
|
23
|
+
const aspect = String(aspectRatio || 'auto');
|
|
24
|
+
if (aspect && aspect !== 'auto') params.aspect_ratio = aspect;
|
|
25
|
+
const res = String(resolution || '').toLowerCase();
|
|
26
|
+
if (res === '1k' || res === '2k') params.resolution = res;
|
|
27
|
+
return params;
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
/** Reference images ride as data URLs — xAI accepts them in place of a host. */
|
|
31
|
+
function referenceUrl(reference) {
|
|
32
|
+
const data = String(reference?.base64 || '');
|
|
33
|
+
if (data.startsWith('data:')) return data;
|
|
34
|
+
return `data:${reference?.mime || 'image/png'};base64,${data}`;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
export async function generateImage({ lane, model, prompt, options = {}, references = [], signal }) {
|
|
38
|
+
const { baseURL, token } = await resolveXaiAuth(lane);
|
|
39
|
+
const refs = references.map(referenceUrl).map((url) => ({ type: 'image_url', url }));
|
|
40
|
+
const body = {
|
|
41
|
+
model,
|
|
42
|
+
prompt,
|
|
43
|
+
n: 1,
|
|
44
|
+
response_format: 'b64_json',
|
|
45
|
+
// One reference edits that image; several composite into a new one.
|
|
46
|
+
...(refs.length === 1 ? { image: refs[0] } : refs.length > 1 ? { images: refs } : {}),
|
|
47
|
+
...sizeParams(options),
|
|
48
|
+
};
|
|
49
|
+
const route = refs.length ? 'images/edits' : 'images/generations';
|
|
50
|
+
const res = await fetch(`${baseURL}/${route}`, {
|
|
51
|
+
method: 'POST',
|
|
52
|
+
headers: { 'Content-Type': 'application/json', Authorization: `Bearer ${token}` },
|
|
53
|
+
body: JSON.stringify(body),
|
|
54
|
+
signal,
|
|
55
|
+
});
|
|
56
|
+
if (!res.ok) throw upstreamError('xAI image', res.status, await readError(res));
|
|
57
|
+
const data = await res.json();
|
|
58
|
+
const entry = data?.data?.[0];
|
|
59
|
+
if (!entry?.b64_json) throw mediaError('xAI returned no image data', 'MEDIA_EMPTY_RESULT', 502);
|
|
60
|
+
return {
|
|
61
|
+
bytes: Buffer.from(entry.b64_json, 'base64'),
|
|
62
|
+
mime: entry.mime_type || 'image/png',
|
|
63
|
+
revisedPrompt: entry.revised_prompt || null,
|
|
64
|
+
};
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
export async function generateVideo({ lane, model, prompt, options = {}, references = [], signal, onProgress }) {
|
|
68
|
+
const { baseURL, token } = await resolveXaiAuth(lane);
|
|
69
|
+
const headers = { 'Content-Type': 'application/json', Authorization: `Bearer ${token}` };
|
|
70
|
+
const duration = Math.min(15, Math.max(1, Math.trunc(Number(options.duration) || 5)));
|
|
71
|
+
const resolution = ['480p', '720p', '1080p'].includes(options.resolution) ? options.resolution : '480p';
|
|
72
|
+
const body = { model, prompt, duration, resolution };
|
|
73
|
+
const aspect = String(options.aspectRatio || 'auto');
|
|
74
|
+
if (aspect && aspect !== 'auto') body.aspect_ratio = aspect;
|
|
75
|
+
// Mode is inferred from the reference count: 1 = image-to-video,
|
|
76
|
+
// 2+ = reference-to-video (capped at the documented 7).
|
|
77
|
+
const refs = references.slice(0, 7).map(referenceUrl);
|
|
78
|
+
if (refs.length === 1) body.image = { url: refs[0] };
|
|
79
|
+
else if (refs.length > 1) body.reference_images = refs.map((url) => ({ url }));
|
|
80
|
+
|
|
81
|
+
const started = await fetch(`${baseURL}/videos/generations`, {
|
|
82
|
+
method: 'POST',
|
|
83
|
+
headers,
|
|
84
|
+
body: JSON.stringify(body),
|
|
85
|
+
signal: AbortSignal.any([signal, AbortSignal.timeout(START_TIMEOUT_MS)].filter(Boolean)),
|
|
86
|
+
});
|
|
87
|
+
if (!started.ok) throw upstreamError('xAI video', started.status, await readError(started));
|
|
88
|
+
const startData = await started.json();
|
|
89
|
+
const requestId = startData?.request_id || startData?.id;
|
|
90
|
+
if (!requestId) throw mediaError('xAI video start returned no request id', 'MEDIA_UPSTREAM_FAILED', 502);
|
|
91
|
+
|
|
92
|
+
const deadline = Date.now() + TOTAL_TIMEOUT_MS;
|
|
93
|
+
for (;;) {
|
|
94
|
+
if (signal?.aborted) throw mediaError('canceled', 'MEDIA_CANCELED', 499);
|
|
95
|
+
if (Date.now() > deadline) throw mediaError('xAI video poll budget exceeded', 'MEDIA_TIMEOUT', 504);
|
|
96
|
+
await new Promise((resolve) => setTimeout(resolve, POLL_INTERVAL_MS));
|
|
97
|
+
const poll = await fetch(`${baseURL}/videos/${requestId}`, { headers, signal });
|
|
98
|
+
if (!poll.ok && poll.status !== 202) throw upstreamError('xAI video poll', poll.status, await readError(poll));
|
|
99
|
+
const data = await poll.json().catch(() => ({}));
|
|
100
|
+
if (typeof data?.progress === 'number' && typeof onProgress === 'function') onProgress(data.progress);
|
|
101
|
+
if (data?.status === 'done') {
|
|
102
|
+
const url = data?.video?.url;
|
|
103
|
+
if (!url) throw mediaError('xAI video finished without a URL', 'MEDIA_EMPTY_RESULT', 502);
|
|
104
|
+
const file = await fetch(url, { signal });
|
|
105
|
+
if (!file.ok) throw mediaError(`xAI video download failed (${file.status})`, 'MEDIA_UPSTREAM_FAILED', file.status);
|
|
106
|
+
return {
|
|
107
|
+
bytes: Buffer.from(await file.arrayBuffer()),
|
|
108
|
+
mime: 'video/mp4',
|
|
109
|
+
durationSeconds: Number(data?.video?.duration) || duration,
|
|
110
|
+
};
|
|
111
|
+
}
|
|
112
|
+
if (data?.status === 'failed' || data?.status === 'expired') {
|
|
113
|
+
throw mediaError(`xAI video ${data.status}${data?.error?.code ? `: ${data.error.code}` : ''}`, 'MEDIA_UPSTREAM_FAILED', 502);
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
}
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Credential resolution for media lanes.
|
|
3
|
+
*
|
|
4
|
+
* OAuth lanes reuse the chat providers' token stores (including their refresh
|
|
5
|
+
* paths) so the Studio never holds a second copy of a session. API-key lanes
|
|
6
|
+
* read the same keychain-backed store the agent providers use.
|
|
7
|
+
*/
|
|
8
|
+
import { getAgentApiKey } from '../shared/provider-api-key.mjs';
|
|
9
|
+
|
|
10
|
+
export { hasGrokOAuthCredentials } from '../agent/orchestrator/providers/oauth-credential-probes.mjs';
|
|
11
|
+
|
|
12
|
+
const XAI_BASE_URL = 'https://api.x.ai/v1';
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* Bearer for the xAI media endpoints. `grok-oauth` refreshes through the chat
|
|
16
|
+
* provider; `xai` uses the stored API key. Both hit the same api.x.ai routes.
|
|
17
|
+
*/
|
|
18
|
+
export async function resolveXaiAuth(laneId) {
|
|
19
|
+
if (laneId === 'xai') {
|
|
20
|
+
const key = getAgentApiKey('xai');
|
|
21
|
+
if (!key) {
|
|
22
|
+
const err = new Error('xAI API key is not configured');
|
|
23
|
+
err.code = 'MEDIA_LANE_UNAUTHENTICATED';
|
|
24
|
+
err.status = 401;
|
|
25
|
+
throw err;
|
|
26
|
+
}
|
|
27
|
+
return { baseURL: XAI_BASE_URL, token: key };
|
|
28
|
+
}
|
|
29
|
+
const { GrokOAuthProvider } = await import('../agent/orchestrator/providers/grok-oauth.mjs');
|
|
30
|
+
const provider = new GrokOAuthProvider({});
|
|
31
|
+
const tokens = await provider.ensureAuth();
|
|
32
|
+
return { baseURL: XAI_BASE_URL, token: tokens.access_token };
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
/** Codex (ChatGPT OAuth) auth record: access token + account id for headers. */
|
|
36
|
+
export async function resolveCodexAuth() {
|
|
37
|
+
const { OpenAIOAuthProvider } = await import('../agent/orchestrator/providers/openai-oauth.mjs');
|
|
38
|
+
const provider = new OpenAIOAuthProvider({});
|
|
39
|
+
return await provider.ensureAuth();
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
export function resolveGeminiKey() {
|
|
43
|
+
const key = getAgentApiKey('gemini');
|
|
44
|
+
if (!key) {
|
|
45
|
+
const err = new Error('Gemini API key is not configured');
|
|
46
|
+
err.code = 'MEDIA_LANE_UNAUTHENTICATED';
|
|
47
|
+
err.status = 401;
|
|
48
|
+
throw err;
|
|
49
|
+
}
|
|
50
|
+
return key;
|
|
51
|
+
}
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Media studio surface (image / video generation).
|
|
3
|
+
*
|
|
4
|
+
* Single entry point for the session runtime: lane catalog, job control, and
|
|
5
|
+
* the on-disk asset store. Adapters are imported lazily by the job runner so a
|
|
6
|
+
* boot that never generates never pays for them.
|
|
7
|
+
*/
|
|
8
|
+
export { MEDIA_KINDS, listMediaLanes, getMediaLane } from './lanes.mjs';
|
|
9
|
+
export { startMediaJob, getMediaJob, listMediaJobs, cancelMediaJob } from './jobs.mjs';
|
|
10
|
+
export {
|
|
11
|
+
listMediaAssets,
|
|
12
|
+
readMediaAsset,
|
|
13
|
+
resolveMediaFile,
|
|
14
|
+
deleteMediaAsset,
|
|
15
|
+
mediaAssetPath,
|
|
16
|
+
openMediaAsset,
|
|
17
|
+
openMediaFolder,
|
|
18
|
+
} from './store.mjs';
|
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Media generation jobs.
|
|
3
|
+
*
|
|
4
|
+
* Image lanes answer in one request, video lanes submit + poll for minutes, so
|
|
5
|
+
* both run as background jobs with a polled snapshot instead of a blocking
|
|
6
|
+
* call. The desktop reads snapshots over the normal capability bridge; nothing
|
|
7
|
+
* here streams, which keeps a long video job independent of window lifetime.
|
|
8
|
+
*/
|
|
9
|
+
import { randomUUID } from 'crypto';
|
|
10
|
+
import { mediaError, resolveMediaRequest } from './lanes.mjs';
|
|
11
|
+
import { saveMediaAsset } from './store.mjs';
|
|
12
|
+
|
|
13
|
+
const JOBS = new Map();
|
|
14
|
+
// Finished jobs stay readable for a while so a slow poller still sees the
|
|
15
|
+
// terminal state, then drop out to bound memory.
|
|
16
|
+
const TERMINAL_TTL_MS = 10 * 60_000;
|
|
17
|
+
const MAX_PROMPT_CHARS = 8_000;
|
|
18
|
+
|
|
19
|
+
function snapshot(job) {
|
|
20
|
+
return {
|
|
21
|
+
id: job.id,
|
|
22
|
+
status: job.status,
|
|
23
|
+
kind: job.kind,
|
|
24
|
+
lane: job.lane,
|
|
25
|
+
model: job.model,
|
|
26
|
+
prompt: job.prompt,
|
|
27
|
+
options: job.options,
|
|
28
|
+
progress: job.progress,
|
|
29
|
+
assetId: job.assetId,
|
|
30
|
+
error: job.error,
|
|
31
|
+
errorCode: job.errorCode,
|
|
32
|
+
startedAt: job.startedAt,
|
|
33
|
+
endedAt: job.endedAt,
|
|
34
|
+
};
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
function sweep() {
|
|
38
|
+
const now = Date.now();
|
|
39
|
+
for (const [id, job] of JOBS) {
|
|
40
|
+
if (job.endedAt && now - job.endedAt > TERMINAL_TTL_MS) JOBS.delete(id);
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
async function runAdapter({ lane, kind, model, prompt, options, references, signal, onProgress }) {
|
|
45
|
+
if (lane.id === 'grok-oauth' || lane.id === 'xai') {
|
|
46
|
+
const adapter = await import('./adapters/xai-media.mjs');
|
|
47
|
+
return kind === 'video'
|
|
48
|
+
? await adapter.generateVideo({ lane: lane.id, model, prompt, options, references, signal, onProgress })
|
|
49
|
+
: await adapter.generateImage({ lane: lane.id, model, prompt, options, references, signal });
|
|
50
|
+
}
|
|
51
|
+
if (lane.id === 'openai-oauth') {
|
|
52
|
+
const adapter = await import('./adapters/codex-image.mjs');
|
|
53
|
+
return await adapter.generateImage({ model, prompt, options, references, signal });
|
|
54
|
+
}
|
|
55
|
+
if (lane.id === 'gemini') {
|
|
56
|
+
if (kind === 'video') {
|
|
57
|
+
const adapter = await import('./adapters/gemini-video.mjs');
|
|
58
|
+
return await adapter.generateVideo({ model, prompt, options, references, signal, onProgress });
|
|
59
|
+
}
|
|
60
|
+
const adapter = await import('./adapters/gemini-image.mjs');
|
|
61
|
+
return await adapter.generateImage({ model, prompt, options, references, signal });
|
|
62
|
+
}
|
|
63
|
+
throw mediaError(`lane "${lane.id}" has no adapter`, 'MEDIA_LANE_UNKNOWN');
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* Validate + start one generation. Returns the initial snapshot immediately;
|
|
68
|
+
* the caller polls getMediaJob for progress and the finished asset id.
|
|
69
|
+
*/
|
|
70
|
+
export function startMediaJob({ lane: laneId, kind, model, prompt, options = {}, references = [] } = {}) {
|
|
71
|
+
const text = String(prompt || '').trim();
|
|
72
|
+
if (!text) throw mediaError('prompt is required', 'MEDIA_PROMPT_REQUIRED');
|
|
73
|
+
if (text.length > MAX_PROMPT_CHARS) throw mediaError('prompt is too long', 'MEDIA_PROMPT_TOO_LONG');
|
|
74
|
+
const resolved = resolveMediaRequest({ lane: laneId, kind, model });
|
|
75
|
+
// Reference images: the MODEL publishes its own cap (Veo takes one start
|
|
76
|
+
// frame, Grok ref2v takes seven), so the bound follows the resolved model.
|
|
77
|
+
const modelEntry = resolved.spec.models.find((entry) => entry.id === resolved.model);
|
|
78
|
+
const maxRefs = Number(
|
|
79
|
+
modelEntry?.controls?.maxReferences ?? resolved.spec.controls?.maxReferences ?? 0,
|
|
80
|
+
) || (resolved.kind === 'video' ? 7 : 5);
|
|
81
|
+
const refs = (Array.isArray(references) ? references : [])
|
|
82
|
+
.filter((ref) => typeof ref?.base64 === 'string' && ref.base64.length > 0)
|
|
83
|
+
.slice(0, maxRefs)
|
|
84
|
+
.map((ref) => ({ base64: ref.base64, mime: String(ref.mime || 'image/png') }));
|
|
85
|
+
sweep();
|
|
86
|
+
|
|
87
|
+
const controller = new AbortController();
|
|
88
|
+
const job = {
|
|
89
|
+
id: randomUUID(),
|
|
90
|
+
status: 'running',
|
|
91
|
+
kind: resolved.kind,
|
|
92
|
+
lane: resolved.lane.id,
|
|
93
|
+
model: resolved.model,
|
|
94
|
+
prompt: text,
|
|
95
|
+
options: { ...options },
|
|
96
|
+
referenceCount: refs.length,
|
|
97
|
+
progress: 0,
|
|
98
|
+
assetId: null,
|
|
99
|
+
error: null,
|
|
100
|
+
errorCode: null,
|
|
101
|
+
startedAt: Date.now(),
|
|
102
|
+
endedAt: null,
|
|
103
|
+
controller,
|
|
104
|
+
};
|
|
105
|
+
JOBS.set(job.id, job);
|
|
106
|
+
|
|
107
|
+
(async () => {
|
|
108
|
+
try {
|
|
109
|
+
const result = await runAdapter({
|
|
110
|
+
lane: resolved.lane,
|
|
111
|
+
kind: resolved.kind,
|
|
112
|
+
model: resolved.model,
|
|
113
|
+
prompt: text,
|
|
114
|
+
options,
|
|
115
|
+
references: refs,
|
|
116
|
+
signal: controller.signal,
|
|
117
|
+
onProgress: (value) => {
|
|
118
|
+
const raw = Number(value);
|
|
119
|
+
if (!Number.isFinite(raw)) return;
|
|
120
|
+
// Lanes report either a 0-1 fraction or a percentage, and a poll can
|
|
121
|
+
// come back stale — normalize and never let the rail walk backwards.
|
|
122
|
+
const next = Math.round(Math.max(0, Math.min(100, raw > 1 ? raw : raw * 100)));
|
|
123
|
+
if (next > job.progress) job.progress = next;
|
|
124
|
+
},
|
|
125
|
+
});
|
|
126
|
+
const asset = saveMediaAsset({
|
|
127
|
+
kind: resolved.kind,
|
|
128
|
+
lane: resolved.lane.id,
|
|
129
|
+
model: resolved.model,
|
|
130
|
+
prompt: text,
|
|
131
|
+
options: job.options,
|
|
132
|
+
mime: result.mime,
|
|
133
|
+
bytes: result.bytes,
|
|
134
|
+
meta: {
|
|
135
|
+
...(result.revisedPrompt ? { revisedPrompt: result.revisedPrompt } : {}),
|
|
136
|
+
...(result.durationSeconds ? { durationSeconds: result.durationSeconds } : {}),
|
|
137
|
+
},
|
|
138
|
+
});
|
|
139
|
+
job.assetId = asset.id;
|
|
140
|
+
job.progress = 100;
|
|
141
|
+
job.status = 'done';
|
|
142
|
+
} catch (err) {
|
|
143
|
+
const canceled = controller.signal.aborted || err?.code === 'MEDIA_CANCELED' || err?.name === 'AbortError';
|
|
144
|
+
job.status = canceled ? 'canceled' : 'failed';
|
|
145
|
+
job.error = canceled ? 'canceled' : String(err?.message || err).slice(0, 500);
|
|
146
|
+
job.errorCode = err?.code || null;
|
|
147
|
+
} finally {
|
|
148
|
+
job.endedAt = Date.now();
|
|
149
|
+
}
|
|
150
|
+
})();
|
|
151
|
+
|
|
152
|
+
return snapshot(job);
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
export function getMediaJob(id) {
|
|
156
|
+
const job = JOBS.get(String(id || ''));
|
|
157
|
+
return job ? snapshot(job) : null;
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
export function listMediaJobs() {
|
|
161
|
+
sweep();
|
|
162
|
+
return [...JOBS.values()]
|
|
163
|
+
.sort((a, b) => b.startedAt - a.startedAt)
|
|
164
|
+
.map(snapshot);
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
export function cancelMediaJob(id) {
|
|
168
|
+
const job = JOBS.get(String(id || ''));
|
|
169
|
+
if (!job) return { id, canceled: false };
|
|
170
|
+
if (job.status === 'running') {
|
|
171
|
+
try { job.controller.abort(); } catch {}
|
|
172
|
+
}
|
|
173
|
+
return { id: job.id, canceled: job.status === 'running' };
|
|
174
|
+
}
|
|
@@ -0,0 +1,230 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Media generation lanes (image / video).
|
|
3
|
+
*
|
|
4
|
+
* A lane is one credential path we can actually generate with. Auth state is
|
|
5
|
+
* resolved from the SAME stores the chat providers use (OAuth token stores and
|
|
6
|
+
* the keychain-backed API keys), so the Studio only ever offers lanes the user
|
|
7
|
+
* already signed in to. No lane owns a fallback to another lane's credential —
|
|
8
|
+
* an unauthenticated lane fails closed.
|
|
9
|
+
*/
|
|
10
|
+
import { getAgentApiKey } from '../shared/provider-api-key.mjs';
|
|
11
|
+
// Credential probes are file-level checks: the catalog must not drag the whole
|
|
12
|
+
// provider graph in just to render lane availability.
|
|
13
|
+
import {
|
|
14
|
+
hasGrokOAuthCredentials,
|
|
15
|
+
hasOpenAIOAuthCredentials,
|
|
16
|
+
} from '../agent/orchestrator/providers/oauth-credential-probes.mjs';
|
|
17
|
+
|
|
18
|
+
export const MEDIA_KINDS = Object.freeze(['image', 'video']);
|
|
19
|
+
|
|
20
|
+
const GROK_IMAGE_MODELS = Object.freeze([
|
|
21
|
+
Object.freeze({ id: 'grok-imagine-image', label: 'Imagine Image' }),
|
|
22
|
+
Object.freeze({ id: 'grok-imagine-image-quality', label: 'Imagine Image Q' }),
|
|
23
|
+
]);
|
|
24
|
+
// Per-model control overrides: video contracts differ per model (fixed clip
|
|
25
|
+
// lengths on Veo, 1080p only on Grok 1.5, no length knob on Omni), so each
|
|
26
|
+
// entry carries what it actually accepts instead of one lane-wide guess.
|
|
27
|
+
const GROK_VIDEO_MODELS = Object.freeze([
|
|
28
|
+
Object.freeze({
|
|
29
|
+
id: 'grok-imagine-video',
|
|
30
|
+
label: 'Imagine Video',
|
|
31
|
+
controls: Object.freeze({ resolution: Object.freeze(['480p', '720p']) }),
|
|
32
|
+
}),
|
|
33
|
+
]);
|
|
34
|
+
// grok-imagine-video-1.5 is intentionally absent: upstream rejects prompt-only
|
|
35
|
+
// text-to-video on it ("Text-to-video is not supported for this model"), and it
|
|
36
|
+
// only works through an image/reference input we do not accept yet.
|
|
37
|
+
|
|
38
|
+
// Aspect/resolution vocabularies are lane-native: xAI takes aspect_ratio +
|
|
39
|
+
// resolution, Gemini/Codex take pixel sizes. The UI renders whatever the lane
|
|
40
|
+
// declares instead of inventing a cross-provider size model.
|
|
41
|
+
const GROK_ASPECTS = Object.freeze(['auto', '1:1', '16:9', '9:16', '4:3', '3:4', '3:2', '2:3']);
|
|
42
|
+
|
|
43
|
+
const LANES = Object.freeze([
|
|
44
|
+
Object.freeze({
|
|
45
|
+
id: 'grok-oauth',
|
|
46
|
+
label: 'Grok Imagine (OAuth)',
|
|
47
|
+
auth: Object.freeze({ type: 'oauth', provider: 'grok-oauth' }),
|
|
48
|
+
image: Object.freeze({
|
|
49
|
+
models: GROK_IMAGE_MODELS,
|
|
50
|
+
defaultModel: 'grok-imagine-image',
|
|
51
|
+
controls: Object.freeze({ aspectRatio: GROK_ASPECTS, resolution: Object.freeze(['1k', '2k']), maxReferences: 5 }),
|
|
52
|
+
}),
|
|
53
|
+
video: Object.freeze({
|
|
54
|
+
models: GROK_VIDEO_MODELS,
|
|
55
|
+
defaultModel: 'grok-imagine-video',
|
|
56
|
+
controls: Object.freeze({
|
|
57
|
+
aspectRatio: GROK_ASPECTS,
|
|
58
|
+
resolution: Object.freeze(['480p', '720p', '1080p']),
|
|
59
|
+
durationRange: Object.freeze([1, 15]),
|
|
60
|
+
// 1 ref = image-to-video, 2-7 = reference-to-video.
|
|
61
|
+
maxReferences: 7,
|
|
62
|
+
}),
|
|
63
|
+
}),
|
|
64
|
+
}),
|
|
65
|
+
Object.freeze({
|
|
66
|
+
id: 'xai',
|
|
67
|
+
label: 'Grok Imagine (API key)',
|
|
68
|
+
auth: Object.freeze({ type: 'api-key', provider: 'xai' }),
|
|
69
|
+
image: Object.freeze({
|
|
70
|
+
models: GROK_IMAGE_MODELS,
|
|
71
|
+
defaultModel: 'grok-imagine-image',
|
|
72
|
+
controls: Object.freeze({ aspectRatio: GROK_ASPECTS, resolution: Object.freeze(['1k', '2k']), maxReferences: 5 }),
|
|
73
|
+
}),
|
|
74
|
+
video: Object.freeze({
|
|
75
|
+
models: GROK_VIDEO_MODELS,
|
|
76
|
+
defaultModel: 'grok-imagine-video',
|
|
77
|
+
controls: Object.freeze({
|
|
78
|
+
aspectRatio: GROK_ASPECTS,
|
|
79
|
+
resolution: Object.freeze(['480p', '720p', '1080p']),
|
|
80
|
+
durationRange: Object.freeze([1, 15]),
|
|
81
|
+
maxReferences: 7,
|
|
82
|
+
}),
|
|
83
|
+
}),
|
|
84
|
+
}),
|
|
85
|
+
Object.freeze({
|
|
86
|
+
id: 'openai-oauth',
|
|
87
|
+
label: 'ChatGPT Image (OAuth)',
|
|
88
|
+
auth: Object.freeze({ type: 'oauth', provider: 'openai-oauth' }),
|
|
89
|
+
image: Object.freeze({
|
|
90
|
+
// The hosted image_generation tool runs on the Codex responses backend,
|
|
91
|
+
// so the id is a chat model driving the tool. The catalog exposes it as
|
|
92
|
+
// an IMAGE choice (quality vs fast) — listing raw chat models here read
|
|
93
|
+
// as "why is a chat model in an image picker".
|
|
94
|
+
models: Object.freeze([
|
|
95
|
+
Object.freeze({ id: 'gpt-5.6-sol', label: 'GPT Image' }),
|
|
96
|
+
Object.freeze({ id: 'gpt-5.4-mini', label: 'GPT Image Fast' }),
|
|
97
|
+
]),
|
|
98
|
+
defaultModel: 'gpt-5.6-sol',
|
|
99
|
+
controls: Object.freeze({
|
|
100
|
+
size: Object.freeze(['auto', '1024x1024', '1536x1024', '1024x1536']),
|
|
101
|
+
quality: Object.freeze(['auto', 'low', 'medium', 'high']),
|
|
102
|
+
maxReferences: 5,
|
|
103
|
+
}),
|
|
104
|
+
}),
|
|
105
|
+
}),
|
|
106
|
+
Object.freeze({
|
|
107
|
+
id: 'gemini',
|
|
108
|
+
label: 'Gemini Image (API key)',
|
|
109
|
+
auth: Object.freeze({ type: 'api-key', provider: 'gemini' }),
|
|
110
|
+
image: Object.freeze({
|
|
111
|
+
models: Object.freeze([
|
|
112
|
+
Object.freeze({ id: 'gemini-3-pro-image', label: 'Nano Banana Pro' }),
|
|
113
|
+
Object.freeze({ id: 'gemini-3.1-flash-image', label: 'Nano Banana 2' }),
|
|
114
|
+
Object.freeze({ id: 'gemini-2.5-flash-image', label: 'Nano Banana 2.5' }),
|
|
115
|
+
]),
|
|
116
|
+
defaultModel: 'gemini-3.1-flash-image',
|
|
117
|
+
controls: Object.freeze({
|
|
118
|
+
aspectRatio: Object.freeze(['auto', '1:1', '16:9', '9:16', '4:3', '3:4']),
|
|
119
|
+
// Reference caps the image edit path at 3 inline references.
|
|
120
|
+
maxReferences: 3,
|
|
121
|
+
}),
|
|
122
|
+
}),
|
|
123
|
+
video: Object.freeze({
|
|
124
|
+
// Omni Flash answers inline on the Interactions API; the Veo trio runs as
|
|
125
|
+
// a long-running predict. Both are paid-tier only on this key.
|
|
126
|
+
models: Object.freeze([
|
|
127
|
+
Object.freeze({
|
|
128
|
+
id: 'gemini-omni-flash-preview',
|
|
129
|
+
label: 'Omni Flash',
|
|
130
|
+
// Omni picks its own clip length; only the frame shape is ours.
|
|
131
|
+
controls: Object.freeze({
|
|
132
|
+
resolution: Object.freeze([]),
|
|
133
|
+
durations: Object.freeze([]),
|
|
134
|
+
maxReferences: 3,
|
|
135
|
+
}),
|
|
136
|
+
}),
|
|
137
|
+
Object.freeze({ id: 'veo-3.1-fast-generate-preview', label: 'Veo 3.1 Fast' }),
|
|
138
|
+
Object.freeze({ id: 'veo-3.1-generate-preview', label: 'Veo 3.1' }),
|
|
139
|
+
Object.freeze({ id: 'veo-3.1-lite-generate-preview', label: 'Veo 3.1 Lite' }),
|
|
140
|
+
]),
|
|
141
|
+
defaultModel: 'gemini-omni-flash-preview',
|
|
142
|
+
controls: Object.freeze({
|
|
143
|
+
aspectRatio: Object.freeze(['16:9', '9:16']),
|
|
144
|
+
resolution: Object.freeze(['720p', '1080p']),
|
|
145
|
+
// Veo 3.1 accepts discrete clip lengths, not a free range.
|
|
146
|
+
durations: Object.freeze([4, 6, 8]),
|
|
147
|
+
// Veo takes a single start frame; Omni overrides this below.
|
|
148
|
+
maxReferences: 1,
|
|
149
|
+
}),
|
|
150
|
+
}),
|
|
151
|
+
}),
|
|
152
|
+
]);
|
|
153
|
+
|
|
154
|
+
const LANE_BY_ID = new Map(LANES.map((lane) => [lane.id, lane]));
|
|
155
|
+
|
|
156
|
+
function laneAuthenticated(lane) {
|
|
157
|
+
try {
|
|
158
|
+
if (lane.auth.type === 'api-key') return Boolean(getAgentApiKey(lane.auth.provider));
|
|
159
|
+
if (lane.auth.provider === 'grok-oauth') return hasGrokOAuthCredentials();
|
|
160
|
+
if (lane.auth.provider === 'openai-oauth') return hasOpenAIOAuthCredentials();
|
|
161
|
+
} catch {
|
|
162
|
+
return false;
|
|
163
|
+
}
|
|
164
|
+
return false;
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
function laneKindView(lane, kind) {
|
|
168
|
+
const spec = lane[kind];
|
|
169
|
+
if (!spec) return null;
|
|
170
|
+
const laneControls = spec.controls || {};
|
|
171
|
+
return {
|
|
172
|
+
// Each model publishes its EFFECTIVE controls (lane defaults + its own
|
|
173
|
+
// overrides) so the UI never offers a knob the model rejects.
|
|
174
|
+
models: spec.models.map((model) => ({
|
|
175
|
+
id: model.id,
|
|
176
|
+
label: model.label,
|
|
177
|
+
controls: JSON.parse(JSON.stringify({ ...laneControls, ...(model.controls || {}) })),
|
|
178
|
+
})),
|
|
179
|
+
defaultModel: spec.defaultModel,
|
|
180
|
+
controls: JSON.parse(JSON.stringify(laneControls)),
|
|
181
|
+
};
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
/** Lane catalog with live auth state; the caller filters on `authenticated`. */
|
|
185
|
+
export function listMediaLanes() {
|
|
186
|
+
return LANES.map((lane) => ({
|
|
187
|
+
id: lane.id,
|
|
188
|
+
label: lane.label,
|
|
189
|
+
authType: lane.auth.type,
|
|
190
|
+
authProvider: lane.auth.provider,
|
|
191
|
+
authenticated: laneAuthenticated(lane),
|
|
192
|
+
kinds: MEDIA_KINDS.filter((kind) => Boolean(lane[kind])),
|
|
193
|
+
image: laneKindView(lane, 'image'),
|
|
194
|
+
video: laneKindView(lane, 'video'),
|
|
195
|
+
}));
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
export function getMediaLane(laneId) {
|
|
199
|
+
return LANE_BY_ID.get(String(laneId || '').trim()) || null;
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
/**
|
|
203
|
+
* Resolve + validate a generation request against the lane contract. Throws a
|
|
204
|
+
* coded error rather than letting an unsupported model reach the network.
|
|
205
|
+
*/
|
|
206
|
+
export function resolveMediaRequest({ lane: laneId, kind, model } = {}) {
|
|
207
|
+
const kindName = String(kind || '').trim();
|
|
208
|
+
if (!MEDIA_KINDS.includes(kindName)) {
|
|
209
|
+
throw mediaError(`unsupported media kind "${kindName}"`, 'MEDIA_KIND_UNSUPPORTED');
|
|
210
|
+
}
|
|
211
|
+
const lane = getMediaLane(laneId);
|
|
212
|
+
if (!lane) throw mediaError(`unknown media lane "${laneId}"`, 'MEDIA_LANE_UNKNOWN');
|
|
213
|
+
const spec = lane[kindName];
|
|
214
|
+
if (!spec) throw mediaError(`${lane.id} does not support ${kindName}`, 'MEDIA_KIND_UNSUPPORTED');
|
|
215
|
+
if (!laneAuthenticated(lane)) {
|
|
216
|
+
throw mediaError(`${lane.label} is not authenticated — sign in from Settings → Providers first`, 'MEDIA_LANE_UNAUTHENTICATED');
|
|
217
|
+
}
|
|
218
|
+
const requested = String(model || '').trim() || spec.defaultModel;
|
|
219
|
+
if (!spec.models.some((entry) => entry.id === requested)) {
|
|
220
|
+
throw mediaError(`model "${requested}" is not available on ${lane.id}/${kindName}`, 'MEDIA_MODEL_UNSUPPORTED');
|
|
221
|
+
}
|
|
222
|
+
return { lane, kind: kindName, model: requested, spec };
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
export function mediaError(message, code, status = 400) {
|
|
226
|
+
const err = new Error(message);
|
|
227
|
+
err.code = code;
|
|
228
|
+
err.status = status;
|
|
229
|
+
return err;
|
|
230
|
+
}
|