vrex-flow-engine 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- flow_engine/__init__.py +4 -0
- flow_engine/bridge/__init__.py +6 -0
- flow_engine/bridge/flow_client.py +444 -0
- flow_engine/bridge/flow_sdk.py +1743 -0
- flow_engine/bridge/ws_server.py +84 -0
- flow_engine/catalog.py +332 -0
- flow_engine/cli/__init__.py +1 -0
- flow_engine/cli/app.py +149 -0
- flow_engine/cli/config.py +81 -0
- flow_engine/cli/dashboard.py +125 -0
- flow_engine/cli/health.py +40 -0
- flow_engine/cli/processes.py +109 -0
- flow_engine/cli/supervisor.py +184 -0
- flow_engine/config.py +43 -0
- flow_engine/ingest.py +172 -0
- flow_engine/job_store.py +28 -0
- flow_engine/jobs.py +245 -0
- flow_engine/main.py +187 -0
- flow_engine/media.py +227 -0
- flow_engine/media_store.py +202 -0
- flow_engine/openai/__init__.py +1 -0
- flow_engine/openai/_util.py +59 -0
- flow_engine/openai/images.py +134 -0
- flow_engine/openai/models.py +38 -0
- flow_engine/openai/uploads.py +70 -0
- flow_engine/openai/videos.py +289 -0
- flow_engine/pool.py +138 -0
- flow_engine/posthog_client.py +117 -0
- flow_engine/session.py +89 -0
- flow_engine/video_context.py +77 -0
- flow_engine/video_context_store.py +49 -0
- vrex_flow_engine-0.2.0.dist-info/METADATA +62 -0
- vrex_flow_engine-0.2.0.dist-info/RECORD +36 -0
- vrex_flow_engine-0.2.0.dist-info/WHEEL +5 -0
- vrex_flow_engine-0.2.0.dist-info/entry_points.txt +3 -0
- vrex_flow_engine-0.2.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1743 @@
|
|
|
1
|
+
"""Minimal Google Flow SDK wrapper.
|
|
2
|
+
|
|
3
|
+
Ported from flowkit (`/tmp/flowkit-ref/agent/services/flow_client.py`).
|
|
4
|
+
Trimmed to what Run 4 ships: `create_project` (TRPC) + `gen_image` (api_request
|
|
5
|
+
with IMAGE_GENERATION captcha). Video / upload / upscale / check_async land in
|
|
6
|
+
later runs.
|
|
7
|
+
|
|
8
|
+
The wrapper intentionally preserves `raw` on every return so callers (and the
|
|
9
|
+
request-worker that persists it to the DB) can inspect Flow's error payload
|
|
10
|
+
when the user's paygate tier or model name drifts.
|
|
11
|
+
"""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import logging
|
|
15
|
+
import re
|
|
16
|
+
import time
|
|
17
|
+
import uuid
|
|
18
|
+
from typing import Any, Optional
|
|
19
|
+
|
|
20
|
+
from flow_engine.bridge.flow_client import FlowClient
|
|
21
|
+
|
|
22
|
+
logger = logging.getLogger(__name__)
|
|
23
|
+
|
|
24
|
+
# Endpoints -----------------------------------------------------------------
|
|
25
|
+
|
|
26
|
+
FLOW_API_BASE = "https://aisandbox-pa.googleapis.com"
|
|
27
|
+
TRPC_CREATE_PROJECT = "https://labs.google/fx/api/trpc/project.createProject"
|
|
28
|
+
TRPC_SEARCH_PROJECTS = "https://labs.google/fx/api/trpc/project.searchUserProjects"
|
|
29
|
+
# Returns the full modelConfig (image/video families + keys + per-tier credit
|
|
30
|
+
# availability), userData (serviceTier/paygateTier/credits), and the preset
|
|
31
|
+
# audio voice list — used to build the model catalog dynamically.
|
|
32
|
+
TRPC_PROJECT_INITIAL_DATA = "https://labs.google/fx/api/trpc/flow.projectInitialData"
|
|
33
|
+
VIDEO_I2V_URL = f"{FLOW_API_BASE}/v1/video:batchAsyncGenerateVideoStartImage"
|
|
34
|
+
# Reference-to-video endpoint: takes referenceImages[] (multi-ref, asset-typed)
|
|
35
|
+
# instead of a single startImage. See gen_video_r2v() for the body assembly.
|
|
36
|
+
VIDEO_R2V_URL = f"{FLOW_API_BASE}/v1/video:batchAsyncGenerateVideoReferenceImages"
|
|
37
|
+
# Text-to-video: same body as r2v minus referenceImages/referenceAudio.
|
|
38
|
+
VIDEO_T2V_URL = f"{FLOW_API_BASE}/v1/video:batchAsyncGenerateVideoText"
|
|
39
|
+
# Extend (continue a Veo clip) and edit (video-to-video) — both take a
|
|
40
|
+
# `videoInput.mediaId` pointing at a source Veo video and return the workflow
|
|
41
|
+
# schema (poll via the new primaryMediaId). Verified from a live labs.google
|
|
42
|
+
# capture (2026-05). NOTE: only Veo-generated videos can be extended.
|
|
43
|
+
VIDEO_EXTEND_URL = f"{FLOW_API_BASE}/v1/video:batchAsyncGenerateVideoExtendVideo"
|
|
44
|
+
VIDEO_EDIT_URL = f"{FLOW_API_BASE}/v1/video:batchAsyncGenerateVideoEditVideo"
|
|
45
|
+
VIDEO_POLL_URL = f"{FLOW_API_BASE}/v1/video:batchCheckAsyncVideoGenerationStatus"
|
|
46
|
+
UPLOAD_IMAGE_URL = f"{FLOW_API_BASE}/v1/flow/uploadImage"
|
|
47
|
+
# Video upload is a resumable, cookie-authed flow against labs.google (NOT
|
|
48
|
+
# aisandbox-pa): a zero-body POST ?action=start returns a session URL, then a
|
|
49
|
+
# binary PUT ?action=upload streams the bytes. Both run inside the extension's
|
|
50
|
+
# browser session (see background.js handleUploadVideo). Afterwards we register
|
|
51
|
+
# the clip's trim window via the TRPC videoFx.updateVideoOffset mutation so Flow
|
|
52
|
+
# treats the upload as a usable workflow media — mirroring the web client.
|
|
53
|
+
UPLOAD_VIDEO_START_URL = "https://labs.google/fx/api/upload-video?action=start"
|
|
54
|
+
UPLOAD_VIDEO_UPLOAD_URL = "https://labs.google/fx/api/upload-video?action=upload"
|
|
55
|
+
TRPC_UPDATE_VIDEO_OFFSET = "https://labs.google/fx/api/trpc/videoFx.updateVideoOffset"
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
# Veo 3.1 reference-to-video (r2v). Verified against a live labs.google web
|
|
59
|
+
# request (2026-05): Flow web dispatches `veo_3_1_r2v_lite_low_priority` to
|
|
60
|
+
# `batchAsyncGenerateVideoReferenceImages` and gets a workflow-mode response
|
|
61
|
+
# (poll via /v1/media/<primaryMediaId>). Duration is fixed by the model
|
|
62
|
+
# (~8s); it is not a request field.
|
|
63
|
+
VEO_R2V_MODEL_KEY = "veo_3_1_r2v_lite_low_priority"
|
|
64
|
+
|
|
65
|
+
R2V_VALID_ASPECTS: set[str] = {
|
|
66
|
+
"VIDEO_ASPECT_RATIO_PORTRAIT",
|
|
67
|
+
"VIDEO_ASPECT_RATIO_LANDSCAPE",
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _media_get_url(media_id: str) -> str:
|
|
72
|
+
"""Endpoint that returns inline encoded video bytes for a workflow's
|
|
73
|
+
primary media. Used to poll Low Priority (workflow-schema) submissions —
|
|
74
|
+
they have no operation name and don't appear in ``batchCheckAsync``."""
|
|
75
|
+
return f"{FLOW_API_BASE}/v1/media/{media_id}?clientContext.tool=PINHOLE"
|
|
76
|
+
|
|
77
|
+
# Image model keys, indexed by the user-facing nickname used in
|
|
78
|
+
# flowkit's models.json. Pro is Flow's premium / higher-quality image
|
|
79
|
+
# model; "Banana 2" (NARWHAL) is the lighter / faster option. The
|
|
80
|
+
# frontend Settings panel lets the user pick which one drives gen_image
|
|
81
|
+
# + edit_image at request time. Update when Google rotates model names.
|
|
82
|
+
IMAGE_MODELS: dict[str, str] = {
|
|
83
|
+
"NANO_BANANA_PRO": "GEM_PIX_2",
|
|
84
|
+
"NANO_BANANA_2": "NARWHAL",
|
|
85
|
+
}
|
|
86
|
+
DEFAULT_IMAGE_MODEL_KEY = "NANO_BANANA_PRO"
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def resolve_image_model(key: Optional[str]) -> str:
|
|
90
|
+
"""Map a nickname (`NANO_BANANA_PRO` / `NANO_BANANA_2`) to the actual
|
|
91
|
+
Flow model identifier. Falls back to the Pro default for unknown /
|
|
92
|
+
missing keys so a stale frontend can't break dispatch."""
|
|
93
|
+
if isinstance(key, str) and key in IMAGE_MODELS:
|
|
94
|
+
return IMAGE_MODELS[key]
|
|
95
|
+
return IMAGE_MODELS[DEFAULT_IMAGE_MODEL_KEY]
|
|
96
|
+
|
|
97
|
+
# Video model keys nested by [tier][quality][aspect]. All values verified
|
|
98
|
+
# against real Flow web request bodies (curl exports from labs.google's
|
|
99
|
+
# Network tab) — do NOT speculate suffixes here, only use observed keys.
|
|
100
|
+
#
|
|
101
|
+
# `quality` is "fast" (default), "lite", "quality", or — Ultra only —
|
|
102
|
+
# "lite_relaxed" / "fast_relaxed" (0-credit low-priority queue).
|
|
103
|
+
# - Lite (`veo_3_1_i2v_lite`) is shared by Tier 1 and Tier 2; verified
|
|
104
|
+
# from PRO PLAN and ULTRA PLAN curls (see video_model.md and
|
|
105
|
+
# video_model_ultra.md). Multi-aspect — same key for both 16:9 and
|
|
106
|
+
# 9:16; the model adapts via the aspectRatio field.
|
|
107
|
+
# - Quality (`veo_3_1_i2v_s` / `veo_3_1_i2v_s_portrait`) is also shared
|
|
108
|
+
# across both tiers; the difference is the `userPaygateTier` in
|
|
109
|
+
# clientContext (rate limits / queue priority), not the model key.
|
|
110
|
+
# - Tier 2 Fast naming pattern: Tier 1 Fast key + `_ultra` suffix
|
|
111
|
+
# (e.g. `veo_3_1_i2v_s_fast` → `veo_3_1_i2v_s_fast_ultra`,
|
|
112
|
+
# `veo_3_1_i2v_s_fast_portrait` → `veo_3_1_i2v_s_fast_portrait_ultra`).
|
|
113
|
+
# - Tier 2 "low priority" 0-credit models (Ultra-only fallback when the
|
|
114
|
+
# user wants to keep their daily credit budget): Lite uses the
|
|
115
|
+
# `_low_priority` suffix (`veo_3_1_i2v_lite_low_priority`); Fast uses
|
|
116
|
+
# the `_relaxed` suffix on the ultra family (`veo_3_1_i2v_s_fast_ultra_relaxed`).
|
|
117
|
+
# Verified from ULTRA PLAN curls. PORTRAIT keys for these are not yet
|
|
118
|
+
# observed — we reuse the LANDSCAPE key for both aspects (Lite is
|
|
119
|
+
# genuinely multi-aspect; Fast Relaxed portrait will need a real curl
|
|
120
|
+
# to confirm, but Flow's portrait variants typically follow the
|
|
121
|
+
# `_portrait` suffix convention if separate keys are required).
|
|
122
|
+
VIDEO_MODEL_KEYS: dict[str, dict[str, dict[str, str]]] = {
|
|
123
|
+
# Tier 1 (Pro) — three quality levels, all verified from real PRO
|
|
124
|
+
# PLAN curls (see video_model.md). Lite shares `veo_3_1_i2v_lite`
|
|
125
|
+
# with Tier 2; Quality shares `veo_3_1_i2v_s` with Tier 2 — paygate
|
|
126
|
+
# tier in clientContext drives any per-tier difference. No 0-credit
|
|
127
|
+
# low-priority option here — that's a Tier 2 (Ultra) perk.
|
|
128
|
+
"PAYGATE_TIER_ONE": {
|
|
129
|
+
"lite": {
|
|
130
|
+
"VIDEO_ASPECT_RATIO_LANDSCAPE": "veo_3_1_i2v_lite",
|
|
131
|
+
"VIDEO_ASPECT_RATIO_PORTRAIT": "veo_3_1_i2v_lite",
|
|
132
|
+
},
|
|
133
|
+
"fast": {
|
|
134
|
+
"VIDEO_ASPECT_RATIO_LANDSCAPE": "veo_3_1_i2v_s_fast",
|
|
135
|
+
"VIDEO_ASPECT_RATIO_PORTRAIT": "veo_3_1_i2v_s_fast_portrait",
|
|
136
|
+
},
|
|
137
|
+
"quality": {
|
|
138
|
+
"VIDEO_ASPECT_RATIO_LANDSCAPE": "veo_3_1_i2v_s",
|
|
139
|
+
"VIDEO_ASPECT_RATIO_PORTRAIT": "veo_3_1_i2v_s_portrait",
|
|
140
|
+
},
|
|
141
|
+
},
|
|
142
|
+
# Tier 2 (Ultra) — five quality levels:
|
|
143
|
+
# - lite: `veo_3_1_i2v_lite` (5 credits, multi-aspect)
|
|
144
|
+
# - fast: `_fast_ultra` family (10 credits, default, balanced)
|
|
145
|
+
# - quality: `veo_3_1_i2v_s*` family (highest fidelity, slowest)
|
|
146
|
+
# - lite_relaxed: `veo_3_1_i2v_lite_low_priority` (0 credits,
|
|
147
|
+
# low-priority queue, Ultra-only)
|
|
148
|
+
# - fast_relaxed: `veo_3_1_i2v_s_fast_ultra_relaxed` (0 credits,
|
|
149
|
+
# low-priority queue, Ultra-only).
|
|
150
|
+
# PORTRAIT keys for the `_relaxed` family are not yet verified
|
|
151
|
+
# from a real curl; we reuse the LANDSCAPE key as a best-effort
|
|
152
|
+
# fallback. If Flow rejects portrait dispatches, capture a portrait
|
|
153
|
+
# curl and add the proper key here.
|
|
154
|
+
"PAYGATE_TIER_TWO": {
|
|
155
|
+
"lite": {
|
|
156
|
+
"VIDEO_ASPECT_RATIO_LANDSCAPE": "veo_3_1_i2v_lite",
|
|
157
|
+
"VIDEO_ASPECT_RATIO_PORTRAIT": "veo_3_1_i2v_lite",
|
|
158
|
+
},
|
|
159
|
+
"fast": {
|
|
160
|
+
"VIDEO_ASPECT_RATIO_LANDSCAPE": "veo_3_1_i2v_s_fast_ultra",
|
|
161
|
+
"VIDEO_ASPECT_RATIO_PORTRAIT": "veo_3_1_i2v_s_fast_portrait_ultra",
|
|
162
|
+
},
|
|
163
|
+
"quality": {
|
|
164
|
+
"VIDEO_ASPECT_RATIO_LANDSCAPE": "veo_3_1_i2v_s",
|
|
165
|
+
"VIDEO_ASPECT_RATIO_PORTRAIT": "veo_3_1_i2v_s_portrait",
|
|
166
|
+
},
|
|
167
|
+
"lite_relaxed": {
|
|
168
|
+
"VIDEO_ASPECT_RATIO_LANDSCAPE": "veo_3_1_i2v_lite_low_priority",
|
|
169
|
+
"VIDEO_ASPECT_RATIO_PORTRAIT": "veo_3_1_i2v_lite_low_priority",
|
|
170
|
+
},
|
|
171
|
+
"fast_relaxed": {
|
|
172
|
+
"VIDEO_ASPECT_RATIO_LANDSCAPE": "veo_3_1_i2v_s_fast_ultra_relaxed",
|
|
173
|
+
"VIDEO_ASPECT_RATIO_PORTRAIT": "veo_3_1_i2v_s_fast_ultra_relaxed",
|
|
174
|
+
},
|
|
175
|
+
},
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
DEFAULT_VIDEO_QUALITY = "fast"
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def resolve_video_model(
|
|
182
|
+
paygate_tier: str, aspect_ratio: str, quality: Optional[str] = None
|
|
183
|
+
) -> Optional[str]:
|
|
184
|
+
"""Resolve a Flow video model key from tier + aspect + quality.
|
|
185
|
+
|
|
186
|
+
Falls back through (quality → fast) → (tier → TIER_ONE) → None so
|
|
187
|
+
a stale frontend or unknown tier can't break dispatch silently.
|
|
188
|
+
"""
|
|
189
|
+
q = (quality or DEFAULT_VIDEO_QUALITY).lower()
|
|
190
|
+
tier_map = (
|
|
191
|
+
VIDEO_MODEL_KEYS.get(paygate_tier)
|
|
192
|
+
or VIDEO_MODEL_KEYS.get("PAYGATE_TIER_ONE")
|
|
193
|
+
or {}
|
|
194
|
+
)
|
|
195
|
+
quality_map = tier_map.get(q) or tier_map.get(DEFAULT_VIDEO_QUALITY) or {}
|
|
196
|
+
return quality_map.get(aspect_ratio)
|
|
197
|
+
|
|
198
|
+
# project_id must match the shape Google Flow returns (UUID-ish). Validated at
|
|
199
|
+
# handler boundaries to prevent path traversal into arbitrary API URLs.
|
|
200
|
+
_PROJECT_ID_RE = re.compile(r"^[A-Za-z0-9_-]{1,128}$")
|
|
201
|
+
|
|
202
|
+
# Flow CDN URLs embed the UUID media_id in the path:
|
|
203
|
+
# https://flow-content.google/video/<UUID>?Expires=...&Signature=...
|
|
204
|
+
# When the polling response omits `metadata.video.mediaId` (it usually does for
|
|
205
|
+
# video — only `mediaGenerationId` which is a base64 protobuf, NOT a UUID),
|
|
206
|
+
# we recover the UUID from the URL exactly like flowkit does.
|
|
207
|
+
_UUID_IN_URL_RE = re.compile(
|
|
208
|
+
r"/([0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12})",
|
|
209
|
+
re.IGNORECASE,
|
|
210
|
+
)
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def _media_id_from_url(url: Optional[str]) -> Optional[str]:
|
|
214
|
+
if not isinstance(url, str):
|
|
215
|
+
return None
|
|
216
|
+
m = _UUID_IN_URL_RE.search(url)
|
|
217
|
+
return m.group(1) if m else None
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def _first_media_entry(resp: Any, mid: str) -> dict[str, Any]:
|
|
221
|
+
"""Pull the media object for ``mid`` from a batchCheckAsync media response.
|
|
222
|
+
|
|
223
|
+
The body is either a bare ``[{...}]`` list or ``{media:[{...}]}`` — handle
|
|
224
|
+
both. Falls back to the first entry when the name doesn't match.
|
|
225
|
+
"""
|
|
226
|
+
data = resp.get("data") if isinstance(resp, dict) else None
|
|
227
|
+
media = (
|
|
228
|
+
data if isinstance(data, list)
|
|
229
|
+
else (data.get("media") if isinstance(data, dict) else None)
|
|
230
|
+
)
|
|
231
|
+
if not isinstance(media, list):
|
|
232
|
+
return {}
|
|
233
|
+
for m in media:
|
|
234
|
+
if isinstance(m, dict) and m.get("name") == mid:
|
|
235
|
+
return m
|
|
236
|
+
return media[0] if media and isinstance(media[0], dict) else {}
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def _media_url_from_entry(m: dict[str, Any]) -> Optional[str]:
|
|
240
|
+
"""Best-effort playable URL from a finished media object (if Flow inlines
|
|
241
|
+
one in the status payload; otherwise the caller fetches the bytes)."""
|
|
242
|
+
video = m.get("video") if isinstance(m.get("video"), dict) else {}
|
|
243
|
+
gen = video.get("generatedVideo") if isinstance(video.get("generatedVideo"), dict) else {}
|
|
244
|
+
for cand in (
|
|
245
|
+
video.get("fifeUrl"), gen.get("fifeUrl"),
|
|
246
|
+
video.get("servingBaseUri"), gen.get("servingBaseUri"),
|
|
247
|
+
m.get("fifeUrl"),
|
|
248
|
+
):
|
|
249
|
+
if isinstance(cand, str) and cand:
|
|
250
|
+
return cand
|
|
251
|
+
return None
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def _extract_inner_api_error(resp: Any) -> Optional[str]:
|
|
255
|
+
"""Surface a Flow API error if the response envelope indicates one.
|
|
256
|
+
|
|
257
|
+
`flow_client.api_request` returns ``{"id", "status", "data"}`` on a
|
|
258
|
+
completed round-trip even when the underlying call failed — the HTTP
|
|
259
|
+
status from Google lives on ``resp["status"]`` and the structured error
|
|
260
|
+
body on ``resp["data"]["error"]``. Top-level ``resp["error"]`` is only
|
|
261
|
+
set by the client itself for transport-level failures (which we already
|
|
262
|
+
handle). When ``status >= 400`` or ``data.error.status`` is set, the
|
|
263
|
+
request must be reported as failed — silently treating an empty
|
|
264
|
+
``media_ids`` list as success masked Flow's content-filter rejections
|
|
265
|
+
(e.g. ``PUBLIC_ERROR_PROMINENT_PEOPLE_FILTER_FAILED``).
|
|
266
|
+
"""
|
|
267
|
+
if not isinstance(resp, dict):
|
|
268
|
+
return None
|
|
269
|
+
status = resp.get("status")
|
|
270
|
+
data = resp.get("data") if isinstance(resp.get("data"), dict) else None
|
|
271
|
+
err = data.get("error") if isinstance(data, dict) else None
|
|
272
|
+
has_status_err = isinstance(status, int) and status >= 400
|
|
273
|
+
has_data_err = isinstance(err, dict)
|
|
274
|
+
if not (has_status_err or has_data_err):
|
|
275
|
+
return None
|
|
276
|
+
if has_data_err:
|
|
277
|
+
reasons: list[str] = []
|
|
278
|
+
for detail in err.get("details") or []:
|
|
279
|
+
if isinstance(detail, dict):
|
|
280
|
+
r = detail.get("reason")
|
|
281
|
+
if isinstance(r, str) and r:
|
|
282
|
+
reasons.append(r)
|
|
283
|
+
msg = err.get("message") or err.get("status") or "API error"
|
|
284
|
+
return f"{reasons[0]}: {msg}" if reasons else str(msg)
|
|
285
|
+
return f"API_{status}"
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
def is_valid_project_id(project_id: str) -> bool:
|
|
289
|
+
return bool(_PROJECT_ID_RE.fullmatch(project_id))
|
|
290
|
+
|
|
291
|
+
# Captcha action strings recognised by Google Flow.
|
|
292
|
+
CAPTCHA_IMAGE = "IMAGE_GENERATION"
|
|
293
|
+
CAPTCHA_VIDEO = "VIDEO_GENERATION"
|
|
294
|
+
|
|
295
|
+
# Default max operations to poll in parallel. Conservative; flowkit passes
|
|
296
|
+
# the full list at once.
|
|
297
|
+
_MAX_VIDEO_OPS = 4
|
|
298
|
+
|
|
299
|
+
# Image variants per dispatch are capped server-side as defence-in-depth — the
|
|
300
|
+
# UI clamps to 4 too. Any value above this is silently coerced down.
|
|
301
|
+
MAX_VARIANT_COUNT = 4
|
|
302
|
+
|
|
303
|
+
# Minimal static headers that have worked against labs.google in flowkit.
|
|
304
|
+
_TRPC_HEADERS = {
|
|
305
|
+
"content-type": "application/json",
|
|
306
|
+
"accept": "*/*",
|
|
307
|
+
}
|
|
308
|
+
_API_HEADERS = {
|
|
309
|
+
"content-type": "text/plain;charset=UTF-8",
|
|
310
|
+
"accept": "*/*",
|
|
311
|
+
"origin": "https://labs.google",
|
|
312
|
+
"referer": "https://labs.google/",
|
|
313
|
+
}
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
def _client_context(project_id: str, paygate_tier: str) -> dict:
|
|
317
|
+
"""Skeleton clientContext — extension fills in recaptchaContext.token.
|
|
318
|
+
|
|
319
|
+
`paygate_tier` is REQUIRED (no default). Pre-v1.1.5 the default was
|
|
320
|
+
`"PAYGATE_TIER_ONE"` which silently downgraded Ultra users when any
|
|
321
|
+
upstream code path forgot to pass tier. Now we raise loudly on
|
|
322
|
+
invalid / unknown values so a code regression can't quietly serve
|
|
323
|
+
Pro to an Ultra account.
|
|
324
|
+
"""
|
|
325
|
+
if paygate_tier not in _VALID_TIERS:
|
|
326
|
+
raise ValueError(
|
|
327
|
+
f"invalid paygate_tier {paygate_tier!r} — must be one of {sorted(_VALID_TIERS)}"
|
|
328
|
+
)
|
|
329
|
+
return {
|
|
330
|
+
"projectId": str(project_id),
|
|
331
|
+
"recaptchaContext": {
|
|
332
|
+
"applicationType": "RECAPTCHA_APPLICATION_TYPE_WEB",
|
|
333
|
+
"token": "",
|
|
334
|
+
},
|
|
335
|
+
"sessionId": f";{int(time.time() * 1000)}",
|
|
336
|
+
"tool": "PINHOLE",
|
|
337
|
+
"userPaygateTier": paygate_tier,
|
|
338
|
+
}
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
def _generate_images_url(project_id: str) -> str:
|
|
342
|
+
return f"{FLOW_API_BASE}/v1/projects/{project_id}/flowMedia:batchGenerateImages"
|
|
343
|
+
|
|
344
|
+
|
|
345
|
+
class FlowSDK:
|
|
346
|
+
"""High-level helpers on top of a FlowClient. Stateless; one per request."""
|
|
347
|
+
|
|
348
|
+
def __init__(self, client: Optional[FlowClient] = None) -> None:
|
|
349
|
+
if client is None:
|
|
350
|
+
raise ValueError(
|
|
351
|
+
"FlowSDK requires an explicit FlowClient — the module-level "
|
|
352
|
+
"singleton has been removed; obtain a client via pool.get_by_api_key()."
|
|
353
|
+
)
|
|
354
|
+
self._client = client
|
|
355
|
+
|
|
356
|
+
# ── project listing (TRPC search) ──────────────────────────────────────
|
|
357
|
+
async def search_user_projects(
|
|
358
|
+
self,
|
|
359
|
+
cursor: Optional[str] = None,
|
|
360
|
+
page_size: int = 20,
|
|
361
|
+
tool: str = "PINHOLE",
|
|
362
|
+
) -> dict[str, Any]:
|
|
363
|
+
"""Fetch one page of the user's Flow project list.
|
|
364
|
+
|
|
365
|
+
Returns ``{raw, projects, next_page_token}`` on success or
|
|
366
|
+
``{raw, error}`` on failure. Each project is a dict with
|
|
367
|
+
``project_id`` and ``project_title`` plus optional
|
|
368
|
+
``thumbnail_media_key`` and ``creation_time``.
|
|
369
|
+
"""
|
|
370
|
+
import json as _json
|
|
371
|
+
from urllib.parse import quote
|
|
372
|
+
|
|
373
|
+
input_json: dict[str, Any] = {
|
|
374
|
+
"json": {
|
|
375
|
+
"pageSize": page_size,
|
|
376
|
+
"toolName": tool,
|
|
377
|
+
"cursor": cursor,
|
|
378
|
+
},
|
|
379
|
+
}
|
|
380
|
+
# TRPC encodes the literal `undefined` cursor via meta on the
|
|
381
|
+
# first call. Subsequent calls pass a string cursor directly.
|
|
382
|
+
if cursor is None:
|
|
383
|
+
input_json["meta"] = {"values": {"cursor": ["undefined"]}}
|
|
384
|
+
|
|
385
|
+
url = (
|
|
386
|
+
f"{TRPC_SEARCH_PROJECTS}?input="
|
|
387
|
+
+ quote(_json.dumps(input_json, separators=(",", ":")))
|
|
388
|
+
)
|
|
389
|
+
resp = await self._client.trpc_request(
|
|
390
|
+
url=url,
|
|
391
|
+
method="GET",
|
|
392
|
+
headers=_TRPC_HEADERS,
|
|
393
|
+
body=None,
|
|
394
|
+
)
|
|
395
|
+
if isinstance(resp, dict) and resp.get("error"):
|
|
396
|
+
return {"raw": resp, "error": resp["error"]}
|
|
397
|
+
|
|
398
|
+
# Response shape (per the canonical TRPC envelope):
|
|
399
|
+
# data.result.data.json.result.{projects, nextPageToken}
|
|
400
|
+
try:
|
|
401
|
+
inner = resp["data"]["result"]["data"]["json"]["result"]
|
|
402
|
+
except (KeyError, TypeError):
|
|
403
|
+
return {"raw": resp, "error": "unexpected_search_response_shape"}
|
|
404
|
+
|
|
405
|
+
raw_projects = inner.get("projects") or []
|
|
406
|
+
projects: list[dict[str, Any]] = []
|
|
407
|
+
for p in raw_projects:
|
|
408
|
+
if not isinstance(p, dict):
|
|
409
|
+
continue
|
|
410
|
+
pid = p.get("projectId")
|
|
411
|
+
if not isinstance(pid, str) or not pid:
|
|
412
|
+
continue
|
|
413
|
+
info = p.get("projectInfo") if isinstance(p.get("projectInfo"), dict) else {}
|
|
414
|
+
projects.append({
|
|
415
|
+
"project_id": pid,
|
|
416
|
+
"project_title": info.get("projectTitle") or "Untitled",
|
|
417
|
+
"thumbnail_media_key": info.get("thumbnailMediaKey"),
|
|
418
|
+
"creation_time": p.get("creationTime"),
|
|
419
|
+
})
|
|
420
|
+
return {
|
|
421
|
+
"raw": resp,
|
|
422
|
+
"projects": projects,
|
|
423
|
+
"next_page_token": inner.get("nextPageToken"),
|
|
424
|
+
}
|
|
425
|
+
|
|
426
|
+
async def list_user_projects_all(
|
|
427
|
+
self, tool: str = "PINHOLE", max_pages: int = 10
|
|
428
|
+
) -> dict[str, Any]:
|
|
429
|
+
"""Paginate `search_user_projects` until exhausted (or max_pages).
|
|
430
|
+
|
|
431
|
+
Returns ``{projects, truncated}`` — truncated is True when we hit
|
|
432
|
+
the page cap with a non-null next_page_token still present.
|
|
433
|
+
"""
|
|
434
|
+
all_projects: list[dict[str, Any]] = []
|
|
435
|
+
cursor: Optional[str] = None
|
|
436
|
+
for _ in range(max_pages):
|
|
437
|
+
page = await self.search_user_projects(cursor=cursor, tool=tool)
|
|
438
|
+
if page.get("error"):
|
|
439
|
+
return {
|
|
440
|
+
"projects": all_projects,
|
|
441
|
+
"truncated": True,
|
|
442
|
+
"error": page["error"],
|
|
443
|
+
}
|
|
444
|
+
all_projects.extend(page.get("projects") or [])
|
|
445
|
+
cursor = page.get("next_page_token")
|
|
446
|
+
if not cursor:
|
|
447
|
+
return {"projects": all_projects, "truncated": False}
|
|
448
|
+
return {"projects": all_projects, "truncated": True}
|
|
449
|
+
|
|
450
|
+
# ── project creation (TRPC) ────────────────────────────────────────────
|
|
451
|
+
async def create_project(
|
|
452
|
+
self, title: str, tool: str = "PINHOLE"
|
|
453
|
+
) -> dict[str, Any]:
|
|
454
|
+
body = {"json": {"projectTitle": title, "toolName": tool}}
|
|
455
|
+
resp = await self._client.trpc_request(
|
|
456
|
+
url=TRPC_CREATE_PROJECT,
|
|
457
|
+
method="POST",
|
|
458
|
+
headers=_TRPC_HEADERS,
|
|
459
|
+
body=body,
|
|
460
|
+
)
|
|
461
|
+
if isinstance(resp, dict) and resp.get("error"):
|
|
462
|
+
return {"raw": resp, "error": resp["error"]}
|
|
463
|
+
|
|
464
|
+
project_id = _extract_project_id(resp)
|
|
465
|
+
out: dict[str, Any] = {"raw": resp}
|
|
466
|
+
if project_id is None:
|
|
467
|
+
out["error"] = "no_project_id_in_response"
|
|
468
|
+
else:
|
|
469
|
+
out["project_id"] = project_id
|
|
470
|
+
return out
|
|
471
|
+
|
|
472
|
+
# ── model catalog (TRPC) ───────────────────────────────────────────────
|
|
473
|
+
async def get_project_initial_data(self, project_id: str) -> dict[str, Any]:
|
|
474
|
+
"""Fetch the project's initial data, which carries the full model
|
|
475
|
+
catalog. Returns ``{raw, data}`` on success (``data`` is the inner
|
|
476
|
+
``json`` block with ``modelConfig`` / ``userData`` /
|
|
477
|
+
``projectContents.externalReferenceMedia``) or ``{raw, error}``.
|
|
478
|
+
"""
|
|
479
|
+
import json as _json
|
|
480
|
+
from urllib.parse import quote
|
|
481
|
+
|
|
482
|
+
input_json = {"json": {"projectId": str(project_id)}}
|
|
483
|
+
url = (
|
|
484
|
+
f"{TRPC_PROJECT_INITIAL_DATA}?input="
|
|
485
|
+
+ quote(_json.dumps(input_json, separators=(",", ":")))
|
|
486
|
+
)
|
|
487
|
+
resp = await self._client.trpc_request(
|
|
488
|
+
url=url, method="GET", headers=_TRPC_HEADERS, body=None
|
|
489
|
+
)
|
|
490
|
+
if isinstance(resp, dict) and resp.get("error"):
|
|
491
|
+
return {"raw": resp, "error": resp["error"]}
|
|
492
|
+
try:
|
|
493
|
+
data = resp["data"]["result"]["data"]["json"]
|
|
494
|
+
except (KeyError, TypeError):
|
|
495
|
+
return {"raw": resp, "error": "unexpected_initial_data_shape"}
|
|
496
|
+
return {"raw": resp, "data": data}
|
|
497
|
+
|
|
498
|
+
# ── video generation (async via operations) ────────────────────────────
|
|
499
|
+
async def gen_video(
|
|
500
|
+
self,
|
|
501
|
+
prompt: str,
|
|
502
|
+
project_id: str,
|
|
503
|
+
start_media_id: Optional[str] = None,
|
|
504
|
+
aspect_ratio: str = "VIDEO_ASPECT_RATIO_LANDSCAPE",
|
|
505
|
+
paygate_tier: Optional[str] = None,
|
|
506
|
+
scene_id: Optional[str] = None,
|
|
507
|
+
start_media_ids: Optional[list[str]] = None,
|
|
508
|
+
video_quality: Optional[str] = None,
|
|
509
|
+
model_key: Optional[str] = None,
|
|
510
|
+
) -> dict[str, Any]:
|
|
511
|
+
"""Kick off i2v operation(s). Returns ``{raw, operation_names}`` on
|
|
512
|
+
success or ``{raw, error}`` on failure. Operations are async — the
|
|
513
|
+
caller polls ``check_async`` until they complete.
|
|
514
|
+
|
|
515
|
+
``start_media_ids`` (optional list) — when provided, dispatch ONE
|
|
516
|
+
item per source image so a 4-variant upstream image produces 4
|
|
517
|
+
videos in a single batch (one operation per source). Falls back to
|
|
518
|
+
``start_media_id`` (single) if the list is missing/empty.
|
|
519
|
+
|
|
520
|
+
``video_quality`` ("fast" / "lite" / "quality" / "lite_relaxed"
|
|
521
|
+
/ "fast_relaxed") routes to a different Veo checkpoint. Defaults
|
|
522
|
+
to "fast" — the `_s_fast` family. The first three are available
|
|
523
|
+
on both Tier 1 (Pro) and Tier 2 (Ultra); the `_relaxed` variants
|
|
524
|
+
are 0-credit low-priority queues and are Ultra-only. See
|
|
525
|
+
``VIDEO_MODEL_KEYS`` for the per-tier mapping.
|
|
526
|
+
|
|
527
|
+
``paygate_tier`` is required. Pre-v1.1.5 it defaulted to
|
|
528
|
+
``"PAYGATE_TIER_ONE"`` which silently downgraded Ultra users.
|
|
529
|
+
Raise loudly instead — the worker should always have a tier
|
|
530
|
+
from the live extension signal before reaching here.
|
|
531
|
+
"""
|
|
532
|
+
if paygate_tier is None:
|
|
533
|
+
raise ValueError("paygate_tier is required — see docs/migrations/clear-polluted-paygate-tier.sql")
|
|
534
|
+
# Prefer a catalog-resolved key (dynamic, tier-correct); fall back to
|
|
535
|
+
# the static tier/quality/aspect map only when none was supplied.
|
|
536
|
+
if not model_key:
|
|
537
|
+
model_key = resolve_video_model(paygate_tier, aspect_ratio, video_quality)
|
|
538
|
+
if not model_key:
|
|
539
|
+
return {
|
|
540
|
+
"raw": None,
|
|
541
|
+
"error": (
|
|
542
|
+
f"no_video_model_for_tier_{paygate_tier}"
|
|
543
|
+
f"_quality_{video_quality or DEFAULT_VIDEO_QUALITY}"
|
|
544
|
+
f"_aspect_{aspect_ratio}"
|
|
545
|
+
),
|
|
546
|
+
}
|
|
547
|
+
|
|
548
|
+
# Normalise into a non-empty list of source media ids. Single
|
|
549
|
+
# `start_media_id` is the common case; `start_media_ids` is for
|
|
550
|
+
# batch-i2v from a multi-variant upstream image.
|
|
551
|
+
sources: list[str] = []
|
|
552
|
+
if start_media_ids:
|
|
553
|
+
sources = [m for m in start_media_ids if isinstance(m, str) and m]
|
|
554
|
+
if not sources and isinstance(start_media_id, str) and start_media_id:
|
|
555
|
+
sources = [start_media_id]
|
|
556
|
+
if not sources:
|
|
557
|
+
return {"raw": None, "error": "missing_start_media_id"}
|
|
558
|
+
|
|
559
|
+
ts = int(time.time() * 1000)
|
|
560
|
+
ctx = _client_context(project_id, paygate_tier)
|
|
561
|
+
items: list[dict[str, Any]] = []
|
|
562
|
+
for i, mid in enumerate(sources):
|
|
563
|
+
items.append({
|
|
564
|
+
"aspectRatio": aspect_ratio,
|
|
565
|
+
# Distinct seed per item so Flow doesn't dedupe.
|
|
566
|
+
"seed": (ts + i * 9973) % 1_000_000,
|
|
567
|
+
"textInput": {"structuredPrompt": {"parts": [{"text": prompt}]}},
|
|
568
|
+
"videoModelKey": model_key,
|
|
569
|
+
"startImage": {"mediaId": mid},
|
|
570
|
+
"metadata": {"sceneId": scene_id or str(uuid.uuid4())},
|
|
571
|
+
})
|
|
572
|
+
body = {
|
|
573
|
+
"clientContext": ctx,
|
|
574
|
+
"mediaGenerationContext": {"batchId": str(uuid.uuid4())},
|
|
575
|
+
"requests": items,
|
|
576
|
+
"useV2ModelConfig": True,
|
|
577
|
+
}
|
|
578
|
+
|
|
579
|
+
resp = await self._client.api_request(
|
|
580
|
+
url=VIDEO_I2V_URL,
|
|
581
|
+
method="POST",
|
|
582
|
+
headers=dict(_API_HEADERS),
|
|
583
|
+
body=body,
|
|
584
|
+
captcha_action=CAPTCHA_VIDEO,
|
|
585
|
+
)
|
|
586
|
+
if isinstance(resp, dict) and resp.get("error"):
|
|
587
|
+
return {"raw": resp, "error": resp["error"]}
|
|
588
|
+
inner_err = _extract_inner_api_error(resp)
|
|
589
|
+
if inner_err:
|
|
590
|
+
return {"raw": resp, "error": inner_err}
|
|
591
|
+
|
|
592
|
+
op_names = extract_operation_names(resp)
|
|
593
|
+
if not op_names:
|
|
594
|
+
return {"raw": resp, "error": "no_operations_in_response"}
|
|
595
|
+
out: dict[str, Any] = {"raw": resp, "operation_names": op_names}
|
|
596
|
+
# NEW low-priority workflow models return `data.workflows[]` with a
|
|
597
|
+
# `primaryMediaId` per workflow instead of operations. Surface the
|
|
598
|
+
# pairing so the poller can hit `/v1/media/<id>` directly.
|
|
599
|
+
workflows = extract_video_workflows(resp)
|
|
600
|
+
if workflows:
|
|
601
|
+
out["workflows"] = workflows
|
|
602
|
+
return out
|
|
603
|
+
|
|
604
|
+
# ── Veo 3.1 reference-to-video (r2v) ───────────────────────────────────
|
|
605
|
+
async def gen_video_r2v(
|
|
606
|
+
self,
|
|
607
|
+
prompt: str,
|
|
608
|
+
project_id: str,
|
|
609
|
+
ref_media_ids: list[str],
|
|
610
|
+
aspect_ratio: str = "VIDEO_ASPECT_RATIO_PORTRAIT",
|
|
611
|
+
paygate_tier: Optional[str] = None,
|
|
612
|
+
seed: Optional[int] = None,
|
|
613
|
+
reference_audio_ids: Optional[list[str]] = None,
|
|
614
|
+
model_key: Optional[str] = None,
|
|
615
|
+
scene_id: Optional[str] = None,
|
|
616
|
+
) -> dict[str, Any]:
|
|
617
|
+
"""Kick off Veo 3.1 reference-to-video generation. Uses
|
|
618
|
+
/video:batchAsyncGenerateVideoReferenceImages with a referenceImages[]
|
|
619
|
+
payload — distinct from the (now-stale) i2v startImage path.
|
|
620
|
+
|
|
621
|
+
Returns ``{raw, operation_names}`` (+ ``workflows`` for the
|
|
622
|
+
low-priority workflow schema) on success or ``{raw, error}`` on
|
|
623
|
+
failure. Polls via the shared ``check_async`` path.
|
|
624
|
+
|
|
625
|
+
``ref_media_ids`` MUST be non-empty (the model is
|
|
626
|
+
reference-conditioned, not text-only).
|
|
627
|
+
``aspect_ratio`` ∈ {PORTRAIT, LANDSCAPE}.
|
|
628
|
+
``reference_audio_ids`` — optional audio reference media ids (e.g.
|
|
629
|
+
``"algenib"``). When provided, attached as ``referenceAudio``; omitted
|
|
630
|
+
entirely otherwise.
|
|
631
|
+
``paygate_tier`` required — same as gen_video.
|
|
632
|
+
"""
|
|
633
|
+
if paygate_tier is None:
|
|
634
|
+
raise ValueError("paygate_tier is required")
|
|
635
|
+
if aspect_ratio not in R2V_VALID_ASPECTS:
|
|
636
|
+
return {
|
|
637
|
+
"raw": None,
|
|
638
|
+
"error": f"r2v_aspect_unsupported_{aspect_ratio}",
|
|
639
|
+
}
|
|
640
|
+
cleaned_refs = [m for m in (ref_media_ids or []) if isinstance(m, str) and m]
|
|
641
|
+
if not cleaned_refs:
|
|
642
|
+
return {"raw": None, "error": "missing_ref_media_ids"}
|
|
643
|
+
|
|
644
|
+
ts = int(time.time() * 1000)
|
|
645
|
+
used_seed = seed if seed is not None else ts % 1_000_000
|
|
646
|
+
ctx = _client_context(project_id, paygate_tier)
|
|
647
|
+
request_item: dict[str, Any] = {
|
|
648
|
+
"aspectRatio": aspect_ratio,
|
|
649
|
+
"textInput": {"structuredPrompt": {"parts": [{"text": prompt}]}},
|
|
650
|
+
"videoModelKey": model_key or VEO_R2V_MODEL_KEY,
|
|
651
|
+
"seed": used_seed,
|
|
652
|
+
"metadata": {"sceneId": scene_id} if scene_id else {},
|
|
653
|
+
"referenceImages": [
|
|
654
|
+
{"mediaId": mid, "imageUsageType": "IMAGE_USAGE_TYPE_ASSET"}
|
|
655
|
+
for mid in cleaned_refs
|
|
656
|
+
],
|
|
657
|
+
}
|
|
658
|
+
cleaned_audio = [a for a in (reference_audio_ids or []) if isinstance(a, str) and a]
|
|
659
|
+
if cleaned_audio:
|
|
660
|
+
request_item["referenceAudio"] = [{"mediaId": a} for a in cleaned_audio]
|
|
661
|
+
body = {
|
|
662
|
+
"mediaGenerationContext": {
|
|
663
|
+
"batchId": str(uuid.uuid4()),
|
|
664
|
+
# V2 config flags silent-audio outputs as failures so the
|
|
665
|
+
# caller can retry instead of getting a degraded video.
|
|
666
|
+
"audioFailurePreference": "BLOCK_SILENCED_VIDEOS",
|
|
667
|
+
},
|
|
668
|
+
"clientContext": {**ctx, "sessionId": f";{ts}"},
|
|
669
|
+
"requests": [request_item],
|
|
670
|
+
"useV2ModelConfig": True,
|
|
671
|
+
}
|
|
672
|
+
|
|
673
|
+
resp = await self._client.api_request(
|
|
674
|
+
url=VIDEO_R2V_URL,
|
|
675
|
+
method="POST",
|
|
676
|
+
headers=dict(_API_HEADERS),
|
|
677
|
+
body=body,
|
|
678
|
+
captcha_action=CAPTCHA_VIDEO,
|
|
679
|
+
)
|
|
680
|
+
if isinstance(resp, dict) and resp.get("error"):
|
|
681
|
+
return {"raw": resp, "error": resp["error"]}
|
|
682
|
+
inner_err = _extract_inner_api_error(resp)
|
|
683
|
+
if inner_err:
|
|
684
|
+
return {"raw": resp, "error": inner_err}
|
|
685
|
+
|
|
686
|
+
op_names = extract_operation_names(resp)
|
|
687
|
+
if not op_names:
|
|
688
|
+
return {"raw": resp, "error": "no_operations_in_response"}
|
|
689
|
+
out: dict[str, Any] = {"raw": resp, "operation_names": op_names}
|
|
690
|
+
workflows = extract_video_workflows(resp)
|
|
691
|
+
if workflows:
|
|
692
|
+
out["workflows"] = workflows
|
|
693
|
+
if scene_id:
|
|
694
|
+
out["scene_id"] = scene_id
|
|
695
|
+
return out
|
|
696
|
+
|
|
697
|
+
# ── Veo 3.1 text-to-video (t2v) ────────────────────────────────────────
|
|
698
|
+
async def gen_video_t2v(
|
|
699
|
+
self,
|
|
700
|
+
prompt: str,
|
|
701
|
+
project_id: str,
|
|
702
|
+
aspect_ratio: str = "VIDEO_ASPECT_RATIO_LANDSCAPE",
|
|
703
|
+
paygate_tier: Optional[str] = None,
|
|
704
|
+
seed: Optional[int] = None,
|
|
705
|
+
model_key: Optional[str] = None,
|
|
706
|
+
scene_id: Optional[str] = None,
|
|
707
|
+
) -> dict[str, Any]:
|
|
708
|
+
"""Kick off text-to-video. Same workflow path as r2v but with no image
|
|
709
|
+
or audio inputs — hits /video:batchAsyncGenerateVideoText.
|
|
710
|
+
|
|
711
|
+
``model_key`` is required (resolved from the catalog) — there is no
|
|
712
|
+
static fallback for t2v.
|
|
713
|
+
|
|
714
|
+
``scene_id`` (optional) pins the output into a client-chosen scene via
|
|
715
|
+
``metadata.sceneId`` — Flow otherwise auto-creates a scene and never
|
|
716
|
+
tells us its id, which would leave the clip un-extendable. When set, the
|
|
717
|
+
scene id is echoed back in the result so the caller can remember it.
|
|
718
|
+
"""
|
|
719
|
+
if paygate_tier is None:
|
|
720
|
+
raise ValueError("paygate_tier is required")
|
|
721
|
+
if aspect_ratio not in R2V_VALID_ASPECTS:
|
|
722
|
+
return {"raw": None, "error": f"t2v_aspect_unsupported_{aspect_ratio}"}
|
|
723
|
+
if not model_key:
|
|
724
|
+
return {"raw": None, "error": "t2v_model_key_required"}
|
|
725
|
+
|
|
726
|
+
ts = int(time.time() * 1000)
|
|
727
|
+
used_seed = seed if seed is not None else ts % 1_000_000
|
|
728
|
+
ctx = _client_context(project_id, paygate_tier)
|
|
729
|
+
request_item = {
|
|
730
|
+
"aspectRatio": aspect_ratio,
|
|
731
|
+
"textInput": {"structuredPrompt": {"parts": [{"text": prompt}]}},
|
|
732
|
+
"videoModelKey": model_key,
|
|
733
|
+
"seed": used_seed,
|
|
734
|
+
"metadata": {"sceneId": scene_id} if scene_id else {},
|
|
735
|
+
}
|
|
736
|
+
body = {
|
|
737
|
+
"mediaGenerationContext": {
|
|
738
|
+
"batchId": str(uuid.uuid4()),
|
|
739
|
+
"audioFailurePreference": "BLOCK_SILENCED_VIDEOS",
|
|
740
|
+
},
|
|
741
|
+
"clientContext": {**ctx, "sessionId": f";{ts}"},
|
|
742
|
+
"requests": [request_item],
|
|
743
|
+
"useV2ModelConfig": True,
|
|
744
|
+
}
|
|
745
|
+
|
|
746
|
+
resp = await self._client.api_request(
|
|
747
|
+
url=VIDEO_T2V_URL,
|
|
748
|
+
method="POST",
|
|
749
|
+
headers=dict(_API_HEADERS),
|
|
750
|
+
body=body,
|
|
751
|
+
captcha_action=CAPTCHA_VIDEO,
|
|
752
|
+
)
|
|
753
|
+
if isinstance(resp, dict) and resp.get("error"):
|
|
754
|
+
return {"raw": resp, "error": resp["error"]}
|
|
755
|
+
inner_err = _extract_inner_api_error(resp)
|
|
756
|
+
if inner_err:
|
|
757
|
+
return {"raw": resp, "error": inner_err}
|
|
758
|
+
|
|
759
|
+
op_names = extract_operation_names(resp)
|
|
760
|
+
if not op_names:
|
|
761
|
+
return {"raw": resp, "error": "no_operations_in_response"}
|
|
762
|
+
out: dict[str, Any] = {"raw": resp, "operation_names": op_names}
|
|
763
|
+
workflows = extract_video_workflows(resp)
|
|
764
|
+
if workflows:
|
|
765
|
+
out["workflows"] = workflows
|
|
766
|
+
if scene_id:
|
|
767
|
+
out["scene_id"] = scene_id
|
|
768
|
+
return out
|
|
769
|
+
|
|
770
|
+
# ── Veo 3.1 extend (continue a clip) ───────────────────────────────────
|
|
771
|
+
async def gen_video_extend(
|
|
772
|
+
self,
|
|
773
|
+
prompt: str,
|
|
774
|
+
project_id: str,
|
|
775
|
+
source_media_id: str,
|
|
776
|
+
scene_id: str,
|
|
777
|
+
position: int = 1,
|
|
778
|
+
aspect_ratio: str = "VIDEO_ASPECT_RATIO_LANDSCAPE",
|
|
779
|
+
paygate_tier: Optional[str] = None,
|
|
780
|
+
seed: Optional[int] = None,
|
|
781
|
+
model_key: Optional[str] = None,
|
|
782
|
+
) -> dict[str, Any]:
|
|
783
|
+
"""Extend an existing Veo clip (``batchAsyncGenerateVideoExtendVideo``).
|
|
784
|
+
|
|
785
|
+
``source_media_id`` is the clip to continue — for extend this is the
|
|
786
|
+
source video's *operation/generation id* (``video.operation.name``),
|
|
787
|
+
NOT its workflow primaryMediaId. ``scene_id`` is the scene the source
|
|
788
|
+
belongs to and ``position`` is the slot the extension occupies in that
|
|
789
|
+
scene's timeline (1 for the first extension). Only Veo-generated videos
|
|
790
|
+
can be extended.
|
|
791
|
+
|
|
792
|
+
Returns the workflow-schema shape (``{raw, operation_names, workflows}``)
|
|
793
|
+
like the other video dispatchers; poll via the shared ``check_async``.
|
|
794
|
+
"""
|
|
795
|
+
if paygate_tier is None:
|
|
796
|
+
raise ValueError("paygate_tier is required")
|
|
797
|
+
if not model_key:
|
|
798
|
+
return {"raw": None, "error": "extend_model_key_required"}
|
|
799
|
+
if not isinstance(source_media_id, str) or not source_media_id:
|
|
800
|
+
return {"raw": None, "error": "missing_source_media_id"}
|
|
801
|
+
if not isinstance(scene_id, str) or not scene_id:
|
|
802
|
+
return {"raw": None, "error": "missing_scene_id"}
|
|
803
|
+
|
|
804
|
+
ts = int(time.time() * 1000)
|
|
805
|
+
used_seed = seed if seed is not None else ts % 1_000_000
|
|
806
|
+
ctx = _client_context(project_id, paygate_tier)
|
|
807
|
+
body = {
|
|
808
|
+
"mediaGenerationContext": {
|
|
809
|
+
"batchId": str(uuid.uuid4()),
|
|
810
|
+
"audioFailurePreference": "BLOCK_SILENCED_VIDEOS",
|
|
811
|
+
"sceneContext": {"sceneId": scene_id, "position": position},
|
|
812
|
+
},
|
|
813
|
+
"clientContext": {**ctx, "sessionId": f";{ts}"},
|
|
814
|
+
"requests": [
|
|
815
|
+
{
|
|
816
|
+
"aspectRatio": aspect_ratio,
|
|
817
|
+
"textInput": {"structuredPrompt": {"parts": [{"text": prompt}]}},
|
|
818
|
+
"videoModelKey": model_key,
|
|
819
|
+
"seed": used_seed,
|
|
820
|
+
"metadata": {"sceneId": scene_id},
|
|
821
|
+
"videoInput": {"mediaId": source_media_id},
|
|
822
|
+
}
|
|
823
|
+
],
|
|
824
|
+
"useV2ModelConfig": True,
|
|
825
|
+
}
|
|
826
|
+
return await self._dispatch_video(VIDEO_EXTEND_URL, body)
|
|
827
|
+
|
|
828
|
+
# ── Veo 3.1 edit (video-to-video) ──────────────────────────────────────
|
|
829
|
+
async def gen_video_edit(
|
|
830
|
+
self,
|
|
831
|
+
prompt: str,
|
|
832
|
+
project_id: str,
|
|
833
|
+
source_media_id: str,
|
|
834
|
+
workflow_id: str,
|
|
835
|
+
end_frame_index: int,
|
|
836
|
+
start_frame_index: int = 0,
|
|
837
|
+
aspect_ratio: str = "VIDEO_ASPECT_RATIO_LANDSCAPE",
|
|
838
|
+
paygate_tier: Optional[str] = None,
|
|
839
|
+
seed: Optional[int] = None,
|
|
840
|
+
model_key: Optional[str] = None,
|
|
841
|
+
) -> dict[str, Any]:
|
|
842
|
+
"""Edit an existing video (``batchAsyncGenerateVideoEditVideo``, v2v).
|
|
843
|
+
|
|
844
|
+
``source_media_id`` is the source video's workflow primaryMediaId (its
|
|
845
|
+
media ``name``). ``workflow_id`` is the source's workflow; the edit
|
|
846
|
+
produces a new media version under it. ``start_frame_index`` /
|
|
847
|
+
``end_frame_index`` select the slice to edit — frames are 30fps, so a
|
|
848
|
+
full 8s clip is 0..240. Verified key: ``abra_edit``.
|
|
849
|
+
|
|
850
|
+
Returns the workflow-schema shape; poll via the shared ``check_async``.
|
|
851
|
+
"""
|
|
852
|
+
if paygate_tier is None:
|
|
853
|
+
raise ValueError("paygate_tier is required")
|
|
854
|
+
if not model_key:
|
|
855
|
+
return {"raw": None, "error": "edit_model_key_required"}
|
|
856
|
+
if not isinstance(source_media_id, str) or not source_media_id:
|
|
857
|
+
return {"raw": None, "error": "missing_source_media_id"}
|
|
858
|
+
if not isinstance(workflow_id, str) or not workflow_id:
|
|
859
|
+
return {"raw": None, "error": "missing_workflow_id"}
|
|
860
|
+
|
|
861
|
+
ts = int(time.time() * 1000)
|
|
862
|
+
used_seed = seed if seed is not None else ts % 1_000_000
|
|
863
|
+
ctx = _client_context(project_id, paygate_tier)
|
|
864
|
+
body = {
|
|
865
|
+
"mediaGenerationContext": {
|
|
866
|
+
"batchId": str(uuid.uuid4()),
|
|
867
|
+
"audioFailurePreference": "BLOCK_SILENCED_VIDEOS",
|
|
868
|
+
},
|
|
869
|
+
"clientContext": {**ctx, "sessionId": f";{ts}"},
|
|
870
|
+
"requests": [
|
|
871
|
+
{
|
|
872
|
+
"aspectRatio": aspect_ratio,
|
|
873
|
+
"textInput": {"structuredPrompt": {"parts": [{"text": prompt}]}},
|
|
874
|
+
"videoModelKey": model_key,
|
|
875
|
+
"seed": used_seed,
|
|
876
|
+
"metadata": {"workflowId": workflow_id},
|
|
877
|
+
"videoInput": {
|
|
878
|
+
"mediaId": source_media_id,
|
|
879
|
+
"startFrameIndex": start_frame_index,
|
|
880
|
+
"endFrameIndex": end_frame_index,
|
|
881
|
+
},
|
|
882
|
+
}
|
|
883
|
+
],
|
|
884
|
+
}
|
|
885
|
+
return await self._dispatch_video(VIDEO_EDIT_URL, body)
|
|
886
|
+
|
|
887
|
+
async def resolve_scene(
|
|
888
|
+
self, project_id: str, workflow_ids: list[str]
|
|
889
|
+
) -> dict[str, Any]:
|
|
890
|
+
"""Resolve the scene that holds a set of workflows, via
|
|
891
|
+
``POST /v1/flow/projects/{id}/scenes`` with ``{workflowIds:[...]}``.
|
|
892
|
+
|
|
893
|
+
This is how the web client learns a clip's scene id before extending it
|
|
894
|
+
— Flow places every generated video in an auto-created scene but never
|
|
895
|
+
echoes the id anywhere else. Returns ``{raw, scene_id, position}`` where
|
|
896
|
+
``position`` is the next free slot in the scene's timeline (the count of
|
|
897
|
+
workflows already in it), or ``{raw, error}``.
|
|
898
|
+
"""
|
|
899
|
+
cleaned = [w for w in workflow_ids if isinstance(w, str) and w]
|
|
900
|
+
if not cleaned:
|
|
901
|
+
return {"raw": None, "error": "missing_workflow_ids"}
|
|
902
|
+
url = f"{FLOW_API_BASE}/v1/flow/projects/{project_id}/scenes"
|
|
903
|
+
resp = await self._client.api_request(
|
|
904
|
+
url=url,
|
|
905
|
+
method="POST",
|
|
906
|
+
headers=dict(_API_HEADERS),
|
|
907
|
+
body={"workflowIds": cleaned},
|
|
908
|
+
)
|
|
909
|
+
if isinstance(resp, dict) and resp.get("error"):
|
|
910
|
+
return {"raw": resp, "error": resp["error"]}
|
|
911
|
+
inner_err = _extract_inner_api_error(resp)
|
|
912
|
+
if inner_err:
|
|
913
|
+
return {"raw": resp, "error": inner_err}
|
|
914
|
+
data = resp.get("data") if isinstance(resp, dict) else None
|
|
915
|
+
scene = data.get("scene") if isinstance(data, dict) else None
|
|
916
|
+
scene_id = scene.get("sceneId") if isinstance(scene, dict) else None
|
|
917
|
+
if not isinstance(scene_id, str) or not scene_id:
|
|
918
|
+
return {"raw": resp, "error": "no_scene_id_in_response"}
|
|
919
|
+
sws = data.get("sceneWorkflows") if isinstance(data, dict) else None
|
|
920
|
+
position = len(sws) if isinstance(sws, list) else 1
|
|
921
|
+
return {"raw": resp, "scene_id": scene_id, "position": position}
|
|
922
|
+
|
|
923
|
+
async def fetch_video_meta(
|
|
924
|
+
self, media_ids: list[str], project_id: str
|
|
925
|
+
) -> dict[str, Any]:
|
|
926
|
+
"""Fetch rich metadata for finished video media via the media-based
|
|
927
|
+
``batchCheckAsync`` (``{media:[{name,projectId}]}``) — distinct from the
|
|
928
|
+
operation-based poll. This is the only call that returns the workflow id,
|
|
929
|
+
operation/generation id, scene id, and dimensions needed to drive
|
|
930
|
+
extend/edit. Returns ``{raw, meta: {media_id: {...}}}``.
|
|
931
|
+
"""
|
|
932
|
+
cleaned = [m for m in media_ids if isinstance(m, str) and m]
|
|
933
|
+
if not cleaned:
|
|
934
|
+
return {"raw": None, "meta": {}}
|
|
935
|
+
body = {"media": [{"name": m, "projectId": project_id} for m in cleaned]}
|
|
936
|
+
resp = await self._client.api_request(
|
|
937
|
+
url=VIDEO_POLL_URL,
|
|
938
|
+
method="POST",
|
|
939
|
+
headers=dict(_API_HEADERS),
|
|
940
|
+
body=body,
|
|
941
|
+
)
|
|
942
|
+
meta: dict[str, dict[str, Any]] = {}
|
|
943
|
+
data = resp.get("data") if isinstance(resp, dict) else None
|
|
944
|
+
media = data.get("media") if isinstance(data, dict) else None
|
|
945
|
+
for m in media or []:
|
|
946
|
+
if not isinstance(m, dict):
|
|
947
|
+
continue
|
|
948
|
+
name = m.get("name")
|
|
949
|
+
if not isinstance(name, str):
|
|
950
|
+
continue
|
|
951
|
+
video = m.get("video") if isinstance(m.get("video"), dict) else {}
|
|
952
|
+
op = video.get("operation") if isinstance(video.get("operation"), dict) else {}
|
|
953
|
+
dims = video.get("dimensions") if isinstance(video.get("dimensions"), dict) else {}
|
|
954
|
+
meta[name] = {
|
|
955
|
+
"workflow_id": m.get("workflowId"),
|
|
956
|
+
"operation_id": op.get("name"),
|
|
957
|
+
"scene_id": m.get("sceneId") or m.get("sceneIdContext"),
|
|
958
|
+
"length": dims.get("length"),
|
|
959
|
+
}
|
|
960
|
+
return {"raw": resp, "meta": meta}
|
|
961
|
+
|
|
962
|
+
async def _dispatch_video(self, url: str, body: dict[str, Any]) -> dict[str, Any]:
|
|
963
|
+
"""Shared submit + response-shaping for the workflow-schema video
|
|
964
|
+
dispatchers (extend / edit). Sends the captcha-gated request and
|
|
965
|
+
returns ``{raw, operation_names, workflows}`` or ``{raw, error}``."""
|
|
966
|
+
resp = await self._client.api_request(
|
|
967
|
+
url=url,
|
|
968
|
+
method="POST",
|
|
969
|
+
headers=dict(_API_HEADERS),
|
|
970
|
+
body=body,
|
|
971
|
+
captcha_action=CAPTCHA_VIDEO,
|
|
972
|
+
)
|
|
973
|
+
if isinstance(resp, dict) and resp.get("error"):
|
|
974
|
+
return {"raw": resp, "error": resp["error"]}
|
|
975
|
+
inner_err = _extract_inner_api_error(resp)
|
|
976
|
+
if inner_err:
|
|
977
|
+
return {"raw": resp, "error": inner_err}
|
|
978
|
+
op_names = extract_operation_names(resp)
|
|
979
|
+
if not op_names:
|
|
980
|
+
return {"raw": resp, "error": "no_operations_in_response"}
|
|
981
|
+
out: dict[str, Any] = {"raw": resp, "operation_names": op_names}
|
|
982
|
+
workflows = extract_video_workflows(resp)
|
|
983
|
+
if workflows:
|
|
984
|
+
out["workflows"] = workflows
|
|
985
|
+
return out
|
|
986
|
+
|
|
987
|
+
async def check_async(
|
|
988
|
+
self,
|
|
989
|
+
operation_names: list[str],
|
|
990
|
+
workflows: Optional[list[dict[str, Any]]] = None,
|
|
991
|
+
project_id: Optional[str] = None,
|
|
992
|
+
) -> dict[str, Any]:
|
|
993
|
+
"""Poll one or more video operations. No captcha.
|
|
994
|
+
|
|
995
|
+
Returns ``{raw, operations: [{name, done, media_entries}]}`` — one
|
|
996
|
+
entry per input operation. ``media_entries`` is a list of
|
|
997
|
+
``{media_id, url, mediaType}`` ready for ``media.ingest_urls``.
|
|
998
|
+
|
|
999
|
+
``workflows`` (optional) carries ``{name, primary_media_id}`` pairs
|
|
1000
|
+
from the NEW low-priority response. When provided, every workflow
|
|
1001
|
+
entry is polled against ``/v1/media/<primary_media_id>`` and the
|
|
1002
|
+
result is merged into the same ``operations`` shape so the caller
|
|
1003
|
+
is schema-agnostic.
|
|
1004
|
+
"""
|
|
1005
|
+
ops_summary: list[dict[str, Any]] = []
|
|
1006
|
+
raw_old: Any = None
|
|
1007
|
+
# Names that came from workflows are NOT valid operation handles —
|
|
1008
|
+
# don't dispatch them to batchCheckAsync (Flow would 400).
|
|
1009
|
+
workflow_names = {w["name"] for w in (workflows or []) if isinstance(w, dict) and w.get("name")}
|
|
1010
|
+
old_names = [n for n in operation_names if n not in workflow_names]
|
|
1011
|
+
if old_names:
|
|
1012
|
+
body = {
|
|
1013
|
+
"operations": [
|
|
1014
|
+
{"operation": {"name": name}} for name in old_names
|
|
1015
|
+
]
|
|
1016
|
+
}
|
|
1017
|
+
raw_old = await self._client.api_request(
|
|
1018
|
+
url=VIDEO_POLL_URL,
|
|
1019
|
+
method="POST",
|
|
1020
|
+
headers=dict(_API_HEADERS),
|
|
1021
|
+
body=body,
|
|
1022
|
+
)
|
|
1023
|
+
if isinstance(raw_old, dict) and raw_old.get("error"):
|
|
1024
|
+
return {"raw": raw_old, "error": raw_old["error"]}
|
|
1025
|
+
ops_summary.extend(
|
|
1026
|
+
extract_video_operations(raw_old, requested=old_names)
|
|
1027
|
+
)
|
|
1028
|
+
|
|
1029
|
+
raw_workflows: list[dict[str, Any]] = []
|
|
1030
|
+
if workflows:
|
|
1031
|
+
wf_summary, raw_workflows = await self._poll_workflows(
|
|
1032
|
+
workflows, project_id
|
|
1033
|
+
)
|
|
1034
|
+
ops_summary.extend(wf_summary)
|
|
1035
|
+
|
|
1036
|
+
# Preserve the original input order so callers (worker) can keep
|
|
1037
|
+
# positional alignment with their per-op state.
|
|
1038
|
+
order = {name: i for i, name in enumerate(operation_names)}
|
|
1039
|
+
ops_summary.sort(key=lambda op: order.get(op.get("name"), 1 << 30))
|
|
1040
|
+
|
|
1041
|
+
raw_out: dict[str, Any] = {}
|
|
1042
|
+
if raw_old is not None:
|
|
1043
|
+
raw_out["operations_poll"] = raw_old
|
|
1044
|
+
if raw_workflows:
|
|
1045
|
+
raw_out["workflow_polls"] = raw_workflows
|
|
1046
|
+
return {"raw": raw_out or raw_old, "operations": ops_summary}
|
|
1047
|
+
|
|
1048
|
+
async def _poll_workflows(
|
|
1049
|
+
self,
|
|
1050
|
+
workflows: list[dict[str, Any]],
|
|
1051
|
+
project_id: Optional[str] = None,
|
|
1052
|
+
) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]:
|
|
1053
|
+
"""Single poll pass for workflow-mode (Low Priority) submissions.
|
|
1054
|
+
|
|
1055
|
+
Flow's web client checks these via
|
|
1056
|
+
``POST /v1/video:batchCheckAsyncVideoGenerationStatus`` with a
|
|
1057
|
+
``{media:[{name, projectId}]}`` body and reads
|
|
1058
|
+
``mediaMetadata.mediaStatus.mediaGenerationStatus``
|
|
1059
|
+
(SCHEDULED → ACTIVE → SUCCESSFUL). The older ``GET /v1/media/<id>``
|
|
1060
|
+
status probe now returns ``400 INVALID_ARGUMENT`` while the media is
|
|
1061
|
+
still scheduled, which aborted otherwise-fine jobs — so status MUST
|
|
1062
|
+
come from the batch-check call. Only once the media is SUCCESSFUL do
|
|
1063
|
+
we fetch the encoded MP4 bytes (``GET /v1/media/<id>`` → base64).
|
|
1064
|
+
|
|
1065
|
+
Returns ``(ops_summary, raw_polls)`` mirroring the OLD-schema
|
|
1066
|
+
``check_async`` contract: one entry per workflow with
|
|
1067
|
+
``{name, done, media_entries, status, error}``. The poll loop in
|
|
1068
|
+
the worker calls this repeatedly via ``check_async`` until ``done``.
|
|
1069
|
+
"""
|
|
1070
|
+
ops_summary: list[dict[str, Any]] = []
|
|
1071
|
+
raw_polls: list[dict[str, Any]] = []
|
|
1072
|
+
for wf in workflows:
|
|
1073
|
+
if not isinstance(wf, dict):
|
|
1074
|
+
continue
|
|
1075
|
+
name = wf.get("name")
|
|
1076
|
+
mid = wf.get("primary_media_id")
|
|
1077
|
+
if not isinstance(name, str) or not isinstance(mid, str) or not mid:
|
|
1078
|
+
continue
|
|
1079
|
+
pending = {
|
|
1080
|
+
"name": name, "done": False, "media_entries": [],
|
|
1081
|
+
"status": None, "error": None,
|
|
1082
|
+
}
|
|
1083
|
+
# 1. Status via the media-based batchCheckAsync (same call the web
|
|
1084
|
+
# client uses). projectId is required by Flow.
|
|
1085
|
+
try:
|
|
1086
|
+
status_resp = await self._client.api_request(
|
|
1087
|
+
url=VIDEO_POLL_URL,
|
|
1088
|
+
method="POST",
|
|
1089
|
+
headers=dict(_API_HEADERS),
|
|
1090
|
+
body={"media": [{"name": mid, "projectId": project_id}]},
|
|
1091
|
+
)
|
|
1092
|
+
except Exception as exc: # noqa: BLE001
|
|
1093
|
+
logger.warning("workflow status poll error for %s: %s", mid[:8], exc)
|
|
1094
|
+
ops_summary.append(pending)
|
|
1095
|
+
continue
|
|
1096
|
+
raw_polls.append({"name": name, "media_id": mid, "resp": status_resp})
|
|
1097
|
+
|
|
1098
|
+
m = _first_media_entry(status_resp, mid)
|
|
1099
|
+
m_status = (
|
|
1100
|
+
((m.get("mediaMetadata") or {}).get("mediaStatus") or {})
|
|
1101
|
+
.get("mediaGenerationStatus")
|
|
1102
|
+
)
|
|
1103
|
+
if m_status == "MEDIA_GENERATION_STATUS_FAILED":
|
|
1104
|
+
ops_summary.append({
|
|
1105
|
+
**pending, "done": True, "status": m_status,
|
|
1106
|
+
"error": "MEDIA_GENERATION_STATUS_FAILED",
|
|
1107
|
+
})
|
|
1108
|
+
continue
|
|
1109
|
+
if m_status != "MEDIA_GENERATION_STATUS_SUCCESSFUL":
|
|
1110
|
+
# SCHEDULED / ACTIVE / PENDING / unknown — keep polling.
|
|
1111
|
+
ops_summary.append(pending)
|
|
1112
|
+
continue
|
|
1113
|
+
|
|
1114
|
+
# 2. SUCCESSFUL. Prefer a direct URL if the status payload carries
|
|
1115
|
+
# one; otherwise fetch the encoded MP4 bytes.
|
|
1116
|
+
url = _media_url_from_entry(m)
|
|
1117
|
+
if url:
|
|
1118
|
+
ops_summary.append({
|
|
1119
|
+
"name": name, "done": True,
|
|
1120
|
+
"media_entries": [
|
|
1121
|
+
{"media_id": mid, "url": url, "mediaType": "video"}
|
|
1122
|
+
],
|
|
1123
|
+
"status": m_status, "error": None,
|
|
1124
|
+
})
|
|
1125
|
+
continue
|
|
1126
|
+
entry = await self._fetch_workflow_bytes(mid)
|
|
1127
|
+
if entry is None:
|
|
1128
|
+
# Status is SUCCESSFUL but bytes not materialised yet — keep
|
|
1129
|
+
# polling rather than failing.
|
|
1130
|
+
ops_summary.append(pending)
|
|
1131
|
+
continue
|
|
1132
|
+
ops_summary.append({
|
|
1133
|
+
"name": name, "done": True, "media_entries": [entry],
|
|
1134
|
+
"status": m_status, "error": None,
|
|
1135
|
+
})
|
|
1136
|
+
return ops_summary, raw_polls
|
|
1137
|
+
|
|
1138
|
+
async def _fetch_workflow_bytes(self, mid: str) -> Optional[dict[str, Any]]:
|
|
1139
|
+
"""Fetch a finished workflow clip's encoded MP4 via ``GET /v1/media/<id>``.
|
|
1140
|
+
|
|
1141
|
+
Returns a media entry (``{media_id, url, mediaType, encoded_video}``)
|
|
1142
|
+
once Flow serves the base64 MP4, or ``None`` while the bytes are not
|
|
1143
|
+
yet available (any error / non-``ftyp`` payload) so the caller keeps
|
|
1144
|
+
polling.
|
|
1145
|
+
"""
|
|
1146
|
+
import base64 as _b64
|
|
1147
|
+
|
|
1148
|
+
try:
|
|
1149
|
+
resp = await self._client.api_request(
|
|
1150
|
+
url=_media_get_url(mid),
|
|
1151
|
+
method="GET",
|
|
1152
|
+
headers=dict(_API_HEADERS),
|
|
1153
|
+
body=None,
|
|
1154
|
+
)
|
|
1155
|
+
except Exception as exc: # noqa: BLE001
|
|
1156
|
+
logger.warning("workflow byte fetch error for %s: %s", mid[:8], exc)
|
|
1157
|
+
return None
|
|
1158
|
+
if not isinstance(resp, dict):
|
|
1159
|
+
return None
|
|
1160
|
+
status_code = resp.get("status")
|
|
1161
|
+
if isinstance(status_code, int) and status_code >= 400:
|
|
1162
|
+
return None # not ready yet
|
|
1163
|
+
data = resp.get("data") if isinstance(resp.get("data"), dict) else {}
|
|
1164
|
+
video_block = data.get("video") if isinstance(data.get("video"), dict) else {}
|
|
1165
|
+
encoded = (
|
|
1166
|
+
video_block.get("encodedVideo") if isinstance(video_block, dict) else None
|
|
1167
|
+
)
|
|
1168
|
+
if not isinstance(encoded, str) or not encoded:
|
|
1169
|
+
return None
|
|
1170
|
+
try:
|
|
1171
|
+
binary = _b64.b64decode(encoded, validate=False)
|
|
1172
|
+
except Exception: # noqa: BLE001
|
|
1173
|
+
return None
|
|
1174
|
+
# MP4 box layout: bytes 4..8 == "ftyp" on a complete file.
|
|
1175
|
+
if not (len(binary) >= 12 and binary[4:8] == b"ftyp"):
|
|
1176
|
+
return None
|
|
1177
|
+
fife = (
|
|
1178
|
+
video_block.get("fifeUrl") if isinstance(video_block, dict) else None
|
|
1179
|
+
) or data.get("fifeUrl")
|
|
1180
|
+
return {
|
|
1181
|
+
"media_id": mid,
|
|
1182
|
+
"url": fife if isinstance(fife, str) else None,
|
|
1183
|
+
"mediaType": "video",
|
|
1184
|
+
"encoded_video": encoded,
|
|
1185
|
+
}
|
|
1186
|
+
|
|
1187
|
+
# ── image generation (api_request + captcha) ───────────────────────────
|
|
1188
|
+
async def gen_image(
|
|
1189
|
+
self,
|
|
1190
|
+
prompt: str,
|
|
1191
|
+
project_id: str,
|
|
1192
|
+
aspect_ratio: str = "IMAGE_ASPECT_RATIO_LANDSCAPE",
|
|
1193
|
+
paygate_tier: Optional[str] = None,
|
|
1194
|
+
ref_media_ids: Optional[list[str]] = None,
|
|
1195
|
+
variant_count: int = 1,
|
|
1196
|
+
character_media_ids: Optional[list[str]] = None, # legacy alias
|
|
1197
|
+
prompts: Optional[list[str]] = None,
|
|
1198
|
+
image_model: Optional[str] = None,
|
|
1199
|
+
) -> dict[str, Any]:
|
|
1200
|
+
"""Generate ``variant_count`` images (1-4). When ``ref_media_ids`` is
|
|
1201
|
+
provided, every request item is augmented with ``imageInputs`` so Flow
|
|
1202
|
+
conditions the result on those upstream images (any combination of
|
|
1203
|
+
character / image / visual_asset upstream nodes — all become
|
|
1204
|
+
``IMAGE_INPUT_TYPE_REFERENCE`` inputs).
|
|
1205
|
+
|
|
1206
|
+
Multiple variants are produced by replicating the request item with
|
|
1207
|
+
distinct seeds — Flow returns one entry in ``data.media[]`` per
|
|
1208
|
+
request item.
|
|
1209
|
+
|
|
1210
|
+
``paygate_tier`` is required. See ``gen_video`` for rationale.
|
|
1211
|
+
"""
|
|
1212
|
+
if paygate_tier is None:
|
|
1213
|
+
raise ValueError("paygate_tier is required — caller must resolve before dispatch")
|
|
1214
|
+
n = max(1, min(int(variant_count), MAX_VARIANT_COUNT))
|
|
1215
|
+
ts = int(time.time() * 1000)
|
|
1216
|
+
ctx = _client_context(project_id, paygate_tier)
|
|
1217
|
+
model_name = resolve_image_model(image_model)
|
|
1218
|
+
# Accept the legacy `character_media_ids` kwarg as a fallback.
|
|
1219
|
+
merged_refs = ref_media_ids if ref_media_ids is not None else character_media_ids
|
|
1220
|
+
image_inputs = None
|
|
1221
|
+
if merged_refs:
|
|
1222
|
+
image_inputs = [
|
|
1223
|
+
{"name": mid, "imageInputType": "IMAGE_INPUT_TYPE_REFERENCE"}
|
|
1224
|
+
for mid in merged_refs
|
|
1225
|
+
]
|
|
1226
|
+
|
|
1227
|
+
# Per-variant prompts: when the caller provides `prompts`, each
|
|
1228
|
+
# request_item gets its own text so the 4 variants render with
|
|
1229
|
+
# different poses instead of 4 seeds of one stance. Missing /
|
|
1230
|
+
# short list falls back to the single `prompt` for that slot.
|
|
1231
|
+
per_item_prompts: list[str] = []
|
|
1232
|
+
for i in range(n):
|
|
1233
|
+
if prompts and i < len(prompts) and isinstance(prompts[i], str) and prompts[i]:
|
|
1234
|
+
per_item_prompts.append(prompts[i])
|
|
1235
|
+
else:
|
|
1236
|
+
per_item_prompts.append(prompt)
|
|
1237
|
+
|
|
1238
|
+
requests_arr: list[dict[str, Any]] = []
|
|
1239
|
+
for i in range(n):
|
|
1240
|
+
seed = (ts + i * 9973) % 1_000_000 # any deterministic spread is fine
|
|
1241
|
+
item: dict[str, Any] = {
|
|
1242
|
+
"clientContext": {**ctx, "sessionId": f";{ts + i}"},
|
|
1243
|
+
"seed": seed,
|
|
1244
|
+
"structuredPrompt": {"parts": [{"text": per_item_prompts[i]}]},
|
|
1245
|
+
"imageAspectRatio": aspect_ratio,
|
|
1246
|
+
"imageModelName": model_name,
|
|
1247
|
+
}
|
|
1248
|
+
if image_inputs is not None:
|
|
1249
|
+
item["imageInputs"] = list(image_inputs)
|
|
1250
|
+
requests_arr.append(item)
|
|
1251
|
+
|
|
1252
|
+
body = {
|
|
1253
|
+
"clientContext": ctx,
|
|
1254
|
+
"mediaGenerationContext": {"batchId": str(uuid.uuid4())},
|
|
1255
|
+
"useNewMedia": True,
|
|
1256
|
+
"requests": requests_arr,
|
|
1257
|
+
}
|
|
1258
|
+
|
|
1259
|
+
resp = await self._client.api_request(
|
|
1260
|
+
url=_generate_images_url(project_id),
|
|
1261
|
+
method="POST",
|
|
1262
|
+
headers=dict(_API_HEADERS),
|
|
1263
|
+
body=body,
|
|
1264
|
+
captcha_action=CAPTCHA_IMAGE,
|
|
1265
|
+
)
|
|
1266
|
+
if isinstance(resp, dict) and resp.get("error"):
|
|
1267
|
+
return {"raw": resp, "error": resp["error"]}
|
|
1268
|
+
inner_err = _extract_inner_api_error(resp)
|
|
1269
|
+
if inner_err:
|
|
1270
|
+
return {"raw": resp, "error": inner_err}
|
|
1271
|
+
|
|
1272
|
+
entries = extract_media_entries(resp)
|
|
1273
|
+
media_ids = [e["media_id"] for e in entries]
|
|
1274
|
+
return {"raw": resp, "media_ids": media_ids, "media_entries": entries}
|
|
1275
|
+
|
|
1276
|
+
# ── image refine (edit_image) ──────────────────────────────────────────
|
|
1277
|
+
async def edit_image(
|
|
1278
|
+
self,
|
|
1279
|
+
prompt: str,
|
|
1280
|
+
project_id: str,
|
|
1281
|
+
source_media_id: str,
|
|
1282
|
+
ref_media_ids: Optional[list[str]] = None,
|
|
1283
|
+
aspect_ratio: str = "IMAGE_ASPECT_RATIO_LANDSCAPE",
|
|
1284
|
+
paygate_tier: Optional[str] = None,
|
|
1285
|
+
image_model: Optional[str] = None,
|
|
1286
|
+
) -> dict[str, Any]:
|
|
1287
|
+
"""Refine an existing image with an optional list of reference media.
|
|
1288
|
+
|
|
1289
|
+
Order of ``imageInputs`` matters — flowkit puts BASE_IMAGE first so
|
|
1290
|
+
Flow knows which is the canonical source.
|
|
1291
|
+
|
|
1292
|
+
``paygate_tier`` is required. See ``gen_video`` for rationale.
|
|
1293
|
+
"""
|
|
1294
|
+
if paygate_tier is None:
|
|
1295
|
+
raise ValueError("paygate_tier is required — caller must resolve before dispatch")
|
|
1296
|
+
ts = int(time.time() * 1000)
|
|
1297
|
+
ctx = _client_context(project_id, paygate_tier)
|
|
1298
|
+
model_name = resolve_image_model(image_model)
|
|
1299
|
+
|
|
1300
|
+
image_inputs: list[dict[str, Any]] = [
|
|
1301
|
+
{"name": source_media_id, "imageInputType": "IMAGE_INPUT_TYPE_BASE_IMAGE"}
|
|
1302
|
+
]
|
|
1303
|
+
for mid in ref_media_ids or []:
|
|
1304
|
+
if isinstance(mid, str) and mid:
|
|
1305
|
+
image_inputs.append(
|
|
1306
|
+
{"name": mid, "imageInputType": "IMAGE_INPUT_TYPE_REFERENCE"}
|
|
1307
|
+
)
|
|
1308
|
+
|
|
1309
|
+
request_item = {
|
|
1310
|
+
"clientContext": {**ctx, "sessionId": f";{ts}"},
|
|
1311
|
+
"seed": ts % 1_000_000,
|
|
1312
|
+
"structuredPrompt": {"parts": [{"text": prompt}]},
|
|
1313
|
+
"imageAspectRatio": aspect_ratio,
|
|
1314
|
+
"imageModelName": model_name,
|
|
1315
|
+
"imageInputs": image_inputs,
|
|
1316
|
+
}
|
|
1317
|
+
body = {
|
|
1318
|
+
"clientContext": ctx,
|
|
1319
|
+
"mediaGenerationContext": {"batchId": str(uuid.uuid4())},
|
|
1320
|
+
"useNewMedia": True,
|
|
1321
|
+
"requests": [request_item],
|
|
1322
|
+
}
|
|
1323
|
+
|
|
1324
|
+
resp = await self._client.api_request(
|
|
1325
|
+
url=_generate_images_url(project_id),
|
|
1326
|
+
method="POST",
|
|
1327
|
+
headers=dict(_API_HEADERS),
|
|
1328
|
+
body=body,
|
|
1329
|
+
captcha_action=CAPTCHA_IMAGE,
|
|
1330
|
+
)
|
|
1331
|
+
if isinstance(resp, dict) and resp.get("error"):
|
|
1332
|
+
return {"raw": resp, "error": resp["error"]}
|
|
1333
|
+
inner_err = _extract_inner_api_error(resp)
|
|
1334
|
+
if inner_err:
|
|
1335
|
+
return {"raw": resp, "error": inner_err}
|
|
1336
|
+
|
|
1337
|
+
entries = extract_media_entries(resp)
|
|
1338
|
+
media_ids = [e["media_id"] for e in entries]
|
|
1339
|
+
return {"raw": resp, "media_ids": media_ids, "media_entries": entries}
|
|
1340
|
+
|
|
1341
|
+
# ── image upload (api_request, no captcha) ─────────────────────────────
|
|
1342
|
+
async def upload_image(
|
|
1343
|
+
self,
|
|
1344
|
+
image_base64: str,
|
|
1345
|
+
mime_type: str,
|
|
1346
|
+
project_id: str,
|
|
1347
|
+
file_name: str = "upload.png",
|
|
1348
|
+
) -> dict[str, Any]:
|
|
1349
|
+
"""Upload a user-provided image into a Flow project. Returns
|
|
1350
|
+
``{raw, media_id}`` on success or ``{raw, error}`` on failure.
|
|
1351
|
+
|
|
1352
|
+
``image_base64`` should be a base64-encoded payload (no data: prefix).
|
|
1353
|
+
Flow accepts the image bytes inline in the JSON body.
|
|
1354
|
+
"""
|
|
1355
|
+
body = {
|
|
1356
|
+
"clientContext": {
|
|
1357
|
+
"projectId": str(project_id),
|
|
1358
|
+
"tool": "PINHOLE",
|
|
1359
|
+
},
|
|
1360
|
+
"fileName": file_name,
|
|
1361
|
+
"imageBytes": image_base64,
|
|
1362
|
+
"isHidden": False,
|
|
1363
|
+
"isUserUploaded": True,
|
|
1364
|
+
"mimeType": mime_type,
|
|
1365
|
+
}
|
|
1366
|
+
resp = await self._client.api_request(
|
|
1367
|
+
url=UPLOAD_IMAGE_URL,
|
|
1368
|
+
method="POST",
|
|
1369
|
+
headers=dict(_API_HEADERS),
|
|
1370
|
+
body=body,
|
|
1371
|
+
)
|
|
1372
|
+
if isinstance(resp, dict) and resp.get("error"):
|
|
1373
|
+
return {"raw": resp, "error": resp["error"]}
|
|
1374
|
+
|
|
1375
|
+
media_id = _extract_uploaded_media_id(resp)
|
|
1376
|
+
if media_id is None:
|
|
1377
|
+
# Flow returned 200 but no usable media handle. Most common cause
|
|
1378
|
+
# is a silent content-filter rejection (logos/watermarks/branded
|
|
1379
|
+
# imagery from product CDNs); next most common is a Flow schema
|
|
1380
|
+
# change. Log the full payload so the operator can tell which.
|
|
1381
|
+
logger.error(
|
|
1382
|
+
"upload_image: no media_id in response (project_id=%s, "
|
|
1383
|
+
"file=%s, mime=%s) — raw=%r",
|
|
1384
|
+
project_id, file_name, mime_type, resp,
|
|
1385
|
+
)
|
|
1386
|
+
return {"raw": resp, "error": "no_media_id_in_upload_response"}
|
|
1387
|
+
return {"raw": resp, "media_id": media_id}
|
|
1388
|
+
|
|
1389
|
+
# ── video upload (resumable, via extension) ────────────────────────────
|
|
1390
|
+
async def upload_video(
|
|
1391
|
+
self,
|
|
1392
|
+
video_base64: str,
|
|
1393
|
+
mime_type: str,
|
|
1394
|
+
project_id: str,
|
|
1395
|
+
file_name: str = "upload.mp4",
|
|
1396
|
+
end_offset: str = "8s",
|
|
1397
|
+
) -> dict[str, Any]:
|
|
1398
|
+
"""Upload a user-provided video into a Flow project, returning
|
|
1399
|
+
``{raw, media_id, workflow_id, width, height}`` on success or
|
|
1400
|
+
``{raw, error}`` on failure.
|
|
1401
|
+
|
|
1402
|
+
Unlike images, video upload can't go through ``api_request`` — it's a
|
|
1403
|
+
resumable, cookie-authed handshake against labs.google with a binary
|
|
1404
|
+
PUT body. The extension runs that handshake in the browser session (a
|
|
1405
|
+
``upload_video`` bridge method) and hands back Flow's final
|
|
1406
|
+
``{mediaServerId, workflowServerId, videoWidth, videoHeight}`` payload.
|
|
1407
|
+
We then register the trim window via ``videoFx.updateVideoOffset`` so
|
|
1408
|
+
the upload becomes a referenceable workflow media — exactly what the
|
|
1409
|
+
Flow web client does after a drag-and-drop upload.
|
|
1410
|
+
|
|
1411
|
+
``video_base64`` should be a base64-encoded payload (no data: prefix).
|
|
1412
|
+
"""
|
|
1413
|
+
resp = await self._client.upload_video(
|
|
1414
|
+
project_id=project_id,
|
|
1415
|
+
file_name=file_name,
|
|
1416
|
+
content_type=mime_type,
|
|
1417
|
+
data_base64=video_base64,
|
|
1418
|
+
)
|
|
1419
|
+
if isinstance(resp, dict) and resp.get("error"):
|
|
1420
|
+
return {"raw": resp, "error": resp["error"]}
|
|
1421
|
+
|
|
1422
|
+
final = resp.get("data") if isinstance(resp, dict) else None
|
|
1423
|
+
final = final if isinstance(final, dict) else {}
|
|
1424
|
+
media_id = final.get("mediaServerId")
|
|
1425
|
+
if not isinstance(media_id, str) or not media_id:
|
|
1426
|
+
logger.error(
|
|
1427
|
+
"upload_video: no mediaServerId in response (project_id=%s, "
|
|
1428
|
+
"file=%s, mime=%s) — raw=%r",
|
|
1429
|
+
project_id, file_name, mime_type, resp,
|
|
1430
|
+
)
|
|
1431
|
+
return {"raw": resp, "error": "no_media_id_in_upload_response"}
|
|
1432
|
+
|
|
1433
|
+
# Register the trim window (videoFx.updateVideoOffset). This is what
|
|
1434
|
+
# promotes the raw upload into a usable workflow media; the media id is
|
|
1435
|
+
# already valid, so a failure here is soft — surface it in raw for
|
|
1436
|
+
# diagnosis but still return the id.
|
|
1437
|
+
offset_resp = await self._client.trpc_request(
|
|
1438
|
+
url=TRPC_UPDATE_VIDEO_OFFSET,
|
|
1439
|
+
method="POST",
|
|
1440
|
+
headers=_TRPC_HEADERS,
|
|
1441
|
+
body={
|
|
1442
|
+
"json": {
|
|
1443
|
+
"mediaId": media_id,
|
|
1444
|
+
"startOffset": "0s",
|
|
1445
|
+
"endOffset": end_offset,
|
|
1446
|
+
}
|
|
1447
|
+
},
|
|
1448
|
+
)
|
|
1449
|
+
workflow_id = final.get("workflowServerId")
|
|
1450
|
+
return {
|
|
1451
|
+
"raw": {"upload": resp, "offset": offset_resp},
|
|
1452
|
+
"media_id": media_id,
|
|
1453
|
+
"workflow_id": workflow_id if isinstance(workflow_id, str) else None,
|
|
1454
|
+
"width": final.get("videoWidth"),
|
|
1455
|
+
"height": final.get("videoHeight"),
|
|
1456
|
+
}
|
|
1457
|
+
|
|
1458
|
+
|
|
1459
|
+
def _extract_project_id(resp: Any) -> Optional[str]:
|
|
1460
|
+
"""TRPC createProject nests the projectId quite deeply."""
|
|
1461
|
+
try:
|
|
1462
|
+
data = resp.get("data") if isinstance(resp, dict) else None
|
|
1463
|
+
return data["result"]["data"]["json"]["result"]["projectId"] # type: ignore[index]
|
|
1464
|
+
except (KeyError, TypeError):
|
|
1465
|
+
return None
|
|
1466
|
+
|
|
1467
|
+
|
|
1468
|
+
_VALID_TIERS = {"PAYGATE_TIER_ONE", "PAYGATE_TIER_TWO"}
|
|
1469
|
+
|
|
1470
|
+
|
|
1471
|
+
def _extract_uploaded_media_id(resp: Any) -> Optional[str]:
|
|
1472
|
+
"""uploadImage returns ``data.media.name`` as the new media_id."""
|
|
1473
|
+
if not isinstance(resp, dict):
|
|
1474
|
+
return None
|
|
1475
|
+
data = resp.get("data")
|
|
1476
|
+
if not isinstance(data, dict):
|
|
1477
|
+
return None
|
|
1478
|
+
media = data.get("media")
|
|
1479
|
+
if isinstance(media, dict):
|
|
1480
|
+
name = media.get("name")
|
|
1481
|
+
if isinstance(name, str) and name:
|
|
1482
|
+
return name
|
|
1483
|
+
return None
|
|
1484
|
+
|
|
1485
|
+
|
|
1486
|
+
def extract_operation_names(resp: Any) -> list[str]:
|
|
1487
|
+
"""Pull ``operation.name`` out of a ``batchAsyncGenerateVideo*`` response.
|
|
1488
|
+
|
|
1489
|
+
Supports two shapes:
|
|
1490
|
+
|
|
1491
|
+
* **OLD** (Lite / Fast / Quality) — ``data.operations[].operation.name``.
|
|
1492
|
+
* **NEW** (Low Priority — ``_low_priority`` / ``_relaxed`` models) —
|
|
1493
|
+
``data.workflows[].name``. Workflows don't have ``operation.name``;
|
|
1494
|
+
callers that need to poll must also read ``primaryMediaId`` from
|
|
1495
|
+
``workflows[].metadata`` (see ``extract_video_workflows``).
|
|
1496
|
+
"""
|
|
1497
|
+
if not isinstance(resp, dict):
|
|
1498
|
+
return []
|
|
1499
|
+
data = resp.get("data")
|
|
1500
|
+
if not isinstance(data, dict):
|
|
1501
|
+
return []
|
|
1502
|
+
names: list[str] = []
|
|
1503
|
+
ops = data.get("operations")
|
|
1504
|
+
if isinstance(ops, list):
|
|
1505
|
+
for op in ops:
|
|
1506
|
+
if not isinstance(op, dict):
|
|
1507
|
+
continue
|
|
1508
|
+
inner = op.get("operation") if isinstance(op.get("operation"), dict) else None
|
|
1509
|
+
if inner is None:
|
|
1510
|
+
# Some variants inline the name at top level.
|
|
1511
|
+
name = op.get("name")
|
|
1512
|
+
else:
|
|
1513
|
+
name = inner.get("name")
|
|
1514
|
+
if isinstance(name, str) and name:
|
|
1515
|
+
names.append(name)
|
|
1516
|
+
if names:
|
|
1517
|
+
return names
|
|
1518
|
+
# NEW workflow schema. The tracking key is the workflow id — take it from
|
|
1519
|
+
# `media[].workflowId` (preferred, pairs with extract_video_workflows) or
|
|
1520
|
+
# fall back to `workflows[].name`. The two normally coincide; keeping them
|
|
1521
|
+
# consistent ensures check_async routes these to workflow polling, not the
|
|
1522
|
+
# operation poll (which would 400 on a workflow id).
|
|
1523
|
+
media = data.get("media")
|
|
1524
|
+
if isinstance(media, list):
|
|
1525
|
+
for m in media:
|
|
1526
|
+
wf = m.get("workflowId") if isinstance(m, dict) else None
|
|
1527
|
+
if isinstance(wf, str) and wf and wf not in names:
|
|
1528
|
+
names.append(wf)
|
|
1529
|
+
if names:
|
|
1530
|
+
return names
|
|
1531
|
+
workflows = data.get("workflows")
|
|
1532
|
+
if isinstance(workflows, list):
|
|
1533
|
+
for wf in workflows:
|
|
1534
|
+
if not isinstance(wf, dict):
|
|
1535
|
+
continue
|
|
1536
|
+
name = wf.get("name")
|
|
1537
|
+
if isinstance(name, str) and name:
|
|
1538
|
+
names.append(name)
|
|
1539
|
+
return names
|
|
1540
|
+
|
|
1541
|
+
|
|
1542
|
+
def extract_video_workflows(resp: Any) -> list[dict[str, Any]]:
|
|
1543
|
+
"""Pull workflow entries out of a NEW-schema video submit response.
|
|
1544
|
+
|
|
1545
|
+
Returns ``[{"name": <workflow_name>, "primary_media_id": <new media uuid>},
|
|
1546
|
+
...]`` — the media id to poll for the freshly generated output. Empty list
|
|
1547
|
+
when the response is OLD-schema (operations-based) or has no workflows.
|
|
1548
|
+
|
|
1549
|
+
Source of the media id matters: ``data.media[].name`` is the NEW output, and
|
|
1550
|
+
we pair it to its workflow via ``media[].workflowId``. For **edit** the
|
|
1551
|
+
workflow's ``metadata.primaryMediaId`` still points at the SOURCE clip, so
|
|
1552
|
+
polling that would return the un-edited input — we must poll ``media[].name``
|
|
1553
|
+
instead. For t2v/r2v/extend the two coincide, so ``media[]`` is correct in
|
|
1554
|
+
all cases. ``workflows[].metadata.primaryMediaId`` is only a fallback for a
|
|
1555
|
+
response that omits ``media[]``.
|
|
1556
|
+
"""
|
|
1557
|
+
if not isinstance(resp, dict):
|
|
1558
|
+
return []
|
|
1559
|
+
data = resp.get("data")
|
|
1560
|
+
if not isinstance(data, dict):
|
|
1561
|
+
return []
|
|
1562
|
+
|
|
1563
|
+
out: list[dict[str, Any]] = []
|
|
1564
|
+
media = data.get("media")
|
|
1565
|
+
if isinstance(media, list):
|
|
1566
|
+
for m in media:
|
|
1567
|
+
if not isinstance(m, dict):
|
|
1568
|
+
continue
|
|
1569
|
+
mid = m.get("name")
|
|
1570
|
+
wf = m.get("workflowId")
|
|
1571
|
+
if isinstance(mid, str) and mid and isinstance(wf, str) and wf:
|
|
1572
|
+
out.append({"name": wf, "primary_media_id": mid})
|
|
1573
|
+
if out:
|
|
1574
|
+
return out
|
|
1575
|
+
|
|
1576
|
+
# Fallback: derive from workflows[].metadata.primaryMediaId (no media[]).
|
|
1577
|
+
workflows = data.get("workflows")
|
|
1578
|
+
if not isinstance(workflows, list):
|
|
1579
|
+
return []
|
|
1580
|
+
for wf in workflows:
|
|
1581
|
+
if not isinstance(wf, dict):
|
|
1582
|
+
continue
|
|
1583
|
+
name = wf.get("name")
|
|
1584
|
+
meta = wf.get("metadata") if isinstance(wf.get("metadata"), dict) else {}
|
|
1585
|
+
primary = meta.get("primaryMediaId") if isinstance(meta, dict) else None
|
|
1586
|
+
if isinstance(name, str) and name and isinstance(primary, str) and primary:
|
|
1587
|
+
out.append({"name": name, "primary_media_id": primary})
|
|
1588
|
+
return out
|
|
1589
|
+
|
|
1590
|
+
|
|
1591
|
+
def extract_video_operations(
|
|
1592
|
+
resp: Any, *, requested: list[str]
|
|
1593
|
+
) -> list[dict[str, Any]]:
|
|
1594
|
+
"""Summarise a ``batchCheckAsync`` response.
|
|
1595
|
+
|
|
1596
|
+
Flow's response shape is::
|
|
1597
|
+
|
|
1598
|
+
{"data": {"operations": [{
|
|
1599
|
+
"status": "MEDIA_GENERATION_STATUS_{PENDING,SUCCESSFUL,FAILED}",
|
|
1600
|
+
"operation": {"name": "<id>", "metadata": {"video": {
|
|
1601
|
+
"mediaId": "<uuid>", "fifeUrl": "https://flow-content..."
|
|
1602
|
+
}}}
|
|
1603
|
+
}]}}
|
|
1604
|
+
|
|
1605
|
+
flowkit treats ``MEDIA_GENERATION_STATUS_SUCCESSFUL`` as terminal-success;
|
|
1606
|
+
we mirror that.
|
|
1607
|
+
|
|
1608
|
+
Returns one entry per *requested* operation name, in order. Missing
|
|
1609
|
+
operations are reported as ``done=False`` so the caller can keep polling.
|
|
1610
|
+
"""
|
|
1611
|
+
by_name: dict[str, dict[str, Any]] = {}
|
|
1612
|
+
if isinstance(resp, dict):
|
|
1613
|
+
data = resp.get("data")
|
|
1614
|
+
if isinstance(data, dict):
|
|
1615
|
+
ops = data.get("operations")
|
|
1616
|
+
if isinstance(ops, list):
|
|
1617
|
+
for op in ops:
|
|
1618
|
+
if not isinstance(op, dict):
|
|
1619
|
+
continue
|
|
1620
|
+
inner = op.get("operation") if isinstance(op.get("operation"), dict) else op
|
|
1621
|
+
name = inner.get("name") if isinstance(inner, dict) else None
|
|
1622
|
+
if not isinstance(name, str):
|
|
1623
|
+
continue
|
|
1624
|
+
meta = (inner.get("metadata") or {}) if isinstance(inner, dict) else {}
|
|
1625
|
+
video_meta = meta.get("video") if isinstance(meta.get("video"), dict) else {}
|
|
1626
|
+
media_id = video_meta.get("mediaId") if isinstance(video_meta, dict) else None
|
|
1627
|
+
fife = video_meta.get("fifeUrl") if isinstance(video_meta, dict) else None
|
|
1628
|
+
# Flow's video poll response usually omits `mediaId` and only
|
|
1629
|
+
# provides `mediaGenerationId` (base64 protobuf, NOT a UUID).
|
|
1630
|
+
# The actual UUID is embedded in the `fifeUrl` path. Recover it.
|
|
1631
|
+
if not (isinstance(media_id, str) and media_id):
|
|
1632
|
+
recovered = _media_id_from_url(fife if isinstance(fife, str) else None)
|
|
1633
|
+
if recovered is None and isinstance(video_meta, dict):
|
|
1634
|
+
recovered = _media_id_from_url(video_meta.get("servingBaseUri"))
|
|
1635
|
+
if recovered is not None:
|
|
1636
|
+
media_id = recovered
|
|
1637
|
+
# Flow puts the status at the *top* of each op envelope,
|
|
1638
|
+
# not on the inner operation object — bug we hit before.
|
|
1639
|
+
status = op.get("status") if isinstance(op.get("status"), str) else None
|
|
1640
|
+
# Per-op terminal failure (e.g. PUBLIC_ERROR_AUDIO_FILTERED).
|
|
1641
|
+
# Flow puts the error on the inner operation object as
|
|
1642
|
+
# ``{code, message}``. We surface it so the worker can bail
|
|
1643
|
+
# instead of polling for the full timeout.
|
|
1644
|
+
op_err: Optional[str] = None
|
|
1645
|
+
inner_err = inner.get("error") if isinstance(inner, dict) else None
|
|
1646
|
+
if isinstance(inner_err, dict):
|
|
1647
|
+
msg = inner_err.get("message") or inner_err.get("status") or "operation_failed"
|
|
1648
|
+
op_err = str(msg)
|
|
1649
|
+
if status == "MEDIA_GENERATION_STATUS_FAILED" and op_err is None:
|
|
1650
|
+
op_err = "MEDIA_GENERATION_STATUS_FAILED"
|
|
1651
|
+
done_flag = (
|
|
1652
|
+
status == "MEDIA_GENERATION_STATUS_SUCCESSFUL"
|
|
1653
|
+
or status == "MEDIA_GENERATION_STATUS_FAILED"
|
|
1654
|
+
or bool(inner.get("done"))
|
|
1655
|
+
or bool(media_id and fife)
|
|
1656
|
+
)
|
|
1657
|
+
entries = []
|
|
1658
|
+
if (
|
|
1659
|
+
done_flag
|
|
1660
|
+
and op_err is None
|
|
1661
|
+
and isinstance(media_id, str)
|
|
1662
|
+
):
|
|
1663
|
+
entries.append(
|
|
1664
|
+
{
|
|
1665
|
+
"media_id": media_id,
|
|
1666
|
+
"url": fife if isinstance(fife, str) else None,
|
|
1667
|
+
"mediaType": "video",
|
|
1668
|
+
}
|
|
1669
|
+
)
|
|
1670
|
+
by_name[name] = {
|
|
1671
|
+
"name": name,
|
|
1672
|
+
"done": done_flag,
|
|
1673
|
+
"media_entries": entries,
|
|
1674
|
+
"status": status,
|
|
1675
|
+
"error": op_err,
|
|
1676
|
+
}
|
|
1677
|
+
|
|
1678
|
+
out: list[dict[str, Any]] = []
|
|
1679
|
+
for name in requested:
|
|
1680
|
+
out.append(
|
|
1681
|
+
by_name.get(
|
|
1682
|
+
name, {"name": name, "done": False, "media_entries": []}
|
|
1683
|
+
)
|
|
1684
|
+
)
|
|
1685
|
+
return out
|
|
1686
|
+
|
|
1687
|
+
|
|
1688
|
+
def _extract_media_ids(resp: Any) -> list[str]:
|
|
1689
|
+
return [e["media_id"] for e in extract_media_entries(resp)]
|
|
1690
|
+
|
|
1691
|
+
|
|
1692
|
+
def extract_media_entries(resp: Any) -> list[dict[str, Any]]:
|
|
1693
|
+
"""Pull media entries out of a ``batchGenerateImages`` response.
|
|
1694
|
+
|
|
1695
|
+
Returns a list of ``{media_id, url, mediaType}`` dicts suitable for
|
|
1696
|
+
``media.ingest_urls``. ``url`` may be missing if Flow didn't include a
|
|
1697
|
+
``fifeUrl`` for some reason — caller should handle that.
|
|
1698
|
+
"""
|
|
1699
|
+
if not isinstance(resp, dict):
|
|
1700
|
+
return []
|
|
1701
|
+
data = resp.get("data")
|
|
1702
|
+
if not isinstance(data, dict):
|
|
1703
|
+
return []
|
|
1704
|
+
media = data.get("media")
|
|
1705
|
+
if not isinstance(media, list):
|
|
1706
|
+
return []
|
|
1707
|
+
out: list[dict[str, Any]] = []
|
|
1708
|
+
for m in media:
|
|
1709
|
+
if not isinstance(m, dict):
|
|
1710
|
+
continue
|
|
1711
|
+
media_id = m.get("name")
|
|
1712
|
+
if not isinstance(media_id, str) or not media_id:
|
|
1713
|
+
continue
|
|
1714
|
+
url: Optional[str] = None
|
|
1715
|
+
kind = "image"
|
|
1716
|
+
image = m.get("image") if isinstance(m.get("image"), dict) else None
|
|
1717
|
+
video = m.get("video") if isinstance(m.get("video"), dict) else None
|
|
1718
|
+
if image is not None:
|
|
1719
|
+
gen = image.get("generatedImage")
|
|
1720
|
+
if isinstance(gen, dict):
|
|
1721
|
+
candidate = gen.get("fifeUrl")
|
|
1722
|
+
if isinstance(candidate, str):
|
|
1723
|
+
url = candidate
|
|
1724
|
+
kind = "image"
|
|
1725
|
+
elif video is not None:
|
|
1726
|
+
gen = video.get("generatedVideo") or video.get("generatedImage")
|
|
1727
|
+
if isinstance(gen, dict):
|
|
1728
|
+
candidate = gen.get("fifeUrl")
|
|
1729
|
+
if isinstance(candidate, str):
|
|
1730
|
+
url = candidate
|
|
1731
|
+
kind = "video"
|
|
1732
|
+
out.append({"media_id": media_id, "url": url, "mediaType": kind})
|
|
1733
|
+
return out
|
|
1734
|
+
|
|
1735
|
+
|
|
1736
|
+
_sdk: Optional[FlowSDK] = None
|
|
1737
|
+
|
|
1738
|
+
|
|
1739
|
+
def get_flow_sdk() -> FlowSDK:
|
|
1740
|
+
global _sdk
|
|
1741
|
+
if _sdk is None:
|
|
1742
|
+
_sdk = FlowSDK()
|
|
1743
|
+
return _sdk
|